log-foundry 0.10.2.dev67__tar.gz → 0.10.2.dev69__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {log_foundry-0.10.2.dev67 → log_foundry-0.10.2.dev69}/PKG-INFO +50 -10
- {log_foundry-0.10.2.dev67 → log_foundry-0.10.2.dev69}/README.md +49 -9
- {log_foundry-0.10.2.dev67 → log_foundry-0.10.2.dev69}/pyproject.toml +1 -1
- {log_foundry-0.10.2.dev67 → log_foundry-0.10.2.dev69}/src/log_foundry/__init__.py +9 -1
- {log_foundry-0.10.2.dev67 → log_foundry-0.10.2.dev69}/src/log_foundry/api.py +28 -2
- {log_foundry-0.10.2.dev67 → log_foundry-0.10.2.dev69}/src/log_foundry/context.py +21 -0
- {log_foundry-0.10.2.dev67 → log_foundry-0.10.2.dev69}/src/log_foundry/decorator.py +260 -11
- {log_foundry-0.10.2.dev67 → log_foundry-0.10.2.dev69}/src/log_foundry/model.py +20 -0
- {log_foundry-0.10.2.dev67 → log_foundry-0.10.2.dev69}/src/log_foundry/sinks/multi.py +36 -8
- {log_foundry-0.10.2.dev67 → log_foundry-0.10.2.dev69}/src/log_foundry/worker.py +19 -0
- {log_foundry-0.10.2.dev67 → log_foundry-0.10.2.dev69}/LICENSE +0 -0
- {log_foundry-0.10.2.dev67 → log_foundry-0.10.2.dev69}/src/log_foundry/_diag.py +0 -0
- {log_foundry-0.10.2.dev67 → log_foundry-0.10.2.dev69}/src/log_foundry/_fork.py +0 -0
- {log_foundry-0.10.2.dev67 → log_foundry-0.10.2.dev69}/src/log_foundry/_lifecycle.py +0 -0
- {log_foundry-0.10.2.dev67 → log_foundry-0.10.2.dev69}/src/log_foundry/config.py +0 -0
- {log_foundry-0.10.2.dev67 → log_foundry-0.10.2.dev69}/src/log_foundry/console.py +0 -0
- {log_foundry-0.10.2.dev67 → log_foundry-0.10.2.dev69}/src/log_foundry/ids.py +0 -0
- {log_foundry-0.10.2.dev67 → log_foundry-0.10.2.dev69}/src/log_foundry/py.typed +0 -0
- {log_foundry-0.10.2.dev67 → log_foundry-0.10.2.dev69}/src/log_foundry/results.py +0 -0
- {log_foundry-0.10.2.dev67 → log_foundry-0.10.2.dev69}/src/log_foundry/sanitize.py +0 -0
- {log_foundry-0.10.2.dev67 → log_foundry-0.10.2.dev69}/src/log_foundry/sinks/__init__.py +0 -0
- {log_foundry-0.10.2.dev67 → log_foundry-0.10.2.dev69}/src/log_foundry/sinks/_batch.py +0 -0
- {log_foundry-0.10.2.dev67 → log_foundry-0.10.2.dev69}/src/log_foundry/sinks/_chunk.py +0 -0
- {log_foundry-0.10.2.dev67 → log_foundry-0.10.2.dev69}/src/log_foundry/sinks/_retry.py +0 -0
- {log_foundry-0.10.2.dev67 → log_foundry-0.10.2.dev69}/src/log_foundry/sinks/_socket.py +0 -0
- {log_foundry-0.10.2.dev67 → log_foundry-0.10.2.dev69}/src/log_foundry/sinks/_time.py +0 -0
- {log_foundry-0.10.2.dev67 → log_foundry-0.10.2.dev69}/src/log_foundry/sinks/base.py +0 -0
- {log_foundry-0.10.2.dev67 → log_foundry-0.10.2.dev69}/src/log_foundry/sinks/callback.py +0 -0
- {log_foundry-0.10.2.dev67 → log_foundry-0.10.2.dev69}/src/log_foundry/sinks/clickhouse.py +0 -0
- {log_foundry-0.10.2.dev67 → log_foundry-0.10.2.dev69}/src/log_foundry/sinks/datadog.py +0 -0
- {log_foundry-0.10.2.dev67 → log_foundry-0.10.2.dev69}/src/log_foundry/sinks/elasticsearch.py +0 -0
- {log_foundry-0.10.2.dev67 → log_foundry-0.10.2.dev69}/src/log_foundry/sinks/eventhubs.py +0 -0
- {log_foundry-0.10.2.dev67 → log_foundry-0.10.2.dev69}/src/log_foundry/sinks/file.py +0 -0
- {log_foundry-0.10.2.dev67 → log_foundry-0.10.2.dev69}/src/log_foundry/sinks/filtering.py +0 -0
- {log_foundry-0.10.2.dev67 → log_foundry-0.10.2.dev69}/src/log_foundry/sinks/firehose.py +0 -0
- {log_foundry-0.10.2.dev67 → log_foundry-0.10.2.dev69}/src/log_foundry/sinks/honeycomb.py +0 -0
- {log_foundry-0.10.2.dev67 → log_foundry-0.10.2.dev69}/src/log_foundry/sinks/http.py +0 -0
- {log_foundry-0.10.2.dev67 → log_foundry-0.10.2.dev69}/src/log_foundry/sinks/kafka.py +0 -0
- {log_foundry-0.10.2.dev67 → log_foundry-0.10.2.dev69}/src/log_foundry/sinks/kinesis.py +0 -0
- {log_foundry-0.10.2.dev67 → log_foundry-0.10.2.dev69}/src/log_foundry/sinks/logging_sink.py +0 -0
- {log_foundry-0.10.2.dev67 → log_foundry-0.10.2.dev69}/src/log_foundry/sinks/logstash.py +0 -0
- {log_foundry-0.10.2.dev67 → log_foundry-0.10.2.dev69}/src/log_foundry/sinks/loki.py +0 -0
- {log_foundry-0.10.2.dev67 → log_foundry-0.10.2.dev69}/src/log_foundry/sinks/memory.py +0 -0
- {log_foundry-0.10.2.dev67 → log_foundry-0.10.2.dev69}/src/log_foundry/sinks/mongodb.py +0 -0
- {log_foundry-0.10.2.dev67 → log_foundry-0.10.2.dev69}/src/log_foundry/sinks/nats.py +0 -0
- {log_foundry-0.10.2.dev67 → log_foundry-0.10.2.dev69}/src/log_foundry/sinks/newrelic.py +0 -0
- {log_foundry-0.10.2.dev67 → log_foundry-0.10.2.dev69}/src/log_foundry/sinks/null.py +0 -0
- {log_foundry-0.10.2.dev67 → log_foundry-0.10.2.dev69}/src/log_foundry/sinks/postgres.py +0 -0
- {log_foundry-0.10.2.dev67 → log_foundry-0.10.2.dev69}/src/log_foundry/sinks/pubsub.py +0 -0
- {log_foundry-0.10.2.dev67 → log_foundry-0.10.2.dev69}/src/log_foundry/sinks/rabbitmq.py +0 -0
- {log_foundry-0.10.2.dev67 → log_foundry-0.10.2.dev69}/src/log_foundry/sinks/redis.py +0 -0
- {log_foundry-0.10.2.dev67 → log_foundry-0.10.2.dev69}/src/log_foundry/sinks/sentry.py +0 -0
- {log_foundry-0.10.2.dev67 → log_foundry-0.10.2.dev69}/src/log_foundry/sinks/sns.py +0 -0
- {log_foundry-0.10.2.dev67 → log_foundry-0.10.2.dev69}/src/log_foundry/sinks/splunk.py +0 -0
- {log_foundry-0.10.2.dev67 → log_foundry-0.10.2.dev69}/src/log_foundry/sinks/sqlite.py +0 -0
- {log_foundry-0.10.2.dev67 → log_foundry-0.10.2.dev69}/src/log_foundry/sinks/sqs.py +0 -0
- {log_foundry-0.10.2.dev67 → log_foundry-0.10.2.dev69}/src/log_foundry/sinks/stdout.py +0 -0
- {log_foundry-0.10.2.dev67 → log_foundry-0.10.2.dev69}/src/log_foundry/sinks/syslog.py +0 -0
- {log_foundry-0.10.2.dev67 → log_foundry-0.10.2.dev69}/src/log_foundry/sinks/transform.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: log-foundry
|
|
3
|
-
Version: 0.10.2.
|
|
3
|
+
Version: 0.10.2.dev69
|
|
4
4
|
Summary: Generate logs for your console and JSON events for downstream consumption.
|
|
5
5
|
License-Expression: MIT
|
|
6
6
|
License-File: LICENSE
|
|
@@ -233,6 +233,15 @@ is FIFO, everything submitted before the call is necessarily ahead of that marke
|
|
|
233
233
|
exactly why the guarantee is "events submitted before this call", and why concurrent
|
|
234
234
|
submissions from other threads may or may not be included.
|
|
235
235
|
|
|
236
|
+
**`flush()` also sweeps the spans that are still open**, so an in-span event does not have to wait
|
|
237
|
+
for its span to close to be delivered. The span stays open and usable afterwards: its events go
|
|
238
|
+
now and its `span.end` arrives later, in its own batch. Two consequences worth knowing. Boundary
|
|
239
|
+
events swept this way carry the baggage as of the **flush** rather than as of the close, since
|
|
240
|
+
that completion has to happen before they leave. And the sweep reaches only the **calling
|
|
241
|
+
context's** spans — `contextvars` offers no way to enumerate another thread's or task's context,
|
|
242
|
+
so a `flush()` in a handler that fanned out to tasks does not reach what those tasks have
|
|
243
|
+
buffered; their events arrive when their own spans close.
|
|
244
|
+
|
|
236
245
|
## Usage
|
|
237
246
|
|
|
238
247
|
### `configure(...)`
|
|
@@ -413,15 +422,23 @@ def enqueue_check(location: str) -> None:
|
|
|
413
422
|
|
|
414
423
|
```python
|
|
415
424
|
@lf.trace
|
|
416
|
-
def
|
|
425
|
+
def _handler(event, context):
|
|
417
426
|
lf.continue_trace(event.get("traceparent"), baggage=event.get("baggage"))
|
|
418
427
|
lf.info("inspecting") # same trace_id as the producer; parent is its span
|
|
428
|
+
return inspect(event)
|
|
429
|
+
|
|
430
|
+
|
|
431
|
+
def handler(event, context): # the entry point, deliberately not decorated
|
|
419
432
|
try:
|
|
420
|
-
return
|
|
433
|
+
return _handler(event, context)
|
|
421
434
|
finally:
|
|
422
|
-
lf.flush()
|
|
435
|
+
lf.flush() # the span has closed, so its events are drained
|
|
423
436
|
```
|
|
424
437
|
|
|
438
|
+
`flush()` goes **outside** the traced function, not in its `finally`. An in-span event lives on the
|
|
439
|
+
span until the span *closes*, and `flush()` drains the queue — so a `flush()` inside the span has
|
|
440
|
+
nothing to drain yet.
|
|
441
|
+
|
|
425
442
|
| Call | Does |
|
|
426
443
|
|---|---|
|
|
427
444
|
| `continue_trace(traceparent=None, *, trace_id=None, parent_span_id=None, baggage=None)` | Adopt an inbound context. Returns a `ContinueResult`: truthy if adopted, else falsy with `reason` of `"nothing-supplied"` or `"rejected"`. The verdict is about the **trace context** — `baggage=` is merged independently and does not make it truthy. Never raises. |
|
|
@@ -953,6 +970,7 @@ if (
|
|
|
953
970
|
h.dropped or h.failed_batches or h.stopped_reason or h.incomplete_swaps
|
|
954
971
|
or (h.sink and (h.sink.dropped or h.sink.failed))
|
|
955
972
|
or (h.retired and h.submitted_after_shutdown)
|
|
973
|
+
or h.orphan_lost or h.in_span_lost
|
|
956
974
|
):
|
|
957
975
|
... # logs were silently lost — worth an alert
|
|
958
976
|
```
|
|
@@ -974,8 +992,14 @@ They tell you different things, and they want different responses:
|
|
|
974
992
|
| `retired` + `submitted_after_shutdown` | `shutdown()` was called and the process **kept logging**. Those events are queued where nothing will drain them — total loss, for as long as the process runs. | Use `flush()`, not `shutdown()`, in a process that logs again. This is the serverless mistake below. |
|
|
975
993
|
| `incomplete_swaps` | A late `configure(sink=...)` could not confirm the previous sink was drained. The swap took effect; that sink was left open and some queued events may have gone to the new one. | Investigate the previous sink — it was hung or failing. Configure the sink before the first log where you can. |
|
|
976
994
|
| `inherited_sink` | This process is delivering to a sink it **inherited across a `fork`** and may not release, so it will not be closed here. Not a loss and not an alert term. | Nothing, usually. It explains a handle still open after `shutdown()`, and tells you a deployment shares one sink across a fork at all. `True` for a shared `StdoutSink` too, whose `close()` only flushes — so a `True` is not by itself evidence that anything is held. If you want the child to own its transport, build the sink in the worker process (see Forking). |
|
|
995
|
+
| `orphan_lost` | An event logged **with no active span** never reached the sink. That call emits on your own thread with no worker behind it, so no other field here can carry it — it is not a batch, there was no retry, and there may be no worker at all. Covers a sink that failed to *construct* as well as one that raised. | Fix the destination, or the data. The stderr line names the exception type. If a process logs this way at all, this is the field to alert on: nothing else describes that path. |
|
|
996
|
+
| `in_span_lost` | An event logged **inside a span** could not be built — a value that could not be turned into an event. Always the data, never the destination: the in-span path cannot fail at delivery, which is `failed_batches`. | Fix the call site. Passing a non-string message (an exception object, say) is the common cause. |
|
|
977
997
|
| `closing_sinks` | Swapped-out sinks inside `close()` **right now** — a live gauge, not a counter, and the only field that falls as well as rises. Non-zero on a single read is normal during a swap. | Nothing, unless it stays non-zero. That means a destination is stuck in `close()` and will not release its resources. |
|
|
978
998
|
|
|
999
|
+
`orphan_lost` and `in_span_lost` are deliberately two fields and their sum is deliberately not
|
|
1000
|
+
reported. They aggregate different failure populations — one can mean the destination *or* the
|
|
1001
|
+
data, the other can only mean the data — so a single number would hide which fix applies.
|
|
1002
|
+
|
|
979
1003
|
`h.sink` is a `SinkLosses(dropped, failed)` or `None` — `None` when no worker exists yet, or when
|
|
980
1004
|
the configured sink reports nothing (`losses()` is optional). Note the two `dropped` fields count
|
|
981
1005
|
different things: the worker's is backpressure at *its* queue, the sink's is an event that never
|
|
@@ -996,14 +1020,21 @@ logged, and after a clean `shutdown()`, so a plain truthiness check is safe. Wit
|
|
|
996
1020
|
thread showed up only indirectly, as `dropped` climbing once the queue filled — the wrong signal,
|
|
997
1021
|
pointing at the wrong fix.
|
|
998
1022
|
|
|
1023
|
+
`orphan_lost` climbing is on its own a reason to look: unlike `dropped`, it is never
|
|
1024
|
+
backpressure and never transient. Each increment is one event that reached no destination, and on
|
|
1025
|
+
a process that logs only outside a span it is the *only* field that can say so.
|
|
1026
|
+
|
|
999
1027
|
`retired` is deliberately **not** alerted on by itself. A process that shuts down and then stops
|
|
1000
1028
|
logging is doing the right thing; it is the *pair* — retired, and still being handed events — that
|
|
1001
1029
|
means every log line since the shutdown has gone nowhere. That state used to read as perfectly
|
|
1002
1030
|
healthy: `stopped_reason` is `None` after a clean shutdown, and the queue simply grows.
|
|
1003
1031
|
|
|
1004
|
-
`retired`
|
|
1005
|
-
that only ever calls `info()`/`error()` outside a span emits
|
|
1006
|
-
worker, so
|
|
1032
|
+
`retired`, `orphan_lost` and `in_span_lost` are the fields reported for a process that has **no
|
|
1033
|
+
worker at all**. A program that only ever calls `info()`/`error()` outside a span emits
|
|
1034
|
+
synchronously and builds no background worker, so the rest describe something that does not exist
|
|
1035
|
+
and read zero — which is why that path needs counters of its own. Until it had them, such a process
|
|
1036
|
+
reported `queued=0 dropped=0 failed_batches=0 stopped_reason=None` over total, permanent loss, and
|
|
1037
|
+
the only thing that said otherwise was a line on stderr. Its
|
|
1007
1038
|
`shutdown()` still closes the sink, exactly once and without starting a thread, and `retired` reads
|
|
1008
1039
|
`True` afterwards rather than staying vacuously `False`. `submitted_after_shutdown` stays `0` there
|
|
1009
1040
|
by design: a later level call is *refused* at the closed sink and announced on stderr — if the sink
|
|
@@ -1124,19 +1155,28 @@ from log_foundry.sinks.sqs import SQSSink
|
|
|
1124
1155
|
lf.configure(service="billing-api", env="prod", sink=SQSSink(queue_url=QUEUE_URL))
|
|
1125
1156
|
|
|
1126
1157
|
@lf.trace
|
|
1127
|
-
def
|
|
1158
|
+
def _handler(event, context):
|
|
1128
1159
|
lf.info("received", records=len(event["Records"]))
|
|
1160
|
+
return do_work(event)
|
|
1161
|
+
|
|
1162
|
+
|
|
1163
|
+
def handler(event, context):
|
|
1164
|
+
# NOT decorated, so the span closes when `_handler` returns and its events reach the queue
|
|
1165
|
+
# before `flush()` runs. A `flush()` *inside* the traced function has nothing to drain yet.
|
|
1129
1166
|
try:
|
|
1130
|
-
return
|
|
1167
|
+
return _handler(event, context)
|
|
1131
1168
|
finally:
|
|
1132
1169
|
# In `finally`: the failed invocation is the one worth logging. NEVER shutdown() here —
|
|
1133
1170
|
# the worker does not come back, and every later invocation on this warm container
|
|
1134
1171
|
# would log nothing.
|
|
1135
1172
|
drained = lf.flush()
|
|
1136
1173
|
h = lf.health()
|
|
1137
|
-
if not drained or h.failed_batches or h.dropped or h.stopped_reason or h.retired
|
|
1174
|
+
if (not drained or h.failed_batches or h.dropped or h.stopped_reason or h.retired
|
|
1175
|
+
or h.orphan_lost or h.in_span_lost):
|
|
1138
1176
|
# `drained` covers this invocation's tail; the counters cover anything the worker
|
|
1139
1177
|
# lost earlier — a batch its own interval trigger already gave up on, for instance.
|
|
1178
|
+
# `orphan_lost`/`in_span_lost` cover the two paths no worker field can describe: a
|
|
1179
|
+
# log made outside any span, and an event that could not be built.
|
|
1140
1180
|
# `h.retired` catches the mistake above: inside a handler it can only mean something
|
|
1141
1181
|
# called shutdown(), and from here on this container logs nothing.
|
|
1142
1182
|
# Emitting this through your platform's own logger keeps it outside the pipeline
|
|
@@ -197,6 +197,15 @@ is FIFO, everything submitted before the call is necessarily ahead of that marke
|
|
|
197
197
|
exactly why the guarantee is "events submitted before this call", and why concurrent
|
|
198
198
|
submissions from other threads may or may not be included.
|
|
199
199
|
|
|
200
|
+
**`flush()` also sweeps the spans that are still open**, so an in-span event does not have to wait
|
|
201
|
+
for its span to close to be delivered. The span stays open and usable afterwards: its events go
|
|
202
|
+
now and its `span.end` arrives later, in its own batch. Two consequences worth knowing. Boundary
|
|
203
|
+
events swept this way carry the baggage as of the **flush** rather than as of the close, since
|
|
204
|
+
that completion has to happen before they leave. And the sweep reaches only the **calling
|
|
205
|
+
context's** spans — `contextvars` offers no way to enumerate another thread's or task's context,
|
|
206
|
+
so a `flush()` in a handler that fanned out to tasks does not reach what those tasks have
|
|
207
|
+
buffered; their events arrive when their own spans close.
|
|
208
|
+
|
|
200
209
|
## Usage
|
|
201
210
|
|
|
202
211
|
### `configure(...)`
|
|
@@ -377,15 +386,23 @@ def enqueue_check(location: str) -> None:
|
|
|
377
386
|
|
|
378
387
|
```python
|
|
379
388
|
@lf.trace
|
|
380
|
-
def
|
|
389
|
+
def _handler(event, context):
|
|
381
390
|
lf.continue_trace(event.get("traceparent"), baggage=event.get("baggage"))
|
|
382
391
|
lf.info("inspecting") # same trace_id as the producer; parent is its span
|
|
392
|
+
return inspect(event)
|
|
393
|
+
|
|
394
|
+
|
|
395
|
+
def handler(event, context): # the entry point, deliberately not decorated
|
|
383
396
|
try:
|
|
384
|
-
return
|
|
397
|
+
return _handler(event, context)
|
|
385
398
|
finally:
|
|
386
|
-
lf.flush()
|
|
399
|
+
lf.flush() # the span has closed, so its events are drained
|
|
387
400
|
```
|
|
388
401
|
|
|
402
|
+
`flush()` goes **outside** the traced function, not in its `finally`. An in-span event lives on the
|
|
403
|
+
span until the span *closes*, and `flush()` drains the queue — so a `flush()` inside the span has
|
|
404
|
+
nothing to drain yet.
|
|
405
|
+
|
|
389
406
|
| Call | Does |
|
|
390
407
|
|---|---|
|
|
391
408
|
| `continue_trace(traceparent=None, *, trace_id=None, parent_span_id=None, baggage=None)` | Adopt an inbound context. Returns a `ContinueResult`: truthy if adopted, else falsy with `reason` of `"nothing-supplied"` or `"rejected"`. The verdict is about the **trace context** — `baggage=` is merged independently and does not make it truthy. Never raises. |
|
|
@@ -917,6 +934,7 @@ if (
|
|
|
917
934
|
h.dropped or h.failed_batches or h.stopped_reason or h.incomplete_swaps
|
|
918
935
|
or (h.sink and (h.sink.dropped or h.sink.failed))
|
|
919
936
|
or (h.retired and h.submitted_after_shutdown)
|
|
937
|
+
or h.orphan_lost or h.in_span_lost
|
|
920
938
|
):
|
|
921
939
|
... # logs were silently lost — worth an alert
|
|
922
940
|
```
|
|
@@ -938,8 +956,14 @@ They tell you different things, and they want different responses:
|
|
|
938
956
|
| `retired` + `submitted_after_shutdown` | `shutdown()` was called and the process **kept logging**. Those events are queued where nothing will drain them — total loss, for as long as the process runs. | Use `flush()`, not `shutdown()`, in a process that logs again. This is the serverless mistake below. |
|
|
939
957
|
| `incomplete_swaps` | A late `configure(sink=...)` could not confirm the previous sink was drained. The swap took effect; that sink was left open and some queued events may have gone to the new one. | Investigate the previous sink — it was hung or failing. Configure the sink before the first log where you can. |
|
|
940
958
|
| `inherited_sink` | This process is delivering to a sink it **inherited across a `fork`** and may not release, so it will not be closed here. Not a loss and not an alert term. | Nothing, usually. It explains a handle still open after `shutdown()`, and tells you a deployment shares one sink across a fork at all. `True` for a shared `StdoutSink` too, whose `close()` only flushes — so a `True` is not by itself evidence that anything is held. If you want the child to own its transport, build the sink in the worker process (see Forking). |
|
|
959
|
+
| `orphan_lost` | An event logged **with no active span** never reached the sink. That call emits on your own thread with no worker behind it, so no other field here can carry it — it is not a batch, there was no retry, and there may be no worker at all. Covers a sink that failed to *construct* as well as one that raised. | Fix the destination, or the data. The stderr line names the exception type. If a process logs this way at all, this is the field to alert on: nothing else describes that path. |
|
|
960
|
+
| `in_span_lost` | An event logged **inside a span** could not be built — a value that could not be turned into an event. Always the data, never the destination: the in-span path cannot fail at delivery, which is `failed_batches`. | Fix the call site. Passing a non-string message (an exception object, say) is the common cause. |
|
|
941
961
|
| `closing_sinks` | Swapped-out sinks inside `close()` **right now** — a live gauge, not a counter, and the only field that falls as well as rises. Non-zero on a single read is normal during a swap. | Nothing, unless it stays non-zero. That means a destination is stuck in `close()` and will not release its resources. |
|
|
942
962
|
|
|
963
|
+
`orphan_lost` and `in_span_lost` are deliberately two fields and their sum is deliberately not
|
|
964
|
+
reported. They aggregate different failure populations — one can mean the destination *or* the
|
|
965
|
+
data, the other can only mean the data — so a single number would hide which fix applies.
|
|
966
|
+
|
|
943
967
|
`h.sink` is a `SinkLosses(dropped, failed)` or `None` — `None` when no worker exists yet, or when
|
|
944
968
|
the configured sink reports nothing (`losses()` is optional). Note the two `dropped` fields count
|
|
945
969
|
different things: the worker's is backpressure at *its* queue, the sink's is an event that never
|
|
@@ -960,14 +984,21 @@ logged, and after a clean `shutdown()`, so a plain truthiness check is safe. Wit
|
|
|
960
984
|
thread showed up only indirectly, as `dropped` climbing once the queue filled — the wrong signal,
|
|
961
985
|
pointing at the wrong fix.
|
|
962
986
|
|
|
987
|
+
`orphan_lost` climbing is on its own a reason to look: unlike `dropped`, it is never
|
|
988
|
+
backpressure and never transient. Each increment is one event that reached no destination, and on
|
|
989
|
+
a process that logs only outside a span it is the *only* field that can say so.
|
|
990
|
+
|
|
963
991
|
`retired` is deliberately **not** alerted on by itself. A process that shuts down and then stops
|
|
964
992
|
logging is doing the right thing; it is the *pair* — retired, and still being handed events — that
|
|
965
993
|
means every log line since the shutdown has gone nowhere. That state used to read as perfectly
|
|
966
994
|
healthy: `stopped_reason` is `None` after a clean shutdown, and the queue simply grows.
|
|
967
995
|
|
|
968
|
-
`retired`
|
|
969
|
-
that only ever calls `info()`/`error()` outside a span emits
|
|
970
|
-
worker, so
|
|
996
|
+
`retired`, `orphan_lost` and `in_span_lost` are the fields reported for a process that has **no
|
|
997
|
+
worker at all**. A program that only ever calls `info()`/`error()` outside a span emits
|
|
998
|
+
synchronously and builds no background worker, so the rest describe something that does not exist
|
|
999
|
+
and read zero — which is why that path needs counters of its own. Until it had them, such a process
|
|
1000
|
+
reported `queued=0 dropped=0 failed_batches=0 stopped_reason=None` over total, permanent loss, and
|
|
1001
|
+
the only thing that said otherwise was a line on stderr. Its
|
|
971
1002
|
`shutdown()` still closes the sink, exactly once and without starting a thread, and `retired` reads
|
|
972
1003
|
`True` afterwards rather than staying vacuously `False`. `submitted_after_shutdown` stays `0` there
|
|
973
1004
|
by design: a later level call is *refused* at the closed sink and announced on stderr — if the sink
|
|
@@ -1088,19 +1119,28 @@ from log_foundry.sinks.sqs import SQSSink
|
|
|
1088
1119
|
lf.configure(service="billing-api", env="prod", sink=SQSSink(queue_url=QUEUE_URL))
|
|
1089
1120
|
|
|
1090
1121
|
@lf.trace
|
|
1091
|
-
def
|
|
1122
|
+
def _handler(event, context):
|
|
1092
1123
|
lf.info("received", records=len(event["Records"]))
|
|
1124
|
+
return do_work(event)
|
|
1125
|
+
|
|
1126
|
+
|
|
1127
|
+
def handler(event, context):
|
|
1128
|
+
# NOT decorated, so the span closes when `_handler` returns and its events reach the queue
|
|
1129
|
+
# before `flush()` runs. A `flush()` *inside* the traced function has nothing to drain yet.
|
|
1093
1130
|
try:
|
|
1094
|
-
return
|
|
1131
|
+
return _handler(event, context)
|
|
1095
1132
|
finally:
|
|
1096
1133
|
# In `finally`: the failed invocation is the one worth logging. NEVER shutdown() here —
|
|
1097
1134
|
# the worker does not come back, and every later invocation on this warm container
|
|
1098
1135
|
# would log nothing.
|
|
1099
1136
|
drained = lf.flush()
|
|
1100
1137
|
h = lf.health()
|
|
1101
|
-
if not drained or h.failed_batches or h.dropped or h.stopped_reason or h.retired
|
|
1138
|
+
if (not drained or h.failed_batches or h.dropped or h.stopped_reason or h.retired
|
|
1139
|
+
or h.orphan_lost or h.in_span_lost):
|
|
1102
1140
|
# `drained` covers this invocation's tail; the counters cover anything the worker
|
|
1103
1141
|
# lost earlier — a batch its own interval trigger already gave up on, for instance.
|
|
1142
|
+
# `orphan_lost`/`in_span_lost` cover the two paths no worker field can describe: a
|
|
1143
|
+
# log made outside any span, and an event that could not be built.
|
|
1104
1144
|
# `h.retired` catches the mistake above: inside a handler it can only mean something
|
|
1105
1145
|
# called shutdown(), and from here on this container logs nothing.
|
|
1106
1146
|
# Emitting this through your platform's own logger keeps it outside the pipeline
|
|
@@ -72,9 +72,17 @@ def health() -> Health:
|
|
|
72
72
|
h = log_foundry.health()
|
|
73
73
|
if h.dropped or h.failed_batches or h.stopped_reason or h.incomplete_swaps or (
|
|
74
74
|
h.sink and (h.sink.dropped or h.sink.failed)
|
|
75
|
-
) or (h.retired and h.submitted_after_shutdown):
|
|
75
|
+
) or (h.retired and h.submitted_after_shutdown) or h.orphan_lost or h.in_span_lost:
|
|
76
76
|
... # raise an alert; logs were silently lost
|
|
77
77
|
|
|
78
|
+
``orphan_lost`` and ``in_span_lost`` are the two terms that do **not** describe the worker
|
|
79
|
+
(SPEC-036 FR-003). A level call made with no active span emits on the caller's own thread, and
|
|
80
|
+
an event that cannot be *built* never reaches a queue at all — so every other field here
|
|
81
|
+
describes machinery those two losses never touched, and a process that only logs outside a
|
|
82
|
+
span read all zeros over total loss until they existed. They stay separate because one can
|
|
83
|
+
mean the destination or the data and the other can only mean the data; their sum is a number
|
|
84
|
+
nobody can act on.
|
|
85
|
+
|
|
78
86
|
``retired`` alone is not a fault — a process that shuts down and then stops logging is
|
|
79
87
|
doing the right thing, which is why it is paired with the count rather than alerted on.
|
|
80
88
|
``closing_sinks`` is deliberately absent for the same kind of reason: it is briefly non-zero
|
|
@@ -8,7 +8,7 @@ from log_foundry import _diag, context
|
|
|
8
8
|
from log_foundry.config import _ensure_sink
|
|
9
9
|
from log_foundry.console import ConsoleWriter
|
|
10
10
|
from log_foundry.context import set_baggage
|
|
11
|
-
from log_foundry.decorator import _note_orphan_emit
|
|
11
|
+
from log_foundry.decorator import _note_in_span_loss, _note_orphan_emit, _note_orphan_loss
|
|
12
12
|
from log_foundry.ids import new_span_id, new_trace_id
|
|
13
13
|
from log_foundry.model import Span, build_event
|
|
14
14
|
|
|
@@ -71,6 +71,30 @@ def _log(level: str, message: str, echo: bool, fields: dict[str, object]) -> Non
|
|
|
71
71
|
A guard that reads "this cannot fail" is a guard that stops being true when the code under
|
|
72
72
|
it changes, and this one had already stopped.
|
|
73
73
|
|
|
74
|
+
**An append to a span that has already closed takes the orphan route** (SPEC-036 FR-004).
|
|
75
|
+
``contextvars`` copies the same ``Span`` object into every task created inside a span, so a
|
|
76
|
+
fire-and-forget ``create_task`` can log after its parent returned — and since ``_flush``
|
|
77
|
+
detaches at submit, that append lands in a buffer nothing will ever emit. This is the only
|
|
78
|
+
place that can notice, because nothing in the library reads a span again after it closes; a
|
|
79
|
+
post-hoc check of the buffer has no observer. A span that has closed is not a span, so the
|
|
80
|
+
event becomes a fresh one-event span and inherits the orphan path's accounting entire —
|
|
81
|
+
delivered on success, ``orphan_lost`` on failure. The flag is read before the event is built
|
|
82
|
+
and the append happens after, so a thread *sharing* the span can still slip between the two and
|
|
83
|
+
strand its event in the post-swap list; that window is recorded in ``architecture.md`` §13
|
|
84
|
+
rather than closed, since closing it costs a per-span lock on the hottest path. That is why FR-004 adds no ``Health`` field
|
|
85
|
+
of its own, and why the destination is named here rather than left implied.
|
|
86
|
+
|
|
87
|
+
The cost is stated rather than hidden: the event gets a **fresh** ``trace_id``, so it leaves
|
|
88
|
+
its trace, not merely its span. ``contextvars`` cannot deliver the alternative once the span
|
|
89
|
+
is gone, and the choice is against losing the event outright.
|
|
90
|
+
|
|
91
|
+
**Both losses are counted, and counted apart** (SPEC-036 FR-003). Each guard records before it
|
|
92
|
+
announces, so the stderr line SPEC-025 already writes is unchanged and this adds a counter
|
|
93
|
+
rather than a second announcement. The orphan increment sits in the ``except`` rather than
|
|
94
|
+
after ``sink.emit``, which is what makes a sink that fails to *construct* count too. They stay
|
|
95
|
+
two fields because the populations differ: this branch can fail at the destination, the
|
|
96
|
+
in-span branch can only fail at the data.
|
|
97
|
+
|
|
74
98
|
The echo runs after the emit, so a closed or redirected stream never costs the event itself.
|
|
75
99
|
|
|
76
100
|
Args:
|
|
@@ -89,11 +113,12 @@ def _log(level: str, message: str, echo: bool, fields: dict[str, object]) -> Non
|
|
|
89
113
|
baggage = context._live_baggage()
|
|
90
114
|
span = context.current_span()
|
|
91
115
|
event: dict[str, object] | None = None
|
|
92
|
-
if span is not None:
|
|
116
|
+
if span is not None and not span.closed:
|
|
93
117
|
try:
|
|
94
118
|
event = build_event(span, level, message, fields=fields, baggage=baggage)
|
|
95
119
|
span.events.append(event)
|
|
96
120
|
except Exception as exc:
|
|
121
|
+
_note_in_span_loss()
|
|
97
122
|
_diag.absorbed("building an in-span log", exc, "the event was lost")
|
|
98
123
|
else:
|
|
99
124
|
try:
|
|
@@ -109,6 +134,7 @@ def _log(level: str, message: str, echo: bool, fields: dict[str, object]) -> Non
|
|
|
109
134
|
_note_orphan_emit(sink)
|
|
110
135
|
sink.emit([event])
|
|
111
136
|
except Exception as exc:
|
|
137
|
+
_note_orphan_loss()
|
|
112
138
|
_diag.absorbed("emitting an orphan log", exc, "the event was lost")
|
|
113
139
|
if echo and event is not None:
|
|
114
140
|
try:
|
|
@@ -55,6 +55,27 @@ def current_span() -> Span | None:
|
|
|
55
55
|
return stack[-1] if stack else None
|
|
56
56
|
|
|
57
57
|
|
|
58
|
+
def _live_span_stack() -> tuple[Span, ...]:
|
|
59
|
+
"""Returns every span open on this context, outermost first (SPEC-036 FR-001).
|
|
60
|
+
|
|
61
|
+
The library's non-copying read, named as :func:`_live_baggage` is and distinguished from the
|
|
62
|
+
public accessors for the same reason (SPEC-034 FR-003): a caller gets a copy, the library
|
|
63
|
+
reads the live object. There is nothing to copy here in any case — the stack is a tuple, so
|
|
64
|
+
handing it out cannot let anyone mutate it. :func:`current_span` returns only the innermost,
|
|
65
|
+
which is what an event needs and not what a sweep does.
|
|
66
|
+
|
|
67
|
+
Args:
|
|
68
|
+
None.
|
|
69
|
+
|
|
70
|
+
Returns:
|
|
71
|
+
Every span open on this context, outermost first. Empty when none is.
|
|
72
|
+
|
|
73
|
+
Raises:
|
|
74
|
+
None.
|
|
75
|
+
"""
|
|
76
|
+
return _span_stack.get()
|
|
77
|
+
|
|
78
|
+
|
|
58
79
|
def push_span(span: Span) -> contextvars.Token[tuple[Span, ...]]:
|
|
59
80
|
"""Pushes a span onto the stack.
|
|
60
81
|
|
|
@@ -60,6 +60,26 @@ shutdown's event has every backoff collapsed to zero.
|
|
|
60
60
|
"""
|
|
61
61
|
_orphan_retired = False
|
|
62
62
|
|
|
63
|
+
_sweep_lock = threading.Lock()
|
|
64
|
+
"""Serializes the span sweep, so two threads cannot deliver one span's buffer twice.
|
|
65
|
+
|
|
66
|
+
The detach is a load and a store, and ``contextvars`` copies the same ``Span`` object into every
|
|
67
|
+
task and into any ``copy_context()`` thread, so the two-reader case is ordinary rather than exotic
|
|
68
|
+
(SPEC-036 FR-001 AC-10). A flush is not a hot path; a sink lock it is not competing with.
|
|
69
|
+
"""
|
|
70
|
+
_loss_lock = threading.Lock()
|
|
71
|
+
"""Guards the two loss counters, and deliberately not ``_worker_lock`` (SPEC-036 FR-003 AC-5).
|
|
72
|
+
|
|
73
|
+
SPEC-028's ordering rule: a counter takes its own lock, because the orphan path runs on arbitrary
|
|
74
|
+
application threads and ``_worker_lock`` is held across ``Worker(_ensure_sink())`` in
|
|
75
|
+
:func:`_get_worker` — a blocking build a counter increment must never queue behind. It cannot
|
|
76
|
+
deadlock here either: the increment sits in ``api._log``'s ``except``, where
|
|
77
|
+
:func:`_note_orphan_emit` has already released ``_worker_lock`` and the propagating exception has
|
|
78
|
+
released any sink lock.
|
|
79
|
+
"""
|
|
80
|
+
_orphan_lost = 0
|
|
81
|
+
_in_span_lost = 0
|
|
82
|
+
|
|
63
83
|
F = TypeVar("F", bound=Callable[..., Any])
|
|
64
84
|
|
|
65
85
|
|
|
@@ -184,6 +204,14 @@ def continue_trace(
|
|
|
184
204
|
_diag.rejected("parent_span_id given with no trace_id to join", parent_span_id)
|
|
185
205
|
announced = True
|
|
186
206
|
|
|
207
|
+
if adopted is not None and _current_span_was_swept():
|
|
208
|
+
_diag.rejected(
|
|
209
|
+
"the current span has already been flushed; trace context refused",
|
|
210
|
+
traceparent if traceparent is not None else str(trace_id),
|
|
211
|
+
)
|
|
212
|
+
adopted = None
|
|
213
|
+
announced = True
|
|
214
|
+
|
|
187
215
|
if adopted is not None:
|
|
188
216
|
context.set_adopted_context(*adopted)
|
|
189
217
|
_reparent_current_span(*adopted)
|
|
@@ -201,6 +229,32 @@ def continue_trace(
|
|
|
201
229
|
return ContinueResult(ok=False, reason="rejected" if announced else "nothing-supplied")
|
|
202
230
|
|
|
203
231
|
|
|
232
|
+
def _current_span_was_swept() -> bool:
|
|
233
|
+
"""Reports whether an in-span ``flush()`` has already shipped this span's events.
|
|
234
|
+
|
|
235
|
+
Read from ``context.current_span()`` — what :func:`_reparent_current_span` itself reads —
|
|
236
|
+
**and only when that span is a root**, which is the other half of that function's own guard.
|
|
237
|
+
A swept *child* is not a reason to refuse: the re-parent would have returned early on it and
|
|
238
|
+
rewritten nothing, so a refusal there prevents no corruption. It would still be wrong to
|
|
239
|
+
refuse — the two guards must agree, or the refusal fires where the thing it guards does not
|
|
240
|
+
run — though the adoption it spares reaches less than it appears to: SPEC-024 clears the
|
|
241
|
+
adopted context at the **root** span's close, so one made inside a child does not survive to
|
|
242
|
+
the next root span either. ``continue_trace``'s documented placement on the entry
|
|
243
|
+
point's first line is untouched either way: nothing has been swept that early.
|
|
244
|
+
|
|
245
|
+
Args:
|
|
246
|
+
None.
|
|
247
|
+
|
|
248
|
+
Returns:
|
|
249
|
+
Whether the current span is a root that has been swept.
|
|
250
|
+
|
|
251
|
+
Raises:
|
|
252
|
+
None.
|
|
253
|
+
"""
|
|
254
|
+
span = context.current_span()
|
|
255
|
+
return span is not None and span.parent_span_id is None and span.swept
|
|
256
|
+
|
|
257
|
+
|
|
204
258
|
def _reparent_current_span(trace_id: str, parent_span_id: str | None) -> None:
|
|
205
259
|
"""Moves an already-open root span into the adopted trace, events included.
|
|
206
260
|
|
|
@@ -699,33 +753,191 @@ def _adopt_declined_swap(new_sink: Sink) -> None:
|
|
|
699
753
|
_orphan_sink = new_sink
|
|
700
754
|
|
|
701
755
|
|
|
756
|
+
def _sweep_open_spans() -> None:
|
|
757
|
+
"""Hands the worker every event buffered on an open span in this context (SPEC-036 FR-001).
|
|
758
|
+
|
|
759
|
+
An in-span event lives on ``span.events`` until the span *closes*, and ``Worker.flush``
|
|
760
|
+
drains the *queue* — so a ``flush()`` called inside a ``@trace``d function, which is where
|
|
761
|
+
the README's serverless recipe put it, had by construction nothing to drain. Measured: zero
|
|
762
|
+
of two events delivered, every counter clean, and ``FlushResult`` reporting ``reason=None``.
|
|
763
|
+
|
|
764
|
+
The span stays **open**: its events go now and its ``span.end`` arrives later in its own
|
|
765
|
+
batch. Closing and reopening was rejected — it would emit a ``span.end`` the function never
|
|
766
|
+
reached, with a fabricated ``duration_ms`` and ``status``.
|
|
767
|
+
|
|
768
|
+
Two things must happen before the events leave, and both are why this is not a one-liner.
|
|
769
|
+
The boundary events are backfilled **first**, because SPEC-015 completes them at close by
|
|
770
|
+
iterating ``span.events`` and a swept buffer would ship ``span.start`` with ``fields={}`` —
|
|
771
|
+
the very defect that spec exists to fix, recreated by any in-span flush. They therefore carry
|
|
772
|
+
the baggage as of the flush rather than as of the close, which is a real semantic change and
|
|
773
|
+
the alternative is mutating an event the worker already owns (SPEC-028). And the buffer is
|
|
774
|
+
**detached by swap**, never cleared: ``clear()`` empties the same list object the worker was
|
|
775
|
+
handed.
|
|
776
|
+
|
|
777
|
+
The worker is created when there is something to submit, and **resolved before the buffer is
|
|
778
|
+
detached**. That ordering is the whole of the difference between a lost batch and a delivered
|
|
779
|
+
one: ``_get_worker`` can raise — it ends in ``Thread.start()`` — and a detach that has already
|
|
780
|
+
happened leaves the events in a discarded local while the span reads empty and ``flush()``
|
|
781
|
+
reports success. Measured with the failure injected: 3 of 4 events destroyed, every counter
|
|
782
|
+
zero, on a span that was still open and would have delivered them at its close.
|
|
783
|
+
``Worker.submit`` raises nothing, so once it is reached the batch is safe. Creating the worker
|
|
784
|
+
at all narrows SPEC-013's refusal rather than contradicting it — that exists so an *empty*
|
|
785
|
+
flush does not stand up a thread, and a sweep that found buffered events is not an empty
|
|
786
|
+
flush. A cold-start Lambda flushing before it returns is exactly this case: the worker is
|
|
787
|
+
built when the first span *closes*, so inside the first traced call there is none.
|
|
788
|
+
|
|
789
|
+
Concurrent sweeps are serialized on ``_sweep_lock``. The detach is a load and a store with a
|
|
790
|
+
real gap between them, and two threads sharing one ``Span`` — which ``contextvars`` makes
|
|
791
|
+
ordinary — can both read the same buffer and deliver it twice: measured, all 8 events
|
|
792
|
+
duplicated with the window held open, and 9 of 25 runs with only a GIL yield between them.
|
|
793
|
+
Rarely preempted on today's build is not a guarantee, and the floor is ``>=3.12`` where a
|
|
794
|
+
free-threading build removes even that. A flush is not a hot path, so a single lock is the
|
|
795
|
+
right cost.
|
|
796
|
+
|
|
797
|
+
**The detach stays one statement, and the two orderings above are not in tension.** A draft
|
|
798
|
+
hoisted the load to the top of the loop so a test could park on it — which put
|
|
799
|
+
``_get_worker()``, and therefore ``Thread.start()``, *inside* the load-to-store gap: measured,
|
|
800
|
+
a sweep racing a close then delivered the whole batch twice, two ``span.end`` events among
|
|
801
|
+
them, in 67 of 100 unforced trials against 0 before.
|
|
802
|
+
|
|
803
|
+
One statement makes that gap **narrow, not closed**, and the difference matters. It compiles
|
|
804
|
+
to ``LOAD_ATTR … STORE_ATTR`` with no ``CALL`` between, so CPython's eval breaker never runs
|
|
805
|
+
there and today's GIL cannot switch inside it — 0 of 500 unforced trials. Forced with an
|
|
806
|
+
opcode-level preemption it reproduces 10 of 10, and a free-threaded build removes the
|
|
807
|
+
accident entirely while ``requires-python`` has no upper bound. So :func:`_flush` takes this
|
|
808
|
+
same lock rather than relying on the width of a window: that is the *detach-vs-detach* race,
|
|
809
|
+
and a process-global lock is the right instrument for it. The **append** race
|
|
810
|
+
(``api._log`` versus a detach) is a different window needing a per-span lock, and
|
|
811
|
+
``architecture.md`` §13 declines it on cost.
|
|
812
|
+
|
|
813
|
+
It reaches only the calling context's spans. ``contextvars`` offers no way to enumerate
|
|
814
|
+
another thread's or task's context, so a ``flush()`` in a handler that fanned out does not
|
|
815
|
+
reach what those tasks buffered.
|
|
816
|
+
|
|
817
|
+
Args:
|
|
818
|
+
None.
|
|
819
|
+
|
|
820
|
+
Returns:
|
|
821
|
+
None.
|
|
822
|
+
|
|
823
|
+
Raises:
|
|
824
|
+
Exception: Whatever building the worker raises. :func:`_flush_worker` guards it, because a
|
|
825
|
+
flush is the call most likely to be made in a ``finally``.
|
|
826
|
+
"""
|
|
827
|
+
with _sweep_lock:
|
|
828
|
+
for span in context._live_span_stack():
|
|
829
|
+
if not span.events:
|
|
830
|
+
span.swept = True
|
|
831
|
+
continue
|
|
832
|
+
worker = _get_worker()
|
|
833
|
+
backfill_baggage(span, context._live_baggage())
|
|
834
|
+
span.swept = True
|
|
835
|
+
buffered, span.events = span.events, []
|
|
836
|
+
worker.submit(buffered)
|
|
837
|
+
|
|
838
|
+
|
|
702
839
|
def _flush_worker(timeout: float | None = 5.0) -> FlushResult:
|
|
703
840
|
"""Drains the process worker without retiring it, backing ``flush()`` (SPEC-013 FR-003).
|
|
704
841
|
|
|
705
|
-
This deliberately does not call :func:`_get_worker
|
|
706
|
-
|
|
707
|
-
|
|
842
|
+
~~This deliberately does not call :func:`_get_worker`~~ — narrowed by SPEC-036 FR-001. The
|
|
843
|
+
refusal still holds for an *empty* flush: a process that never logged has nothing to drain,
|
|
844
|
+
and building a worker — with the thread and ``atexit`` registration that brings — in order to
|
|
845
|
+
flush nothing would be pure cost. What changed is that :func:`_sweep_open_spans`, which runs
|
|
846
|
+
first, does build one when it finds buffered events on an open span, because submitting them
|
|
847
|
+
into a worker that does not exist delivers nothing and still reports success.
|
|
708
848
|
|
|
709
849
|
Args:
|
|
710
850
|
timeout: Seconds to wait for the drain, or ``None`` to wait indefinitely.
|
|
711
851
|
|
|
712
852
|
Returns:
|
|
713
853
|
A :class:`FlushResult`, truthy when everything outstanding was delivered and when no
|
|
714
|
-
worker exists — a process that never logged has nothing to drain, so it has lost
|
|
715
|
-
|
|
854
|
+
worker exists — a process that never logged has nothing to drain, so it has lost nothing.
|
|
855
|
+
A sweep that could not hand its buffers over reports ``"abandoned"``, the existing token
|
|
856
|
+
for "this call did not deliver them" (SPEC-036 FR-001): the events are still on their open
|
|
857
|
+
spans and their close may yet carry them, but the caller asked *now*, and on the
|
|
858
|
+
cold-start path this exists for there may be no close — reporting success there is the
|
|
859
|
+
exact shape the spec was written to remove. The drain still runs, so whatever was
|
|
860
|
+
submitted before the failure is not held back by it.
|
|
716
861
|
|
|
717
862
|
Raises:
|
|
718
863
|
None. A flush is the call most likely to be made in a ``finally``, so the library must
|
|
719
864
|
never be the reason a caller's function fails; a failure is reported by the return
|
|
720
865
|
value instead (FR-003).
|
|
721
866
|
"""
|
|
867
|
+
swept = True
|
|
868
|
+
try:
|
|
869
|
+
_sweep_open_spans()
|
|
870
|
+
except Exception as exc:
|
|
871
|
+
_diag.absorbed("sweeping open spans for a flush", exc, "buffered events were not swept")
|
|
872
|
+
swept = False
|
|
722
873
|
worker = _worker
|
|
723
874
|
if worker is None:
|
|
724
|
-
return FlushResult(ok=True)
|
|
875
|
+
return FlushResult(ok=True) if swept else FlushResult(ok=False, reason="abandoned")
|
|
725
876
|
try:
|
|
726
|
-
|
|
877
|
+
if swept:
|
|
878
|
+
return worker.flush(timeout)
|
|
879
|
+
worker.flush(timeout)
|
|
727
880
|
except Exception:
|
|
728
881
|
return FlushResult(ok=False, reason="thread-died")
|
|
882
|
+
return FlushResult(ok=False, reason="abandoned")
|
|
883
|
+
|
|
884
|
+
|
|
885
|
+
def _note_orphan_loss() -> None:
|
|
886
|
+
"""Counts one event lost on the synchronous path (SPEC-036 FR-003).
|
|
887
|
+
|
|
888
|
+
Called from ``api._log``'s orphan ``except``, which wraps the span construction, the event
|
|
889
|
+
build, ``_ensure_sink`` and the emit — so a sink that failed to *construct* is counted here
|
|
890
|
+
too, which an increment placed after ``sink.emit`` would miss.
|
|
891
|
+
|
|
892
|
+
Args:
|
|
893
|
+
None.
|
|
894
|
+
|
|
895
|
+
Returns:
|
|
896
|
+
None.
|
|
897
|
+
|
|
898
|
+
Raises:
|
|
899
|
+
None.
|
|
900
|
+
"""
|
|
901
|
+
global _orphan_lost
|
|
902
|
+
with _loss_lock:
|
|
903
|
+
_orphan_lost += 1
|
|
904
|
+
|
|
905
|
+
|
|
906
|
+
def _note_in_span_loss() -> None:
|
|
907
|
+
"""Counts one event lost while being built inside a span (SPEC-036 FR-003).
|
|
908
|
+
|
|
909
|
+
Separate from :func:`_note_orphan_loss` because the two aggregate different failure
|
|
910
|
+
populations: this path cannot fail at ``emit``, so a non-zero count here always means the
|
|
911
|
+
data, never the destination.
|
|
912
|
+
|
|
913
|
+
Args:
|
|
914
|
+
None.
|
|
915
|
+
|
|
916
|
+
Returns:
|
|
917
|
+
None.
|
|
918
|
+
|
|
919
|
+
Raises:
|
|
920
|
+
None.
|
|
921
|
+
"""
|
|
922
|
+
global _in_span_lost
|
|
923
|
+
with _loss_lock:
|
|
924
|
+
_in_span_lost += 1
|
|
925
|
+
|
|
926
|
+
|
|
927
|
+
def _read_losses() -> tuple[int, int]:
|
|
928
|
+
"""Reads both loss counters under the lock they are written under.
|
|
929
|
+
|
|
930
|
+
Args:
|
|
931
|
+
None.
|
|
932
|
+
|
|
933
|
+
Returns:
|
|
934
|
+
The orphan and in-span loss counts, in that order.
|
|
935
|
+
|
|
936
|
+
Raises:
|
|
937
|
+
None.
|
|
938
|
+
"""
|
|
939
|
+
with _loss_lock:
|
|
940
|
+
return _orphan_lost, _in_span_lost
|
|
729
941
|
|
|
730
942
|
|
|
731
943
|
def _delivering_to_an_inherited_sink() -> bool:
|
|
@@ -786,6 +998,12 @@ def _worker_health() -> Health:
|
|
|
786
998
|
defines that count as submissions queued where nothing will drain them, and a later orphan
|
|
787
999
|
log is refused at the closed sink and announced instead. The two are not the same claim.
|
|
788
1000
|
|
|
1001
|
+
The two loss counters are synthesized on **both** branches, for the reason ``retired`` is on
|
|
1002
|
+
one: they describe the caller's own path, not the worker's, and ``Worker`` cannot report them
|
|
1003
|
+
because it does not know they exist — ``worker.py`` imports nothing from this module, and the
|
|
1004
|
+
reverse read would be a cycle. A process that only ever logged outside a span has no worker
|
|
1005
|
+
and is exactly the process whose loss they exist to show (SPEC-036 FR-003 AC-7).
|
|
1006
|
+
|
|
789
1007
|
The synthesis also survives a worker built *after* that shutdown, which is why it is an
|
|
790
1008
|
``or`` rather than a fallback. An orphan-only ``shutdown()`` leaves ``_worker`` unset, so a
|
|
791
1009
|
later ``@trace`` constructs a fresh worker whose own ``retired`` is ``False`` — and reading
|
|
@@ -806,6 +1024,7 @@ def _worker_health() -> Health:
|
|
|
806
1024
|
None.
|
|
807
1025
|
"""
|
|
808
1026
|
worker = _worker
|
|
1027
|
+
orphan_lost, in_span_lost = _read_losses()
|
|
809
1028
|
if worker is None:
|
|
810
1029
|
return Health(
|
|
811
1030
|
queued=0,
|
|
@@ -814,11 +1033,14 @@ def _worker_health() -> Health:
|
|
|
814
1033
|
retired=_orphan_retired,
|
|
815
1034
|
closing_sinks=_lifecycle.closing_count(),
|
|
816
1035
|
inherited_sink=_delivering_to_an_inherited_sink(),
|
|
1036
|
+
orphan_lost=orphan_lost,
|
|
1037
|
+
in_span_lost=in_span_lost,
|
|
817
1038
|
)
|
|
818
1039
|
health = worker.health()
|
|
819
|
-
|
|
820
|
-
|
|
821
|
-
|
|
1040
|
+
retired = _orphan_retired or health.retired
|
|
1041
|
+
return replace(
|
|
1042
|
+
health, retired=retired, orphan_lost=orphan_lost, in_span_lost=in_span_lost
|
|
1043
|
+
)
|
|
822
1044
|
|
|
823
1045
|
|
|
824
1046
|
def _flush(span: Span) -> None:
|
|
@@ -828,6 +1050,24 @@ def _flush(span: Span) -> None:
|
|
|
828
1050
|
``_ensure_sink``, so a zero-config ``@trace`` still falls back to ``StdoutSink`` rather
|
|
829
1051
|
than crashing.
|
|
830
1052
|
|
|
1053
|
+
**The buffer is detached, not handed over** (SPEC-036 FR-004). ``submit`` used to take the
|
|
1054
|
+
live list, so a task outliving its parent span appended to a list the worker already owned:
|
|
1055
|
+
under the flush interval the event was delivered but ordered after ``span.end``, and over it
|
|
1056
|
+
silently lost — a pure race on a timer. The swap gives the worker the old list and leaves the
|
|
1057
|
+
span a fresh one, so the outcome stops depending on timing. A copy would do the same and
|
|
1058
|
+
allocates a second list per span on the hottest path in the library; the swap is free.
|
|
1059
|
+
|
|
1060
|
+
The late append is now landing in a buffer nothing will emit, which is *also* loss — that
|
|
1061
|
+
half is ``api._log``'s, keyed on :attr:`Span.closed`.
|
|
1062
|
+
|
|
1063
|
+
It takes ``_sweep_lock`` for the detach, because :func:`_sweep_open_spans` performs the same
|
|
1064
|
+
detach on the same attribute and a span can be swept and closed concurrently — measured, the
|
|
1065
|
+
whole batch delivered twice with two ``span.end`` events among them. The hold covers one
|
|
1066
|
+
statement and not the submit; ``Worker.submit`` is a ``put_nowait`` that never blocks, and the
|
|
1067
|
+
cost of the lock on the traced path measured within noise (+0.4% single-threaded, +0.9% across
|
|
1068
|
+
eight threads, 20,000 spans each). The **append** window this does not close is a different
|
|
1069
|
+
one, needs a per-span lock, and is declined in ``architecture.md`` §13.
|
|
1070
|
+
|
|
831
1071
|
Args:
|
|
832
1072
|
span: The finished span whose buffered events are submitted.
|
|
833
1073
|
|
|
@@ -837,7 +1077,9 @@ def _flush(span: Span) -> None:
|
|
|
837
1077
|
Raises:
|
|
838
1078
|
Exception: Whatever creating the worker or submitting raises; :func:`_end` is the guard.
|
|
839
1079
|
"""
|
|
840
|
-
|
|
1080
|
+
with _sweep_lock:
|
|
1081
|
+
events, span.events = span.events, []
|
|
1082
|
+
_get_worker().submit(events)
|
|
841
1083
|
|
|
842
1084
|
|
|
843
1085
|
type _SpanScope = tuple[
|
|
@@ -947,6 +1189,12 @@ def _close_span(span: Span, status: str, exc: BaseException | None) -> None:
|
|
|
947
1189
|
baggage known at their construction, empty for ``span.start``, so SPEC-015 completes them
|
|
948
1190
|
from the baggage live now, while the batch is still ours.
|
|
949
1191
|
|
|
1192
|
+
``closed`` is set **before** the flush, not after (SPEC-036 FR-004). ``_flush`` can raise —
|
|
1193
|
+
building the worker, or the queue path — and :func:`_end`'s guard absorbs it, so setting the
|
|
1194
|
+
flag afterwards would leave it ``False`` on exactly the runs where a later append most needs
|
|
1195
|
+
the orphan route. ``sinks/base.py`` settles the general form of this: set it before releasing
|
|
1196
|
+
anything.
|
|
1197
|
+
|
|
950
1198
|
Args:
|
|
951
1199
|
span: The span being closed.
|
|
952
1200
|
status: The span outcome, ``"ok"`` or ``"error"``.
|
|
@@ -961,6 +1209,7 @@ def _close_span(span: Span, status: str, exc: BaseException | None) -> None:
|
|
|
961
1209
|
"""
|
|
962
1210
|
span.events.append(end_event(span, status, exc))
|
|
963
1211
|
backfill_baggage(span, context._live_baggage())
|
|
1212
|
+
span.closed = True
|
|
964
1213
|
_flush(span)
|
|
965
1214
|
|
|
966
1215
|
|
|
@@ -30,6 +30,24 @@ class Span:
|
|
|
30
30
|
``start_ts`` is a ``time.monotonic()`` reading rather than wall-clock, so
|
|
31
31
|
``duration_ms`` can never go negative across a clock change. ``defaults`` are per-
|
|
32
32
|
decorator field overrides, and ``events`` is the queue flushed together at span end.
|
|
33
|
+
|
|
34
|
+
``closed`` marks a span whose events have already been handed to the worker (SPEC-036
|
|
35
|
+
FR-004). ``contextvars`` copies the *same* ``Span`` object into every task created inside a
|
|
36
|
+
span, so a fire-and-forget ``create_task`` can outlive its parent and append to a buffer
|
|
37
|
+
nothing will emit again. It is read by ``api._log`` at append time, which is the only place
|
|
38
|
+
that can notice: nothing in the library looks at a span after ``_close_span`` returns.
|
|
39
|
+
|
|
40
|
+
``swept`` marks a span whose buffered events an in-span ``flush()`` has already handed to the
|
|
41
|
+
worker (SPEC-036 FR-001). Nothing in the delivery path needs it — the sweep is correct without
|
|
42
|
+
it — but ``continue_trace`` does: ``_reparent_current_span`` adopts a context by rewriting the
|
|
43
|
+
events still *buffered* on the open root span, and swept events have left that buffer, so an
|
|
44
|
+
adoption after a sweep would leave one span carrying two trace ids. That is the SPEC-024
|
|
45
|
+
category, wrong data rather than lost data, so the flag lets the adoption refuse instead.
|
|
46
|
+
|
|
47
|
+
The guarantee is **single-threaded**. The flag is not a synchronization primitive:
|
|
48
|
+
``continue_trace`` reads it and then re-parents across an ordinary function call, so a sweep
|
|
49
|
+
arriving in that gap still splits the span. It is set before the detach so a concurrent reader
|
|
50
|
+
errs toward refusing, and the residual is recorded in ``architecture.md`` §13.
|
|
33
51
|
"""
|
|
34
52
|
|
|
35
53
|
trace_id: str
|
|
@@ -39,6 +57,8 @@ class Span:
|
|
|
39
57
|
start_ts: float
|
|
40
58
|
defaults: dict[str, object] = field(default_factory=dict)
|
|
41
59
|
events: list[dict[str, object]] = field(default_factory=list)
|
|
60
|
+
closed: bool = False
|
|
61
|
+
swept: bool = False
|
|
42
62
|
|
|
43
63
|
|
|
44
64
|
def _iso_now() -> str:
|
|
@@ -52,12 +52,21 @@ class MultiSink:
|
|
|
52
52
|
"""
|
|
53
53
|
self._sinks = sinks
|
|
54
54
|
self.failed = 0
|
|
55
|
+
self._silent_failed = 0
|
|
55
56
|
self._counter_lock = threading.Lock()
|
|
56
57
|
self._stop_signal: threading.Event | None = None
|
|
57
58
|
|
|
58
59
|
def emit(self, batch: list[dict[str, object]]) -> None:
|
|
59
60
|
"""Forwards the batch to every child in construction order, isolating failures (FR-002).
|
|
60
61
|
|
|
62
|
+
A failing child that reports **nothing** of its own is counted here, in events, because
|
|
63
|
+
otherwise it is invisible: ``losses()`` sums only children with a ``losses()``, so a
|
|
64
|
+
destination that has delivered nothing since the process started reported zero loss
|
|
65
|
+
forever (SPEC-036 FR-005). Which children are silent is decided by ``read_losses`` per
|
|
66
|
+
call, the same probe the aggregate uses, so a child that gains a ``losses()`` later moves
|
|
67
|
+
categories on its own — at the cost that the aggregate can then *fall*, which is why the
|
|
68
|
+
counter is per child rather than a single total.
|
|
69
|
+
|
|
61
70
|
Partial success stays isolated: a retry there would re-deliver the batch to the children
|
|
62
71
|
that already took it, and duplicates are worse than the one failure already counted on
|
|
63
72
|
``failed`` and written to stderr. Total failure delivered nothing, so it has no
|
|
@@ -84,8 +93,11 @@ class MultiSink:
|
|
|
84
93
|
try:
|
|
85
94
|
sink.emit(batch)
|
|
86
95
|
except Exception as err:
|
|
96
|
+
silent = read_losses(sink) is None
|
|
87
97
|
with self._counter_lock:
|
|
88
98
|
self.failed += 1
|
|
99
|
+
if silent:
|
|
100
|
+
self._silent_failed += len(batch)
|
|
89
101
|
if first_error is None:
|
|
90
102
|
first_error = err
|
|
91
103
|
_diag.absorbed(
|
|
@@ -147,18 +159,32 @@ class MultiSink:
|
|
|
147
159
|
"""Sums the children's losses so a fan-out reports the whole tree (SPEC-026 FR-002).
|
|
148
160
|
|
|
149
161
|
Nesting is handled for free, since a child ``MultiSink`` is just another sink with a
|
|
150
|
-
``losses()``.
|
|
162
|
+
``losses()``. ~~``MultiSink.failed`` is deliberately absent from the total: it counts
|
|
151
163
|
child calls that raised, not events, so adding it to a per-event figure would produce a
|
|
152
|
-
number with no unit, and the children already report their own loss in events
|
|
164
|
+
number with no unit, and the children already report their own loss in events.~~ —
|
|
165
|
+
superseded in part by SPEC-036 FR-005. ``failed`` is still absent, and still for that
|
|
166
|
+
reason. What was wrong was concluding that the fan-out therefore reports nothing: a child
|
|
167
|
+
with no ``losses()`` contributed nothing to the total, so a permanently dead destination
|
|
168
|
+
was invisible with ``health()`` reading all zeros. ``_silent_failed`` is the same loss in
|
|
169
|
+
the **right unit** — events from failing children that report nothing — so a reporting
|
|
170
|
+
child is still left to report itself and nothing is counted twice.
|
|
171
|
+
|
|
172
|
+
The figure is a **total**, not a breakdown: it cannot say which child is failing. The
|
|
173
|
+
per-batch stderr line names the class (:meth:`emit`), which is where that lives.
|
|
174
|
+
|
|
175
|
+
``close()`` deliberately does not move ``_silent_failed``. Its failure path has no batch,
|
|
176
|
+
so there is no event count to add, and a ``+1`` there would mix units in exactly the way
|
|
177
|
+
that keeps ``failed`` out of this sum. A client buffer lost to a failed close is of
|
|
178
|
+
unknown size, and an invented number is worse than an absent one.
|
|
153
179
|
|
|
154
180
|
Args:
|
|
155
181
|
None.
|
|
156
182
|
|
|
157
183
|
Returns:
|
|
158
|
-
The summed losses, or ``None`` when no child reported anything
|
|
159
|
-
|
|
160
|
-
silent children has not given a clean bill of health
|
|
161
|
-
|
|
184
|
+
The summed losses, or ``None`` when no child reported anything **and** no silent child
|
|
185
|
+
has lost a batch. FR-003 separates "reports nothing" from "reports no loss", and a tree
|
|
186
|
+
of silent children that has lost nothing has still not given a clean bill of health —
|
|
187
|
+
but one that *has* lost something now says so, which is the whole of FR-005.
|
|
162
188
|
|
|
163
189
|
Raises:
|
|
164
190
|
None. A child without ``losses()`` contributes zero, and a child whose accessor raises
|
|
@@ -167,11 +193,13 @@ class MultiSink:
|
|
|
167
193
|
"""
|
|
168
194
|
children = [read_losses(sink) for sink in self._sinks]
|
|
169
195
|
reported = [child for child in children if child is not None]
|
|
170
|
-
|
|
196
|
+
with self._counter_lock:
|
|
197
|
+
silent_failed = self._silent_failed
|
|
198
|
+
if not reported and not silent_failed:
|
|
171
199
|
return None
|
|
172
200
|
return SinkLosses(
|
|
173
201
|
dropped=sum(child.dropped for child in reported),
|
|
174
|
-
failed=sum(child.failed for child in reported),
|
|
202
|
+
failed=sum(child.failed for child in reported) + silent_failed,
|
|
175
203
|
)
|
|
176
204
|
|
|
177
205
|
def close(self) -> None:
|
|
@@ -144,6 +144,23 @@ class Health:
|
|
|
144
144
|
``shutdown()``, and it is the signal that a deployment shares a sink across a fork at
|
|
145
145
|
all. ``True`` for a shared ``StdoutSink`` too, whose ``close()`` only flushes — so a
|
|
146
146
|
``True`` is not by itself evidence that anything is held.
|
|
147
|
+
orphan_lost: Events lost on the **synchronous** path — a level call made with no active
|
|
148
|
+
span, which emits on the caller's own thread with no worker between it and the sink
|
|
149
|
+
(SPEC-036 FR-003). Until this field existed that loss was counted nowhere: ``health()``
|
|
150
|
+
describes a worker, and this path has none, so a process logging only this way read all
|
|
151
|
+
zeros over total loss. Not ``failed_batches``, which means batches a worker abandoned
|
|
152
|
+
after spending a retry budget and kept running past; there is no batch, no retry and no
|
|
153
|
+
worker here. Not ``SinkLosses`` either — the sink did not absorb anything, it raised,
|
|
154
|
+
which is what SPEC-026 requires of it. It covers everything inside the orphan guard, a
|
|
155
|
+
sink that failed to *construct* included, so it climbing means **the destination or the
|
|
156
|
+
data**.
|
|
157
|
+
in_span_lost: Events lost while being built *inside* a span (SPEC-037 AC-5c, deferred to
|
|
158
|
+
SPEC-036 FR-003 so the pair was designed together). The in-span path cannot lose an
|
|
159
|
+
event at ``emit`` — that is ``failed_batches`` — so this climbing means **the data**,
|
|
160
|
+
always: a value that could not be built into an event. Two fields rather than one
|
|
161
|
+
because they aggregate different failure populations and so fail SPEC-026's test,
|
|
162
|
+
*would one number hide which fix applies*. Their sum is deliberately not reported: with
|
|
163
|
+
different populations it is a number nobody can act on.
|
|
147
164
|
"""
|
|
148
165
|
|
|
149
166
|
queued: int
|
|
@@ -156,6 +173,8 @@ class Health:
|
|
|
156
173
|
incomplete_swaps: int = 0
|
|
157
174
|
closing_sinks: int = 0
|
|
158
175
|
inherited_sink: bool = False
|
|
176
|
+
orphan_lost: int = 0
|
|
177
|
+
in_span_lost: int = 0
|
|
159
178
|
|
|
160
179
|
|
|
161
180
|
class _FlushMarker:
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{log_foundry-0.10.2.dev67 → log_foundry-0.10.2.dev69}/src/log_foundry/sinks/elasticsearch.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|