log-foundry 0.10.2.dev128__tar.gz → 1.0.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {log_foundry-0.10.2.dev128 → log_foundry-1.0.0}/PKG-INFO +34 -24
- {log_foundry-0.10.2.dev128 → log_foundry-1.0.0}/README.md +33 -23
- {log_foundry-0.10.2.dev128 → log_foundry-1.0.0}/pyproject.toml +6 -4
- {log_foundry-0.10.2.dev128 → log_foundry-1.0.0}/src/log_foundry/__init__.py +2 -2
- {log_foundry-0.10.2.dev128 → log_foundry-1.0.0}/src/log_foundry/_lifecycle.py +744 -732
- {log_foundry-0.10.2.dev128 → log_foundry-1.0.0}/src/log_foundry/config.py +8 -2
- {log_foundry-0.10.2.dev128 → log_foundry-1.0.0}/src/log_foundry/console.py +1 -1
- {log_foundry-0.10.2.dev128 → log_foundry-1.0.0}/src/log_foundry/ids.py +3 -1
- {log_foundry-0.10.2.dev128 → log_foundry-1.0.0}/src/log_foundry/sinks/_chunk.py +6 -1
- {log_foundry-0.10.2.dev128 → log_foundry-1.0.0}/src/log_foundry/sinks/_retry.py +68 -0
- {log_foundry-0.10.2.dev128 → log_foundry-1.0.0}/src/log_foundry/sinks/_socket.py +26 -6
- {log_foundry-0.10.2.dev128 → log_foundry-1.0.0}/src/log_foundry/sinks/base.py +7 -2
- {log_foundry-0.10.2.dev128 → log_foundry-1.0.0}/src/log_foundry/sinks/clickhouse.py +35 -7
- {log_foundry-0.10.2.dev128 → log_foundry-1.0.0}/src/log_foundry/sinks/elasticsearch.py +13 -2
- {log_foundry-0.10.2.dev128 → log_foundry-1.0.0}/src/log_foundry/sinks/file.py +49 -29
- {log_foundry-0.10.2.dev128 → log_foundry-1.0.0}/src/log_foundry/sinks/http.py +115 -9
- {log_foundry-0.10.2.dev128 → log_foundry-1.0.0}/src/log_foundry/sinks/kinesis.py +8 -3
- {log_foundry-0.10.2.dev128 → log_foundry-1.0.0}/src/log_foundry/sinks/logging_sink.py +5 -1
- {log_foundry-0.10.2.dev128 → log_foundry-1.0.0}/src/log_foundry/sinks/logstash.py +44 -10
- {log_foundry-0.10.2.dev128 → log_foundry-1.0.0}/src/log_foundry/sinks/memory.py +5 -2
- {log_foundry-0.10.2.dev128 → log_foundry-1.0.0}/src/log_foundry/sinks/mongodb.py +117 -2
- {log_foundry-0.10.2.dev128 → log_foundry-1.0.0}/src/log_foundry/sinks/nats.py +25 -6
- {log_foundry-0.10.2.dev128 → log_foundry-1.0.0}/src/log_foundry/sinks/postgres.py +22 -8
- {log_foundry-0.10.2.dev128 → log_foundry-1.0.0}/src/log_foundry/sinks/pubsub.py +13 -5
- {log_foundry-0.10.2.dev128 → log_foundry-1.0.0}/src/log_foundry/sinks/rabbitmq.py +53 -1
- {log_foundry-0.10.2.dev128 → log_foundry-1.0.0}/src/log_foundry/sinks/redis.py +16 -3
- {log_foundry-0.10.2.dev128 → log_foundry-1.0.0}/src/log_foundry/sinks/sentry.py +30 -4
- {log_foundry-0.10.2.dev128 → log_foundry-1.0.0}/src/log_foundry/sinks/syslog.py +12 -2
- {log_foundry-0.10.2.dev128 → log_foundry-1.0.0}/src/log_foundry/worker.py +238 -533
- {log_foundry-0.10.2.dev128 → log_foundry-1.0.0}/LICENSE +0 -0
- {log_foundry-0.10.2.dev128 → log_foundry-1.0.0}/src/log_foundry/_diag.py +0 -0
- {log_foundry-0.10.2.dev128 → log_foundry-1.0.0}/src/log_foundry/_fork.py +0 -0
- {log_foundry-0.10.2.dev128 → log_foundry-1.0.0}/src/log_foundry/api.py +0 -0
- {log_foundry-0.10.2.dev128 → log_foundry-1.0.0}/src/log_foundry/context.py +0 -0
- {log_foundry-0.10.2.dev128 → log_foundry-1.0.0}/src/log_foundry/decorator.py +0 -0
- {log_foundry-0.10.2.dev128 → log_foundry-1.0.0}/src/log_foundry/model.py +0 -0
- {log_foundry-0.10.2.dev128 → log_foundry-1.0.0}/src/log_foundry/py.typed +0 -0
- {log_foundry-0.10.2.dev128 → log_foundry-1.0.0}/src/log_foundry/results.py +0 -0
- {log_foundry-0.10.2.dev128 → log_foundry-1.0.0}/src/log_foundry/sanitize.py +0 -0
- {log_foundry-0.10.2.dev128 → log_foundry-1.0.0}/src/log_foundry/sinks/__init__.py +0 -0
- {log_foundry-0.10.2.dev128 → log_foundry-1.0.0}/src/log_foundry/sinks/_batch.py +0 -0
- {log_foundry-0.10.2.dev128 → log_foundry-1.0.0}/src/log_foundry/sinks/_time.py +0 -0
- {log_foundry-0.10.2.dev128 → log_foundry-1.0.0}/src/log_foundry/sinks/callback.py +0 -0
- {log_foundry-0.10.2.dev128 → log_foundry-1.0.0}/src/log_foundry/sinks/datadog.py +0 -0
- {log_foundry-0.10.2.dev128 → log_foundry-1.0.0}/src/log_foundry/sinks/eventhubs.py +0 -0
- {log_foundry-0.10.2.dev128 → log_foundry-1.0.0}/src/log_foundry/sinks/filtering.py +0 -0
- {log_foundry-0.10.2.dev128 → log_foundry-1.0.0}/src/log_foundry/sinks/firehose.py +0 -0
- {log_foundry-0.10.2.dev128 → log_foundry-1.0.0}/src/log_foundry/sinks/honeycomb.py +0 -0
- {log_foundry-0.10.2.dev128 → log_foundry-1.0.0}/src/log_foundry/sinks/kafka.py +0 -0
- {log_foundry-0.10.2.dev128 → log_foundry-1.0.0}/src/log_foundry/sinks/loki.py +0 -0
- {log_foundry-0.10.2.dev128 → log_foundry-1.0.0}/src/log_foundry/sinks/multi.py +0 -0
- {log_foundry-0.10.2.dev128 → log_foundry-1.0.0}/src/log_foundry/sinks/newrelic.py +0 -0
- {log_foundry-0.10.2.dev128 → log_foundry-1.0.0}/src/log_foundry/sinks/null.py +0 -0
- {log_foundry-0.10.2.dev128 → log_foundry-1.0.0}/src/log_foundry/sinks/sns.py +0 -0
- {log_foundry-0.10.2.dev128 → log_foundry-1.0.0}/src/log_foundry/sinks/splunk.py +0 -0
- {log_foundry-0.10.2.dev128 → log_foundry-1.0.0}/src/log_foundry/sinks/sqlite.py +0 -0
- {log_foundry-0.10.2.dev128 → log_foundry-1.0.0}/src/log_foundry/sinks/sqs.py +0 -0
- {log_foundry-0.10.2.dev128 → log_foundry-1.0.0}/src/log_foundry/sinks/stdout.py +0 -0
- {log_foundry-0.10.2.dev128 → log_foundry-1.0.0}/src/log_foundry/sinks/transform.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: log-foundry
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 1.0.0
|
|
4
4
|
Summary: Generate logs for your console and JSON events for downstream consumption.
|
|
5
5
|
License-Expression: MIT
|
|
6
6
|
License-File: LICENSE
|
|
@@ -578,7 +578,8 @@ nothing.
|
|
|
578
578
|
Wire one up by passing an instance to `configure(sink=...)`; if you never do, the first decorated
|
|
579
579
|
call falls back to `StdoutSink()`. The **protocol** is a top-level export, alongside
|
|
580
580
|
`SinkDeliveryError`, `SinkLosses`, `read_losses` and `flush_sink` — the last two being the probes
|
|
581
|
-
a wrapper sink needs to ask a child for its losses and to push its client-side buffer
|
|
581
|
+
a wrapper sink needs to ask a child for its losses and to push its client-side buffer — plus the
|
|
582
|
+
two default bounds, `DEFAULT_SHUTDOWN_TIMEOUT` and `DEFAULT_SWAP_TIMEOUT`. The
|
|
582
583
|
**concrete sinks** are not exported, so import each from its own module, e.g.
|
|
583
584
|
`from log_foundry.sinks.sqs import SQSSink`.
|
|
584
585
|
|
|
@@ -892,8 +893,8 @@ disconnected, so a sustained outage moves `health().failed_batches` instead of b
|
|
|
892
893
|
| `KafkaSink` | `log_foundry.sinks.kafka` | `kafka` | `KafkaSink(topic, *, flush_timeout=10.0, bootstrap_servers="…", key_field="trace_id", producer_config=None)` — `producer_config` is merged **beneath** the sink's own keys, so it reaches `message.timeout.ms` and friends without displacing `bootstrap.servers`; passing it with `producer=` is a `ValueError` |
|
|
893
894
|
| `RedisStreamsSink` | `log_foundry.sinks.redis` | `redis` | `RedisStreamsSink(stream, *, url=None, maxlen=None)` — `XADD`. `maxlen` caps the stream (`approximate=True`); trimming happens **at Redis**, after delivery, so it is invisible to `health()` — which is why the default is unbounded |
|
|
894
895
|
| `RedisListSink` | `log_foundry.sinks.redis` | `redis` | `RedisListSink(key, *, url=None, maxlen=None)` — `RPUSH` + `LTRIM` to the newest `maxlen`; same destination-side trimming caveat |
|
|
895
|
-
| `RabbitMQSink` | `log_foundry.sinks.rabbitmq` | `amqp` | `RabbitMQSink(*, exchange, routing_key, url=None)` — persistent messages |
|
|
896
|
-
| `NATSSink` | `log_foundry.sinks.nats` | `nats` | `NATSSink(subject, *, jetstream=False, servers=None, publish_timeout=10.0, connect_timeout=None, max_reconnect_attempts=None, reconnect_time_wait=None, drain_timeout=None)` — `publish_timeout` bounds one whole `emit` and applies to an injected `client=` too; the four `None` timeouts are forwarded to `nats.connect` only when set, and passing one with `client=` is a `ValueError` |
|
|
896
|
+
| `RabbitMQSink` | `log_foundry.sinks.rabbitmq` | `amqp` | `RabbitMQSink(*, exchange, routing_key, url=None, blocked_connection_timeout=None, socket_timeout=None, stack_timeout=None)` — persistent messages; `pika` leaves `blocked_connection_timeout` unset, so a broker under a memory or disk alarm blocks every publish indefinitely — the sink applies `DEFAULT_BLOCKED_CONNECTION_TIMEOUT` (30 s) unless the URL's query names one, and an explicit keyword overrides the URL. **Read from `pika` 1.4.4 and not executed against a broker** — verify it against yours |
|
|
897
|
+
| `NATSSink` | `log_foundry.sinks.nats` | `nats` | `NATSSink(subject, *, jetstream=False, servers=None, publish_timeout=10.0, connect_timeout=None, max_reconnect_attempts=None, reconnect_time_wait=None, drain_timeout=None)` — a non-positive `max_reconnect_attempts` is a `ValueError` at construction, since `nats-py` retires a server from its pool only under `max_reconnect_attempts > 0` and a non-positive value therefore never returns; `publish_timeout` bounds one whole `emit` and applies to an injected `client=` too; the four `None` timeouts are forwarded to `nats.connect` only when set, and passing one with `client=` is a `ValueError` |
|
|
897
898
|
| `GooglePubSubSink` | `log_foundry.sinks.pubsub` | `gcp-pubsub` | `GooglePubSubSink(topic)` |
|
|
898
899
|
| `AzureEventHubsSink` | `log_foundry.sinks.eventhubs` | `azure-eventhubs` | `AzureEventHubsSink(*, connection_str="…", eventhub=None)` |
|
|
899
900
|
|
|
@@ -908,9 +909,9 @@ Write-only inserts (querying is the downstream tool's job); each needs its own e
|
|
|
908
909
|
|
|
909
910
|
| Sink | Import from | Extra | Configure |
|
|
910
911
|
|---|---|---|---|
|
|
911
|
-
| `MongoDBSink` | `log_foundry.sinks.mongodb` | `mongo` | `MongoDBSink(*, uri="…", database="…", collection="…")` |
|
|
912
|
+
| `MongoDBSink` | `log_foundry.sinks.mongodb` | `mongo` | `MongoDBSink(*, uri="…", database="…", collection="…", socket_timeout=None, server_selection_timeout=None)` — `pymongo`'s `socketTimeoutMS` default is `None`, so a read that never returns holds the drain thread; the sink applies `DEFAULT_SOCKET_TIMEOUT` (30 s) only when neither the keyword nor the URI query names one |
|
|
912
913
|
| `PostgresSink` | `log_foundry.sinks.postgres` | `postgres` | `PostgresSink(table, *, dsn="…", create_table=False, connect_timeout=5)` — JSONB `event` column + extracted columns. Reconnects an **owned** connection the server has closed; a `connection=` you inject is never reopened. `connect_timeout` is passed to libpq explicitly, so it **overrides** any `connect_timeout` in your DSN |
|
|
913
|
-
| `ClickHouseSink` | `log_foundry.sinks.clickhouse` | `clickhouse` | `ClickHouseSink(table, *, dsn="…", create_table=False)` — MergeTree, columnar insert |
|
|
914
|
+
| `ClickHouseSink` | `log_foundry.sinks.clickhouse` | `clickhouse` | `ClickHouseSink(table, *, dsn="…", create_table=False, chunk_size=1000, send_receive_timeout=None)` — MergeTree, columnar insert; the driver's own 300 s `send_receive_timeout` is finite so it is left alone, and forwarded only when you set it |
|
|
914
915
|
|
|
915
916
|
`PostgresSink` / `ClickHouseSink` default `create_table=False` (you own the schema and indexes); set
|
|
916
917
|
it `True` for an idempotent `CREATE TABLE IF NOT EXISTS` convenience.
|
|
@@ -1086,7 +1087,7 @@ They tell you different things, and they want different responses:
|
|
|
1086
1087
|
|
|
1087
1088
|
| Field | Means | What to do |
|
|
1088
1089
|
|---|---|---|
|
|
1089
|
-
| `dropped` | The queue filled — the destination is not keeping up. Delivery continues. | Make the destination keep up: scale the sink, or reduce what you log. The worker's batch size, flush interval and queue depth are **not** reachable from the public API, so tuning them is not an option this version offers you. |
|
|
1090
|
+
| `dropped` | The queue filled — the destination is not keeping up. Delivery continues. It counts **submissions** — one per span close, or per open span a `flush()` sweeps — so the events lost are a multiple of it. | Make the destination keep up: scale the sink, or reduce what you log. The worker's batch size, flush interval and queue depth are **not** reachable from the public API, so tuning them is not an option this version offers you. |
|
|
1090
1091
|
| `failed_batches` | A sink stayed broken through the whole retry budget. Delivery continues. | Fix the destination. |
|
|
1091
1092
|
| `stopped_reason` | The background thread **died** on that exception type. Nothing further will be delivered, ever. | Restart the process; investigate the named exception. |
|
|
1092
1093
|
| `sink.dropped` | The sink discarded events **before** attempting delivery — an oversized record, or one the client refused outright. | Read the stderr line: it names the cause. An oversized record means shrink what you log; a refused local produce/publish (Kafka, Pub/Sub) points at the client — a saturated buffer, a bad topic, a credential. |
|
|
@@ -1095,16 +1096,19 @@ They tell you different things, and they want different responses:
|
|
|
1095
1096
|
| `incomplete_swaps` | A late `configure(sink=...)` could not confirm the previous sink was drained. The swap took effect; that sink was left open and some queued events may have gone to the new one. | Investigate the previous sink — it was hung or failing. Configure the sink before the first log where you can. |
|
|
1096
1097
|
| `inherited_sink` | This process is delivering to a sink it **inherited across a `fork`** and may not release, so it will not be closed here. Not a loss and not an alert term. | Nothing, usually. It explains a handle still open after `shutdown()`, and tells you a deployment shares one sink across a fork at all. `True` for a shared `StdoutSink` too, whose `close()` only flushes — so a `True` is not by itself evidence that anything is held. If you want the child to own its transport, build the sink in the worker process (see Forking). |
|
|
1097
1098
|
| `orphan_lost` | An event logged **with no open span** never reached the sink. That call emits on your own thread with no worker behind it, so no other field here can carry it — it is not a batch, there was no retry, and there may be no worker at all. Covers a sink that failed to *construct* as well as one that raised. | Fix the destination, or the data. The stderr line names the exception type. If a process logs this way at all, this is the field to alert on: nothing else describes that path. |
|
|
1098
|
-
| `in_span_lost` | An event logged **inside a span**
|
|
1099
|
+
| `in_span_lost` | An event logged **inside a span** was lost before delivery, for one of two reasons the count tells apart: **the data** — a value that could not be turned into an event, one per call — or **no drain thread at all** — the process could not start the worker, and the span's whole buffer is counted at once (SPEC-050). Never a destination that failed at delivery; that is `failed_batches`. | One event at a time: fix the call site — a non-string message (an exception object, say) is the common cause. A whole buffer at once: the process cannot start a thread; the stderr line names it. |
|
|
1099
1100
|
| `closing_sinks` | Swapped-out sinks inside `close()` **right now** — a live gauge, not a counter. It and `queued` are the two fields here that fall as well as rise; the other integer counters only climb. Non-zero on a single read is normal during a swap. | Nothing, unless it stays non-zero. That means a destination is stuck in `close()` and will not release its resources. |
|
|
1100
1101
|
|
|
1101
1102
|
`orphan_lost` and `in_span_lost` are deliberately two fields and their sum is deliberately not
|
|
1102
1103
|
reported. They aggregate different failure populations — one can mean the destination *or* the
|
|
1103
|
-
data
|
|
1104
|
-
|
|
1105
|
-
|
|
1106
|
-
|
|
1107
|
-
|
|
1104
|
+
data; the other never the destination — the data, or no drain thread at all (SPEC-050) — so a
|
|
1105
|
+
single number would hide which fix applies.
|
|
1106
|
+
|
|
1107
|
+
`h.sink` is a `SinkLosses`, carrying `dropped` and `failed`, or `None` — `None` when no sink has
|
|
1108
|
+
been configured at all, and `None` when the configured sink reports nothing (`losses()` is
|
|
1109
|
+
optional, and a sink whose `losses()` raises reports `None` too). It is **not** `None` merely
|
|
1110
|
+
because no worker exists: since SPEC-054 it is answered
|
|
1111
|
+
from the configuration, so it reports on the orphan path too. Note the two `dropped` fields count
|
|
1108
1112
|
different things: the worker's is backpressure at *its* queue, the sink's is an event that never
|
|
1109
1113
|
reached the wire. They are separate because the remedies do not overlap — and `sink.dropped` is
|
|
1110
1114
|
itself two causes, which is why the diagnostic line matters. Most sinks drop only what can never
|
|
@@ -1132,10 +1136,13 @@ logging is doing the right thing; it is the *pair* — retired, and still being
|
|
|
1132
1136
|
means every log line since the shutdown has gone nowhere. That state used to read as perfectly
|
|
1133
1137
|
healthy: `stopped_reason` is `None` after a clean shutdown, and the queue simply grows.
|
|
1134
1138
|
|
|
1135
|
-
|
|
1136
|
-
|
|
1137
|
-
|
|
1138
|
-
|
|
1139
|
+
**Six of the twelve fields report for a process that has no worker at all** — `sink`, `retired`,
|
|
1140
|
+
`closing_sinks`, `inherited_sink`, `orphan_lost` and `in_span_lost`. That set is not a list to
|
|
1141
|
+
maintain by hand: it is exactly the fields in `_lifecycle`'s `Health(...)` assembly that are *not*
|
|
1142
|
+
guarded by `counters is None`, and re-reading that call is how to check it. A program that only
|
|
1143
|
+
ever calls `info()`/`error()` outside a span emits synchronously and builds no background worker,
|
|
1144
|
+
so the other six describe something that does not exist and read their empty value — zero for the
|
|
1145
|
+
five counters, `None` for `stopped_reason` — which is why that path needs counters of its own. Until it had them, such a process
|
|
1139
1146
|
reported `queued=0 dropped=0 failed_batches=0 stopped_reason=None` over total, permanent loss, and
|
|
1140
1147
|
the only thing that said otherwise was a line on stderr. Its
|
|
1141
1148
|
`shutdown()` still closes the sink, exactly once and without starting a thread, and `retired` reads
|
|
@@ -1410,8 +1417,8 @@ failure lands on whoever caused it rather than on a shared branch. Contributors
|
|
|
1410
1417
|
with `scripts/spec-lint.sh`, before pushing.
|
|
1411
1418
|
|
|
1412
1419
|
On a push to `main` the full `ci.yml` matrix still runs, but as the first job of
|
|
1413
|
-
[`release.yml`](.github/workflows/release.yml), which `uses:` this same workflow before it
|
|
1414
|
-
allowed to publish — so on main the checks are listed under the *Release* workflow, named
|
|
1420
|
+
[`release.yml`](.github/workflows/release.yml), which `uses:` this same workflow before it would
|
|
1421
|
+
be allowed to publish — so on main the checks are listed under the *Release* workflow, named
|
|
1415
1422
|
`test / test (py3.12)` and `test / test (py3.13)`. `ci.yml` deliberately carries no `push`
|
|
1416
1423
|
trigger of its own; it had one, and the result was that every merge ran the identical matrix
|
|
1417
1424
|
twice. A `v*` tag is gated the same way, by that same reusable call.
|
|
@@ -1468,12 +1475,14 @@ sdist and a wheel:
|
|
|
1468
1475
|
|
|
1469
1476
|
| Trigger | Version built | Published to PyPI as |
|
|
1470
1477
|
|---|---|---|
|
|
1471
|
-
| merge to `main` | `X.Y.Z.devN` |
|
|
1478
|
+
| merge to `main` | `X.Y.Z.devN` | **nothing, for now** — `publish-dev` is disabled |
|
|
1472
1479
|
| push tag `vX.Y.Z` | `X.Y.Z` | stable release |
|
|
1473
1480
|
|
|
1474
|
-
Dev pre-releases
|
|
1475
|
-
first time it
|
|
1476
|
-
|
|
1481
|
+
Dev pre-releases **kept** the upload path exercised on every merge, so a real release was never
|
|
1482
|
+
the first time it ran. That property is suspended along with the job: the next `vX.Y.Z` tag is
|
|
1483
|
+
the first attempt at the upload path since `publish-dev` was disabled. `pip install log-foundry`
|
|
1484
|
+
resolves to the latest **stable** version either way — pip ignores pre-releases unless you pass
|
|
1485
|
+
`--pre`.
|
|
1477
1486
|
|
|
1478
1487
|
Cutting a release is one tag:
|
|
1479
1488
|
|
|
@@ -1490,7 +1499,8 @@ the tagged commit are not seen. A release body cannot be amended once published
|
|
|
1490
1499
|
repository has immutable releases — so check the file is there and named for the tag before
|
|
1491
1500
|
pushing it. [`docs/release-notes/`](docs/release-notes/) holds the ones written so far, and
|
|
1492
1501
|
[`docs/spec-delivery/RELEASES.md`](docs/spec-delivery/RELEASES.md) records which specs each
|
|
1493
|
-
|
|
1502
|
+
version carried — its top row is written in the commit that version's tag is cut from, so it can
|
|
1503
|
+
name a tag that does not exist yet.
|
|
1494
1504
|
|
|
1495
1505
|
Uploads authenticate with PyPI [Trusted Publishing](https://docs.pypi.org/trusted-publishers/)
|
|
1496
1506
|
(OIDC) through the `pypi` GitHub Environment — there is no API token stored in the repository.
|
|
@@ -527,7 +527,8 @@ nothing.
|
|
|
527
527
|
Wire one up by passing an instance to `configure(sink=...)`; if you never do, the first decorated
|
|
528
528
|
call falls back to `StdoutSink()`. The **protocol** is a top-level export, alongside
|
|
529
529
|
`SinkDeliveryError`, `SinkLosses`, `read_losses` and `flush_sink` — the last two being the probes
|
|
530
|
-
a wrapper sink needs to ask a child for its losses and to push its client-side buffer
|
|
530
|
+
a wrapper sink needs to ask a child for its losses and to push its client-side buffer — plus the
|
|
531
|
+
two default bounds, `DEFAULT_SHUTDOWN_TIMEOUT` and `DEFAULT_SWAP_TIMEOUT`. The
|
|
531
532
|
**concrete sinks** are not exported, so import each from its own module, e.g.
|
|
532
533
|
`from log_foundry.sinks.sqs import SQSSink`.
|
|
533
534
|
|
|
@@ -841,8 +842,8 @@ disconnected, so a sustained outage moves `health().failed_batches` instead of b
|
|
|
841
842
|
| `KafkaSink` | `log_foundry.sinks.kafka` | `kafka` | `KafkaSink(topic, *, flush_timeout=10.0, bootstrap_servers="…", key_field="trace_id", producer_config=None)` — `producer_config` is merged **beneath** the sink's own keys, so it reaches `message.timeout.ms` and friends without displacing `bootstrap.servers`; passing it with `producer=` is a `ValueError` |
|
|
842
843
|
| `RedisStreamsSink` | `log_foundry.sinks.redis` | `redis` | `RedisStreamsSink(stream, *, url=None, maxlen=None)` — `XADD`. `maxlen` caps the stream (`approximate=True`); trimming happens **at Redis**, after delivery, so it is invisible to `health()` — which is why the default is unbounded |
|
|
843
844
|
| `RedisListSink` | `log_foundry.sinks.redis` | `redis` | `RedisListSink(key, *, url=None, maxlen=None)` — `RPUSH` + `LTRIM` to the newest `maxlen`; same destination-side trimming caveat |
|
|
844
|
-
| `RabbitMQSink` | `log_foundry.sinks.rabbitmq` | `amqp` | `RabbitMQSink(*, exchange, routing_key, url=None)` — persistent messages |
|
|
845
|
-
| `NATSSink` | `log_foundry.sinks.nats` | `nats` | `NATSSink(subject, *, jetstream=False, servers=None, publish_timeout=10.0, connect_timeout=None, max_reconnect_attempts=None, reconnect_time_wait=None, drain_timeout=None)` — `publish_timeout` bounds one whole `emit` and applies to an injected `client=` too; the four `None` timeouts are forwarded to `nats.connect` only when set, and passing one with `client=` is a `ValueError` |
|
|
845
|
+
| `RabbitMQSink` | `log_foundry.sinks.rabbitmq` | `amqp` | `RabbitMQSink(*, exchange, routing_key, url=None, blocked_connection_timeout=None, socket_timeout=None, stack_timeout=None)` — persistent messages; `pika` leaves `blocked_connection_timeout` unset, so a broker under a memory or disk alarm blocks every publish indefinitely — the sink applies `DEFAULT_BLOCKED_CONNECTION_TIMEOUT` (30 s) unless the URL's query names one, and an explicit keyword overrides the URL. **Read from `pika` 1.4.4 and not executed against a broker** — verify it against yours |
|
|
846
|
+
| `NATSSink` | `log_foundry.sinks.nats` | `nats` | `NATSSink(subject, *, jetstream=False, servers=None, publish_timeout=10.0, connect_timeout=None, max_reconnect_attempts=None, reconnect_time_wait=None, drain_timeout=None)` — a non-positive `max_reconnect_attempts` is a `ValueError` at construction, since `nats-py` retires a server from its pool only under `max_reconnect_attempts > 0` and a non-positive value therefore never returns; `publish_timeout` bounds one whole `emit` and applies to an injected `client=` too; the four `None` timeouts are forwarded to `nats.connect` only when set, and passing one with `client=` is a `ValueError` |
|
|
846
847
|
| `GooglePubSubSink` | `log_foundry.sinks.pubsub` | `gcp-pubsub` | `GooglePubSubSink(topic)` |
|
|
847
848
|
| `AzureEventHubsSink` | `log_foundry.sinks.eventhubs` | `azure-eventhubs` | `AzureEventHubsSink(*, connection_str="…", eventhub=None)` |
|
|
848
849
|
|
|
@@ -857,9 +858,9 @@ Write-only inserts (querying is the downstream tool's job); each needs its own e
|
|
|
857
858
|
|
|
858
859
|
| Sink | Import from | Extra | Configure |
|
|
859
860
|
|---|---|---|---|
|
|
860
|
-
| `MongoDBSink` | `log_foundry.sinks.mongodb` | `mongo` | `MongoDBSink(*, uri="…", database="…", collection="…")` |
|
|
861
|
+
| `MongoDBSink` | `log_foundry.sinks.mongodb` | `mongo` | `MongoDBSink(*, uri="…", database="…", collection="…", socket_timeout=None, server_selection_timeout=None)` — `pymongo`'s `socketTimeoutMS` default is `None`, so a read that never returns holds the drain thread; the sink applies `DEFAULT_SOCKET_TIMEOUT` (30 s) only when neither the keyword nor the URI query names one |
|
|
861
862
|
| `PostgresSink` | `log_foundry.sinks.postgres` | `postgres` | `PostgresSink(table, *, dsn="…", create_table=False, connect_timeout=5)` — JSONB `event` column + extracted columns. Reconnects an **owned** connection the server has closed; a `connection=` you inject is never reopened. `connect_timeout` is passed to libpq explicitly, so it **overrides** any `connect_timeout` in your DSN |
|
|
862
|
-
| `ClickHouseSink` | `log_foundry.sinks.clickhouse` | `clickhouse` | `ClickHouseSink(table, *, dsn="…", create_table=False)` — MergeTree, columnar insert |
|
|
863
|
+
| `ClickHouseSink` | `log_foundry.sinks.clickhouse` | `clickhouse` | `ClickHouseSink(table, *, dsn="…", create_table=False, chunk_size=1000, send_receive_timeout=None)` — MergeTree, columnar insert; the driver's own 300 s `send_receive_timeout` is finite so it is left alone, and forwarded only when you set it |
|
|
863
864
|
|
|
864
865
|
`PostgresSink` / `ClickHouseSink` default `create_table=False` (you own the schema and indexes); set
|
|
865
866
|
it `True` for an idempotent `CREATE TABLE IF NOT EXISTS` convenience.
|
|
@@ -1035,7 +1036,7 @@ They tell you different things, and they want different responses:
|
|
|
1035
1036
|
|
|
1036
1037
|
| Field | Means | What to do |
|
|
1037
1038
|
|---|---|---|
|
|
1038
|
-
| `dropped` | The queue filled — the destination is not keeping up. Delivery continues. | Make the destination keep up: scale the sink, or reduce what you log. The worker's batch size, flush interval and queue depth are **not** reachable from the public API, so tuning them is not an option this version offers you. |
|
|
1039
|
+
| `dropped` | The queue filled — the destination is not keeping up. Delivery continues. It counts **submissions** — one per span close, or per open span a `flush()` sweeps — so the events lost are a multiple of it. | Make the destination keep up: scale the sink, or reduce what you log. The worker's batch size, flush interval and queue depth are **not** reachable from the public API, so tuning them is not an option this version offers you. |
|
|
1039
1040
|
| `failed_batches` | A sink stayed broken through the whole retry budget. Delivery continues. | Fix the destination. |
|
|
1040
1041
|
| `stopped_reason` | The background thread **died** on that exception type. Nothing further will be delivered, ever. | Restart the process; investigate the named exception. |
|
|
1041
1042
|
| `sink.dropped` | The sink discarded events **before** attempting delivery — an oversized record, or one the client refused outright. | Read the stderr line: it names the cause. An oversized record means shrink what you log; a refused local produce/publish (Kafka, Pub/Sub) points at the client — a saturated buffer, a bad topic, a credential. |
|
|
@@ -1044,16 +1045,19 @@ They tell you different things, and they want different responses:
|
|
|
1044
1045
|
| `incomplete_swaps` | A late `configure(sink=...)` could not confirm the previous sink was drained. The swap took effect; that sink was left open and some queued events may have gone to the new one. | Investigate the previous sink — it was hung or failing. Configure the sink before the first log where you can. |
|
|
1045
1046
|
| `inherited_sink` | This process is delivering to a sink it **inherited across a `fork`** and may not release, so it will not be closed here. Not a loss and not an alert term. | Nothing, usually. It explains a handle still open after `shutdown()`, and tells you a deployment shares one sink across a fork at all. `True` for a shared `StdoutSink` too, whose `close()` only flushes — so a `True` is not by itself evidence that anything is held. If you want the child to own its transport, build the sink in the worker process (see Forking). |
|
|
1046
1047
|
| `orphan_lost` | An event logged **with no open span** never reached the sink. That call emits on your own thread with no worker behind it, so no other field here can carry it — it is not a batch, there was no retry, and there may be no worker at all. Covers a sink that failed to *construct* as well as one that raised. | Fix the destination, or the data. The stderr line names the exception type. If a process logs this way at all, this is the field to alert on: nothing else describes that path. |
|
|
1047
|
-
| `in_span_lost` | An event logged **inside a span**
|
|
1048
|
+
| `in_span_lost` | An event logged **inside a span** was lost before delivery, for one of two reasons the count tells apart: **the data** — a value that could not be turned into an event, one per call — or **no drain thread at all** — the process could not start the worker, and the span's whole buffer is counted at once (SPEC-050). Never a destination that failed at delivery; that is `failed_batches`. | One event at a time: fix the call site — a non-string message (an exception object, say) is the common cause. A whole buffer at once: the process cannot start a thread; the stderr line names it. |
|
|
1048
1049
|
| `closing_sinks` | Swapped-out sinks inside `close()` **right now** — a live gauge, not a counter. It and `queued` are the two fields here that fall as well as rise; the other integer counters only climb. Non-zero on a single read is normal during a swap. | Nothing, unless it stays non-zero. That means a destination is stuck in `close()` and will not release its resources. |
|
|
1049
1050
|
|
|
1050
1051
|
`orphan_lost` and `in_span_lost` are deliberately two fields and their sum is deliberately not
|
|
1051
1052
|
reported. They aggregate different failure populations — one can mean the destination *or* the
|
|
1052
|
-
data
|
|
1053
|
-
|
|
1054
|
-
|
|
1055
|
-
|
|
1056
|
-
|
|
1053
|
+
data; the other never the destination — the data, or no drain thread at all (SPEC-050) — so a
|
|
1054
|
+
single number would hide which fix applies.
|
|
1055
|
+
|
|
1056
|
+
`h.sink` is a `SinkLosses`, carrying `dropped` and `failed`, or `None` — `None` when no sink has
|
|
1057
|
+
been configured at all, and `None` when the configured sink reports nothing (`losses()` is
|
|
1058
|
+
optional, and a sink whose `losses()` raises reports `None` too). It is **not** `None` merely
|
|
1059
|
+
because no worker exists: since SPEC-054 it is answered
|
|
1060
|
+
from the configuration, so it reports on the orphan path too. Note the two `dropped` fields count
|
|
1057
1061
|
different things: the worker's is backpressure at *its* queue, the sink's is an event that never
|
|
1058
1062
|
reached the wire. They are separate because the remedies do not overlap — and `sink.dropped` is
|
|
1059
1063
|
itself two causes, which is why the diagnostic line matters. Most sinks drop only what can never
|
|
@@ -1081,10 +1085,13 @@ logging is doing the right thing; it is the *pair* — retired, and still being
|
|
|
1081
1085
|
means every log line since the shutdown has gone nowhere. That state used to read as perfectly
|
|
1082
1086
|
healthy: `stopped_reason` is `None` after a clean shutdown, and the queue simply grows.
|
|
1083
1087
|
|
|
1084
|
-
|
|
1085
|
-
|
|
1086
|
-
|
|
1087
|
-
|
|
1088
|
+
**Six of the twelve fields report for a process that has no worker at all** — `sink`, `retired`,
|
|
1089
|
+
`closing_sinks`, `inherited_sink`, `orphan_lost` and `in_span_lost`. That set is not a list to
|
|
1090
|
+
maintain by hand: it is exactly the fields in `_lifecycle`'s `Health(...)` assembly that are *not*
|
|
1091
|
+
guarded by `counters is None`, and re-reading that call is how to check it. A program that only
|
|
1092
|
+
ever calls `info()`/`error()` outside a span emits synchronously and builds no background worker,
|
|
1093
|
+
so the other six describe something that does not exist and read their empty value — zero for the
|
|
1094
|
+
five counters, `None` for `stopped_reason` — which is why that path needs counters of its own. Until it had them, such a process
|
|
1088
1095
|
reported `queued=0 dropped=0 failed_batches=0 stopped_reason=None` over total, permanent loss, and
|
|
1089
1096
|
the only thing that said otherwise was a line on stderr. Its
|
|
1090
1097
|
`shutdown()` still closes the sink, exactly once and without starting a thread, and `retired` reads
|
|
@@ -1359,8 +1366,8 @@ failure lands on whoever caused it rather than on a shared branch. Contributors
|
|
|
1359
1366
|
with `scripts/spec-lint.sh`, before pushing.
|
|
1360
1367
|
|
|
1361
1368
|
On a push to `main` the full `ci.yml` matrix still runs, but as the first job of
|
|
1362
|
-
[`release.yml`](.github/workflows/release.yml), which `uses:` this same workflow before it
|
|
1363
|
-
allowed to publish — so on main the checks are listed under the *Release* workflow, named
|
|
1369
|
+
[`release.yml`](.github/workflows/release.yml), which `uses:` this same workflow before it would
|
|
1370
|
+
be allowed to publish — so on main the checks are listed under the *Release* workflow, named
|
|
1364
1371
|
`test / test (py3.12)` and `test / test (py3.13)`. `ci.yml` deliberately carries no `push`
|
|
1365
1372
|
trigger of its own; it had one, and the result was that every merge ran the identical matrix
|
|
1366
1373
|
twice. A `v*` tag is gated the same way, by that same reusable call.
|
|
@@ -1417,12 +1424,14 @@ sdist and a wheel:
|
|
|
1417
1424
|
|
|
1418
1425
|
| Trigger | Version built | Published to PyPI as |
|
|
1419
1426
|
|---|---|---|
|
|
1420
|
-
| merge to `main` | `X.Y.Z.devN` |
|
|
1427
|
+
| merge to `main` | `X.Y.Z.devN` | **nothing, for now** — `publish-dev` is disabled |
|
|
1421
1428
|
| push tag `vX.Y.Z` | `X.Y.Z` | stable release |
|
|
1422
1429
|
|
|
1423
|
-
Dev pre-releases
|
|
1424
|
-
first time it
|
|
1425
|
-
|
|
1430
|
+
Dev pre-releases **kept** the upload path exercised on every merge, so a real release was never
|
|
1431
|
+
the first time it ran. That property is suspended along with the job: the next `vX.Y.Z` tag is
|
|
1432
|
+
the first attempt at the upload path since `publish-dev` was disabled. `pip install log-foundry`
|
|
1433
|
+
resolves to the latest **stable** version either way — pip ignores pre-releases unless you pass
|
|
1434
|
+
`--pre`.
|
|
1426
1435
|
|
|
1427
1436
|
Cutting a release is one tag:
|
|
1428
1437
|
|
|
@@ -1439,7 +1448,8 @@ the tagged commit are not seen. A release body cannot be amended once published
|
|
|
1439
1448
|
repository has immutable releases — so check the file is there and named for the tag before
|
|
1440
1449
|
pushing it. [`docs/release-notes/`](docs/release-notes/) holds the ones written so far, and
|
|
1441
1450
|
[`docs/spec-delivery/RELEASES.md`](docs/spec-delivery/RELEASES.md) records which specs each
|
|
1442
|
-
|
|
1451
|
+
version carried — its top row is written in the commit that version's tag is cut from, so it can
|
|
1452
|
+
name a tag that does not exist yet.
|
|
1443
1453
|
|
|
1444
1454
|
Uploads authenticate with PyPI [Trusted Publishing](https://docs.pypi.org/trusted-publishers/)
|
|
1445
1455
|
(OIDC) through the `pypi` GitHub Environment — there is no API token stored in the repository.
|
|
@@ -30,9 +30,11 @@ dependencies = [
|
|
|
30
30
|
classifiers = [
|
|
31
31
|
# Asserts what the NEXT STABLE TAG will be, not what the current version is. Safe only
|
|
32
32
|
# because PyPI's default project page renders the latest stable release, which is still
|
|
33
|
-
# `v0.10.1` and carries none of this
|
|
34
|
-
# `.devN` pre-releases
|
|
35
|
-
#
|
|
33
|
+
# `v0.10.1` and carries none of this. The artifacts that carried it in the meantime were the
|
|
34
|
+
# `.devN` pre-releases SECURITY.md declares unsupported and `pip install` does not select;
|
|
35
|
+
# their own per-version pages DO show it, so this was narrow rather than free. With
|
|
36
|
+
# `publish-dev` disabled no new artifact carries it at all, which narrows the exposure
|
|
37
|
+
# further — the claim is not more true, it just reaches nobody until the next tag.
|
|
36
38
|
# Revisit if the next stable tag is not `v1.0.0`: README still says three public shapes
|
|
37
39
|
# change once, before the API is frozen, and Production/Stable on the default page would
|
|
38
40
|
# then contradict it.
|
|
@@ -74,7 +76,7 @@ keywords = [
|
|
|
74
76
|
# vulnerability-reporting channel. The repository is still named `log-forge` — the ORIGINAL name,
|
|
75
77
|
# which PyPI rejected for the distribution — so these URLs deliberately do not match the package
|
|
76
78
|
# name. See the note on `name` above before "correcting" them.
|
|
77
|
-
version = "0.
|
|
79
|
+
version = "1.0.0"
|
|
78
80
|
|
|
79
81
|
[project.urls]
|
|
80
82
|
Homepage = "https://github.com/agriffi10/log-forge"
|
|
@@ -87,8 +87,8 @@ def health() -> Health:
|
|
|
87
87
|
an event that cannot be *built* never reaches a queue at all — so every other field here
|
|
88
88
|
describes machinery those two losses never touched, and a process that only logs outside a
|
|
89
89
|
span read all zeros over total loss until they existed. They stay separate because one can
|
|
90
|
-
mean the destination or the data and the other
|
|
91
|
-
nobody can act on.
|
|
90
|
+
mean the destination or the data and the other never the destination — the data, or no drain
|
|
91
|
+
thread at all (SPEC-050 FR-003); their sum is a number nobody can act on.
|
|
92
92
|
|
|
93
93
|
``retired`` alone is not a fault — a process that shuts down and then stops logging is
|
|
94
94
|
doing the right thing, which is why it is paired with the count rather than alerted on.
|