log-foundry 0.10.2.dev89__tar.gz → 0.10.2.dev91__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (59) hide show
  1. {log_foundry-0.10.2.dev89 → log_foundry-0.10.2.dev91}/PKG-INFO +28 -13
  2. {log_foundry-0.10.2.dev89 → log_foundry-0.10.2.dev91}/README.md +27 -12
  3. {log_foundry-0.10.2.dev89 → log_foundry-0.10.2.dev91}/pyproject.toml +38 -3
  4. {log_foundry-0.10.2.dev89 → log_foundry-0.10.2.dev91}/src/log_foundry/sinks/kafka.py +34 -4
  5. {log_foundry-0.10.2.dev89 → log_foundry-0.10.2.dev91}/src/log_foundry/sinks/nats.py +48 -3
  6. {log_foundry-0.10.2.dev89 → log_foundry-0.10.2.dev91}/LICENSE +0 -0
  7. {log_foundry-0.10.2.dev89 → log_foundry-0.10.2.dev91}/src/log_foundry/__init__.py +0 -0
  8. {log_foundry-0.10.2.dev89 → log_foundry-0.10.2.dev91}/src/log_foundry/_diag.py +0 -0
  9. {log_foundry-0.10.2.dev89 → log_foundry-0.10.2.dev91}/src/log_foundry/_fork.py +0 -0
  10. {log_foundry-0.10.2.dev89 → log_foundry-0.10.2.dev91}/src/log_foundry/_lifecycle.py +0 -0
  11. {log_foundry-0.10.2.dev89 → log_foundry-0.10.2.dev91}/src/log_foundry/api.py +0 -0
  12. {log_foundry-0.10.2.dev89 → log_foundry-0.10.2.dev91}/src/log_foundry/config.py +0 -0
  13. {log_foundry-0.10.2.dev89 → log_foundry-0.10.2.dev91}/src/log_foundry/console.py +0 -0
  14. {log_foundry-0.10.2.dev89 → log_foundry-0.10.2.dev91}/src/log_foundry/context.py +0 -0
  15. {log_foundry-0.10.2.dev89 → log_foundry-0.10.2.dev91}/src/log_foundry/decorator.py +0 -0
  16. {log_foundry-0.10.2.dev89 → log_foundry-0.10.2.dev91}/src/log_foundry/ids.py +0 -0
  17. {log_foundry-0.10.2.dev89 → log_foundry-0.10.2.dev91}/src/log_foundry/model.py +0 -0
  18. {log_foundry-0.10.2.dev89 → log_foundry-0.10.2.dev91}/src/log_foundry/py.typed +0 -0
  19. {log_foundry-0.10.2.dev89 → log_foundry-0.10.2.dev91}/src/log_foundry/results.py +0 -0
  20. {log_foundry-0.10.2.dev89 → log_foundry-0.10.2.dev91}/src/log_foundry/sanitize.py +0 -0
  21. {log_foundry-0.10.2.dev89 → log_foundry-0.10.2.dev91}/src/log_foundry/sinks/__init__.py +0 -0
  22. {log_foundry-0.10.2.dev89 → log_foundry-0.10.2.dev91}/src/log_foundry/sinks/_batch.py +0 -0
  23. {log_foundry-0.10.2.dev89 → log_foundry-0.10.2.dev91}/src/log_foundry/sinks/_chunk.py +0 -0
  24. {log_foundry-0.10.2.dev89 → log_foundry-0.10.2.dev91}/src/log_foundry/sinks/_retry.py +0 -0
  25. {log_foundry-0.10.2.dev89 → log_foundry-0.10.2.dev91}/src/log_foundry/sinks/_socket.py +0 -0
  26. {log_foundry-0.10.2.dev89 → log_foundry-0.10.2.dev91}/src/log_foundry/sinks/_time.py +0 -0
  27. {log_foundry-0.10.2.dev89 → log_foundry-0.10.2.dev91}/src/log_foundry/sinks/base.py +0 -0
  28. {log_foundry-0.10.2.dev89 → log_foundry-0.10.2.dev91}/src/log_foundry/sinks/callback.py +0 -0
  29. {log_foundry-0.10.2.dev89 → log_foundry-0.10.2.dev91}/src/log_foundry/sinks/clickhouse.py +0 -0
  30. {log_foundry-0.10.2.dev89 → log_foundry-0.10.2.dev91}/src/log_foundry/sinks/datadog.py +0 -0
  31. {log_foundry-0.10.2.dev89 → log_foundry-0.10.2.dev91}/src/log_foundry/sinks/elasticsearch.py +0 -0
  32. {log_foundry-0.10.2.dev89 → log_foundry-0.10.2.dev91}/src/log_foundry/sinks/eventhubs.py +0 -0
  33. {log_foundry-0.10.2.dev89 → log_foundry-0.10.2.dev91}/src/log_foundry/sinks/file.py +0 -0
  34. {log_foundry-0.10.2.dev89 → log_foundry-0.10.2.dev91}/src/log_foundry/sinks/filtering.py +0 -0
  35. {log_foundry-0.10.2.dev89 → log_foundry-0.10.2.dev91}/src/log_foundry/sinks/firehose.py +0 -0
  36. {log_foundry-0.10.2.dev89 → log_foundry-0.10.2.dev91}/src/log_foundry/sinks/honeycomb.py +0 -0
  37. {log_foundry-0.10.2.dev89 → log_foundry-0.10.2.dev91}/src/log_foundry/sinks/http.py +0 -0
  38. {log_foundry-0.10.2.dev89 → log_foundry-0.10.2.dev91}/src/log_foundry/sinks/kinesis.py +0 -0
  39. {log_foundry-0.10.2.dev89 → log_foundry-0.10.2.dev91}/src/log_foundry/sinks/logging_sink.py +0 -0
  40. {log_foundry-0.10.2.dev89 → log_foundry-0.10.2.dev91}/src/log_foundry/sinks/logstash.py +0 -0
  41. {log_foundry-0.10.2.dev89 → log_foundry-0.10.2.dev91}/src/log_foundry/sinks/loki.py +0 -0
  42. {log_foundry-0.10.2.dev89 → log_foundry-0.10.2.dev91}/src/log_foundry/sinks/memory.py +0 -0
  43. {log_foundry-0.10.2.dev89 → log_foundry-0.10.2.dev91}/src/log_foundry/sinks/mongodb.py +0 -0
  44. {log_foundry-0.10.2.dev89 → log_foundry-0.10.2.dev91}/src/log_foundry/sinks/multi.py +0 -0
  45. {log_foundry-0.10.2.dev89 → log_foundry-0.10.2.dev91}/src/log_foundry/sinks/newrelic.py +0 -0
  46. {log_foundry-0.10.2.dev89 → log_foundry-0.10.2.dev91}/src/log_foundry/sinks/null.py +0 -0
  47. {log_foundry-0.10.2.dev89 → log_foundry-0.10.2.dev91}/src/log_foundry/sinks/postgres.py +0 -0
  48. {log_foundry-0.10.2.dev89 → log_foundry-0.10.2.dev91}/src/log_foundry/sinks/pubsub.py +0 -0
  49. {log_foundry-0.10.2.dev89 → log_foundry-0.10.2.dev91}/src/log_foundry/sinks/rabbitmq.py +0 -0
  50. {log_foundry-0.10.2.dev89 → log_foundry-0.10.2.dev91}/src/log_foundry/sinks/redis.py +0 -0
  51. {log_foundry-0.10.2.dev89 → log_foundry-0.10.2.dev91}/src/log_foundry/sinks/sentry.py +0 -0
  52. {log_foundry-0.10.2.dev89 → log_foundry-0.10.2.dev91}/src/log_foundry/sinks/sns.py +0 -0
  53. {log_foundry-0.10.2.dev89 → log_foundry-0.10.2.dev91}/src/log_foundry/sinks/splunk.py +0 -0
  54. {log_foundry-0.10.2.dev89 → log_foundry-0.10.2.dev91}/src/log_foundry/sinks/sqlite.py +0 -0
  55. {log_foundry-0.10.2.dev89 → log_foundry-0.10.2.dev91}/src/log_foundry/sinks/sqs.py +0 -0
  56. {log_foundry-0.10.2.dev89 → log_foundry-0.10.2.dev91}/src/log_foundry/sinks/stdout.py +0 -0
  57. {log_foundry-0.10.2.dev89 → log_foundry-0.10.2.dev91}/src/log_foundry/sinks/syslog.py +0 -0
  58. {log_foundry-0.10.2.dev89 → log_foundry-0.10.2.dev91}/src/log_foundry/sinks/transform.py +0 -0
  59. {log_foundry-0.10.2.dev89 → log_foundry-0.10.2.dev91}/src/log_foundry/worker.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: log-foundry
3
- Version: 0.10.2.dev89
3
+ Version: 0.10.2.dev91
4
4
  Summary: Generate logs for your console and JSON events for downstream consumption.
5
5
  License-Expression: MIT
6
6
  License-File: LICENSE
@@ -818,23 +818,37 @@ Two things worth knowing:
818
818
  #### Queue & stream
819
819
 
820
820
  Each needs its own extra (lazy-imported). All publish within a bound and close cleanly. **Where the
821
- retry lives differs, and it was measured rather than assumed** (SPEC-041 FR-004): the Redis,
822
- RabbitMQ and Event Hubs sinks retry through `sinks/_retry`, so their backoff is bounded *and* cut
823
- short by a shutdown. `KafkaSink`, `NATSSink` and `GooglePubSubSink` add no retry loop and need
824
- none — each hands off locally and returns without waiting, so nothing of theirs holds the single
825
- drain thread. Their clients retry on their own threads within their own bounds:
826
- `message.timeout.ms` (5 min default) for Kafka, a 600 s deadline for Pub/Sub, and for NATS a
827
- JetStream publish bounded by its 5 s ack timeout with no retry at all. `NATSSink` refuses a batch
828
- outright while its client reports itself disconnected, so a sustained outage moves
829
- `health().failed_batches` instead of being absorbed.
821
+ bound lives differs, and it was measured rather than assumed** (SPEC-041 FR-004, SPEC-047): the
822
+ Redis, RabbitMQ and Event Hubs sinks retry through `sinks/_retry`, so their backoff is bounded
823
+ *and* cut short by a shutdown.
824
+
825
+ `KafkaSink` and `GooglePubSubSink` add no retry loop and need none — each hands off locally and
826
+ returns without waiting, and their clients retry on their own threads within their own bounds.
827
+ (`KafkaSink` holds nothing at all; `GooglePubSubSink` does take one wait on the drain thread, its
828
+ `overflow_timeout`, when a batch exceeds `max_pending` — bounded and interruptible per SPEC-027.) For Kafka that is `message.timeout.ms`, five
829
+ minutes by default (measured: the delivery callback fires at 300.18 s), reachable through
830
+ `producer_config=`; for Pub/Sub, a 600 s deadline.
831
+
832
+ > ~~and for NATS a JetStream publish bounded by its 5 s ack timeout with no retry at all~~ —
833
+ > **superseded by SPEC-047 FR-001.** That was true per *event* and false per *batch*: the awaits
834
+ > were sequential and nothing bounded how many there were, so a stalled server cost `n × 5 s` on
835
+ > the single drain thread (measured: 25.01 s for five events), and `_final_drain` hands the exit
836
+ > backlog over as one batch.
837
+
838
+ `NATSSink` now bounds a whole `emit` with one `publish_timeout` (10 s by default), giving each
839
+ JetStream publish the lesser of the driver's ack timeout and the budget remaining; a core publish
840
+ takes no timeout and is bounded between events. Its connect, reconnect and drain timeouts are
841
+ reachable from the constructor — at the driver's defaults, construction against an unreachable
842
+ server blocks for a measured 120.17 s. It refuses a batch outright while its client reports itself
843
+ disconnected, so a sustained outage moves `health().failed_batches` instead of being absorbed.
830
844
 
831
845
  | Sink | Import from | Extra | Configure |
832
846
  |---|---|---|---|
833
- | `KafkaSink` | `log_foundry.sinks.kafka` | `kafka` | `KafkaSink(topic, *, flush_timeout=10.0, bootstrap_servers="…", key_field="trace_id")` |
847
+ | `KafkaSink` | `log_foundry.sinks.kafka` | `kafka` | `KafkaSink(topic, *, flush_timeout=10.0, bootstrap_servers="…", key_field="trace_id", producer_config=None)` — `producer_config` is merged **beneath** the sink's own keys, so it reaches `message.timeout.ms` and friends without displacing `bootstrap.servers`; passing it with `producer=` is a `ValueError` |
834
848
  | `RedisStreamsSink` | `log_foundry.sinks.redis` | `redis` | `RedisStreamsSink(stream, *, url=None, maxlen=None)` — `XADD`. `maxlen` caps the stream (`approximate=True`); trimming happens **at Redis**, after delivery, so it is invisible to `health()` — which is why the default is unbounded |
835
849
  | `RedisListSink` | `log_foundry.sinks.redis` | `redis` | `RedisListSink(key, *, url=None, maxlen=None)` — `RPUSH` + `LTRIM` to the newest `maxlen`; same destination-side trimming caveat |
836
850
  | `RabbitMQSink` | `log_foundry.sinks.rabbitmq` | `amqp` | `RabbitMQSink(*, exchange, routing_key, url=None)` — persistent messages |
837
- | `NATSSink` | `log_foundry.sinks.nats` | `nats` | `NATSSink(subject, *, jetstream=False, servers=None)` |
851
+ | `NATSSink` | `log_foundry.sinks.nats` | `nats` | `NATSSink(subject, *, jetstream=False, servers=None, publish_timeout=10.0, connect_timeout=None, max_reconnect_attempts=None, reconnect_time_wait=None, drain_timeout=None)` — `publish_timeout` bounds one whole `emit` and applies to an injected `client=` too; the four `None` timeouts are forwarded to `nats.connect` only when set, and passing one with `client=` is a `ValueError` |
838
852
  | `GooglePubSubSink` | `log_foundry.sinks.pubsub` | `gcp-pubsub` | `GooglePubSubSink(topic)` |
839
853
  | `AzureEventHubsSink` | `log_foundry.sinks.eventhubs` | `azure-eventhubs` | `AzureEventHubsSink(*, connection_str="…", eventhub=None)` |
840
854
 
@@ -1296,7 +1310,8 @@ bound each *value* — an event of many bounded values can still be large; see
1296
1310
 
1297
1311
  ```bash
1298
1312
  poetry install --with dev # set up (Python 3.12+)
1299
- poetry run pytest # test
1313
+ poetry run pytest # test (runs in parallel by default; see addopts)
1314
+ poetry run pytest -n 0 # ...serially, when debugging a failure
1300
1315
  poetry run ruff check . # lint (line-length 100)
1301
1316
  poetry run mypy # typecheck (strict, over src/)
1302
1317
  ```
@@ -782,23 +782,37 @@ Two things worth knowing:
782
782
  #### Queue & stream
783
783
 
784
784
  Each needs its own extra (lazy-imported). All publish within a bound and close cleanly. **Where the
785
- retry lives differs, and it was measured rather than assumed** (SPEC-041 FR-004): the Redis,
786
- RabbitMQ and Event Hubs sinks retry through `sinks/_retry`, so their backoff is bounded *and* cut
787
- short by a shutdown. `KafkaSink`, `NATSSink` and `GooglePubSubSink` add no retry loop and need
788
- none — each hands off locally and returns without waiting, so nothing of theirs holds the single
789
- drain thread. Their clients retry on their own threads within their own bounds:
790
- `message.timeout.ms` (5 min default) for Kafka, a 600 s deadline for Pub/Sub, and for NATS a
791
- JetStream publish bounded by its 5 s ack timeout with no retry at all. `NATSSink` refuses a batch
792
- outright while its client reports itself disconnected, so a sustained outage moves
793
- `health().failed_batches` instead of being absorbed.
785
+ bound lives differs, and it was measured rather than assumed** (SPEC-041 FR-004, SPEC-047): the
786
+ Redis, RabbitMQ and Event Hubs sinks retry through `sinks/_retry`, so their backoff is bounded
787
+ *and* cut short by a shutdown.
788
+
789
+ `KafkaSink` and `GooglePubSubSink` add no retry loop and need none — each hands off locally and
790
+ returns without waiting, and their clients retry on their own threads within their own bounds.
791
+ (`KafkaSink` holds nothing at all; `GooglePubSubSink` does take one wait on the drain thread, its
792
+ `overflow_timeout`, when a batch exceeds `max_pending` — bounded and interruptible per SPEC-027.) For Kafka that is `message.timeout.ms`, five
793
+ minutes by default (measured: the delivery callback fires at 300.18 s), reachable through
794
+ `producer_config=`; for Pub/Sub, a 600 s deadline.
795
+
796
+ > ~~and for NATS a JetStream publish bounded by its 5 s ack timeout with no retry at all~~ —
797
+ > **superseded by SPEC-047 FR-001.** That was true per *event* and false per *batch*: the awaits
798
+ > were sequential and nothing bounded how many there were, so a stalled server cost `n × 5 s` on
799
+ > the single drain thread (measured: 25.01 s for five events), and `_final_drain` hands the exit
800
+ > backlog over as one batch.
801
+
802
+ `NATSSink` now bounds a whole `emit` with one `publish_timeout` (10 s by default), giving each
803
+ JetStream publish the lesser of the driver's ack timeout and the budget remaining; a core publish
804
+ takes no timeout and is bounded between events. Its connect, reconnect and drain timeouts are
805
+ reachable from the constructor — at the driver's defaults, construction against an unreachable
806
+ server blocks for a measured 120.17 s. It refuses a batch outright while its client reports itself
807
+ disconnected, so a sustained outage moves `health().failed_batches` instead of being absorbed.
794
808
 
795
809
  | Sink | Import from | Extra | Configure |
796
810
  |---|---|---|---|
797
- | `KafkaSink` | `log_foundry.sinks.kafka` | `kafka` | `KafkaSink(topic, *, flush_timeout=10.0, bootstrap_servers="…", key_field="trace_id")` |
811
+ | `KafkaSink` | `log_foundry.sinks.kafka` | `kafka` | `KafkaSink(topic, *, flush_timeout=10.0, bootstrap_servers="…", key_field="trace_id", producer_config=None)` — `producer_config` is merged **beneath** the sink's own keys, so it reaches `message.timeout.ms` and friends without displacing `bootstrap.servers`; passing it with `producer=` is a `ValueError` |
798
812
  | `RedisStreamsSink` | `log_foundry.sinks.redis` | `redis` | `RedisStreamsSink(stream, *, url=None, maxlen=None)` — `XADD`. `maxlen` caps the stream (`approximate=True`); trimming happens **at Redis**, after delivery, so it is invisible to `health()` — which is why the default is unbounded |
799
813
  | `RedisListSink` | `log_foundry.sinks.redis` | `redis` | `RedisListSink(key, *, url=None, maxlen=None)` — `RPUSH` + `LTRIM` to the newest `maxlen`; same destination-side trimming caveat |
800
814
  | `RabbitMQSink` | `log_foundry.sinks.rabbitmq` | `amqp` | `RabbitMQSink(*, exchange, routing_key, url=None)` — persistent messages |
801
- | `NATSSink` | `log_foundry.sinks.nats` | `nats` | `NATSSink(subject, *, jetstream=False, servers=None)` |
815
+ | `NATSSink` | `log_foundry.sinks.nats` | `nats` | `NATSSink(subject, *, jetstream=False, servers=None, publish_timeout=10.0, connect_timeout=None, max_reconnect_attempts=None, reconnect_time_wait=None, drain_timeout=None)` — `publish_timeout` bounds one whole `emit` and applies to an injected `client=` too; the four `None` timeouts are forwarded to `nats.connect` only when set, and passing one with `client=` is a `ValueError` |
802
816
  | `GooglePubSubSink` | `log_foundry.sinks.pubsub` | `gcp-pubsub` | `GooglePubSubSink(topic)` |
803
817
  | `AzureEventHubsSink` | `log_foundry.sinks.eventhubs` | `azure-eventhubs` | `AzureEventHubsSink(*, connection_str="…", eventhub=None)` |
804
818
 
@@ -1260,7 +1274,8 @@ bound each *value* — an event of many bounded values can still be large; see
1260
1274
 
1261
1275
  ```bash
1262
1276
  poetry install --with dev # set up (Python 3.12+)
1263
- poetry run pytest # test
1277
+ poetry run pytest # test (runs in parallel by default; see addopts)
1278
+ poetry run pytest -n 0 # ...serially, when debugging a failure
1264
1279
  poetry run ruff check . # lint (line-length 100)
1265
1280
  poetry run mypy # typecheck (strict, over src/)
1266
1281
  ```
@@ -20,7 +20,7 @@ dependencies = [
20
20
  ]
21
21
 
22
22
  # Optional features. Install with: pip install log-foundry[aws]
23
- version = "0.10.2.dev89"
23
+ version = "0.10.2.dev91"
24
24
 
25
25
  [project.optional-dependencies]
26
26
  aws = ["boto3>=1.43.61"] # SQSSink, SNSSink, KinesisSink, FirehoseSink
@@ -67,7 +67,7 @@ pytest-cov = "^7.1.0"
67
67
  # CPU, because the bounds this library promises (a 30 s shutdown budget, a backoff, a
68
68
  # wedged drain) are proved by waiting for them. Two tests block 30 s each. Distributing
69
69
  # across processes therefore buys almost the whole wall clock back — measured 125.9 s ->
70
- # 40.8 s at `-n 8 --dist load` — where caching the dependency install could only ever
70
+ # 34.8 s at `-n 12 --dist worksteal` — where caching the dependency install could only ever
71
71
  # save the 18 s that install takes.
72
72
  pytest-xdist = "^3.8"
73
73
  # Bounded to a minor, not left open. Ruff's DEFAULT rule set is not stable across minors —
@@ -114,7 +114,42 @@ build-backend = "poetry_dynamic_versioning.backend"
114
114
  testpaths = ["tests"]
115
115
  pythonpath = ["src"] # src layout: make `import log_foundry` work without an install
116
116
  asyncio_mode = "auto" # async test functions run without an explicit marker
117
- addopts = "--strict-markers"
117
+ # `-n 12 --dist worksteal` lives HERE rather than in the CI workflow, so the gate a developer runs
118
+ # before pushing is the same command CI runs, at the same speed: ~133 s -> ~35 s locally, and the
119
+ # pre-push gate is where those 98 s are actually paid for by a human waiting. It also removes a
120
+ # divergence — with the flags only in CI, an order-dependent failure could appear there and be
121
+ # unreproducible with a bare `poetry run pytest`.
122
+ #
123
+ # The suite is sleep-bound, not CPU-bound (~133 s of wall clock at well under 10% CPU), because
124
+ # this library's promises ARE bounds and a bound is proved by waiting for it; two tests in
125
+ # test_orphan_sink_handoff.py block 30 s each. That is why the count is a fixed 12 rather than
126
+ # `auto`: workers here are overwhelmingly blocked rather than computing, so `auto` — the CPU
127
+ # count, which is 4 on a GitHub runner — leaves most of the win on the table.
128
+ #
129
+ # `--dist worksteal`, NOT `--dist load`, and this was learned the expensive way. `load` hands each
130
+ # worker a fixed slice up front, so the allocation is a deterministic function of the test list —
131
+ # and when SPEC-047 added a handful of NATS tests, both 30 s tests landed in the SAME slice and
132
+ # the parallel run went from 40.8 s to 71.4 s (three runs, spread 0.1 s: not luck, and it would
133
+ # have stayed green in CI while quietly costing 30 s a run). `worksteal` lets an idle worker take
134
+ # work from a busy one, which is exactly the long-tail shape here. Measured on the same tree,
135
+ # 1,899 passed / 7 skipped every time:
136
+ #
137
+ # serial 133.4 s -n 8 --dist worksteal 40.0 s
138
+ # -n 8 --dist load 71.4 s -n 12 --dist worksteal 34.8 s <- chosen
139
+ # -n 12 --dist load 38.4 s -n 16 --dist worksteal 34.1 s (knee)
140
+ #
141
+ # `loadfile` is worse than either: it keeps a module on one worker, pinning those two 30 s tests
142
+ # to one process and flooring the run at ~72 s — measured at both -n 8 and -n 12, though not at
143
+ # every worker count, since `-n 1 --dist loadfile` is 125.4 s.
144
+ #
145
+ # TO OVERRIDE, PASS `-n 0` — it disables distribution entirely (verified: no workers are created)
146
+ # and is the right flag for debugging a failure, for `--pdb`, or for reading interleaved output.
147
+ # `tests/integration` is run that way in CI for a stronger reason than convenience: those tests
148
+ # share nine real services, so distributing them would have them reading each other's writes.
149
+ # That one is ENFORCED, not merely configured -- `tests/integration/conftest.py` refuses a
150
+ # distributed session, so deleting `-n 0` from the workflow fails loudly instead of producing the
151
+ # intermittent cross-talk it used to.
152
+ addopts = "--strict-markers -n 12 --dist worksteal"
118
153
 
119
154
  [tool.ruff]
120
155
  src = ["src", "tests"]
@@ -54,7 +54,7 @@ class KafkaSink:
54
54
  serviced by ``poll()`` and ``flush()``.
55
55
 
56
56
  **Retry (SPEC-041 FR-004).** This sink adds none and needs none: ``produce()`` is a local
57
- hand-off that returns without waiting — measured at 0.0000 s for three messages against a
57
+ hand-off that returns without waiting — measured at 0.0001 s for three messages against a
58
58
  dead broker — so it never holds the worker's single drain thread, and librdkafka's own retry
59
59
  is bounded by ``message.timeout.ms`` (five minutes by default). Measured: with that set to
60
60
  1500 ms the delivery callback fired at 2.01 s, and at 4000 ms it fired at 4.01 s. SPEC-027's
@@ -62,6 +62,14 @@ class KafkaSink:
62
62
  which is also why it deliberately exposes no ``log_foundry_stop_signal`` attribute, since
63
63
  ``_lifecycle.offer_stop_signal`` probes by ``hasattr`` and that absence is the opt-out.
64
64
 
65
+ **The bound is ``librdkafka``'s, and since SPEC-047 FR-003 it is reachable.** ``message.
66
+ timeout.ms`` is what decides how long an accepted message is retried, five minutes by
67
+ default — measured, the delivery callback fires at 300.18 s — and before ``producer_config``
68
+ nothing could change it without abandoning ``bootstrap_servers=`` and injecting a whole
69
+ producer. No log-foundry retry was added instead: ``produce()`` having returned means the
70
+ producer owns delivery, so re-sending duplicates whatever it eventually lands (SPEC-018), and
71
+ a second bound over a five-minute one multiplies the worst case rather than shortening it.
72
+
65
73
  The worst case (SPEC-027 FR-005) is not a retry loop — this sink has none — but its
66
74
  ``close()``: one ``flush_timeout`` wait, 10 s at the default, spent **beside**
67
75
  ``shutdown()``'s budget rather than from it. ``Worker.shutdown`` joins the drain thread
@@ -95,6 +103,7 @@ class KafkaSink:
95
103
  bootstrap_servers: str | None = None,
96
104
  key_field: str = "trace_id",
97
105
  flush_timeout: float = DEFAULT_FLUSH_TIMEOUT,
106
+ producer_config: dict[str, object] | None = None,
98
107
  ) -> None:
99
108
  """Binds the sink to a topic and a producer.
100
109
 
@@ -110,22 +119,43 @@ class KafkaSink:
110
119
  default, as ``Worker._emit`` floors its own retries (SPEC-021): ``0`` reinstates
111
120
  exactly the ``flush(0)`` that switched this sink's exit delivery off, and ``inf``
112
121
  reinstates the unbounded wait.
122
+ producer_config: ``librdkafka`` configuration merged **beneath** this sink's own keys
123
+ when it builds the producer (SPEC-047 FR-003). It is the route to the bound that
124
+ actually governs delivery here: this sink adds no retry — ``produce()`` is a local
125
+ hand-off, measured at 0.0001 s for three messages against a dead broker — and
126
+ ``librdkafka`` retries on its own thread within ``message.timeout.ms``, whose
127
+ five-minute default was measured at a 300.18 s delivery callback. Before this
128
+ argument nothing reached it, so a caller working to a shorter deadline could not fit
129
+ the sink inside one. The default is deliberately left alone: lowering it would drop
130
+ messages that a process surviving a two-minute broker outage delivers today, which
131
+ is the durable-buffer role ``architecture.md`` section 8 gives this sink.
113
132
 
114
133
  Returns:
115
134
  None.
116
135
 
117
136
  Raises:
118
- ValueError: If no producer is injected and no broker list was given.
137
+ ValueError: If no producer is injected and no broker list was given, or if
138
+ ``producer_config`` is passed alongside ``producer=`` — it can only be applied to a
139
+ producer this sink builds, and ignoring it would silently discard the caller's bound
140
+ (SPEC-043's rule).
119
141
  ImportError: If the ``kafka`` extra is not installed.
120
142
  """
121
- if producer is None:
143
+ if producer is not None:
144
+ if producer_config is not None:
145
+ raise ValueError(
146
+ "KafkaSink cannot apply producer_config to an injected producer, which is "
147
+ "already built; pass the settings where the producer is constructed"
148
+ )
149
+ else:
122
150
  if bootstrap_servers is None:
123
151
  raise ValueError(
124
152
  "KafkaSink requires bootstrap_servers when no producer is injected"
125
153
  )
126
154
  from confluent_kafka import Producer # type: ignore[import-not-found]
127
155
 
128
- producer = Producer({"bootstrap.servers": bootstrap_servers})
156
+ producer = Producer(
157
+ {**(producer_config or {}), "bootstrap.servers": bootstrap_servers}
158
+ )
129
159
  self.topic = topic
130
160
  self.producer = producer
131
161
  self.key_field = key_field
@@ -87,8 +87,11 @@ class NATSSink:
87
87
  disconnected.
88
88
 
89
89
  The limit is stated rather than overclaimed: ``is_connected`` does not flip the instant the
90
- server dies, so the first batch in the window before the client notices is still accepted and
91
- still buffered — measured, one of five emits landed in that window. It is the same
90
+ server dies, so a batch in the window before the client notices is still accepted and still
91
+ buffered. ~~Measured, one of five emits landed in that window.~~ **Corrected by SPEC-047
92
+ FR-004:** the window is wider than that reading suggests — the flag was still ``True``
93
+ **40 s after the server was stopped**, which is a lower bound on the window rather than the
94
+ window, because ``nats-py`` notices through a 120 s ping interval or a failed write. It is the same
92
95
  check-then-act window ``MongoDBSink`` and ``KafkaSink`` already document for their own flags;
93
96
  what the guard ends is the far larger case of an outage that has been going on for any
94
97
  appreciable time. Refusing moves no ``losses()`` counter, per SPEC-032: it is a failure
@@ -104,6 +107,10 @@ class NATSSink:
104
107
  jetstream: bool = False,
105
108
  servers: str | None = None,
106
109
  publish_timeout: float = DEFAULT_PUBLISH_TIMEOUT,
110
+ connect_timeout: float | None = None,
111
+ max_reconnect_attempts: int | None = None,
112
+ reconnect_time_wait: float | None = None,
113
+ drain_timeout: float | None = None,
107
114
  ) -> None:
108
115
  """Binds the sink to a subject and connects if no client was injected.
109
116
 
@@ -116,14 +123,52 @@ class NATSSink:
116
123
  :func:`~log_foundry.sinks._retry.usable_timeout` as ``KafkaSink`` floors its flush
117
124
  (SPEC-047 FR-001). It applies to an injected ``client=`` too: it is this sink's own
118
125
  bound over its own loop, not a request made of the driver at connect time.
126
+ connect_timeout: Seconds the driver may spend on one connection attempt, or ``None``
127
+ to pass nothing and leave the driver's own default (SPEC-047 FR-002).
128
+ max_reconnect_attempts: Attempts the driver makes before giving up, or ``None`` to
129
+ pass nothing. This is what governs the *initial* connect too, which is why the
130
+ constructor blocked for a measured 120.17 s against a dead server at the driver's
131
+ defaults (60 attempts x 2 s) before this argument existed. **``0`` and negative
132
+ values do not mean "give up immediately" — they block forever.** ``nats-py`` retires
133
+ a server from its pool only under ``if max_reconnect_attempts > 0``, so a
134
+ non-positive value never retires one and the connect loop does not terminate;
135
+ measured, both ``0`` and ``-1`` were still blocking at 30 s and one probe of ``0``
136
+ ran past 400 s. The smallest value that bounds anything is ``1`` (measured 2.02 s
137
+ against a dead server). Passed through rather than corrected here, because
138
+ overriding a driver's documented behaviour would surprise a caller who knows it.
139
+ reconnect_time_wait: Seconds between reconnect attempts, or ``None`` to pass nothing.
140
+ drain_timeout: Seconds ``Client.drain`` may spend, or ``None`` to pass nothing. It is
141
+ forwarded because it does bound the driver's subscription drain, but it is **not**
142
+ what bounds :meth:`close`: ``drain()`` ends with a ``flush()`` carrying the driver's
143
+ own 10 s default, and that is what fired in both measured cases (10.00 s against a
144
+ stalled server, 0.00 s against a stopped one). Recorded in ``architecture.md``
145
+ section 12.
119
146
 
120
147
  Returns:
121
148
  None.
122
149
 
123
150
  Raises:
151
+ ValueError: If a connect-time argument is passed alongside ``client=``, which cannot
152
+ consume it — SPEC-043's rule that an argument no backend can use is an error rather
153
+ than a silent ignore. ``publish_timeout`` is deliberately outside that set: it is
154
+ this sink's own bound and applies to any client.
124
155
  ImportError: If the ``nats`` extra is not installed.
125
156
  Exception: Whatever the driver raises when connecting.
126
157
  """
158
+ options: dict[str, object] = {
159
+ "connect_timeout": connect_timeout,
160
+ "max_reconnect_attempts": max_reconnect_attempts,
161
+ "reconnect_time_wait": reconnect_time_wait,
162
+ "drain_timeout": drain_timeout,
163
+ }
164
+ supplied = {name: value for name, value in options.items() if value is not None}
165
+ if client is not None and supplied:
166
+ raise ValueError(
167
+ "NATSSink cannot apply "
168
+ + ", ".join(sorted(supplied))
169
+ + " to an injected client, which is already connected; "
170
+ "pass them where the client is built, or drop client="
171
+ )
127
172
  self._subject = subject
128
173
  self._jetstream = jetstream
129
174
  self.publish_timeout = usable_timeout(publish_timeout, DEFAULT_PUBLISH_TIMEOUT)
@@ -135,7 +180,7 @@ class NATSSink:
135
180
  import nats # type: ignore[import-not-found]
136
181
 
137
182
  client = self._loop.run_until_complete(
138
- nats.connect(servers or "nats://localhost:4222")
183
+ nats.connect(servers or "nats://localhost:4222", **supplied)
139
184
  )
140
185
  self._client = client
141
186