log-foundry 0.10.2.dev72__tar.gz → 0.10.2.dev74__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (59) hide show
  1. {log_foundry-0.10.2.dev72 → log_foundry-0.10.2.dev74}/PKG-INFO +35 -9
  2. {log_foundry-0.10.2.dev72 → log_foundry-0.10.2.dev74}/README.md +34 -8
  3. {log_foundry-0.10.2.dev72 → log_foundry-0.10.2.dev74}/pyproject.toml +7 -1
  4. {log_foundry-0.10.2.dev72 → log_foundry-0.10.2.dev74}/src/log_foundry/sinks/kafka.py +9 -0
  5. {log_foundry-0.10.2.dev72 → log_foundry-0.10.2.dev74}/src/log_foundry/sinks/logstash.py +46 -5
  6. {log_foundry-0.10.2.dev72 → log_foundry-0.10.2.dev74}/src/log_foundry/sinks/nats.py +55 -0
  7. {log_foundry-0.10.2.dev72 → log_foundry-0.10.2.dev74}/src/log_foundry/sinks/postgres.py +165 -8
  8. {log_foundry-0.10.2.dev72 → log_foundry-0.10.2.dev74}/src/log_foundry/sinks/pubsub.py +8 -0
  9. {log_foundry-0.10.2.dev72 → log_foundry-0.10.2.dev74}/src/log_foundry/sinks/sentry.py +195 -18
  10. {log_foundry-0.10.2.dev72 → log_foundry-0.10.2.dev74}/LICENSE +0 -0
  11. {log_foundry-0.10.2.dev72 → log_foundry-0.10.2.dev74}/src/log_foundry/__init__.py +0 -0
  12. {log_foundry-0.10.2.dev72 → log_foundry-0.10.2.dev74}/src/log_foundry/_diag.py +0 -0
  13. {log_foundry-0.10.2.dev72 → log_foundry-0.10.2.dev74}/src/log_foundry/_fork.py +0 -0
  14. {log_foundry-0.10.2.dev72 → log_foundry-0.10.2.dev74}/src/log_foundry/_lifecycle.py +0 -0
  15. {log_foundry-0.10.2.dev72 → log_foundry-0.10.2.dev74}/src/log_foundry/api.py +0 -0
  16. {log_foundry-0.10.2.dev72 → log_foundry-0.10.2.dev74}/src/log_foundry/config.py +0 -0
  17. {log_foundry-0.10.2.dev72 → log_foundry-0.10.2.dev74}/src/log_foundry/console.py +0 -0
  18. {log_foundry-0.10.2.dev72 → log_foundry-0.10.2.dev74}/src/log_foundry/context.py +0 -0
  19. {log_foundry-0.10.2.dev72 → log_foundry-0.10.2.dev74}/src/log_foundry/decorator.py +0 -0
  20. {log_foundry-0.10.2.dev72 → log_foundry-0.10.2.dev74}/src/log_foundry/ids.py +0 -0
  21. {log_foundry-0.10.2.dev72 → log_foundry-0.10.2.dev74}/src/log_foundry/model.py +0 -0
  22. {log_foundry-0.10.2.dev72 → log_foundry-0.10.2.dev74}/src/log_foundry/py.typed +0 -0
  23. {log_foundry-0.10.2.dev72 → log_foundry-0.10.2.dev74}/src/log_foundry/results.py +0 -0
  24. {log_foundry-0.10.2.dev72 → log_foundry-0.10.2.dev74}/src/log_foundry/sanitize.py +0 -0
  25. {log_foundry-0.10.2.dev72 → log_foundry-0.10.2.dev74}/src/log_foundry/sinks/__init__.py +0 -0
  26. {log_foundry-0.10.2.dev72 → log_foundry-0.10.2.dev74}/src/log_foundry/sinks/_batch.py +0 -0
  27. {log_foundry-0.10.2.dev72 → log_foundry-0.10.2.dev74}/src/log_foundry/sinks/_chunk.py +0 -0
  28. {log_foundry-0.10.2.dev72 → log_foundry-0.10.2.dev74}/src/log_foundry/sinks/_retry.py +0 -0
  29. {log_foundry-0.10.2.dev72 → log_foundry-0.10.2.dev74}/src/log_foundry/sinks/_socket.py +0 -0
  30. {log_foundry-0.10.2.dev72 → log_foundry-0.10.2.dev74}/src/log_foundry/sinks/_time.py +0 -0
  31. {log_foundry-0.10.2.dev72 → log_foundry-0.10.2.dev74}/src/log_foundry/sinks/base.py +0 -0
  32. {log_foundry-0.10.2.dev72 → log_foundry-0.10.2.dev74}/src/log_foundry/sinks/callback.py +0 -0
  33. {log_foundry-0.10.2.dev72 → log_foundry-0.10.2.dev74}/src/log_foundry/sinks/clickhouse.py +0 -0
  34. {log_foundry-0.10.2.dev72 → log_foundry-0.10.2.dev74}/src/log_foundry/sinks/datadog.py +0 -0
  35. {log_foundry-0.10.2.dev72 → log_foundry-0.10.2.dev74}/src/log_foundry/sinks/elasticsearch.py +0 -0
  36. {log_foundry-0.10.2.dev72 → log_foundry-0.10.2.dev74}/src/log_foundry/sinks/eventhubs.py +0 -0
  37. {log_foundry-0.10.2.dev72 → log_foundry-0.10.2.dev74}/src/log_foundry/sinks/file.py +0 -0
  38. {log_foundry-0.10.2.dev72 → log_foundry-0.10.2.dev74}/src/log_foundry/sinks/filtering.py +0 -0
  39. {log_foundry-0.10.2.dev72 → log_foundry-0.10.2.dev74}/src/log_foundry/sinks/firehose.py +0 -0
  40. {log_foundry-0.10.2.dev72 → log_foundry-0.10.2.dev74}/src/log_foundry/sinks/honeycomb.py +0 -0
  41. {log_foundry-0.10.2.dev72 → log_foundry-0.10.2.dev74}/src/log_foundry/sinks/http.py +0 -0
  42. {log_foundry-0.10.2.dev72 → log_foundry-0.10.2.dev74}/src/log_foundry/sinks/kinesis.py +0 -0
  43. {log_foundry-0.10.2.dev72 → log_foundry-0.10.2.dev74}/src/log_foundry/sinks/logging_sink.py +0 -0
  44. {log_foundry-0.10.2.dev72 → log_foundry-0.10.2.dev74}/src/log_foundry/sinks/loki.py +0 -0
  45. {log_foundry-0.10.2.dev72 → log_foundry-0.10.2.dev74}/src/log_foundry/sinks/memory.py +0 -0
  46. {log_foundry-0.10.2.dev72 → log_foundry-0.10.2.dev74}/src/log_foundry/sinks/mongodb.py +0 -0
  47. {log_foundry-0.10.2.dev72 → log_foundry-0.10.2.dev74}/src/log_foundry/sinks/multi.py +0 -0
  48. {log_foundry-0.10.2.dev72 → log_foundry-0.10.2.dev74}/src/log_foundry/sinks/newrelic.py +0 -0
  49. {log_foundry-0.10.2.dev72 → log_foundry-0.10.2.dev74}/src/log_foundry/sinks/null.py +0 -0
  50. {log_foundry-0.10.2.dev72 → log_foundry-0.10.2.dev74}/src/log_foundry/sinks/rabbitmq.py +0 -0
  51. {log_foundry-0.10.2.dev72 → log_foundry-0.10.2.dev74}/src/log_foundry/sinks/redis.py +0 -0
  52. {log_foundry-0.10.2.dev72 → log_foundry-0.10.2.dev74}/src/log_foundry/sinks/sns.py +0 -0
  53. {log_foundry-0.10.2.dev72 → log_foundry-0.10.2.dev74}/src/log_foundry/sinks/splunk.py +0 -0
  54. {log_foundry-0.10.2.dev72 → log_foundry-0.10.2.dev74}/src/log_foundry/sinks/sqlite.py +0 -0
  55. {log_foundry-0.10.2.dev72 → log_foundry-0.10.2.dev74}/src/log_foundry/sinks/sqs.py +0 -0
  56. {log_foundry-0.10.2.dev72 → log_foundry-0.10.2.dev74}/src/log_foundry/sinks/stdout.py +0 -0
  57. {log_foundry-0.10.2.dev72 → log_foundry-0.10.2.dev74}/src/log_foundry/sinks/syslog.py +0 -0
  58. {log_foundry-0.10.2.dev72 → log_foundry-0.10.2.dev74}/src/log_foundry/sinks/transform.py +0 -0
  59. {log_foundry-0.10.2.dev72 → log_foundry-0.10.2.dev74}/src/log_foundry/worker.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: log-foundry
3
- Version: 0.10.2.dev72
3
+ Version: 0.10.2.dev74
4
4
  Summary: Generate logs for your console and JSON events for downstream consumption.
5
5
  License-Expression: MIT
6
6
  License-File: LICENSE
@@ -694,7 +694,7 @@ completed, not that every chunk of it landed.
694
694
  | `ElasticsearchSink` | `log_foundry.sinks.elasticsearch` | `ElasticsearchSink(url, *, index, auth=None, **http_kwargs)` — POST to `_bulk`, parsing per-item errors (`.item_errors`) |
695
695
  | `OpenSearchSink` | `log_foundry.sinks.elasticsearch` | same signature as `ElasticsearchSink` (identical bulk protocol) |
696
696
  | `LokiSink` | `log_foundry.sinks.loki` | `LokiSink(url, *, labels=("service", "env", "level"), **http_kwargs)` — Grafana Loki push API |
697
- | `LogstashSink` | `log_foundry.sinks.logstash` | `LogstashSink(url=…, **http_kwargs)` for HTTP, **or** `LogstashSink(host=…, port=…, transport="tcp")` for a raw TCP/UDP socket |
697
+ | `LogstashSink` | `log_foundry.sinks.logstash` | `LogstashSink(url=…, body_format="json_array", **http_kwargs)` for HTTP, **or** `LogstashSink(host=…, port=…, transport="tcp")` for a raw TCP/UDP socket. HTTP mode posts a JSON array as `application/json`, which a **stock** `http` input parses into one event per element; pass `body_format="ndjson"` for an input configured with `additional_codecs => {"application/x-ndjson" => "json_lines"}`, which that setting *replaces* the default map to provide |
698
698
  | `SyslogSink` | `log_foundry.sinks.syslog` | `SyslogSink(host, port=514, *, transport="udp", facility="user", app_name="log-foundry", max_datagram_bytes=65507)` — RFC 5424 over UDP/TCP. A UDP frame over the limit is dropped and counted rather than sent, retried and abandoned; TCP is a stream and is unaffected |
699
699
 
700
700
  ```python
@@ -706,7 +706,7 @@ lf.configure(sink=ElasticsearchSink("https://es.internal:9200", index="app-logs"
706
706
  #### SaaS platforms
707
707
 
708
708
  Also HTTP-based. All are zero-dependency **except** `SentrySink`, which prefers the `sentry-sdk`
709
- (the `sentry` extra) and falls back to raw HTTP envelopes when it isn't installed.
709
+ (the `sentry` extra) and falls back to raw HTTP envelopes when it cannot deliver through one.
710
710
 
711
711
  | Sink | Import from | Extra | Configure |
712
712
  |---|---|---|---|
@@ -714,11 +714,28 @@ Also HTTP-based. All are zero-dependency **except** `SentrySink`, which prefers
714
714
  | `SplunkHECSink` | `log_foundry.sinks.splunk` | — | `SplunkHECSink(url, token, *, host=None, source="log-foundry")` — HTTP Event Collector |
715
715
  | `NewRelicSink` | `log_foundry.sinks.newrelic` | — | `NewRelicSink(api_key, *, region="US")` — `region` is `"US"` or `"EU"` |
716
716
  | `HoneycombSink` | `log_foundry.sinks.honeycomb` | — | `HoneycombSink(api_key, dataset, *, url="https://api.honeycomb.io")` |
717
- | `SentrySink` | `log_foundry.sinks.sentry` | `sentry` | `SentrySink(dsn=None, *, min_level="ERROR")` — sends only `min_level`+ events |
717
+ | `SentrySink` | `log_foundry.sinks.sentry` | `sentry` | `SentrySink(dsn=None, *, min_level="ERROR", backend="auto")` — sends only `min_level`+ events |
718
718
 
719
- With the `sentry` extra installed, `SentrySink` captures via `sentry_sdk.capture_event` (initialize
720
- the SDK yourself with `sentry_sdk.init(...)`); without it, pass `dsn=` and events are POSTed as
721
- Sentry envelopes over HTTP.
719
+ `SentrySink` captures via `sentry_sdk.capture_event` when the SDK can deliver — you initialize it
720
+ yourself with `sentry_sdk.init(...)` — and POSTs Sentry envelopes over HTTP to `dsn=` otherwise.
721
+ With neither a deliverable SDK nor a `dsn=`, a batch is **refused** rather than silently dropped.
722
+ `backend=` decides:
723
+
724
+ | `backend=` | What it uses |
725
+ |---|---|
726
+ | `"auto"` (default) | the SDK when it can deliver, otherwise the HTTP fallback; re-decided on every batch, so an `init(...)` that arrives after the sink was built is picked up |
727
+ | `"sdk"` | the SDK only. A client that cannot deliver **refuses the batch** rather than quietly switching transport |
728
+ | `"http"` | the HTTP fallback only. No SDK is held, consulted or flushed |
729
+
730
+ "Can deliver" means the SDK's client reports itself active *and* holds a transport. An
731
+ uninitialized process, an `init()` with no DSN, and a `close()`d client all fail that — the first
732
+ reports itself inactive, the other two report themselves active with nothing to send through, and
733
+ all three drop events silently.
734
+
735
+ An argument the chosen backend can never use is a `ValueError` rather than a silent ignore:
736
+ `opener=` where no HTTP fallback is built, `client=` under `backend="http"`. Until this release
737
+ `opener=` was accepted and then ignored whenever the SDK happened to import — including when it was
738
+ installed as somebody else's transitive dependency.
722
739
 
723
740
  #### AWS — the durable-buffer path (`aws` extra)
724
741
 
@@ -797,7 +814,16 @@ Two things worth knowing:
797
814
 
798
815
  #### Queue & stream
799
816
 
800
- Each needs its own extra (lazy-imported). All publish + retry within a bound and close cleanly.
817
+ Each needs its own extra (lazy-imported). All publish within a bound and close cleanly. **Where the
818
+ retry lives differs, and it was measured rather than assumed** (SPEC-041 FR-004): the Redis,
819
+ RabbitMQ and Event Hubs sinks retry through `sinks/_retry`, so their backoff is bounded *and* cut
820
+ short by a shutdown. `KafkaSink`, `NATSSink` and `GooglePubSubSink` add no retry loop and need
821
+ none — each hands off locally and returns without waiting, so nothing of theirs holds the single
822
+ drain thread. Their clients retry on their own threads within their own bounds:
823
+ `message.timeout.ms` (5 min default) for Kafka, a 600 s deadline for Pub/Sub, and for NATS a
824
+ JetStream publish bounded by its 5 s ack timeout with no retry at all. `NATSSink` refuses a batch
825
+ outright while its client reports itself disconnected, so a sustained outage moves
826
+ `health().failed_batches` instead of being absorbed.
801
827
 
802
828
  | Sink | Import from | Extra | Configure |
803
829
  |---|---|---|---|
@@ -821,7 +847,7 @@ Write-only inserts (querying is the downstream tool's job); each needs its own e
821
847
  | Sink | Import from | Extra | Configure |
822
848
  |---|---|---|---|
823
849
  | `MongoDBSink` | `log_foundry.sinks.mongodb` | `mongo` | `MongoDBSink(*, uri="…", database="…", collection="…")` |
824
- | `PostgresSink` | `log_foundry.sinks.postgres` | `postgres` | `PostgresSink(table, *, dsn="…", create_table=False)` — JSONB `event` column + extracted columns |
850
+ | `PostgresSink` | `log_foundry.sinks.postgres` | `postgres` | `PostgresSink(table, *, dsn="…", create_table=False, connect_timeout=5)` — JSONB `event` column + extracted columns. Reconnects an **owned** connection the server has closed; a `connection=` you inject is never reopened. `connect_timeout` is passed to libpq explicitly, so it **overrides** any `connect_timeout` in your DSN |
825
851
  | `ClickHouseSink` | `log_foundry.sinks.clickhouse` | `clickhouse` | `ClickHouseSink(table, *, dsn="…", create_table=False)` — MergeTree, columnar insert |
826
852
 
827
853
  `PostgresSink` / `ClickHouseSink` default `create_table=False` (you own the schema and indexes); set
@@ -658,7 +658,7 @@ completed, not that every chunk of it landed.
658
658
  | `ElasticsearchSink` | `log_foundry.sinks.elasticsearch` | `ElasticsearchSink(url, *, index, auth=None, **http_kwargs)` — POST to `_bulk`, parsing per-item errors (`.item_errors`) |
659
659
  | `OpenSearchSink` | `log_foundry.sinks.elasticsearch` | same signature as `ElasticsearchSink` (identical bulk protocol) |
660
660
  | `LokiSink` | `log_foundry.sinks.loki` | `LokiSink(url, *, labels=("service", "env", "level"), **http_kwargs)` — Grafana Loki push API |
661
- | `LogstashSink` | `log_foundry.sinks.logstash` | `LogstashSink(url=…, **http_kwargs)` for HTTP, **or** `LogstashSink(host=…, port=…, transport="tcp")` for a raw TCP/UDP socket |
661
+ | `LogstashSink` | `log_foundry.sinks.logstash` | `LogstashSink(url=…, body_format="json_array", **http_kwargs)` for HTTP, **or** `LogstashSink(host=…, port=…, transport="tcp")` for a raw TCP/UDP socket. HTTP mode posts a JSON array as `application/json`, which a **stock** `http` input parses into one event per element; pass `body_format="ndjson"` for an input configured with `additional_codecs => {"application/x-ndjson" => "json_lines"}`, which that setting *replaces* the default map to provide |
662
662
  | `SyslogSink` | `log_foundry.sinks.syslog` | `SyslogSink(host, port=514, *, transport="udp", facility="user", app_name="log-foundry", max_datagram_bytes=65507)` — RFC 5424 over UDP/TCP. A UDP frame over the limit is dropped and counted rather than sent, retried and abandoned; TCP is a stream and is unaffected |
663
663
 
664
664
  ```python
@@ -670,7 +670,7 @@ lf.configure(sink=ElasticsearchSink("https://es.internal:9200", index="app-logs"
670
670
  #### SaaS platforms
671
671
 
672
672
  Also HTTP-based. All are zero-dependency **except** `SentrySink`, which prefers the `sentry-sdk`
673
- (the `sentry` extra) and falls back to raw HTTP envelopes when it isn't installed.
673
+ (the `sentry` extra) and falls back to raw HTTP envelopes when it cannot deliver through one.
674
674
 
675
675
  | Sink | Import from | Extra | Configure |
676
676
  |---|---|---|---|
@@ -678,11 +678,28 @@ Also HTTP-based. All are zero-dependency **except** `SentrySink`, which prefers
678
678
  | `SplunkHECSink` | `log_foundry.sinks.splunk` | — | `SplunkHECSink(url, token, *, host=None, source="log-foundry")` — HTTP Event Collector |
679
679
  | `NewRelicSink` | `log_foundry.sinks.newrelic` | — | `NewRelicSink(api_key, *, region="US")` — `region` is `"US"` or `"EU"` |
680
680
  | `HoneycombSink` | `log_foundry.sinks.honeycomb` | — | `HoneycombSink(api_key, dataset, *, url="https://api.honeycomb.io")` |
681
- | `SentrySink` | `log_foundry.sinks.sentry` | `sentry` | `SentrySink(dsn=None, *, min_level="ERROR")` — sends only `min_level`+ events |
681
+ | `SentrySink` | `log_foundry.sinks.sentry` | `sentry` | `SentrySink(dsn=None, *, min_level="ERROR", backend="auto")` — sends only `min_level`+ events |
682
682
 
683
- With the `sentry` extra installed, `SentrySink` captures via `sentry_sdk.capture_event` (initialize
684
- the SDK yourself with `sentry_sdk.init(...)`); without it, pass `dsn=` and events are POSTed as
685
- Sentry envelopes over HTTP.
683
+ `SentrySink` captures via `sentry_sdk.capture_event` when the SDK can deliver — you initialize it
684
+ yourself with `sentry_sdk.init(...)` — and POSTs Sentry envelopes over HTTP to `dsn=` otherwise.
685
+ With neither a deliverable SDK nor a `dsn=`, a batch is **refused** rather than silently dropped.
686
+ `backend=` decides:
687
+
688
+ | `backend=` | What it uses |
689
+ |---|---|
690
+ | `"auto"` (default) | the SDK when it can deliver, otherwise the HTTP fallback; re-decided on every batch, so an `init(...)` that arrives after the sink was built is picked up |
691
+ | `"sdk"` | the SDK only. A client that cannot deliver **refuses the batch** rather than quietly switching transport |
692
+ | `"http"` | the HTTP fallback only. No SDK is held, consulted or flushed |
693
+
694
+ "Can deliver" means the SDK's client reports itself active *and* holds a transport. An
695
+ uninitialized process, an `init()` with no DSN, and a `close()`d client all fail that — the first
696
+ reports itself inactive, the other two report themselves active with nothing to send through, and
697
+ all three drop events silently.
698
+
699
+ An argument the chosen backend can never use is a `ValueError` rather than a silent ignore:
700
+ `opener=` where no HTTP fallback is built, `client=` under `backend="http"`. Until this release
701
+ `opener=` was accepted and then ignored whenever the SDK happened to import — including when it was
702
+ installed as somebody else's transitive dependency.
686
703
 
687
704
  #### AWS — the durable-buffer path (`aws` extra)
688
705
 
@@ -761,7 +778,16 @@ Two things worth knowing:
761
778
 
762
779
  #### Queue & stream
763
780
 
764
- Each needs its own extra (lazy-imported). All publish + retry within a bound and close cleanly.
781
+ Each needs its own extra (lazy-imported). All publish within a bound and close cleanly. **Where the
782
+ retry lives differs, and it was measured rather than assumed** (SPEC-041 FR-004): the Redis,
783
+ RabbitMQ and Event Hubs sinks retry through `sinks/_retry`, so their backoff is bounded *and* cut
784
+ short by a shutdown. `KafkaSink`, `NATSSink` and `GooglePubSubSink` add no retry loop and need
785
+ none — each hands off locally and returns without waiting, so nothing of theirs holds the single
786
+ drain thread. Their clients retry on their own threads within their own bounds:
787
+ `message.timeout.ms` (5 min default) for Kafka, a 600 s deadline for Pub/Sub, and for NATS a
788
+ JetStream publish bounded by its 5 s ack timeout with no retry at all. `NATSSink` refuses a batch
789
+ outright while its client reports itself disconnected, so a sustained outage moves
790
+ `health().failed_batches` instead of being absorbed.
765
791
 
766
792
  | Sink | Import from | Extra | Configure |
767
793
  |---|---|---|---|
@@ -785,7 +811,7 @@ Write-only inserts (querying is the downstream tool's job); each needs its own e
785
811
  | Sink | Import from | Extra | Configure |
786
812
  |---|---|---|---|
787
813
  | `MongoDBSink` | `log_foundry.sinks.mongodb` | `mongo` | `MongoDBSink(*, uri="…", database="…", collection="…")` |
788
- | `PostgresSink` | `log_foundry.sinks.postgres` | `postgres` | `PostgresSink(table, *, dsn="…", create_table=False)` — JSONB `event` column + extracted columns |
814
+ | `PostgresSink` | `log_foundry.sinks.postgres` | `postgres` | `PostgresSink(table, *, dsn="…", create_table=False, connect_timeout=5)` — JSONB `event` column + extracted columns. Reconnects an **owned** connection the server has closed; a `connection=` you inject is never reopened. `connect_timeout` is passed to libpq explicitly, so it **overrides** any `connect_timeout` in your DSN |
789
815
  | `ClickHouseSink` | `log_foundry.sinks.clickhouse` | `clickhouse` | `ClickHouseSink(table, *, dsn="…", create_table=False)` — MergeTree, columnar insert |
790
816
 
791
817
  `PostgresSink` / `ClickHouseSink` default `create_table=False` (you own the schema and indexes); set
@@ -20,7 +20,7 @@ dependencies = [
20
20
  ]
21
21
 
22
22
  # Optional features. Install with: pip install log-foundry[aws]
23
- version = "0.10.2.dev72"
23
+ version = "0.10.2.dev74"
24
24
 
25
25
  [project.optional-dependencies]
26
26
  aws = ["boto3>=1.43.61"] # SQSSink, SNSSink, KinesisSink, FirehoseSink
@@ -185,6 +185,12 @@ ignore = [
185
185
  # interpolates something it did not generate should still trip it.
186
186
  "tests/integration/test_postgres.py" = ["S608"]
187
187
  "tests/integration/test_clickhouse.py" = ["S608"]
188
+ # The NATS outage test stops and restarts its own compose service, because "the server went
189
+ # away" is the condition under test and no fake reproduces what the real client does about it.
190
+ # Every argument is a literal or a path this module derives from `__file__`; there is no shell
191
+ # and no caller input. `docker` is resolved from PATH deliberately -- CI and a developer laptop
192
+ # put it in different places.
193
+ "tests/integration/test_nats.py" = ["S603", "S607"]
188
194
  # make-sbom.py exists to drive two virtualenvs it creates itself, so subprocess is the whole job.
189
195
  # Every argv element is either a literal, a path under a `tempfile.TemporaryDirectory` this script
190
196
  # just made, or a value read from the repository's own pyproject.toml / poetry.lock. There is no
@@ -53,6 +53,15 @@ class KafkaSink:
53
53
  returns without blocking, while the delivery result arrives asynchronously on a callback
54
54
  serviced by ``poll()`` and ``flush()``.
55
55
 
56
+ **Retry (SPEC-041 FR-004).** This sink adds none and needs none: ``produce()`` is a local
57
+ hand-off that returns without waiting — measured at 0.0000 s for three messages against a
58
+ dead broker — so it never holds the worker's single drain thread, and librdkafka's own retry
59
+ is bounded by ``message.timeout.ms`` (five minutes by default). Measured: with that set to
60
+ 1500 ms the delivery callback fired at 2.01 s, and at 4000 ms it fired at 4.01 s. SPEC-027's
61
+ "cut short by a shutdown" governs a wait *between attempts*, and this sink has none to cut —
62
+ which is also why it deliberately exposes no ``log_foundry_stop_signal`` attribute, since
63
+ ``_lifecycle.offer_stop_signal`` probes by ``hasattr`` and that absence is the opt-out.
64
+
56
65
  The worst case (SPEC-027 FR-005) is not a retry loop — this sink has none — but its
57
66
  ``close()``: one ``flush_timeout`` wait, 10 s at the default, spent **beside**
58
67
  ``shutdown()``'s budget rather than from it. ``Worker.shutdown`` joins the drain thread
@@ -15,16 +15,44 @@ from log_foundry.sinks.http import HTTPSink
15
15
 
16
16
  __all__ = ["LogstashSink"]
17
17
 
18
+ _BODY_FORMATS = frozenset({"json_array", "ndjson"})
19
+ """The two HTTP wire forms, validated rather than passed through.
20
+
21
+ ``HTTPSink._body`` branches on ``== "json_array"`` and treats everything else as NDJSON, so a
22
+ typo -- ``"json-array"`` -- would silently ship the exact body FR-003 exists to stop shipping,
23
+ against a destination that cannot parse it. A parameter documented as taking one of two values
24
+ refuses the third.
25
+ """
26
+
18
27
 
19
28
  class LogstashSink:
20
29
  """A :class:`~log_foundry.sinks.base.Sink` that ships JSON lines to Logstash (FR-005).
21
30
 
22
- Two mutually-exclusive modes are chosen at construction: with a URL the batch goes as JSON
23
- lines over HTTP, reusing the ``HTTPSink`` core, and with a host and port each event goes as
24
- one newline-terminated line over a raw TCP or UDP socket, reusing
31
+ Two mutually-exclusive modes are chosen at construction: with a URL the batch goes over HTTP,
32
+ reusing the ``HTTPSink`` core, and with a host and port each event goes as one
33
+ newline-terminated line over a raw TCP or UDP socket, reusing
25
34
  :class:`~log_foundry.sinks._socket.SocketTransport`. Either backend handles its own bounded
26
35
  retry and raises on total failure of its own accord, so ``emit`` needs no rule of its own.
27
36
 
37
+ **What the Logstash side must be configured as** (SPEC-041 FR-003). In HTTP mode the default
38
+ ``body_format="json_array"`` posts a JSON array as ``application/json``, which a **stock**
39
+ ``http`` input parses into one event per element with no configuration at all. That is a
40
+ change: this sink used to post NDJSON as ``application/x-ndjson``, which the ``http`` input's
41
+ default ``additional_codecs`` map (``{"application/json" => "json"}``) does not cover, so the
42
+ body fell through to the ``plain`` codec. Measured against Logstash 8.15 — a batch of three
43
+ arrived as **one** event with the whole payload as text in ``message`` and every structured
44
+ field gone.
45
+
46
+ ``body_format="ndjson"`` restores the old wire form, and it is not merely legacy: an input
47
+ configured ``additional_codecs => {"application/x-ndjson" => "json_lines"}`` -- the documented
48
+ workaround for the behaviour above -- needs it. That setting **replaces** the default map
49
+ rather than merging with it, so such an input parses NDJSON correctly and a JSON array
50
+ incorrectly. The two configurations are mutually exclusive unless the input lists both, which
51
+ is why the wire form is a parameter rather than a silent change.
52
+
53
+ Socket mode is unaffected: it is newline-delimited either way, and a ``tcp``/``udp`` input
54
+ uses the ``line``/``json_lines`` codec rather than ``additional_codecs``.
55
+
28
56
  In socket mode both IPv4 and IPv6 destinations are supported, over either transport
29
57
  (SPEC-031 FR-002): the UDP socket's address family is resolved from ``host`` rather than
30
58
  assumed, which is what an unconditional ``AF_INET`` used to make impossible.
@@ -58,6 +86,7 @@ class LogstashSink:
58
86
  host: str | None = None,
59
87
  port: int | None = None,
60
88
  transport: str = "tcp",
89
+ body_format: str = "json_array",
61
90
  timeout: float = 5.0,
62
91
  max_retries: int = 3,
63
92
  max_datagram_bytes: int = DEFAULT_MAX_DATAGRAM_BYTES,
@@ -70,6 +99,12 @@ class LogstashSink:
70
99
  host: The destination host, selecting socket mode.
71
100
  port: The destination port, selecting socket mode.
72
101
  transport: ``"tcp"`` or ``"udp"``, in socket mode.
102
+ body_format: ``"json_array"`` (the default) or ``"ndjson"``. It selects the HTTP wire
103
+ form and has **no effect** in socket mode, which is always newline-delimited — but it
104
+ is *validated* in both, because a value this class silently ignores in one mode and
105
+ silently misreads in the other is worse than a refusal. See the class docstring for
106
+ what each one requires of the Logstash side; ``"ndjson"`` is the escape hatch for an
107
+ input already configured with ``additional_codecs``.
73
108
  timeout: Seconds allowed per request or connection.
74
109
  max_retries: Retries the chosen backend makes.
75
110
  max_datagram_bytes: In UDP socket mode, the largest datagram to attempt; a frame over
@@ -81,11 +116,17 @@ class LogstashSink:
81
116
  None.
82
117
 
83
118
  Raises:
84
- ValueError: If neither a URL nor a host and port were given.
119
+ ValueError: If neither a URL nor a host and port were given, or if ``body_format`` is
120
+ neither ``"json_array"`` nor ``"ndjson"``.
85
121
  """
122
+ if body_format not in _BODY_FORMATS:
123
+ raise ValueError(
124
+ f"LogstashSink body_format must be one of {sorted(_BODY_FORMATS)}, "
125
+ f"not {body_format!r}"
126
+ )
86
127
  if url is not None:
87
128
  self._http: HTTPSink | None = HTTPSink(
88
- url, body_format="ndjson", timeout=timeout, max_retries=max_retries,
129
+ url, body_format=body_format, timeout=timeout, max_retries=max_retries,
89
130
  **http_kwargs, # type: ignore[arg-type]
90
131
  )
91
132
  self._socket: SocketTransport | None = None
@@ -21,6 +21,31 @@ class NATSSink:
21
21
  completion from the synchronous ``emit``, which therefore returns only after the batch is
22
22
  handed off. With JetStream enabled it publishes through JetStream for durable
23
23
  acknowledgement.
24
+
25
+ **Retry and the worst-case delay** (SPEC-041 FR-004). This sink adds no retry loop and needs
26
+ none: a core ``publish()`` writes into the client's outbound buffer and returns without
27
+ waiting — measured at 0.00 s for fifty publishes — so it never holds the worker's single
28
+ drain thread and there is no backoff for a shutdown to cut short. SPEC-027's guarantee is met
29
+ because there is no wait, not because a wait is bounded. Under JetStream ``publish()`` awaits
30
+ an ack bounded by the driver's own timeout (5 s by default) and does not retry.
31
+
32
+ **A disconnected client is reported, not absorbed** (FR-004 AC-5). That non-blocking publish
33
+ is exactly what made this sink report success for events that had not left the process: with
34
+ the server stopped, five successive emits each returned in 0.00 s with ``losses()`` reading
35
+ all zeros, and when the client's reconnect budget (60 attempts × 2 s by default) ran out
36
+ first, **one of six events reached the destination with every counter still at zero**. That
37
+ is SPEC-026 FR-001's shape — a sink the worker believes, so its retry never engages and
38
+ ``failed_batches`` never moves. ``emit`` now refuses a batch while the client reports itself
39
+ disconnected.
40
+
41
+ The limit is stated rather than overclaimed: ``is_connected`` does not flip the instant the
42
+ server dies, so the first batch in the window before the client notices is still accepted and
43
+ still buffered — measured, one of five emits landed in that window. It is the same
44
+ check-then-act window ``MongoDBSink`` and ``KafkaSink`` already document for their own flags;
45
+ what the guard ends is the far larger case of an outage that has been going on for any
46
+ appreciable time. Refusing moves no ``losses()`` counter, per SPEC-032: it is a failure
47
+ *reported* to the worker rather than one absorbed, and counting both would report one loss
48
+ twice.
24
49
  """
25
50
 
26
51
  def __init__(
@@ -86,8 +111,38 @@ class NATSSink:
86
111
  raise SinkDeliveryError(
87
112
  f"NATSSink published none of {len(batch)} event(s): the sink is closed"
88
113
  )
114
+ if not self._is_connected():
115
+ raise SinkDeliveryError(
116
+ f"NATSSink published none of {len(batch)} event(s): "
117
+ "the client is disconnected"
118
+ )
89
119
  self._loop.run_until_complete(self._publish_all(batch))
90
120
 
121
+ def _is_connected(self) -> bool:
122
+ """Reports whether the client can currently put anything on the wire (FR-004 AC-5).
123
+
124
+ Probed by name, as ``drain`` and ``flush`` are, because this sink is written against a
125
+ driver it does not own and accepts an injected ``client=``. A client that does not
126
+ publish the attribute is assumed connected: the guard exists to convert a *known*
127
+ non-delivery into a reported one, and inventing a refusal for a client that never
128
+ claimed to be disconnected would fail batches that were going to succeed.
129
+
130
+ Args:
131
+ None.
132
+
133
+ Returns:
134
+ True when the client reports itself connected, or says nothing about it.
135
+
136
+ Raises:
137
+ None. A driver whose attribute access raises is treated as connected, so a probe can
138
+ never be the reason a batch fails.
139
+ """
140
+ try:
141
+ connected = getattr(self._client, "is_connected", True)
142
+ except Exception:
143
+ return True
144
+ return bool(connected)
145
+
91
146
  def losses(self) -> SinkLosses:
92
147
  """Reports events whose publish raised (SPEC-026 FR-002).
93
148
 
@@ -15,6 +15,22 @@ __all__ = ["PostgresSink"]
15
15
 
16
16
  _BACKOFF_BASE = 0.1
17
17
 
18
+ DEFAULT_CONNECT_TIMEOUT = 5
19
+ """Seconds libpq may spend opening one connection (SPEC-041 FR-002).
20
+
21
+ ``psycopg.connect()`` takes no timeout unless one is given, and libpq's default is to wait
22
+ indefinitely. That call runs on the worker's single drain thread, holding this sink's emit lock,
23
+ and it consults no stop signal -- so against a host that blackholes packets rather than refusing
24
+ them it is exactly the unbounded, uninterruptible wait SPEC-027 exists to remove. Measured
25
+ against an unroutable address: **75.01 s** with no timeout, **2.03 s** with ``connect_timeout=2``.
26
+
27
+ Floored at libpq's own minimum of 2 rather than accepted verbatim, because ``0`` means "wait
28
+ forever" -- reinstating the defect -- the same reason ``KafkaSink._usable_timeout`` refuses its
29
+ own degenerate values (SPEC-038 FR-006).
30
+ """
31
+
32
+ _MIN_CONNECT_TIMEOUT = 2
33
+
18
34
  _COLUMNS = ("timestamp", "level", "trace_id", "span_id", "function", "service")
19
35
 
20
36
 
@@ -23,8 +39,15 @@ class PostgresSink:
23
39
 
24
40
  Each event is stored as a ``JSONB`` ``event`` column plus a few extracted columns for
25
41
  indexing. ``psycopg`` v3 is the optional ``postgres`` extra, imported lazily. The sink is
26
- write-only, and the worst-case delay (SPEC-027 FR-005) is ``max_retries`` interruptible waits
27
- per batch, 0.7 s at the defaults.
42
+ write-only. The worst-case delay (SPEC-027 FR-005) has two halves, and only one of them is
43
+ interruptible. The backoffs are ``max_retries`` waits per batch — 0.7 s at the defaults —
44
+ taken through ``_retry.wait`` on the worker's stop event, so a shutdown cuts them short. The
45
+ reconnects are up to ``max_retries + 1`` connects of ``connect_timeout`` each, 20 s more at
46
+ the defaults, and they are **bounded but not interruptible**: the wait is inside libpq, which
47
+ consults nothing. That is the settled line rather than a gap — a shutdown shortens a *wait*
48
+ and never skips *work*, and a reconnect is the work an in-flight batch needs (SPEC-038
49
+ FR-001 AC-4a). Both halves sit inside the existing retry budget rather than a loop of their
50
+ own, which is what bounds them (FR-002 AC-3).
28
51
 
29
52
  The driver requirement satisfied (SPEC-028 FR-002): a ``psycopg`` connection carries one
30
53
  transaction, and this sink's unit of work is a ``cursor`` / ``commit`` / ``rollback`` sequence
@@ -48,6 +71,7 @@ class PostgresSink:
48
71
  create_table: bool = False,
49
72
  chunk_size: int = 1000,
50
73
  max_retries: int = 3,
74
+ connect_timeout: int = DEFAULT_CONNECT_TIMEOUT,
51
75
  ) -> None:
52
76
  """Connects to the database and prepares the insert statement.
53
77
 
@@ -61,6 +85,13 @@ class PostgresSink:
61
85
  max_retries: Retries per batch, floored at zero as ``Worker._emit`` floors its own
62
86
  (SPEC-021) — a negative value returned having attempted no insert at all, and
63
87
  reported success.
88
+ connect_timeout: Seconds libpq may spend opening a connection. Defaults to
89
+ :data:`DEFAULT_CONNECT_TIMEOUT` (5) and is floored at libpq's own minimum of 2, since
90
+ ``0`` means "wait forever" and would reinstate the unbounded connect this argument
91
+ exists to remove. It is passed explicitly and therefore **overrides any
92
+ ``connect_timeout`` in the DSN** — a DSN asking for 30 gets this value instead, so
93
+ set it here rather than there. Applies to the connection opened at construction and
94
+ to every reconnect.
64
95
 
65
96
  Returns:
66
97
  None.
@@ -78,11 +109,12 @@ class PostgresSink:
78
109
  self._lock = threading.Lock()
79
110
  self._counter_lock = threading.Lock()
80
111
  self._owns_connection = connection is None
112
+ self._dsn = dsn
113
+ self.connect_timeout = max(connect_timeout, _MIN_CONNECT_TIMEOUT)
81
114
  if connection is None:
82
- import psycopg # type: ignore[import-not-found]
83
-
84
- connection = psycopg.connect(dsn)
115
+ connection = self._connect()
85
116
  self._conn = connection
117
+ self._reconnect_announced = False
86
118
  columns = ", ".join((*_COLUMNS, "event"))
87
119
  placeholders = ", ".join(["%s"] * len(_COLUMNS) + ["%s::jsonb"])
88
120
  self._insert_sql = f"INSERT INTO {self._table} ({columns}) VALUES ({placeholders})"
@@ -153,6 +185,7 @@ class PostgresSink:
153
185
  """
154
186
  for attempt in range(self.max_retries + 1):
155
187
  try:
188
+ self._reconnect_if_broken()
156
189
  with self._conn.cursor() as cur:
157
190
  for chunk in chunk_list(batch, self._chunk_size):
158
191
  cur.executemany(self._insert_sql, [self._row(event) for event in chunk])
@@ -168,12 +201,126 @@ class PostgresSink:
168
201
  _diag.lost(
169
202
  "event",
170
203
  len(batch),
171
- f"PostgresSink, {self.max_retries + 1} attempts, {type(err).__name__}",
204
+ f"PostgresSink, {self.max_retries + 1} attempts, {type(err).__name__}"
205
+ + (
206
+ ", borrowed connection is broken and this sink may not reopen it"
207
+ if not self._owns_connection and self._is_broken()
208
+ else ""
209
+ ),
172
210
  )
173
211
  raise SinkDeliveryError(
174
212
  f"PostgresSink inserted none of {len(batch)} event(s)"
175
213
  ) from None
176
214
 
215
+ def _connect(self) -> Any:
216
+ """Opens a connection to the configured DSN, bounded by :attr:`connect_timeout`.
217
+
218
+ The bound is passed as a keyword rather than left to the DSN, so it holds **regardless of**
219
+ what the caller's connection string says — which is the same fact ``__init__`` states from
220
+ the caller's side, that this argument overrides a ``connect_timeout`` in the DSN. See
221
+ :data:`DEFAULT_CONNECT_TIMEOUT` for why an unbounded connect here is the defect rather
222
+ than the default.
223
+
224
+ Args:
225
+ None.
226
+
227
+ Returns:
228
+ The new connection.
229
+
230
+ Raises:
231
+ ImportError: If the ``postgres`` extra is not installed.
232
+ Exception: Whatever the driver raises when connecting.
233
+ """
234
+ import psycopg # type: ignore[import-not-found]
235
+
236
+ return psycopg.connect(self._dsn, connect_timeout=self.connect_timeout)
237
+
238
+ def _reconnect_if_broken(self) -> None:
239
+ """Replaces an unusable **owned** connection before an insert attempt (FR-002).
240
+
241
+ A ``psycopg`` connection is permanently unusable once the server closes it, and this sink
242
+ opened one in ``__init__`` and never reopened it — so a single restart, failover or idle
243
+ timeout ended log delivery for the life of the process, with every in-batch retry running
244
+ against the same dead handle. Measured against a real Postgres: one
245
+ ``pg_terminate_backend`` and three subsequent batches were lost, one row delivered.
246
+ Every sibling already recovers (``SocketTransport._reset``, boto3, clickhouse-connect's
247
+ pool, pymongo's pool).
248
+
249
+ **It runs at the top of each attempt, not in the retry branch.** ``max_retries`` is a
250
+ public argument floored at zero, so a reconnect placed where a retry remains never runs
251
+ at all at ``max_retries=0`` and the defect would survive at that setting forever. At the
252
+ top of the attempt, the first emit after a failure reopens — which is FR-002 AC-1
253
+ literally — at every value. It also avoids spending a connect immediately before the
254
+ raise, where it can only cost time.
255
+
256
+ **A borrowed connection is never reopened**, per ``architecture.md`` §13's borrowed-client
257
+ constraint: the caller owns that object and its lifetime, and replacing it here would
258
+ leak theirs and reconnect a session they may be sharing. The exhausted batch's existing
259
+ ``_diag.lost`` line gains a clause rather than a new stderr site, and it is conditioned on
260
+ the connection actually being broken — appending it to every borrowed-connection failure
261
+ would put "not reopened" on a constraint violation, where reopening was never the
262
+ question.
263
+
264
+ The state is read from the connection rather than inferred: ``closed`` and ``broken`` are
265
+ both ``False`` before the first failure and both ``True`` after it, so an ordinary SQL
266
+ error — a constraint violation, a full disk — does not churn the connection. Both are
267
+ probed by name because this sink accepts any ``psycopg``-shaped object it does not own.
268
+
269
+ Both diagnostics here are **announced once per outage, not once per attempt**. Unthrottled
270
+ they fire on every attempt of every batch, so a down server turned one stderr line into
271
+ five per batch, indefinitely — a diagnostic that floods is one an operator stops reading,
272
+ and the batch's own ``_diag.lost`` line already records the loss. The flag covers the
273
+ failed ``close()`` as well as the failed connect, because a failed reconnect leaves the
274
+ old object in place and the next attempt closes it again. It clears on a successful
275
+ reconnect, so a later outage is announced again.
276
+
277
+ Args:
278
+ None.
279
+
280
+ Returns:
281
+ None.
282
+
283
+ Raises:
284
+ None. A failed reconnect is absorbed and left to the next attempt, which is what keeps
285
+ this inside the existing retry budget rather than adding a loop of its own (AC-3):
286
+ the batch's bound is unchanged, and a connect that cannot succeed simply spends an
287
+ attempt visibly, through the counters.
288
+ """
289
+ if not self._owns_connection or not self._is_broken():
290
+ return
291
+ announced, self._reconnect_announced = self._reconnect_announced, True
292
+ try:
293
+ self._conn.close()
294
+ except Exception as err:
295
+ if not announced:
296
+ _diag.absorbed("PostgresSink.close of a broken connection", err)
297
+ try:
298
+ self._conn = self._connect()
299
+ self._reconnect_announced = False
300
+ except Exception as err:
301
+ if not announced:
302
+ _diag.absorbed("PostgresSink.reconnect", err)
303
+
304
+ def _is_broken(self) -> bool:
305
+ """Reports whether the held connection can no longer be used.
306
+
307
+ Args:
308
+ None.
309
+
310
+ Returns:
311
+ True when the driver reports the connection closed or broken.
312
+
313
+ Raises:
314
+ None. A driver whose attribute access raises is treated as usable, so a probe can
315
+ never be the reason a batch fails.
316
+ """
317
+ try:
318
+ return bool(getattr(self._conn, "closed", False)) or bool(
319
+ getattr(self._conn, "broken", False)
320
+ )
321
+ except Exception:
322
+ return False
323
+
177
324
  def _rollback(self) -> None:
178
325
  """Discards the failed transaction, absorbing a rollback that itself fails (FR-002).
179
326
 
@@ -210,6 +357,13 @@ class PostgresSink:
210
357
  Idempotent, and takes the emit lock so the final commit never lands in the middle of
211
358
  another thread's transaction (SPEC-028 FR-002).
212
359
 
360
+ **The final commit is guarded** (SPEC-041 FR-002). It was unconditional, so a connection
361
+ the server had closed made ``close()`` raise — reachable without any exotic timing, since
362
+ a broken connection is only repaired inside an emit attempt and a process that breaks and
363
+ then shuts down never has another. The release still has to happen, and a commit that
364
+ cannot reach the server has nothing to publish, so the failure is announced by type and
365
+ the close proceeds.
366
+
213
367
  Args:
214
368
  None.
215
369
 
@@ -217,13 +371,16 @@ class PostgresSink:
217
371
  None.
218
372
 
219
373
  Raises:
220
- Exception: Whatever the driver raises on commit or close.
374
+ Exception: Whatever the driver raises on close. The commit no longer escapes.
221
375
  """
222
376
  with self._lock:
223
377
  if self._closed:
224
378
  return
225
379
  self._closed = True
226
- self._conn.commit()
380
+ try:
381
+ self._conn.commit()
382
+ except Exception as err:
383
+ _diag.absorbed("PostgresSink.close commit", err)
227
384
  if self._owns_connection:
228
385
  self._conn.close()
229
386
 
@@ -59,6 +59,14 @@ class GooglePubSubSink:
59
59
  imported lazily. ``publish()`` returns a future that resolves asynchronously, so the sink
60
60
  accumulates the batch's futures and resolves them on :meth:`close`.
61
61
 
62
+ **Retry (SPEC-041 FR-004).** The client's own retry is bounded and runs on the client's
63
+ threads, never the worker's drain thread: the generated ``publish`` carries
64
+ ``Retry(initial=0.1, maximum=60.0, multiplier=4, deadline=600.0)`` with a 60 s per-call
65
+ timeout, so a publish gives up after ten minutes at the outside. What this sink contributes
66
+ is the only wait the drain thread ever takes — :meth:`_await_overflow`'s
67
+ ``overflow_timeout``, bounded and interruptible per SPEC-027. Measured with the destination
68
+ stopped and ``overflow_timeout=5.0``: every over-bound emit returned in 5.00 s exactly.
69
+
62
70
  The driver requirement satisfied (SPEC-028 FR-002): this sink takes **no** transport lock —
63
71
  the publisher client owns its own batching and threading, and ``publish()`` is a local
64
72
  hand-off. What it does hold is the pending-futures list, which is genuinely shared between
@@ -5,7 +5,7 @@ from __future__ import annotations
5
5
  import json
6
6
  import threading
7
7
  import uuid
8
- from typing import Any
8
+ from typing import Any, Final, Literal, get_args
9
9
  from urllib.parse import urlparse
10
10
 
11
11
  from log_foundry import _diag, _lifecycle
@@ -14,6 +14,17 @@ from log_foundry.sinks.http import HTTPSink
14
14
 
15
15
  __all__ = ["SentrySink"]
16
16
 
17
+ Backend = Literal["auto", "sdk", "http"]
18
+ """Which transport a :class:`SentrySink` uses. Not exported: callers pass the literals."""
19
+
20
+ _BACKENDS: Final = get_args(Backend)
21
+
22
+ _Selected = Literal["sdk", "http"]
23
+ """A backend actually chosen for one batch -- never ``"auto"``, which selects rather than is."""
24
+
25
+ _ABSENT: Final = object()
26
+ """Sentinel telling an absent member from one whose value is ``None`` (SPEC-043 FR-001)."""
27
+
17
28
  _LEVEL_RANK = {"DEBUG": 10, "INFO": 20, "WARNING": 30, "ERROR": 40, "CRITICAL": 50}
18
29
  _SENTRY_LEVEL = {
19
30
  "DEBUG": "debug", "INFO": "info", "WARNING": "warning", "ERROR": "error", "CRITICAL": "fatal",
@@ -23,12 +34,39 @@ _SENTRY_LEVEL = {
23
34
  class SentrySink:
24
35
  """A :class:`~log_foundry.sinks.base.Sink` that captures qualifying events to Sentry (FR-011).
25
36
 
26
- It uses the ``sentry-sdk`` when the optional extra is installed, imported lazily inside the
27
- sink so importing this module never requires it. Without the SDK it falls back to POSTing a
28
- Sentry envelope over HTTP to the DSN's ingest URL. Only events at or above the configured
29
- minimum level are sent.
37
+ It uses the ``sentry-sdk`` when one is installed *and able to deliver*, imported lazily inside
38
+ the sink so importing this module never requires it. Otherwise it POSTs a Sentry envelope over
39
+ HTTP to the DSN's ingest URL. Only events at or above the configured minimum level are sent.
40
+
41
+ ``backend`` selects explicitly and ``"auto"`` is the default (SPEC-043 FR-002). An explicit
42
+ selection is honoured rather than substituted: under ``"sdk"`` a client that cannot deliver is
43
+ refused, never diverted to HTTP, and under ``"http"`` no client is held, consulted or flushed.
44
+ What is built, and what each ``emit`` then does:
45
+
46
+ =========== ========== ====== ============== =========== ==================================
47
+ ``backend`` client? DSN? ``self.client`` ``_http`` Per emit
48
+ =========== ========== ====== ============== =========== ==================================
49
+ ``auto`` yes yes the client built SDK if it can deliver, else HTTP
50
+ ``auto`` yes no the client ``None`` SDK if it can deliver, else refuse
51
+ ``auto`` no yes ``None`` built HTTP
52
+ ``auto`` no no — — ``ValueError`` at construction
53
+ ``sdk`` yes either the client ``None`` SDK if it can deliver, else refuse
54
+ ``sdk`` no either — — ``ValueError`` at construction
55
+ ``http`` rejected yes ``None`` built HTTP
56
+ ``http`` rejected no — — ``ValueError`` at construction
57
+ =========== ========== ====== ============== =========== ==================================
58
+
59
+ "Can deliver" is judged once per ``emit``, so an application that initialises the SDK after
60
+ building this sink starts using it without rebuilding one. An argument whose only consumer is
61
+ a backend this construction will never select is a ``ValueError`` rather than a silent
62
+ ignore — ``opener`` where no HTTP fallback is built, ``client`` under ``"http"``. That is the
63
+ defect this rule comes from: ``opener`` used to be accepted and then ignored whenever the SDK
64
+ imported. ``max_retries`` is deliberately outside the rule, since its default cannot be told
65
+ from an explicit pass of the same value.
30
66
 
31
67
  Attributes:
68
+ client: The SDK object this sink captures through, or ``None`` when the HTTP fallback is
69
+ the only backend it can select.
32
70
  sent: Events captured or sent to Sentry.
33
71
  skipped: Events below the minimum level, or without a usable level, that were not sent.
34
72
  transport_errors: Events whose send raised something other than an already-counted
@@ -48,16 +86,29 @@ class SentrySink:
48
86
  dsn: str | None = None,
49
87
  *,
50
88
  min_level: str = "ERROR",
89
+ backend: Backend = "auto",
51
90
  client: Any = None,
52
91
  opener: Any = None,
53
92
  max_retries: int = 3,
54
93
  ) -> None:
55
94
  """Selects the SDK or the HTTP-envelope fallback and sets the level floor.
56
95
 
96
+ The order is deliberate: an unknown ``backend`` is rejected first, then the arguments the
97
+ selection cannot use, then the selections nothing can build. Several constructions trip a
98
+ conflict *and* a refusal — any of them with no DSN, for instance — and each raises
99
+ ``ValueError`` either way, so the conflict is reported first because it names an argument
100
+ the caller can drop, where the refusal only says the selection cannot be built.
101
+
102
+ Both backends this construction can select are built here rather than on first use. A
103
+ fallback built lazily would miss the worker's one-shot ``log_foundry_stop_signal`` offer
104
+ and stop being interruptible (SPEC-027), and it would rebind transport state inside
105
+ ``emit``, contradicting this class's SPEC-028 exemption.
106
+
57
107
  Args:
58
108
  dsn: The Sentry DSN. It is required for the fallback, which needs it to know where to
59
109
  POST.
60
110
  min_level: The lowest level worth sending.
111
+ backend: Which transport to use — ``"auto"``, ``"sdk"`` or ``"http"``.
61
112
  client: A ``sentry_sdk``-shaped object to use instead of importing one.
62
113
  opener: A ``urlopen``-shaped callable for the fallback, for tests.
63
114
  max_retries: Retries the fallback's HTTP transport makes.
@@ -66,22 +117,51 @@ class SentrySink:
66
117
  None.
67
118
 
68
119
  Raises:
69
- ValueError: If no SDK is available and no DSN was given.
120
+ ValueError: If ``backend`` is not one of the three names; if an argument cannot be used
121
+ by any backend this construction can select; or if the selection cannot be built —
122
+ ``"sdk"`` with no client available, ``"http"`` with no DSN, or the default with
123
+ neither.
70
124
  """
125
+ if backend not in _BACKENDS:
126
+ raise ValueError(f"SentrySink backend must be one of {_BACKENDS!r}, not {backend!r}")
127
+ if client is not None and backend == "http":
128
+ raise ValueError(
129
+ "SentrySink(backend='http') never captures through a client; drop client= or "
130
+ "select a backend that can use it"
131
+ )
132
+ if opener is not None and (backend == "sdk" or dsn is None):
133
+ remedy = "drop backend='sdk'" if backend == "sdk" else "pass a dsn"
134
+ raise ValueError(
135
+ "SentrySink builds no HTTP fallback for this construction, so opener= would "
136
+ f"never be called; {remedy}"
137
+ )
71
138
  self._dsn = dsn
139
+ self._backend = backend
72
140
  self._min_rank = _LEVEL_RANK.get(min_level.upper(), _LEVEL_RANK["ERROR"])
73
141
  self.sent = 0
74
142
  self.skipped = 0
75
143
  self.transport_errors = 0
76
144
  self._counter_lock = threading.Lock()
77
- self.client = client if client is not None else _import_sdk()
145
+ self.client: Any = None if backend == "http" else (
146
+ client if client is not None else _import_sdk()
147
+ )
78
148
  self._http: HTTPSink | None = None
79
149
  self._auth_header = ""
80
- if self.client is None:
81
- if dsn is None:
150
+ if backend == "sdk":
151
+ if self.client is None:
152
+ raise ValueError(
153
+ "SentrySink(backend='sdk') requires the sentry extra or an injected client="
154
+ )
155
+ elif dsn is None:
156
+ if backend == "http":
157
+ raise ValueError(
158
+ "SentrySink(backend='http') requires a dsn for the HTTP-envelope fallback"
159
+ )
160
+ if self.client is None:
82
161
  raise ValueError(
83
162
  "SentrySink without sentry-sdk requires a dsn for the HTTP-envelope fallback"
84
163
  )
164
+ else:
85
165
  ingest_url, self._auth_header = _parse_dsn(dsn)
86
166
  self._http = HTTPSink(ingest_url, opener=opener, max_retries=max_retries)
87
167
  self._stop_signal: threading.Event | None = None
@@ -94,6 +174,9 @@ class SentrySink:
94
174
  FR-001). Letting the first failure propagate would hand the worker a batch whose earlier
95
175
  events Sentry had already accepted, and the retry would duplicate them.
96
176
 
177
+ The backend is chosen once, before the loop, so one batch cannot split across transports
178
+ partway through and so the client is probed once rather than per event.
179
+
97
180
  Args:
98
181
  batch: The events to consider.
99
182
 
@@ -101,10 +184,12 @@ class SentrySink:
101
184
  None.
102
185
 
103
186
  Raises:
104
- SinkDeliveryError: If every qualifying event failed to land. An event below the
105
- minimum level is skipped rather than lost, so a batch of nothing but skipped events
106
- is a successful emit — there was never anything to deliver.
187
+ SinkDeliveryError: If every qualifying event failed to land, which includes the case
188
+ where no backend can deliver at all (SPEC-043 FR-003). An event below the minimum
189
+ level is skipped rather than lost, so a batch of nothing but skipped events is a
190
+ successful emit — there was never anything to deliver.
107
191
  """
192
+ backend = self._select_backend()
108
193
  attempted = delivered = 0
109
194
  for event in batch:
110
195
  if not self._qualifies(event):
@@ -112,7 +197,7 @@ class SentrySink:
112
197
  self.skipped += 1
113
198
  continue
114
199
  attempted += 1
115
- if not self._capture(event):
200
+ if not self._capture(event, backend):
116
201
  continue
117
202
  with self._counter_lock:
118
203
  self.sent += 1
@@ -133,7 +218,12 @@ class SentrySink:
133
218
  interpreter exit got to it.
134
219
 
135
220
  Only the injected-or-imported SDK client has a queue. The ``urllib`` fallback posts an
136
- envelope per event and holds nothing, so with no client this is correctly a no-op.
221
+ envelope per event and holds nothing, so with no client this is correctly a no-op — and
222
+ ``backend="http"`` holds none, which is what keeps this from pushing an application's own
223
+ Sentry transport on behalf of a sink that never captures through it.
224
+
225
+ A client the per-emit predicate currently reads as unable to deliver is still flushed: it
226
+ may become usable before the next batch, and flushing one that cannot is a no-op anyway.
137
227
 
138
228
  ``Client.flush`` is probed by name, as every optional member the library calls on an object
139
229
  it does not own is: a stand-in ``client=`` satisfying only ``capture_event`` stays valid,
@@ -254,8 +344,20 @@ class SentrySink:
254
344
  with self._counter_lock:
255
345
  return SinkLosses(dropped=0, failed=self.failed + self.transport_errors)
256
346
 
257
- def _capture(self, event: dict[str, object]) -> bool:
258
- """Sends one event by whichever transport is configured.
347
+ def _capture(self, event: dict[str, object], backend: _Selected | None) -> bool:
348
+ """Sends one event by the backend ``emit`` resolved for this batch.
349
+
350
+ A ``None`` backend is refused here, before the ``try``, and moves nothing. Letting it fall
351
+ into the envelope branch instead would reach ``_post_envelope``'s assertion, whose
352
+ ``AssertionError`` the guard below counts as a ``transport_errors`` and announces through
353
+ ``_diag`` — both forbidden for a refusal (SPEC-043 FR-003), because the caller is told by
354
+ the ``SinkDeliveryError`` ``emit`` raises and counting it here would report one loss twice.
355
+ The refusal stays inside the per-event loop so the level filter keeps running ahead of it
356
+ and the raise names the qualifying count.
357
+
358
+ The branch is on the resolved name rather than on ``self.client is not None``: that
359
+ condition is still true under ``"auto"`` when the client cannot deliver, so it would take
360
+ the SDK branch anyway and preserve the very defect this selection exists to fix.
259
361
 
260
362
  One guard covers both branches, catching ``Exception`` rather than an enumerated set.
261
363
  Anything escaping here propagates mid-batch and hands the worker a batch Sentry has
@@ -267,6 +369,7 @@ class SentrySink:
267
369
 
268
370
  Args:
269
371
  event: The event to send.
372
+ backend: The backend ``emit`` resolved, or ``None`` when nothing can deliver.
270
373
 
271
374
  Returns:
272
375
  True when it landed, False when it did not.
@@ -276,8 +379,10 @@ class SentrySink:
276
379
  has already counted and announced it, and counting it again would double-report.
277
380
  Only the exception type is ever written (arch §6).
278
381
  """
382
+ if backend is None:
383
+ return False
279
384
  try:
280
- if self.client is not None:
385
+ if backend == "sdk":
281
386
  self.client.capture_event(self._sentry_event(event))
282
387
  else:
283
388
  self._post_envelope(event)
@@ -290,6 +395,76 @@ class SentrySink:
290
395
  return False
291
396
  return True
292
397
 
398
+ def _select_backend(self) -> _Selected | None:
399
+ """Picks the transport for one batch, or ``None`` when nothing can deliver.
400
+
401
+ An explicit selection is honoured rather than substituted (SPEC-043 FR-002): ``"sdk"``
402
+ against a client that cannot deliver returns ``None`` so the batch is refused, because a
403
+ caller who named a backend and got a different one is this sink's original defect in a
404
+ new place. ``"http"`` never consults the client, which is what keeps a held-but-unusable
405
+ SDK out of the decision entirely.
406
+
407
+ Args:
408
+ None.
409
+
410
+ Returns:
411
+ ``"sdk"``, ``"http"``, or ``None`` when neither backend can deliver.
412
+
413
+ Raises:
414
+ None.
415
+ """
416
+ if self._backend == "http":
417
+ return "http"
418
+ if self.client is not None and self._client_can_deliver():
419
+ return "sdk"
420
+ if self._backend == "sdk":
421
+ return None
422
+ return "http" if self._http is not None else None
423
+
424
+ def _client_can_deliver(self) -> bool:
425
+ """Reports whether the held client has somewhere to send, not merely that it exists.
426
+
427
+ Three states cannot deliver and only one of them reports itself inactive: an
428
+ uninitialised process holds a ``NonRecordingClient`` (``is_active()`` false), while
429
+ ``init()`` without a DSN and a client that has been ``close()``d both report themselves
430
+ active with a ``None`` transport and drop every event silently. So the predicate takes
431
+ both members, and it is the transport that binds — ``is_active()`` is a hardcoded class
432
+ discriminator, kept because it is the SDK's documented answer and a future client may
433
+ diverge, not because anything here can distinguish it from the transport alone.
434
+
435
+ It descends first: ``__init__`` holds the ``sentry_sdk`` **module**, which publishes
436
+ neither member, so a probe that read the held object would call the defective path usable
437
+ and leave the defect in place. ``is_active`` is *called* rather than read, since a bound
438
+ method is truthy and reading one yields a guard that can never fail; a non-callable
439
+ ``is_active`` is treated as absent instead. Absence means usable throughout, which is what
440
+ keeps an injected double, a pre-SPEC-043 client and a pre-2.0 SDK working — hence the
441
+ sentinel, since an absent ``transport`` and a ``None`` one are opposite answers.
442
+
443
+ Args:
444
+ None.
445
+
446
+ Returns:
447
+ True when the client is worth capturing through.
448
+
449
+ Raises:
450
+ None. A probe may never be the reason a batch fails (SPEC-025), so a client that raises
451
+ while being questioned is treated as usable and the fault is announced by type.
452
+ """
453
+ try:
454
+ target = self.client
455
+ descend = getattr(target, "get_client", None)
456
+ if callable(descend):
457
+ descended = descend()
458
+ if descended is not None:
459
+ target = descended
460
+ is_active = getattr(target, "is_active", None)
461
+ if callable(is_active) and not is_active():
462
+ return False
463
+ return getattr(target, "transport", _ABSENT) is not None
464
+ except Exception as err:
465
+ _diag.absorbed("probing the Sentry client", err, "it is treated as usable")
466
+ return True
467
+
293
468
  def _qualifies(self, event: dict[str, object]) -> bool:
294
469
  """Reports whether an event is at or above the configured level floor.
295
470
 
@@ -334,7 +509,9 @@ class SentrySink:
334
509
  """POSTs one event as a Sentry envelope over the HTTP fallback.
335
510
 
336
511
  The assertion narrows the type for mypy rather than checking at runtime: the HTTP
337
- transport is set in the constructor whenever the path that reaches here is in use.
512
+ transport is set in the constructor whenever the path that reaches here is in use, which
513
+ is true because ``_capture`` refuses a ``None`` backend before reaching this branch and
514
+ ``_select_backend`` only answers ``"http"`` where one was built.
338
515
 
339
516
  Args:
340
517
  event: The event to send.