log-foundry 0.10.2.dev71__tar.gz → 0.10.2.dev73__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (59) hide show
  1. {log_foundry-0.10.2.dev71 → log_foundry-0.10.2.dev73}/PKG-INFO +13 -4
  2. {log_foundry-0.10.2.dev71 → log_foundry-0.10.2.dev73}/README.md +12 -3
  3. {log_foundry-0.10.2.dev71 → log_foundry-0.10.2.dev73}/pyproject.toml +14 -1
  4. {log_foundry-0.10.2.dev71 → log_foundry-0.10.2.dev73}/src/log_foundry/sinks/kafka.py +9 -0
  5. {log_foundry-0.10.2.dev71 → log_foundry-0.10.2.dev73}/src/log_foundry/sinks/logstash.py +46 -5
  6. {log_foundry-0.10.2.dev71 → log_foundry-0.10.2.dev73}/src/log_foundry/sinks/nats.py +55 -0
  7. {log_foundry-0.10.2.dev71 → log_foundry-0.10.2.dev73}/src/log_foundry/sinks/postgres.py +165 -8
  8. {log_foundry-0.10.2.dev71 → log_foundry-0.10.2.dev73}/src/log_foundry/sinks/pubsub.py +8 -0
  9. {log_foundry-0.10.2.dev71 → log_foundry-0.10.2.dev73}/LICENSE +0 -0
  10. {log_foundry-0.10.2.dev71 → log_foundry-0.10.2.dev73}/src/log_foundry/__init__.py +0 -0
  11. {log_foundry-0.10.2.dev71 → log_foundry-0.10.2.dev73}/src/log_foundry/_diag.py +0 -0
  12. {log_foundry-0.10.2.dev71 → log_foundry-0.10.2.dev73}/src/log_foundry/_fork.py +0 -0
  13. {log_foundry-0.10.2.dev71 → log_foundry-0.10.2.dev73}/src/log_foundry/_lifecycle.py +0 -0
  14. {log_foundry-0.10.2.dev71 → log_foundry-0.10.2.dev73}/src/log_foundry/api.py +0 -0
  15. {log_foundry-0.10.2.dev71 → log_foundry-0.10.2.dev73}/src/log_foundry/config.py +0 -0
  16. {log_foundry-0.10.2.dev71 → log_foundry-0.10.2.dev73}/src/log_foundry/console.py +0 -0
  17. {log_foundry-0.10.2.dev71 → log_foundry-0.10.2.dev73}/src/log_foundry/context.py +0 -0
  18. {log_foundry-0.10.2.dev71 → log_foundry-0.10.2.dev73}/src/log_foundry/decorator.py +0 -0
  19. {log_foundry-0.10.2.dev71 → log_foundry-0.10.2.dev73}/src/log_foundry/ids.py +0 -0
  20. {log_foundry-0.10.2.dev71 → log_foundry-0.10.2.dev73}/src/log_foundry/model.py +0 -0
  21. {log_foundry-0.10.2.dev71 → log_foundry-0.10.2.dev73}/src/log_foundry/py.typed +0 -0
  22. {log_foundry-0.10.2.dev71 → log_foundry-0.10.2.dev73}/src/log_foundry/results.py +0 -0
  23. {log_foundry-0.10.2.dev71 → log_foundry-0.10.2.dev73}/src/log_foundry/sanitize.py +0 -0
  24. {log_foundry-0.10.2.dev71 → log_foundry-0.10.2.dev73}/src/log_foundry/sinks/__init__.py +0 -0
  25. {log_foundry-0.10.2.dev71 → log_foundry-0.10.2.dev73}/src/log_foundry/sinks/_batch.py +0 -0
  26. {log_foundry-0.10.2.dev71 → log_foundry-0.10.2.dev73}/src/log_foundry/sinks/_chunk.py +0 -0
  27. {log_foundry-0.10.2.dev71 → log_foundry-0.10.2.dev73}/src/log_foundry/sinks/_retry.py +0 -0
  28. {log_foundry-0.10.2.dev71 → log_foundry-0.10.2.dev73}/src/log_foundry/sinks/_socket.py +0 -0
  29. {log_foundry-0.10.2.dev71 → log_foundry-0.10.2.dev73}/src/log_foundry/sinks/_time.py +0 -0
  30. {log_foundry-0.10.2.dev71 → log_foundry-0.10.2.dev73}/src/log_foundry/sinks/base.py +0 -0
  31. {log_foundry-0.10.2.dev71 → log_foundry-0.10.2.dev73}/src/log_foundry/sinks/callback.py +0 -0
  32. {log_foundry-0.10.2.dev71 → log_foundry-0.10.2.dev73}/src/log_foundry/sinks/clickhouse.py +0 -0
  33. {log_foundry-0.10.2.dev71 → log_foundry-0.10.2.dev73}/src/log_foundry/sinks/datadog.py +0 -0
  34. {log_foundry-0.10.2.dev71 → log_foundry-0.10.2.dev73}/src/log_foundry/sinks/elasticsearch.py +0 -0
  35. {log_foundry-0.10.2.dev71 → log_foundry-0.10.2.dev73}/src/log_foundry/sinks/eventhubs.py +0 -0
  36. {log_foundry-0.10.2.dev71 → log_foundry-0.10.2.dev73}/src/log_foundry/sinks/file.py +0 -0
  37. {log_foundry-0.10.2.dev71 → log_foundry-0.10.2.dev73}/src/log_foundry/sinks/filtering.py +0 -0
  38. {log_foundry-0.10.2.dev71 → log_foundry-0.10.2.dev73}/src/log_foundry/sinks/firehose.py +0 -0
  39. {log_foundry-0.10.2.dev71 → log_foundry-0.10.2.dev73}/src/log_foundry/sinks/honeycomb.py +0 -0
  40. {log_foundry-0.10.2.dev71 → log_foundry-0.10.2.dev73}/src/log_foundry/sinks/http.py +0 -0
  41. {log_foundry-0.10.2.dev71 → log_foundry-0.10.2.dev73}/src/log_foundry/sinks/kinesis.py +0 -0
  42. {log_foundry-0.10.2.dev71 → log_foundry-0.10.2.dev73}/src/log_foundry/sinks/logging_sink.py +0 -0
  43. {log_foundry-0.10.2.dev71 → log_foundry-0.10.2.dev73}/src/log_foundry/sinks/loki.py +0 -0
  44. {log_foundry-0.10.2.dev71 → log_foundry-0.10.2.dev73}/src/log_foundry/sinks/memory.py +0 -0
  45. {log_foundry-0.10.2.dev71 → log_foundry-0.10.2.dev73}/src/log_foundry/sinks/mongodb.py +0 -0
  46. {log_foundry-0.10.2.dev71 → log_foundry-0.10.2.dev73}/src/log_foundry/sinks/multi.py +0 -0
  47. {log_foundry-0.10.2.dev71 → log_foundry-0.10.2.dev73}/src/log_foundry/sinks/newrelic.py +0 -0
  48. {log_foundry-0.10.2.dev71 → log_foundry-0.10.2.dev73}/src/log_foundry/sinks/null.py +0 -0
  49. {log_foundry-0.10.2.dev71 → log_foundry-0.10.2.dev73}/src/log_foundry/sinks/rabbitmq.py +0 -0
  50. {log_foundry-0.10.2.dev71 → log_foundry-0.10.2.dev73}/src/log_foundry/sinks/redis.py +0 -0
  51. {log_foundry-0.10.2.dev71 → log_foundry-0.10.2.dev73}/src/log_foundry/sinks/sentry.py +0 -0
  52. {log_foundry-0.10.2.dev71 → log_foundry-0.10.2.dev73}/src/log_foundry/sinks/sns.py +0 -0
  53. {log_foundry-0.10.2.dev71 → log_foundry-0.10.2.dev73}/src/log_foundry/sinks/splunk.py +0 -0
  54. {log_foundry-0.10.2.dev71 → log_foundry-0.10.2.dev73}/src/log_foundry/sinks/sqlite.py +0 -0
  55. {log_foundry-0.10.2.dev71 → log_foundry-0.10.2.dev73}/src/log_foundry/sinks/sqs.py +0 -0
  56. {log_foundry-0.10.2.dev71 → log_foundry-0.10.2.dev73}/src/log_foundry/sinks/stdout.py +0 -0
  57. {log_foundry-0.10.2.dev71 → log_foundry-0.10.2.dev73}/src/log_foundry/sinks/syslog.py +0 -0
  58. {log_foundry-0.10.2.dev71 → log_foundry-0.10.2.dev73}/src/log_foundry/sinks/transform.py +0 -0
  59. {log_foundry-0.10.2.dev71 → log_foundry-0.10.2.dev73}/src/log_foundry/worker.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: log-foundry
3
- Version: 0.10.2.dev71
3
+ Version: 0.10.2.dev73
4
4
  Summary: Generate logs for your console and JSON events for downstream consumption.
5
5
  License-Expression: MIT
6
6
  License-File: LICENSE
@@ -694,7 +694,7 @@ completed, not that every chunk of it landed.
694
694
  | `ElasticsearchSink` | `log_foundry.sinks.elasticsearch` | `ElasticsearchSink(url, *, index, auth=None, **http_kwargs)` — POST to `_bulk`, parsing per-item errors (`.item_errors`) |
695
695
  | `OpenSearchSink` | `log_foundry.sinks.elasticsearch` | same signature as `ElasticsearchSink` (identical bulk protocol) |
696
696
  | `LokiSink` | `log_foundry.sinks.loki` | `LokiSink(url, *, labels=("service", "env", "level"), **http_kwargs)` — Grafana Loki push API |
697
- | `LogstashSink` | `log_foundry.sinks.logstash` | `LogstashSink(url=…, **http_kwargs)` for HTTP, **or** `LogstashSink(host=…, port=…, transport="tcp")` for a raw TCP/UDP socket |
697
+ | `LogstashSink` | `log_foundry.sinks.logstash` | `LogstashSink(url=…, body_format="json_array", **http_kwargs)` for HTTP, **or** `LogstashSink(host=…, port=…, transport="tcp")` for a raw TCP/UDP socket. HTTP mode posts a JSON array as `application/json`, which a **stock** `http` input parses into one event per element; pass `body_format="ndjson"` for an input configured with `additional_codecs => {"application/x-ndjson" => "json_lines"}`, which that setting *replaces* the default map to provide |
698
698
  | `SyslogSink` | `log_foundry.sinks.syslog` | `SyslogSink(host, port=514, *, transport="udp", facility="user", app_name="log-foundry", max_datagram_bytes=65507)` — RFC 5424 over UDP/TCP. A UDP frame over the limit is dropped and counted rather than sent, retried and abandoned; TCP is a stream and is unaffected |
699
699
 
700
700
  ```python
@@ -797,7 +797,16 @@ Two things worth knowing:
797
797
 
798
798
  #### Queue & stream
799
799
 
800
- Each needs its own extra (lazy-imported). All publish + retry within a bound and close cleanly.
800
+ Each needs its own extra (lazy-imported). All publish within a bound and close cleanly. **Where the
801
+ retry lives differs, and it was measured rather than assumed** (SPEC-041 FR-004): the Redis,
802
+ RabbitMQ and Event Hubs sinks retry through `sinks/_retry`, so their backoff is bounded *and* cut
803
+ short by a shutdown. `KafkaSink`, `NATSSink` and `GooglePubSubSink` add no retry loop and need
804
+ none — each hands off locally and returns without waiting, so nothing of theirs holds the single
805
+ drain thread. Their clients retry on their own threads within their own bounds:
806
+ `message.timeout.ms` (5 min default) for Kafka, a 600 s deadline for Pub/Sub, and for NATS a
807
+ JetStream publish bounded by its 5 s ack timeout with no retry at all. `NATSSink` refuses a batch
808
+ outright while its client reports itself disconnected, so a sustained outage moves
809
+ `health().failed_batches` instead of being absorbed.
801
810
 
802
811
  | Sink | Import from | Extra | Configure |
803
812
  |---|---|---|---|
@@ -821,7 +830,7 @@ Write-only inserts (querying is the downstream tool's job); each needs its own e
821
830
  | Sink | Import from | Extra | Configure |
822
831
  |---|---|---|---|
823
832
  | `MongoDBSink` | `log_foundry.sinks.mongodb` | `mongo` | `MongoDBSink(*, uri="…", database="…", collection="…")` |
824
- | `PostgresSink` | `log_foundry.sinks.postgres` | `postgres` | `PostgresSink(table, *, dsn="…", create_table=False)` — JSONB `event` column + extracted columns |
833
+ | `PostgresSink` | `log_foundry.sinks.postgres` | `postgres` | `PostgresSink(table, *, dsn="…", create_table=False, connect_timeout=5)` — JSONB `event` column + extracted columns. Reconnects an **owned** connection the server has closed; a `connection=` you inject is never reopened. `connect_timeout` is passed to libpq explicitly, so it **overrides** any `connect_timeout` in your DSN |
825
834
  | `ClickHouseSink` | `log_foundry.sinks.clickhouse` | `clickhouse` | `ClickHouseSink(table, *, dsn="…", create_table=False)` — MergeTree, columnar insert |
826
835
 
827
836
  `PostgresSink` / `ClickHouseSink` default `create_table=False` (you own the schema and indexes); set
@@ -658,7 +658,7 @@ completed, not that every chunk of it landed.
658
658
  | `ElasticsearchSink` | `log_foundry.sinks.elasticsearch` | `ElasticsearchSink(url, *, index, auth=None, **http_kwargs)` — POST to `_bulk`, parsing per-item errors (`.item_errors`) |
659
659
  | `OpenSearchSink` | `log_foundry.sinks.elasticsearch` | same signature as `ElasticsearchSink` (identical bulk protocol) |
660
660
  | `LokiSink` | `log_foundry.sinks.loki` | `LokiSink(url, *, labels=("service", "env", "level"), **http_kwargs)` — Grafana Loki push API |
661
- | `LogstashSink` | `log_foundry.sinks.logstash` | `LogstashSink(url=…, **http_kwargs)` for HTTP, **or** `LogstashSink(host=…, port=…, transport="tcp")` for a raw TCP/UDP socket |
661
+ | `LogstashSink` | `log_foundry.sinks.logstash` | `LogstashSink(url=…, body_format="json_array", **http_kwargs)` for HTTP, **or** `LogstashSink(host=…, port=…, transport="tcp")` for a raw TCP/UDP socket. HTTP mode posts a JSON array as `application/json`, which a **stock** `http` input parses into one event per element; pass `body_format="ndjson"` for an input configured with `additional_codecs => {"application/x-ndjson" => "json_lines"}`, which that setting *replaces* the default map to provide |
662
662
  | `SyslogSink` | `log_foundry.sinks.syslog` | `SyslogSink(host, port=514, *, transport="udp", facility="user", app_name="log-foundry", max_datagram_bytes=65507)` — RFC 5424 over UDP/TCP. A UDP frame over the limit is dropped and counted rather than sent, retried and abandoned; TCP is a stream and is unaffected |
663
663
 
664
664
  ```python
@@ -761,7 +761,16 @@ Two things worth knowing:
761
761
 
762
762
  #### Queue & stream
763
763
 
764
- Each needs its own extra (lazy-imported). All publish + retry within a bound and close cleanly.
764
+ Each needs its own extra (lazy-imported). All publish within a bound and close cleanly. **Where the
765
+ retry lives differs, and it was measured rather than assumed** (SPEC-041 FR-004): the Redis,
766
+ RabbitMQ and Event Hubs sinks retry through `sinks/_retry`, so their backoff is bounded *and* cut
767
+ short by a shutdown. `KafkaSink`, `NATSSink` and `GooglePubSubSink` add no retry loop and need
768
+ none — each hands off locally and returns without waiting, so nothing of theirs holds the single
769
+ drain thread. Their clients retry on their own threads within their own bounds:
770
+ `message.timeout.ms` (5 min default) for Kafka, a 600 s deadline for Pub/Sub, and for NATS a
771
+ JetStream publish bounded by its 5 s ack timeout with no retry at all. `NATSSink` refuses a batch
772
+ outright while its client reports itself disconnected, so a sustained outage moves
773
+ `health().failed_batches` instead of being absorbed.
765
774
 
766
775
  | Sink | Import from | Extra | Configure |
767
776
  |---|---|---|---|
@@ -785,7 +794,7 @@ Write-only inserts (querying is the downstream tool's job); each needs its own e
785
794
  | Sink | Import from | Extra | Configure |
786
795
  |---|---|---|---|
787
796
  | `MongoDBSink` | `log_foundry.sinks.mongodb` | `mongo` | `MongoDBSink(*, uri="…", database="…", collection="…")` |
788
- | `PostgresSink` | `log_foundry.sinks.postgres` | `postgres` | `PostgresSink(table, *, dsn="…", create_table=False)` — JSONB `event` column + extracted columns |
797
+ | `PostgresSink` | `log_foundry.sinks.postgres` | `postgres` | `PostgresSink(table, *, dsn="…", create_table=False, connect_timeout=5)` — JSONB `event` column + extracted columns. Reconnects an **owned** connection the server has closed; a `connection=` you inject is never reopened. `connect_timeout` is passed to libpq explicitly, so it **overrides** any `connect_timeout` in your DSN |
789
798
  | `ClickHouseSink` | `log_foundry.sinks.clickhouse` | `clickhouse` | `ClickHouseSink(table, *, dsn="…", create_table=False)` — MergeTree, columnar insert |
790
799
 
791
800
  `PostgresSink` / `ClickHouseSink` default `create_table=False` (you own the schema and indexes); set
@@ -20,7 +20,7 @@ dependencies = [
20
20
  ]
21
21
 
22
22
  # Optional features. Install with: pip install log-foundry[aws]
23
- version = "0.10.2.dev71"
23
+ version = "0.10.2.dev73"
24
24
 
25
25
  [project.optional-dependencies]
26
26
  aws = ["boto3>=1.43.61"] # SQSSink, SNSSink, KinesisSink, FirehoseSink
@@ -178,6 +178,19 @@ ignore = [
178
178
  # trip this rule.
179
179
  "src/log_foundry/sinks/postgres.py" = ["S608"]
180
180
  "src/log_foundry/sinks/sqlite.py" = ["S608"]
181
+ # The integration tests build DDL and SELECTs around a table name they generate themselves --
182
+ # a `uuid4().hex` slice, so the table can be created and dropped per test without two runs
183
+ # colliding. Scoped to these two files for the same reason the two sinks above are scoped
184
+ # individually rather than the rule being switched off for `tests/**`: a future test that
185
+ # interpolates something it did not generate should still trip it.
186
+ "tests/integration/test_postgres.py" = ["S608"]
187
+ "tests/integration/test_clickhouse.py" = ["S608"]
188
+ # The NATS outage test stops and restarts its own compose service, because "the server went
189
+ # away" is the condition under test and no fake reproduces what the real client does about it.
190
+ # Every argument is a literal or a path this module derives from `__file__`; there is no shell
191
+ # and no caller input. `docker` is resolved from PATH deliberately -- CI and a developer laptop
192
+ # put it in different places.
193
+ "tests/integration/test_nats.py" = ["S603", "S607"]
181
194
  # make-sbom.py exists to drive two virtualenvs it creates itself, so subprocess is the whole job.
182
195
  # Every argv element is either a literal, a path under a `tempfile.TemporaryDirectory` this script
183
196
  # just made, or a value read from the repository's own pyproject.toml / poetry.lock. There is no
@@ -53,6 +53,15 @@ class KafkaSink:
53
53
  returns without blocking, while the delivery result arrives asynchronously on a callback
54
54
  serviced by ``poll()`` and ``flush()``.
55
55
 
56
+ **Retry (SPEC-041 FR-004).** This sink adds none and needs none: ``produce()`` is a local
57
+ hand-off that returns without waiting — measured at 0.0000 s for three messages against a
58
+ dead broker — so it never holds the worker's single drain thread, and librdkafka's own retry
59
+ is bounded by ``message.timeout.ms`` (five minutes by default). Measured: with that set to
60
+ 1500 ms the delivery callback fired at 2.01 s, and at 4000 ms it fired at 4.01 s. SPEC-027's
61
+ "cut short by a shutdown" governs a wait *between attempts*, and this sink has none to cut —
62
+ which is also why it deliberately exposes no ``log_foundry_stop_signal`` attribute, since
63
+ ``_lifecycle.offer_stop_signal`` probes by ``hasattr`` and that absence is the opt-out.
64
+
56
65
  The worst case (SPEC-027 FR-005) is not a retry loop — this sink has none — but its
57
66
  ``close()``: one ``flush_timeout`` wait, 10 s at the default, spent **beside**
58
67
  ``shutdown()``'s budget rather than from it. ``Worker.shutdown`` joins the drain thread
@@ -15,16 +15,44 @@ from log_foundry.sinks.http import HTTPSink
15
15
 
16
16
  __all__ = ["LogstashSink"]
17
17
 
18
+ _BODY_FORMATS = frozenset({"json_array", "ndjson"})
19
+ """The two HTTP wire forms, validated rather than passed through.
20
+
21
+ ``HTTPSink._body`` branches on ``== "json_array"`` and treats everything else as NDJSON, so a
22
+ typo -- ``"json-array"`` -- would silently ship the exact body FR-003 exists to stop shipping,
23
+ against a destination that cannot parse it. A parameter documented as taking one of two values
24
+ refuses the third.
25
+ """
26
+
18
27
 
19
28
  class LogstashSink:
20
29
  """A :class:`~log_foundry.sinks.base.Sink` that ships JSON lines to Logstash (FR-005).
21
30
 
22
- Two mutually-exclusive modes are chosen at construction: with a URL the batch goes as JSON
23
- lines over HTTP, reusing the ``HTTPSink`` core, and with a host and port each event goes as
24
- one newline-terminated line over a raw TCP or UDP socket, reusing
31
+ Two mutually-exclusive modes are chosen at construction: with a URL the batch goes over HTTP,
32
+ reusing the ``HTTPSink`` core, and with a host and port each event goes as one
33
+ newline-terminated line over a raw TCP or UDP socket, reusing
25
34
  :class:`~log_foundry.sinks._socket.SocketTransport`. Either backend handles its own bounded
26
35
  retry and raises on total failure of its own accord, so ``emit`` needs no rule of its own.
27
36
 
37
+ **What the Logstash side must be configured as** (SPEC-041 FR-003). In HTTP mode the default
38
+ ``body_format="json_array"`` posts a JSON array as ``application/json``, which a **stock**
39
+ ``http`` input parses into one event per element with no configuration at all. That is a
40
+ change: this sink used to post NDJSON as ``application/x-ndjson``, which the ``http`` input's
41
+ default ``additional_codecs`` map (``{"application/json" => "json"}``) does not cover, so the
42
+ body fell through to the ``plain`` codec. Measured against Logstash 8.15 — a batch of three
43
+ arrived as **one** event with the whole payload as text in ``message`` and every structured
44
+ field gone.
45
+
46
+ ``body_format="ndjson"`` restores the old wire form, and it is not merely legacy: an input
47
+ configured ``additional_codecs => {"application/x-ndjson" => "json_lines"}`` -- the documented
48
+ workaround for the behaviour above -- needs it. That setting **replaces** the default map
49
+ rather than merging with it, so such an input parses NDJSON correctly and a JSON array
50
+ incorrectly. The two configurations are mutually exclusive unless the input lists both, which
51
+ is why the wire form is a parameter rather than a silent change.
52
+
53
+ Socket mode is unaffected: it is newline-delimited either way, and a ``tcp``/``udp`` input
54
+ uses the ``line``/``json_lines`` codec rather than ``additional_codecs``.
55
+
28
56
  In socket mode both IPv4 and IPv6 destinations are supported, over either transport
29
57
  (SPEC-031 FR-002): the UDP socket's address family is resolved from ``host`` rather than
30
58
  assumed, which is what an unconditional ``AF_INET`` used to make impossible.
@@ -58,6 +86,7 @@ class LogstashSink:
58
86
  host: str | None = None,
59
87
  port: int | None = None,
60
88
  transport: str = "tcp",
89
+ body_format: str = "json_array",
61
90
  timeout: float = 5.0,
62
91
  max_retries: int = 3,
63
92
  max_datagram_bytes: int = DEFAULT_MAX_DATAGRAM_BYTES,
@@ -70,6 +99,12 @@ class LogstashSink:
70
99
  host: The destination host, selecting socket mode.
71
100
  port: The destination port, selecting socket mode.
72
101
  transport: ``"tcp"`` or ``"udp"``, in socket mode.
102
+ body_format: ``"json_array"`` (the default) or ``"ndjson"``. It selects the HTTP wire
103
+ form and has **no effect** in socket mode, which is always newline-delimited — but it
104
+ is *validated* in both, because a value this class silently ignores in one mode and
105
+ silently misreads in the other is worse than a refusal. See the class docstring for
106
+ what each one requires of the Logstash side; ``"ndjson"`` is the escape hatch for an
107
+ input already configured with ``additional_codecs``.
73
108
  timeout: Seconds allowed per request or connection.
74
109
  max_retries: Retries the chosen backend makes.
75
110
  max_datagram_bytes: In UDP socket mode, the largest datagram to attempt; a frame over
@@ -81,11 +116,17 @@ class LogstashSink:
81
116
  None.
82
117
 
83
118
  Raises:
84
- ValueError: If neither a URL nor a host and port were given.
119
+ ValueError: If neither a URL nor a host and port were given, or if ``body_format`` is
120
+ neither ``"json_array"`` nor ``"ndjson"``.
85
121
  """
122
+ if body_format not in _BODY_FORMATS:
123
+ raise ValueError(
124
+ f"LogstashSink body_format must be one of {sorted(_BODY_FORMATS)}, "
125
+ f"not {body_format!r}"
126
+ )
86
127
  if url is not None:
87
128
  self._http: HTTPSink | None = HTTPSink(
88
- url, body_format="ndjson", timeout=timeout, max_retries=max_retries,
129
+ url, body_format=body_format, timeout=timeout, max_retries=max_retries,
89
130
  **http_kwargs, # type: ignore[arg-type]
90
131
  )
91
132
  self._socket: SocketTransport | None = None
@@ -21,6 +21,31 @@ class NATSSink:
21
21
  completion from the synchronous ``emit``, which therefore returns only after the batch is
22
22
  handed off. With JetStream enabled it publishes through JetStream for durable
23
23
  acknowledgement.
24
+
25
+ **Retry and the worst-case delay** (SPEC-041 FR-004). This sink adds no retry loop and needs
26
+ none: a core ``publish()`` writes into the client's outbound buffer and returns without
27
+ waiting — measured at 0.00 s for fifty publishes — so it never holds the worker's single
28
+ drain thread and there is no backoff for a shutdown to cut short. SPEC-027's guarantee is met
29
+ because there is no wait, not because a wait is bounded. Under JetStream ``publish()`` awaits
30
+ an ack bounded by the driver's own timeout (5 s by default) and does not retry.
31
+
32
+ **A disconnected client is reported, not absorbed** (FR-004 AC-5). That non-blocking publish
33
+ is exactly what made this sink report success for events that had not left the process: with
34
+ the server stopped, five successive emits each returned in 0.00 s with ``losses()`` reading
35
+ all zeros, and when the client's reconnect budget (60 attempts × 2 s by default) ran out
36
+ first, **one of six events reached the destination with every counter still at zero**. That
37
+ is SPEC-026 FR-001's shape — a sink the worker believes, so its retry never engages and
38
+ ``failed_batches`` never moves. ``emit`` now refuses a batch while the client reports itself
39
+ disconnected.
40
+
41
+ The limit is stated rather than overclaimed: ``is_connected`` does not flip the instant the
42
+ server dies, so the first batch in the window before the client notices is still accepted and
43
+ still buffered — measured, one of five emits landed in that window. It is the same
44
+ check-then-act window ``MongoDBSink`` and ``KafkaSink`` already document for their own flags;
45
+ what the guard ends is the far larger case of an outage that has been going on for any
46
+ appreciable time. Refusing moves no ``losses()`` counter, per SPEC-032: it is a failure
47
+ *reported* to the worker rather than one absorbed, and counting both would report one loss
48
+ twice.
24
49
  """
25
50
 
26
51
  def __init__(
@@ -86,8 +111,38 @@ class NATSSink:
86
111
  raise SinkDeliveryError(
87
112
  f"NATSSink published none of {len(batch)} event(s): the sink is closed"
88
113
  )
114
+ if not self._is_connected():
115
+ raise SinkDeliveryError(
116
+ f"NATSSink published none of {len(batch)} event(s): "
117
+ "the client is disconnected"
118
+ )
89
119
  self._loop.run_until_complete(self._publish_all(batch))
90
120
 
121
+ def _is_connected(self) -> bool:
122
+ """Reports whether the client can currently put anything on the wire (FR-004 AC-5).
123
+
124
+ Probed by name, as ``drain`` and ``flush`` are, because this sink is written against a
125
+ driver it does not own and accepts an injected ``client=``. A client that does not
126
+ publish the attribute is assumed connected: the guard exists to convert a *known*
127
+ non-delivery into a reported one, and inventing a refusal for a client that never
128
+ claimed to be disconnected would fail batches that were going to succeed.
129
+
130
+ Args:
131
+ None.
132
+
133
+ Returns:
134
+ True when the client reports itself connected, or says nothing about it.
135
+
136
+ Raises:
137
+ None. A driver whose attribute access raises is treated as connected, so a probe can
138
+ never be the reason a batch fails.
139
+ """
140
+ try:
141
+ connected = getattr(self._client, "is_connected", True)
142
+ except Exception:
143
+ return True
144
+ return bool(connected)
145
+
91
146
  def losses(self) -> SinkLosses:
92
147
  """Reports events whose publish raised (SPEC-026 FR-002).
93
148
 
@@ -15,6 +15,22 @@ __all__ = ["PostgresSink"]
15
15
 
16
16
  _BACKOFF_BASE = 0.1
17
17
 
18
+ DEFAULT_CONNECT_TIMEOUT = 5
19
+ """Seconds libpq may spend opening one connection (SPEC-041 FR-002).
20
+
21
+ ``psycopg.connect()`` takes no timeout unless one is given, and libpq's default is to wait
22
+ indefinitely. That call runs on the worker's single drain thread, holding this sink's emit lock,
23
+ and it consults no stop signal -- so against a host that blackholes packets rather than refusing
24
+ them it is exactly the unbounded, uninterruptible wait SPEC-027 exists to remove. Measured
25
+ against an unroutable address: **75.01 s** with no timeout, **2.03 s** with ``connect_timeout=2``.
26
+
27
+ Floored at libpq's own minimum of 2 rather than accepted verbatim, because ``0`` means "wait
28
+ forever" -- reinstating the defect -- the same reason ``KafkaSink._usable_timeout`` refuses its
29
+ own degenerate values (SPEC-038 FR-006).
30
+ """
31
+
32
+ _MIN_CONNECT_TIMEOUT = 2
33
+
18
34
  _COLUMNS = ("timestamp", "level", "trace_id", "span_id", "function", "service")
19
35
 
20
36
 
@@ -23,8 +39,15 @@ class PostgresSink:
23
39
 
24
40
  Each event is stored as a ``JSONB`` ``event`` column plus a few extracted columns for
25
41
  indexing. ``psycopg`` v3 is the optional ``postgres`` extra, imported lazily. The sink is
26
- write-only, and the worst-case delay (SPEC-027 FR-005) is ``max_retries`` interruptible waits
27
- per batch, 0.7 s at the defaults.
42
+ write-only. The worst-case delay (SPEC-027 FR-005) has two halves, and only one of them is
43
+ interruptible. The backoffs are ``max_retries`` waits per batch — 0.7 s at the defaults —
44
+ taken through ``_retry.wait`` on the worker's stop event, so a shutdown cuts them short. The
45
+ reconnects are up to ``max_retries + 1`` connects of ``connect_timeout`` each, 20 s more at
46
+ the defaults, and they are **bounded but not interruptible**: the wait is inside libpq, which
47
+ consults nothing. That is the settled line rather than a gap — a shutdown shortens a *wait*
48
+ and never skips *work*, and a reconnect is the work an in-flight batch needs (SPEC-038
49
+ FR-001 AC-4a). Both halves sit inside the existing retry budget rather than a loop of their
50
+ own, which is what bounds them (FR-002 AC-3).
28
51
 
29
52
  The driver requirement satisfied (SPEC-028 FR-002): a ``psycopg`` connection carries one
30
53
  transaction, and this sink's unit of work is a ``cursor`` / ``commit`` / ``rollback`` sequence
@@ -48,6 +71,7 @@ class PostgresSink:
48
71
  create_table: bool = False,
49
72
  chunk_size: int = 1000,
50
73
  max_retries: int = 3,
74
+ connect_timeout: int = DEFAULT_CONNECT_TIMEOUT,
51
75
  ) -> None:
52
76
  """Connects to the database and prepares the insert statement.
53
77
 
@@ -61,6 +85,13 @@ class PostgresSink:
61
85
  max_retries: Retries per batch, floored at zero as ``Worker._emit`` floors its own
62
86
  (SPEC-021) — a negative value returned having attempted no insert at all, and
63
87
  reported success.
88
+ connect_timeout: Seconds libpq may spend opening a connection. Defaults to
89
+ :data:`DEFAULT_CONNECT_TIMEOUT` (5) and is floored at libpq's own minimum of 2, since
90
+ ``0`` means "wait forever" and would reinstate the unbounded connect this argument
91
+ exists to remove. It is passed explicitly and therefore **overrides any
92
+ ``connect_timeout`` in the DSN** — a DSN asking for 30 gets this value instead, so
93
+ set it here rather than there. Applies to the connection opened at construction and
94
+ to every reconnect.
64
95
 
65
96
  Returns:
66
97
  None.
@@ -78,11 +109,12 @@ class PostgresSink:
78
109
  self._lock = threading.Lock()
79
110
  self._counter_lock = threading.Lock()
80
111
  self._owns_connection = connection is None
112
+ self._dsn = dsn
113
+ self.connect_timeout = max(connect_timeout, _MIN_CONNECT_TIMEOUT)
81
114
  if connection is None:
82
- import psycopg # type: ignore[import-not-found]
83
-
84
- connection = psycopg.connect(dsn)
115
+ connection = self._connect()
85
116
  self._conn = connection
117
+ self._reconnect_announced = False
86
118
  columns = ", ".join((*_COLUMNS, "event"))
87
119
  placeholders = ", ".join(["%s"] * len(_COLUMNS) + ["%s::jsonb"])
88
120
  self._insert_sql = f"INSERT INTO {self._table} ({columns}) VALUES ({placeholders})"
@@ -153,6 +185,7 @@ class PostgresSink:
153
185
  """
154
186
  for attempt in range(self.max_retries + 1):
155
187
  try:
188
+ self._reconnect_if_broken()
156
189
  with self._conn.cursor() as cur:
157
190
  for chunk in chunk_list(batch, self._chunk_size):
158
191
  cur.executemany(self._insert_sql, [self._row(event) for event in chunk])
@@ -168,12 +201,126 @@ class PostgresSink:
168
201
  _diag.lost(
169
202
  "event",
170
203
  len(batch),
171
- f"PostgresSink, {self.max_retries + 1} attempts, {type(err).__name__}",
204
+ f"PostgresSink, {self.max_retries + 1} attempts, {type(err).__name__}"
205
+ + (
206
+ ", borrowed connection is broken and this sink may not reopen it"
207
+ if not self._owns_connection and self._is_broken()
208
+ else ""
209
+ ),
172
210
  )
173
211
  raise SinkDeliveryError(
174
212
  f"PostgresSink inserted none of {len(batch)} event(s)"
175
213
  ) from None
176
214
 
215
+ def _connect(self) -> Any:
216
+ """Opens a connection to the configured DSN, bounded by :attr:`connect_timeout`.
217
+
218
+ The bound is passed as a keyword rather than left to the DSN, so it holds **regardless of**
219
+ what the caller's connection string says — which is the same fact ``__init__`` states from
220
+ the caller's side, that this argument overrides a ``connect_timeout`` in the DSN. See
221
+ :data:`DEFAULT_CONNECT_TIMEOUT` for why an unbounded connect here is the defect rather
222
+ than the default.
223
+
224
+ Args:
225
+ None.
226
+
227
+ Returns:
228
+ The new connection.
229
+
230
+ Raises:
231
+ ImportError: If the ``postgres`` extra is not installed.
232
+ Exception: Whatever the driver raises when connecting.
233
+ """
234
+ import psycopg # type: ignore[import-not-found]
235
+
236
+ return psycopg.connect(self._dsn, connect_timeout=self.connect_timeout)
237
+
238
+ def _reconnect_if_broken(self) -> None:
239
+ """Replaces an unusable **owned** connection before an insert attempt (FR-002).
240
+
241
+ A ``psycopg`` connection is permanently unusable once the server closes it, and this sink
242
+ opened one in ``__init__`` and never reopened it — so a single restart, failover or idle
243
+ timeout ended log delivery for the life of the process, with every in-batch retry running
244
+ against the same dead handle. Measured against a real Postgres: one
245
+ ``pg_terminate_backend`` and three subsequent batches were lost, one row delivered.
246
+ Every sibling already recovers (``SocketTransport._reset``, boto3, clickhouse-connect's
247
+ pool, pymongo's pool).
248
+
249
+ **It runs at the top of each attempt, not in the retry branch.** ``max_retries`` is a
250
+ public argument floored at zero, so a reconnect placed where a retry remains never runs
251
+ at all at ``max_retries=0`` and the defect would survive at that setting forever. At the
252
+ top of the attempt, the first emit after a failure reopens — which is FR-002 AC-1
253
+ literally — at every value. It also avoids spending a connect immediately before the
254
+ raise, where it can only cost time.
255
+
256
+ **A borrowed connection is never reopened**, per ``architecture.md`` §13's borrowed-client
257
+ constraint: the caller owns that object and its lifetime, and replacing it here would
258
+ leak theirs and reconnect a session they may be sharing. The exhausted batch's existing
259
+ ``_diag.lost`` line gains a clause rather than a new stderr site, and it is conditioned on
260
+ the connection actually being broken — appending it to every borrowed-connection failure
261
+ would put "not reopened" on a constraint violation, where reopening was never the
262
+ question.
263
+
264
+ The state is read from the connection rather than inferred: ``closed`` and ``broken`` are
265
+ both ``False`` before the first failure and both ``True`` after it, so an ordinary SQL
266
+ error — a constraint violation, a full disk — does not churn the connection. Both are
267
+ probed by name because this sink accepts any ``psycopg``-shaped object it does not own.
268
+
269
+ Both diagnostics here are **announced once per outage, not once per attempt**. Unthrottled
270
+ they fire on every attempt of every batch, so a down server turned one stderr line into
271
+ five per batch, indefinitely — a diagnostic that floods is one an operator stops reading,
272
+ and the batch's own ``_diag.lost`` line already records the loss. The flag covers the
273
+ failed ``close()`` as well as the failed connect, because a failed reconnect leaves the
274
+ old object in place and the next attempt closes it again. It clears on a successful
275
+ reconnect, so a later outage is announced again.
276
+
277
+ Args:
278
+ None.
279
+
280
+ Returns:
281
+ None.
282
+
283
+ Raises:
284
+ None. A failed reconnect is absorbed and left to the next attempt, which is what keeps
285
+ this inside the existing retry budget rather than adding a loop of its own (AC-3):
286
+ the batch's bound is unchanged, and a connect that cannot succeed simply spends an
287
+ attempt visibly, through the counters.
288
+ """
289
+ if not self._owns_connection or not self._is_broken():
290
+ return
291
+ announced, self._reconnect_announced = self._reconnect_announced, True
292
+ try:
293
+ self._conn.close()
294
+ except Exception as err:
295
+ if not announced:
296
+ _diag.absorbed("PostgresSink.close of a broken connection", err)
297
+ try:
298
+ self._conn = self._connect()
299
+ self._reconnect_announced = False
300
+ except Exception as err:
301
+ if not announced:
302
+ _diag.absorbed("PostgresSink.reconnect", err)
303
+
304
+ def _is_broken(self) -> bool:
305
+ """Reports whether the held connection can no longer be used.
306
+
307
+ Args:
308
+ None.
309
+
310
+ Returns:
311
+ True when the driver reports the connection closed or broken.
312
+
313
+ Raises:
314
+ None. A driver whose attribute access raises is treated as usable, so a probe can
315
+ never be the reason a batch fails.
316
+ """
317
+ try:
318
+ return bool(getattr(self._conn, "closed", False)) or bool(
319
+ getattr(self._conn, "broken", False)
320
+ )
321
+ except Exception:
322
+ return False
323
+
177
324
  def _rollback(self) -> None:
178
325
  """Discards the failed transaction, absorbing a rollback that itself fails (FR-002).
179
326
 
@@ -210,6 +357,13 @@ class PostgresSink:
210
357
  Idempotent, and takes the emit lock so the final commit never lands in the middle of
211
358
  another thread's transaction (SPEC-028 FR-002).
212
359
 
360
+ **The final commit is guarded** (SPEC-041 FR-002). It was unconditional, so a connection
361
+ the server had closed made ``close()`` raise — reachable without any exotic timing, since
362
+ a broken connection is only repaired inside an emit attempt and a process that breaks and
363
+ then shuts down never has another. The release still has to happen, and a commit that
364
+ cannot reach the server has nothing to publish, so the failure is announced by type and
365
+ the close proceeds.
366
+
213
367
  Args:
214
368
  None.
215
369
 
@@ -217,13 +371,16 @@ class PostgresSink:
217
371
  None.
218
372
 
219
373
  Raises:
220
- Exception: Whatever the driver raises on commit or close.
374
+ Exception: Whatever the driver raises on close. The commit no longer escapes.
221
375
  """
222
376
  with self._lock:
223
377
  if self._closed:
224
378
  return
225
379
  self._closed = True
226
- self._conn.commit()
380
+ try:
381
+ self._conn.commit()
382
+ except Exception as err:
383
+ _diag.absorbed("PostgresSink.close commit", err)
227
384
  if self._owns_connection:
228
385
  self._conn.close()
229
386
 
@@ -59,6 +59,14 @@ class GooglePubSubSink:
59
59
  imported lazily. ``publish()`` returns a future that resolves asynchronously, so the sink
60
60
  accumulates the batch's futures and resolves them on :meth:`close`.
61
61
 
62
+ **Retry (SPEC-041 FR-004).** The client's own retry is bounded and runs on the client's
63
+ threads, never the worker's drain thread: the generated ``publish`` carries
64
+ ``Retry(initial=0.1, maximum=60.0, multiplier=4, deadline=600.0)`` with a 60 s per-call
65
+ timeout, so a publish gives up after ten minutes at the outside. What this sink contributes
66
+ is the only wait the drain thread ever takes — :meth:`_await_overflow`'s
67
+ ``overflow_timeout``, bounded and interruptible per SPEC-027. Measured with the destination
68
+ stopped and ``overflow_timeout=5.0``: every over-bound emit returned in 5.00 s exactly.
69
+
62
70
  The driver requirement satisfied (SPEC-028 FR-002): this sink takes **no** transport lock —
63
71
  the publisher client owns its own batching and threading, and ``publish()`` is a local
64
72
  hand-off. What it does hold is the pending-futures list, which is genuinely shared between