streamgate 2.0.0__tar.gz → 3.0.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (82) hide show
  1. {streamgate-2.0.0 → streamgate-3.0.0}/.github/workflows/ci.yml +1 -1
  2. {streamgate-2.0.0 → streamgate-3.0.0}/CHANGELOG.md +124 -0
  3. streamgate-3.0.0/CONFIGURATION.md +196 -0
  4. {streamgate-2.0.0 → streamgate-3.0.0}/PKG-INFO +20 -6
  5. {streamgate-2.0.0 → streamgate-3.0.0}/README.md +19 -4
  6. {streamgate-2.0.0 → streamgate-3.0.0}/README.zh-CN.md +17 -4
  7. {streamgate-2.0.0 → streamgate-3.0.0}/examples/http_probe/consume.py +4 -0
  8. {streamgate-2.0.0 → streamgate-3.0.0}/examples/http_probe/health_server.py +6 -3
  9. {streamgate-2.0.0 → streamgate-3.0.0}/examples/http_probe/produce.py +4 -0
  10. {streamgate-2.0.0 → streamgate-3.0.0}/examples/prod_pipeline/consume.py +4 -0
  11. {streamgate-2.0.0 → streamgate-3.0.0}/examples/prod_pipeline/health_server.py +6 -3
  12. {streamgate-2.0.0 → streamgate-3.0.0}/examples/prod_pipeline/produce.py +4 -0
  13. {streamgate-2.0.0 → streamgate-3.0.0}/examples/pure_pipeline/consume.py +4 -0
  14. {streamgate-2.0.0 → streamgate-3.0.0}/examples/pure_pipeline/produce.py +4 -0
  15. {streamgate-2.0.0 → streamgate-3.0.0}/examples/redis_admission/produce.py +4 -0
  16. {streamgate-2.0.0 → streamgate-3.0.0}/examples/sqlite_upsert/consume.py +4 -0
  17. {streamgate-2.0.0 → streamgate-3.0.0}/examples/sqlite_upsert/produce.py +4 -0
  18. {streamgate-2.0.0 → streamgate-3.0.0}/pyproject.toml +2 -2
  19. {streamgate-2.0.0 → streamgate-3.0.0}/src/streamgate/__init__.py +4 -6
  20. {streamgate-2.0.0 → streamgate-3.0.0}/src/streamgate/config.py +8 -7
  21. {streamgate-2.0.0 → streamgate-3.0.0}/src/streamgate/consumer/dlq.py +12 -12
  22. {streamgate-2.0.0 → streamgate-3.0.0}/src/streamgate/consumer/loop.py +35 -32
  23. streamgate-3.0.0/src/streamgate/consumer/options.py +123 -0
  24. {streamgate-2.0.0 → streamgate-3.0.0}/src/streamgate/consumer/runner.py +32 -70
  25. {streamgate-2.0.0 → streamgate-3.0.0}/src/streamgate/contrib/http_probe/probe_signal.py +20 -16
  26. {streamgate-2.0.0 → streamgate-3.0.0}/src/streamgate/contrib/mssql_upsert/__init__.py +11 -6
  27. {streamgate-2.0.0 → streamgate-3.0.0}/src/streamgate/contrib/redis_dedup/__init__.py +4 -0
  28. {streamgate-2.0.0 → streamgate-3.0.0}/src/streamgate/contrib/redis_dedup/cache.py +8 -5
  29. {streamgate-2.0.0 → streamgate-3.0.0}/src/streamgate/contrib/redis_dedup/carrier.py +16 -13
  30. {streamgate-2.0.0 → streamgate-3.0.0}/src/streamgate/contrib/redis_dedup/group_cache.py +42 -8
  31. {streamgate-2.0.0 → streamgate-3.0.0}/src/streamgate/contrib/redis_dedup/group_carrier.py +50 -18
  32. {streamgate-2.0.0 → streamgate-3.0.0}/src/streamgate/contrib/sql_upsert/backfill.py +8 -5
  33. {streamgate-2.0.0 → streamgate-3.0.0}/src/streamgate/contrib/sql_upsert/engines.py +6 -2
  34. {streamgate-2.0.0 → streamgate-3.0.0}/src/streamgate/contrib/sql_upsert/factory.py +11 -5
  35. {streamgate-2.0.0 → streamgate-3.0.0}/src/streamgate/contrib/sql_upsert/upsert.py +12 -9
  36. {streamgate-2.0.0 → streamgate-3.0.0}/src/streamgate/contrib/sqlite_upsert/__init__.py +11 -6
  37. {streamgate-2.0.0 → streamgate-3.0.0}/src/streamgate/ingest/producer.py +44 -82
  38. streamgate-3.0.0/src/streamgate/obs/logging.py +37 -0
  39. {streamgate-2.0.0 → streamgate-3.0.0}/src/streamgate/obs/metrics.py +3 -2
  40. {streamgate-2.0.0 → streamgate-3.0.0}/src/streamgate/resilience/health.py +5 -2
  41. {streamgate-2.0.0 → streamgate-3.0.0}/src/streamgate/transport/codec.py +6 -3
  42. {streamgate-2.0.0 → streamgate-3.0.0}/src/streamgate/transport/kafka.py +16 -18
  43. {streamgate-2.0.0 → streamgate-3.0.0}/uv.lock +2 -33
  44. streamgate-2.0.0/CONFIGURATION.md +0 -190
  45. streamgate-2.0.0/src/streamgate/consumer/options.py +0 -261
  46. streamgate-2.0.0/src/streamgate/obs/logging.py +0 -43
  47. {streamgate-2.0.0 → streamgate-3.0.0}/.github/workflows/release.yml +0 -0
  48. {streamgate-2.0.0 → streamgate-3.0.0}/.gitignore +0 -0
  49. {streamgate-2.0.0 → streamgate-3.0.0}/AGENTS.md +0 -0
  50. {streamgate-2.0.0 → streamgate-3.0.0}/CONTRIBUTING.md +0 -0
  51. {streamgate-2.0.0 → streamgate-3.0.0}/LICENSE +0 -0
  52. {streamgate-2.0.0 → streamgate-3.0.0}/examples/docker-compose.yml +0 -0
  53. {streamgate-2.0.0 → streamgate-3.0.0}/examples/http_probe/README.md +0 -0
  54. {streamgate-2.0.0 → streamgate-3.0.0}/examples/http_probe/models.py +0 -0
  55. {streamgate-2.0.0 → streamgate-3.0.0}/examples/prod_pipeline/README.md +0 -0
  56. {streamgate-2.0.0 → streamgate-3.0.0}/examples/prod_pipeline/models.py +0 -0
  57. {streamgate-2.0.0 → streamgate-3.0.0}/examples/pure_pipeline/README.md +0 -0
  58. {streamgate-2.0.0 → streamgate-3.0.0}/examples/pure_pipeline/models.py +0 -0
  59. {streamgate-2.0.0 → streamgate-3.0.0}/examples/redis_admission/README.md +0 -0
  60. {streamgate-2.0.0 → streamgate-3.0.0}/examples/redis_admission/models.py +0 -0
  61. {streamgate-2.0.0 → streamgate-3.0.0}/examples/sqlite_upsert/README.md +0 -0
  62. {streamgate-2.0.0 → streamgate-3.0.0}/examples/sqlite_upsert/models.py +0 -0
  63. {streamgate-2.0.0 → streamgate-3.0.0}/src/streamgate/consumer/__init__.py +0 -0
  64. {streamgate-2.0.0 → streamgate-3.0.0}/src/streamgate/consumer/classifier.py +0 -0
  65. {streamgate-2.0.0 → streamgate-3.0.0}/src/streamgate/contrib/__init__.py +0 -0
  66. {streamgate-2.0.0 → streamgate-3.0.0}/src/streamgate/contrib/_deps.py +0 -0
  67. {streamgate-2.0.0 → streamgate-3.0.0}/src/streamgate/contrib/http_probe/__init__.py +0 -0
  68. {streamgate-2.0.0 → streamgate-3.0.0}/src/streamgate/contrib/redis_dedup/config.py +0 -0
  69. {streamgate-2.0.0 → streamgate-3.0.0}/src/streamgate/contrib/sql_upsert/__init__.py +0 -0
  70. {streamgate-2.0.0 → streamgate-3.0.0}/src/streamgate/contrib/sql_upsert/_dialects/__init__.py +0 -0
  71. {streamgate-2.0.0 → streamgate-3.0.0}/src/streamgate/contrib/sql_upsert/_dialects/mssql.py +0 -0
  72. {streamgate-2.0.0 → streamgate-3.0.0}/src/streamgate/contrib/sql_upsert/_dialects/sqlite.py +0 -0
  73. {streamgate-2.0.0 → streamgate-3.0.0}/src/streamgate/contrib/sql_upsert/classifier.py +0 -0
  74. {streamgate-2.0.0 → streamgate-3.0.0}/src/streamgate/contrib/sql_upsert/config.py +0 -0
  75. {streamgate-2.0.0 → streamgate-3.0.0}/src/streamgate/ingest/__init__.py +0 -0
  76. {streamgate-2.0.0 → streamgate-3.0.0}/src/streamgate/ingest/dedup/__init__.py +0 -0
  77. {streamgate-2.0.0 → streamgate-3.0.0}/src/streamgate/ingest/dedup/in_memory.py +0 -0
  78. {streamgate-2.0.0 → streamgate-3.0.0}/src/streamgate/obs/__init__.py +0 -0
  79. {streamgate-2.0.0 → streamgate-3.0.0}/src/streamgate/protocols.py +0 -0
  80. {streamgate-2.0.0 → streamgate-3.0.0}/src/streamgate/resilience/__init__.py +0 -0
  81. {streamgate-2.0.0 → streamgate-3.0.0}/src/streamgate/resilience/backpressure.py +0 -0
  82. {streamgate-2.0.0 → streamgate-3.0.0}/src/streamgate/transport/__init__.py +0 -0
@@ -48,7 +48,7 @@ jobs:
48
48
 
49
49
  bare-install:
50
50
  # Regression gate for the pure-core contract: a bare install
51
- # (`pip install .`) must pull exactly aiokafka/loguru/pydantic (+transitive),
51
+ # (`pip install .`) must pull exactly aiokafka/pydantic (+transitive),
52
52
  # import cleanly, and run the pure_pipeline example end to end
53
53
  # (ingest → Kafka → consume → sqlite sink) against a real broker.
54
54
  runs-on: ubuntu-latest
@@ -5,6 +5,130 @@ All notable changes to this project will be documented in this file.
5
5
  The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/),
6
6
  and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0.html).
7
7
 
8
+ ## [3.0.0] - 2026-09-14
9
+
10
+ ### BREAKING — loguru removed, logging switched to stdlib `logging`
11
+
12
+ The library no longer depends on loguru. All internal loggers now live under
13
+ the standard-library `streamgate.*` logger tree
14
+ (`streamgate.ingest.producer`, `streamgate.consumer.loop`, ...), and the
15
+ library **only emits records — it never configures logging**: `import
16
+ streamgate` adds no handlers, changes no levels, touches no logging global
17
+ state. Wheel dependencies drop from three to two (`aiokafka`, `pydantic`).
18
+
19
+ **Log event names and structured field names are unchanged** — only the
20
+ transport moved. What breaks and how to migrate:
21
+
22
+ | 2.x behavior | 3.0.0 replacement |
23
+ |--------------|-------------------|
24
+ | `from streamgate import logger` | `logging.getLogger("streamgate")` (module-level loggers are `logging.getLogger(__name__)`) |
25
+ | `configure_logger(level=..., fmt=...)` | removed — configure stdlib logging in your app (`logging.basicConfig` / dictConfig) |
26
+ | auto-configured stderr JSON sink on `import streamgate` | gone — unconfigured hosts see WARNING+ only via stdlib `lastResort`; add `logging.basicConfig(level=logging.INFO, ...)` to restore visibility |
27
+ | JSON parsing of the library's stderr output (`{"timestamp", "level", "event", ...extra}`) | attach an equivalent stdlib JSON formatter yourself (recipe below) |
28
+ | per-level control of library logs | `logging.getLogger("streamgate").setLevel(logging.WARNING)` (or any subtree) |
29
+
30
+ Equivalent JSON formatter recipe (fields `timestamp` / `level` / `event` plus
31
+ flat extras, UTC ISO-8601):
32
+
33
+ ```python
34
+ import json
35
+ import logging
36
+
37
+
38
+ class JsonFormatter(logging.Formatter):
39
+ def format(self, record: logging.LogRecord) -> str:
40
+ entry = {
41
+ "timestamp": self.formatTime(record, datefmt="%Y-%m-%dT%H:%M:%S%z"),
42
+ "level": record.levelname,
43
+ "event": record.getMessage(),
44
+ }
45
+ reserved = set(logging.LogRecord("", 0, "", 0, "", (), None).__dict__) | {
46
+ "asctime", "message", "taskName",
47
+ }
48
+ for key, value in record.__dict__.items():
49
+ if key not in reserved and key not in entry:
50
+ entry[key] = value
51
+ return json.dumps(entry, ensure_ascii=False, default=str)
52
+
53
+
54
+ handler = logging.StreamHandler()
55
+ handler.setFormatter(JsonFormatter())
56
+ logging.getLogger("streamgate").addHandler(handler)
57
+ logging.getLogger("streamgate").setLevel(logging.INFO)
58
+ ```
59
+
60
+ Notes:
61
+
62
+ - `streamgate.obs.logging.emit` is the new internal emission helper; passing a
63
+ structured field whose name collides with a `LogRecord` reserved attribute
64
+ (`message`, `exc_info`, `asctime`, ...) raises `ValueError` (fail-fast) —
65
+ rename such fields. The one internal `exc_info=True` site now uses stdlib's
66
+ native `exc_info` parameter.
67
+ - `LoggingMetricsSink` keeps its level-string dispatch (`debug` / `info` /
68
+ `warning` / `error`, default `info`) and now writes through stdlib logging.
69
+
70
+ ### BREAKING — environment-variable fallbacks removed
71
+
72
+ The library no longer reads environment variables. Configuration has exactly
73
+ two sources of truth: **required settings are true required constructor
74
+ arguments** — missing one raises a native Python `TypeError` (missing
75
+ argument), never a silently guessed default such as `kafka:9092` — and
76
+ **optional settings hold their built-in defaults directly** on the options
77
+ objects (`RuntimeTuning` fields are plain `int` / `float` / `str`, not
78
+ `int | None` placeholders). Every `KAFKA__*` / `CONSUMER__*` / `METRICS__*` /
79
+ `BACKPRESSURE__*` key is gone. Hosts that injected configuration via the
80
+ environment (K8s ConfigMap, docker-compose) must resolve it themselves and
81
+ pass the values explicitly after upgrading: a missing required argument now
82
+ fails loudly at construction instead of silently connecting to the wrong
83
+ cluster, and optional knobs fall back to the built-in defaults instead of the
84
+ env value. Migrate key by key:
85
+
86
+ | Old behavior (≤2.x and earlier in this release) | 3.0.0 |
87
+ |---|---|
88
+ | `Producer(bootstrap_servers=None → KAFKA__BOOTSTRAP_SERVERS → "kafka:9092")` | `Producer(bootstrap_servers, topic, key=None, options=...)` — both truly required, no default |
89
+ | `Consumer(topic=None → KAFKA__TOPIC fail-fast ValueError)` | `Consumer(bootstrap_servers, topic, group_id, handler)` — all truly required; missing ⇒ native `TypeError` |
90
+ | `DlqOptions.enabled=None → CONSUMER__DLQ_ENABLED → True` | `enabled: bool = True` |
91
+ | `DlqOptions.send_retries=None → CONSUMER__DLQ_SEND_RETRIES → 3` | `send_retries: int = 3` |
92
+ | `DlqOptions.topic=None → KAFKA__DLQ_TOPIC` | `topic: str \| None = None` — still required when DLQ is enabled (runtime validation preserved) |
93
+ | `RuntimeTuning` 9 fields `None → CONSUMER__* → default` | fields hold defaults directly (`max_retries=3`, `retry_backoff_base=1.0`, `reconnect_base=1.0`, `reconnect_max=30.0`, `max_poll_records=500`, `session_timeout_ms=30000`, `max_poll_interval_ms=300000`, `auto_offset_reset="earliest"`, `backlog_check_interval=30.0`) |
94
+ | `METRICS__WINDOW_SECONDS` | `metrics_window_seconds=60` passed explicitly on `ProducerOptions` / `ConsumerOptions` |
95
+ | `CONSUMER__BATCH_SIZE` / `CONSUMER__FLUSH_TIMEOUT_SECONDS` | `batch_size` / `flush_timeout` explicit arguments (defaults unchanged: `500` / `5.0`) |
96
+ | `SqliteConsumer(...) / MssqlConsumer(...)` identity arguments optional | `bootstrap_servers` / `topic` / `group_id` are required keyword arguments, mirroring the core `Consumer` |
97
+ | `streamgate.consumer.options.resolve_tuning` / `ResolvedRuntimeTuning` | removed (internal resolution layer); `RuntimeTuning` holds concrete values and its own `backoff_seconds()` |
98
+
99
+ Notes:
100
+
101
+ - All runtime validations are preserved (not fallbacks): DLQ topic required
102
+ when DLQ is enabled, metrics window range 1–600, `batch_size >= 1`,
103
+ `flush_timeout > 0`.
104
+ - `KafkaConfig` / `BackpressureConfig` field defaults are unchanged — they
105
+ are pure config objects and never read the environment; the `Producer`
106
+ constructor values always override `KafkaConfig.bootstrap_servers` /
107
+ `.topic`.
108
+
109
+ ## [2.1.0] - 2026-09-11
110
+
111
+ ### Added
112
+
113
+ - **`RedisGroupDedupCache.confirm_empty(group)`** — an explicit "confirmed
114
+ empty" negative-cache state for group keys. Writes a reserved marker field
115
+ into the group HASH (`_GROUP_EMPTY_MARKER`, TTL follows
116
+ `group_ttl_seconds`, idle-GC); afterwards `group_exists()` is `True` and
117
+ `get_group_fields()` returns `{}`, so enumeration queries stop re-sourcing a
118
+ known-empty group from the DB on every call. `get_identity_meta` /
119
+ `get_group_fields` filter the marker; `group_exists` semantics are unchanged
120
+ ("key present ⇒ group state authoritative, possibly empty"). Purely additive
121
+ — no existing signature changed.
122
+ - **`RedisGroupDedupCarrier.load_group_refill(group)`** — public primitive
123
+ exposing the whole-group cold load + backfill path for reuse by enumeration
124
+ (query) sides instead of copying the mechanism: group-level single-flight,
125
+ concurrency gate, chunked backfill and the "only a successful whole-group
126
+ backfill creates the key" invariant are all shared with `admit` stage 3. It
127
+ returns a `GroupLoadOutcome` (`FOUND` / `EMPTY` / `FAILED` / `GATE_FULL`);
128
+ policy (endpoints, response bodies, HTTP 429/502, `source` labels) stays in
129
+ the caller's adapter layer. `GroupLoadOutcome` / `GroupLoadResult` are now
130
+ exported from `streamgate.contrib.redis_dedup`.
131
+
8
132
  ## [2.0.0] - 2026-09-11
9
133
 
10
134
  ### BREAKING — unified backfill contract (feat!)
@@ -0,0 +1,196 @@
1
+ # Configuration reference
2
+
3
+ streamgate components are configured with typed parameters and Pydantic
4
+ objects. There are exactly **two sources of truth**:
5
+
6
+ 1. **Required settings are true required constructor arguments** —
7
+ `Producer(bootstrap_servers, topic, ...)` and
8
+ `Consumer(bootstrap_servers, topic, group_id, handler)`. Missing one fails
9
+ with a native Python `TypeError` (missing argument) at construction. The
10
+ library never reads environment variables and never silently defaults a
11
+ required setting.
12
+ 2. **Optional settings directly hold their built-in defaults** — advanced
13
+ knobs fold into `ProducerOptions` / `ConsumerOptions` (dedup, backpressure,
14
+ DLQ, tuning, metrics window); omit them and the documented defaults apply.
15
+
16
+ If your host wants env-driven configuration, resolve the environment yourself
17
+ and pass the values explicitly (the examples do exactly that with
18
+ `os.environ.get(..., default)`).
19
+
20
+ Database and Redis connection settings are **not part of the core framework**:
21
+ storage carriers are injected and their configuration belongs to your
22
+ application. The official strategy implementations under `streamgate.contrib`
23
+ ship typed config objects (below) that you construct and pass to the factories.
24
+
25
+ ## Logging (not a configuration surface)
26
+
27
+ streamgate emits logs through the standard-library `logging` module and ships
28
+ **no logging configuration** — the library only emits records under the
29
+ `streamgate.*` logger tree; level, handlers and formats are configured by the
30
+ host application (e.g. `logging.getLogger("streamgate")`).
31
+
32
+ ---
33
+
34
+ ## Consumer (flat arguments)
35
+
36
+ | Argument | Default | Description |
37
+ |----------|---------|-------------|
38
+ | `bootstrap_servers` | **required** | Comma-separated broker list. Missing ⇒ native `TypeError`. |
39
+ | `topic` | **required** | Topic to consume from. Missing ⇒ native `TypeError`. |
40
+ | `group_id` | **required** | Kafka consumer group id. Missing ⇒ native `TypeError`. |
41
+ | `handler` | **required** | Async callable `handler(batch, context)` — the only outlet. |
42
+ | `batch_size` | `500` | Records buffered before a flush. `1` = single-record real-time. |
43
+ | `flush_timeout` | `5.0` | Max wait before a partially filled batch is flushed. |
44
+ | `options` | `None` | `ConsumerOptions` — everything below. |
45
+
46
+ ## ConsumerOptions
47
+
48
+ | Field | Default | Description |
49
+ |-------|---------|-------------|
50
+ | `expected_type` | `None` | Envelope `type` validation (`None` = accept any; mismatched records are quarantined to the DLQ). |
51
+ | `probe` | `None` | Single-record callable with handler-equivalent, idempotent semantics. Provided ⇒ POISON batches are located record-by-record and bad records quarantined precisely; omitted ⇒ the whole batch is quarantined. |
52
+ | `classifier` | `DefaultErrorClassifier` | RETRY/POISON/FATAL mapping for handler exceptions. Connecting a DB? Inject a DB-aware classifier (`contrib.sql_upsert.SQLAlchemyErrorClassifier`). |
53
+ | `persist_hook` | `None` | `DedupCarrier` hooked into the consume loop: `on_persisted` runs after each successful batch (dedup linkage with the producer side). Lifecycle is framework-managed. |
54
+ | `collapse_key` | `None` | In-batch dedup key (keeps the last record) applied to hook notifications. |
55
+ | `log_context` | `None` | Per-record extra fields for quarantine logs. |
56
+ | `codec` | `JsonEnvelopeCodec` | Message envelope codec. |
57
+ | `health_probe` | `None` | Side-channel health component; implements `check_health()` ⇒ observed in snapshots; `start()`/`close()` ⇒ lifecycle managed. |
58
+ | `backlog_ttl_seconds` | `18000` | Backlog alert budget: WARN > TTL/2, ERROR > TTL×0.8. |
59
+ | `metrics_window_seconds` | `60` | Sliding-window length (1–600) for rate/latency snapshot fields; out-of-range values fail at construction. |
60
+ | `dlq` | `DlqOptions()` | DLQ options (below). |
61
+ | `tuning` | `RuntimeTuning()` | Runtime tuning (below). |
62
+
63
+ ## DlqOptions
64
+
65
+ | Field | Default | Description |
66
+ |-------|---------|-------------|
67
+ | `enabled` | `true` | Master switch; `false` falls back to the paused behavior (emergency escape hatch). |
68
+ | `topic` | `None` | Dead-letter topic; required when DLQ is enabled (validated at DLQ-producer construction). |
69
+ | `message_type` | `streamgate_dlq` | Envelope `type` of DLQ messages. |
70
+ | `send_retries` | `3` | Total attempts per DLQ send (exhausting them pauses the batch). |
71
+
72
+ ## RuntimeTuning
73
+
74
+ Every field directly holds its built-in default:
75
+
76
+ | Field | Default | Description |
77
+ |-------|---------|-------------|
78
+ | `max_retries` | `3` | Handler retries per batch. |
79
+ | `retry_backoff_base` | `1.0` | Retry backoff base (seconds, exponential). |
80
+ | `reconnect_base` | `1.0` | Reconnect/paused-recovery backoff base (seconds). |
81
+ | `reconnect_max` | `30.0` | Reconnect backoff cap (seconds). |
82
+ | `max_poll_records` | `500` | Max records per poll. |
83
+ | `session_timeout_ms` | `30000` | Kafka session timeout. |
84
+ | `max_poll_interval_ms` | `300000` | Max time between polls before rebalance. |
85
+ | `auto_offset_reset` | `earliest` | Offset reset policy. |
86
+ | `backlog_check_interval` | `30.0` | Backlog age check period (seconds). |
87
+
88
+ ---
89
+
90
+ ## Contrib strategy configs
91
+
92
+ These are plain constructor arguments. All belong to `streamgate.contrib`
93
+ packages — install the matching extra first (`[redis]` / `[sql]` / `[http]`).
94
+
95
+ ### `contrib.redis_dedup.RedisConfig`
96
+
97
+ | Field | Default | Description |
98
+ |-------|---------|-------------|
99
+ | `url` | `redis://localhost:6379/0` | Redis connection URL. |
100
+ | `key_prefix` | `streamgate:` | Key prefix for dedup keys (`<prefix>dedup:<identity>`). |
101
+ | `identity_ttl_seconds` | `18000` | Single-key placeholder/summary TTL (idle-GC heartbeat: every write renews it). |
102
+ | `group_key_prefix` | `dedup_group:` | Group-carrier key prefix (`<prefix>dedup_group:<group>` HASH; type-isolated from the single-key STRING). |
103
+ | `group_ttl_seconds` | `18000` | Group HASH TTL (idle-GC heartbeat: any group write renews it). |
104
+ | `socket_timeout_ms` | `1000` | Read/query path timeout. |
105
+ | `recv_timeout_ms` | `500` | Push path (reserve/summary write) timeout. |
106
+
107
+ Strategy-level knobs live on `RedisDedupCarrierConfig`
108
+ (`fail_closed_on_unavailable`, `cold_path_max_concurrency`, gate-full /
109
+ unavailable `retry_after` seconds, `log_context` mapping) and on
110
+ `RedisGroupDedupCarrierConfig` (`group_cold_path_max_concurrency`, same
111
+ family of knobs).
112
+
113
+ ### `contrib.sql_upsert.DbConfig`
114
+
115
+ | Field | Default | Description |
116
+ |-------|---------|-------------|
117
+ | `connection_string` | **required** | SQLAlchemy async URL (`sqlite+aiosqlite:///...` or `mssql+aioodbc://...`). Missing ⇒ assembly error with fix instructions. |
118
+ | `echo` | `false` | SQLAlchemy echo logging. |
119
+ | `write_timeout_seconds` | `20` | Driver statement timeout (MSSQL/pyodbc hook; must be < `write_wait_seconds`). |
120
+ | `write_wait_seconds` | `25` | Batch upsert call-level `wait_for` ceiling. |
121
+ | `pool_timeout_seconds` | `3` | Pool checkout timeout (fail fast when exhausted). |
122
+ | `write_pool_size` | `10` | Fixed write-pool connections. |
123
+ | `write_pool_max_overflow` | `20` | Write-pool overflow connections. |
124
+
125
+ The dialect (SQLite vs MSSQL) is selected by the connection string;
126
+ `contrib.sqlite_upsert` and `contrib.mssql_upsert` are the pre-assembled
127
+ factory entries (`SqliteConsumer` / `MssqlConsumer`).
128
+
129
+ ### Backpressure probing (contrib.http_probe)
130
+
131
+ `HttpProbeSignal` is constructed with a `BackpressureConfig` instance (see
132
+ below) — build the config object yourself and pass it to the producer via
133
+ `ProducerOptions(signal=HttpProbeSignal(config), backpressure=config)`.
134
+
135
+ ---
136
+
137
+ ## KafkaConfig (`ProducerOptions.kafka`)
138
+
139
+ The producer takes `bootstrap_servers`/`topic` as **required** flat
140
+ constructor arguments; the remaining connection/self-healing knobs live on
141
+ `KafkaConfig`, injectable via `ProducerOptions.kafka` (defaults below). The
142
+ constructor values always override the `bootstrap_servers` / `topic` fields.
143
+
144
+ | Field | Default | Description |
145
+ |-------|---------|-------------|
146
+ | `bootstrap_servers` | `kafka:9092` | Comma-separated broker list. `http://` / `https://` / `kafka://` scheme prefixes are stripped automatically. Always overridden by the `Producer` constructor value. |
147
+ | `topic` | `None` | Topic to push to (constructor value wins; validated when a raw `KafkaConfig` is used directly). |
148
+ | `acks` | `all` | Producer acks level. |
149
+ | `request_timeout_ms` | `10000` | Producer request timeout (ms). |
150
+ | `enable_idempotence` | `true` | Idempotent producer (safe retries). |
151
+ | `health_check_interval_seconds` | `30.0` | Periodic health probe interval; unhealthy instances are rebuilt. |
152
+ | `reconnect_base_backoff_seconds` | `1.0` | Reconnect exponential backoff base. |
153
+ | `reconnect_max_backoff_seconds` | `30.0` | Reconnect backoff cap. |
154
+ | `reconnect_failure_threshold` | `1` | Consecutive send failures before a rebuild is triggered (raise to tolerate broker jitter). |
155
+ | `send_failure_window_seconds` | `30.0` | A send failure within this window marks the instance unhealthy. |
156
+ | `unhealthy_check_interval_seconds` | `5.0` | High-frequency probe interval while unhealthy/rebuilding. |
157
+
158
+ The dead-letter topic is configured on the consumer side via
159
+ `DlqOptions.topic` (see above) — no longer on `KafkaConfig`.
160
+
161
+ ## BackpressureConfig (`ProducerOptions.backpressure`)
162
+
163
+ The producer defaults to the built-in `ManualBackpressureSignal` (static
164
+ switch, backpressure off). For dynamic probing pass
165
+ `options=ProducerOptions(backpressure=..., signal=...)` —
166
+ `contrib.http_probe.HttpProbeSignal` reads this config object.
167
+
168
+ | Field | Default | Description |
169
+ |-------|---------|-------------|
170
+ | `enabled` | `true` | Master switch; `false` = fully open (emergency rollback). |
171
+ | `consumer_health_url` | `http://localhost:9109/health` | Consumer health endpoint polled by the probe. |
172
+ | `check_interval_seconds` | `30.0` | Poll period while OPEN. |
173
+ | `timeout_seconds` | `2.0` | Per-probe timeout. |
174
+ | `probe_retries` | `3` | Retries per cycle (excluding the first attempt); `0` disables. |
175
+ | `probe_retry_interval_seconds` | `15.0` | Interval between retries. |
176
+ | `trip_seconds` | `9000.0` | Reject when backlog age exceeds this. |
177
+ | `recover_seconds` | `7200.0` | Resume when backlog age falls below this (hysteresis anti-flapping). |
178
+ | `retry_after_seconds` | `60` | `Retry-After` hint (seconds) while rejecting. |
179
+ | `fail_closed_on_unreachable` | `true` | Treat unreachable probe (retries exhausted) as backlog-exceeded. |
180
+ | `reject_on_any_degraded` | `false` | Legacy escape hatch: reject when any component is degraded. |
181
+ | `unhealthy_check_interval_seconds` | `5.0` | Poll period while REJECTING. |
182
+
183
+ ## Metrics window
184
+
185
+ Health-snapshot rate metrics: both `Producer` and `Consumer` maintain
186
+ in-memory sliding windows and expose computed rates / latencies through their
187
+ health snapshots (`push_rate`, `handle_rate`, `produce_latency_ms_avg`,
188
+ ...). Windowed aggregates are computed on read — no background tasks, no extra
189
+ endpoint; rates decay to `0.0` once traffic stops for longer than the window.
190
+
191
+ Configure via `metrics_window_seconds` on `ProducerOptions` /
192
+ `ConsumerOptions`:
193
+
194
+ | Value | Description |
195
+ |-------|-------------|
196
+ | `60` (default) | Sliding-window length (seconds) for all rate/latency fields. Valid range 1–600; out-of-range values fail at construction. Short windows react faster but are noisier. |
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: streamgate
3
- Version: 2.0.0
3
+ Version: 3.0.0
4
4
  Summary: A pure-core Kafka data pipeline framework: conditional admission, reliable delivery, and one obvious consume-loop outlet via a typed handler.
5
5
  Project-URL: Homepage, https://github.com/pwg-code/streamgate
6
6
  Project-URL: Repository, https://github.com/pwg-code/streamgate
@@ -21,7 +21,6 @@ Classifier: Topic :: System :: Distributed Computing
21
21
  Classifier: Typing :: Typed
22
22
  Requires-Python: >=3.10
23
23
  Requires-Dist: aiokafka>=0.10
24
- Requires-Dist: loguru>=0.7
25
24
  Requires-Dist: pydantic>=2.5
26
25
  Provides-Extra: http
27
26
  Requires-Dist: httpx>=0.27; extra == 'http'
@@ -45,7 +44,7 @@ English | [中文](README.zh-CN.md)
45
44
 
46
45
  **streamgate** is a pure-core Kafka data pipeline framework: a dedup-aware door on the way in, reliable delivery through Kafka, and one obvious outlet on the other side — `Producer(bootstrap_servers, topic, ...)` / `Consumer(bootstrap_servers, topic, group_id, handler)`.
47
46
 
48
- The wheel installs exactly three dependencies (`aiokafka`, `loguru`, `pydantic`) and the core contains **zero database, Redis or HTTP-client code**. Official I/O strategy implementations live in [`streamgate.contrib`](#contrib-official-strategy-implementations) — opt-in via extras — and runnable demos live in [`examples/`](examples/).
47
+ The wheel installs exactly two dependencies (`aiokafka`, `pydantic`) and the core contains **zero database, Redis or HTTP-client code**. Logging goes through the standard-library `logging` module — the library emits records and never configures them. Official I/O strategy implementations live in [`streamgate.contrib`](#contrib-official-strategy-implementations) — opt-in via extras — and runnable demos live in [`examples/`](examples/).
49
48
 
50
49
  - 中文提示:本仓库对外文档以英文为主;内部代码注释保留中文。
51
50
 
@@ -68,6 +67,8 @@ aiokafka gives you **transport**; streamgate gives you the **operating semantics
68
67
 
69
68
  A complete produce → Kafka → consume → SQLite round trip that runs on a **bare install** (stdlib outlet, zero extra packages) — [`examples/pure_pipeline/`](examples/pure_pipeline/):
70
69
 
70
+ > Logging note: streamgate emits via stdlib `logging` and never configures it. Without host configuration only WARNING+ shows (stdlib `lastResort`); add one line `logging.basicConfig(level=logging.INFO, ...)` at your entry point to see INFO logs (the examples all do this — see [Logging](#logging)).
71
+
71
72
  ```python
72
73
  import asyncio
73
74
  from pydantic import BaseModel
@@ -218,7 +219,7 @@ await consumer.run()
218
219
  ## Installation
219
220
 
220
221
  ```bash
221
- pip install streamgate # aiokafka + loguru + pydantic. That's it.
222
+ pip install streamgate # aiokafka + pydantic. That's it.
222
223
  pip install "streamgate[redis]" # + Redis distributed dedup carrier
223
224
  pip install "streamgate[sql]" # + SQL upsert outlet (SQLite + MSSQL)
224
225
  pip install "streamgate[http]" # + HTTP probe backpressure
@@ -228,7 +229,20 @@ For anything else (PostgreSQL, Elasticsearch, your own destination...), write th
228
229
 
229
230
  ## Configuration
230
231
 
231
- Both sides take required settings as flat constructor arguments with environment-variable fallbacks (`KAFKA__BOOTSTRAP_SERVERS`, `KAFKA__TOPIC`, `CONSUMER__GROUP_ID`); advanced knobs fold into `ProducerOptions` / `ConsumerOptions` (dedup, backpressure, DLQ, tuning, metrics window) with the same env fallbacks. Required settings fail fast at startup with fix instructions. Database/Redis connection settings belong to your application (see the examples' local configs). See [CONFIGURATION.md](CONFIGURATION.md).
232
+ There are exactly two sources of truth. **Required settings are true required constructor parameters** — `Producer(bootstrap_servers, topic, ...)` and `Consumer(bootstrap_servers, topic, group_id, handler)` fail with a native `TypeError` the moment an argument is missing (they are never silently defaulted or read from the environment). **Optional settings directly hold their built-in defaults** and fold into `ProducerOptions` / `ConsumerOptions` (dedup, backpressure, DLQ, tuning, metrics window) — omit them and the documented defaults apply. The library reads no environment variables; if your host wants env-driven config, resolve it yourself and pass the values explicitly (the examples do exactly that). Database/Redis connection settings belong to your application (see the examples' local configs). See [CONFIGURATION.md](CONFIGURATION.md).
233
+
234
+ ## Logging
235
+
236
+ streamgate logs through the standard-library `logging` module — **the library emits records and never configures logging** (no handlers added, no levels changed at import). Every logger lives under the `streamgate` namespace (`streamgate.ingest.producer`, `streamgate.consumer.loop`, ...), so the whole tree is controlled by the host:
237
+
238
+ ```python
239
+ import logging
240
+
241
+ logging.basicConfig(level=logging.INFO, format="%(asctime)s %(levelname)s %(name)s %(message)s")
242
+ logging.getLogger("streamgate").setLevel(logging.WARNING) # or any subtree: silence INFO noise
243
+ ```
244
+
245
+ Unconfigured hosts see WARNING and above only (stdlib `lastResort` on stderr). Event names (`batch_handle_start`, `poison_message_skipped`, ...) are a stable public contract; structured fields travel as flat `LogRecord` extras. For JSON output, attach any stdlib-compatible JSON formatter of your choice.
232
246
 
233
247
  ## Examples
234
248
 
@@ -236,7 +250,7 @@ See [`examples/`](examples/) for five runnable pipelines plus a `docker-compose.
236
250
 
237
251
  ## Migrating from 0.x
238
252
 
239
- 1.0.0 redesigned **both** sides (breaking): the consumer side replaces `ConsumerWorker` + `ConsumeSpec` + `RecordWriter` with the flat `Consumer` constructor; the producer side replaces `IngestBinding` + `IngestGateway` with the flat `Producer` constructor (`process()` → `push()`, `IngestOutcome` → `PushResult`, `overwrite` → `force`, `admission` → `DedupOptions`, `entity_key`/`slot_key` split into routing `key` + identity `DedupOptions.key`). See the [CHANGELOG](CHANGELOG.md) for the complete old→new mapping (API, result kinds, metrics names, log events, env keys, import paths).
253
+ 1.0.0 redesigned **both** sides (breaking): the consumer side replaces `ConsumerWorker` + `ConsumeSpec` + `RecordWriter` with the flat `Consumer` constructor; the producer side replaces `IngestBinding` + `IngestGateway` with the flat `Producer` constructor (`process()` → `push()`, `IngestOutcome` → `PushResult`, `overwrite` → `force`, `admission` → `DedupOptions`, `entity_key`/`slot_key` split into routing `key` + identity `DedupOptions.key`). 3.0.0 removed the bundled loguru logging (stdlib `logging` instead — `from streamgate import logger` is gone) **and all environment-variable fallbacks**: required settings are now true required constructor parameters and every `KAFKA__*` / `CONSUMER__*` / `METRICS__*` key is gone — options hold their built-in defaults directly. See the [CHANGELOG](CHANGELOG.md) for the complete old→new mapping (API, result kinds, metrics names, log events, import paths) and the per-key migration table.
240
254
 
241
255
  ## Roadmap
242
256
 
@@ -9,7 +9,7 @@ English | [中文](README.zh-CN.md)
9
9
 
10
10
  **streamgate** is a pure-core Kafka data pipeline framework: a dedup-aware door on the way in, reliable delivery through Kafka, and one obvious outlet on the other side — `Producer(bootstrap_servers, topic, ...)` / `Consumer(bootstrap_servers, topic, group_id, handler)`.
11
11
 
12
- The wheel installs exactly three dependencies (`aiokafka`, `loguru`, `pydantic`) and the core contains **zero database, Redis or HTTP-client code**. Official I/O strategy implementations live in [`streamgate.contrib`](#contrib-official-strategy-implementations) — opt-in via extras — and runnable demos live in [`examples/`](examples/).
12
+ The wheel installs exactly two dependencies (`aiokafka`, `pydantic`) and the core contains **zero database, Redis or HTTP-client code**. Logging goes through the standard-library `logging` module — the library emits records and never configures them. Official I/O strategy implementations live in [`streamgate.contrib`](#contrib-official-strategy-implementations) — opt-in via extras — and runnable demos live in [`examples/`](examples/).
13
13
 
14
14
  - 中文提示:本仓库对外文档以英文为主;内部代码注释保留中文。
15
15
 
@@ -32,6 +32,8 @@ aiokafka gives you **transport**; streamgate gives you the **operating semantics
32
32
 
33
33
  A complete produce → Kafka → consume → SQLite round trip that runs on a **bare install** (stdlib outlet, zero extra packages) — [`examples/pure_pipeline/`](examples/pure_pipeline/):
34
34
 
35
+ > Logging note: streamgate emits via stdlib `logging` and never configures it. Without host configuration only WARNING+ shows (stdlib `lastResort`); add one line `logging.basicConfig(level=logging.INFO, ...)` at your entry point to see INFO logs (the examples all do this — see [Logging](#logging)).
36
+
35
37
  ```python
36
38
  import asyncio
37
39
  from pydantic import BaseModel
@@ -182,7 +184,7 @@ await consumer.run()
182
184
  ## Installation
183
185
 
184
186
  ```bash
185
- pip install streamgate # aiokafka + loguru + pydantic. That's it.
187
+ pip install streamgate # aiokafka + pydantic. That's it.
186
188
  pip install "streamgate[redis]" # + Redis distributed dedup carrier
187
189
  pip install "streamgate[sql]" # + SQL upsert outlet (SQLite + MSSQL)
188
190
  pip install "streamgate[http]" # + HTTP probe backpressure
@@ -192,7 +194,20 @@ For anything else (PostgreSQL, Elasticsearch, your own destination...), write th
192
194
 
193
195
  ## Configuration
194
196
 
195
- Both sides take required settings as flat constructor arguments with environment-variable fallbacks (`KAFKA__BOOTSTRAP_SERVERS`, `KAFKA__TOPIC`, `CONSUMER__GROUP_ID`); advanced knobs fold into `ProducerOptions` / `ConsumerOptions` (dedup, backpressure, DLQ, tuning, metrics window) with the same env fallbacks. Required settings fail fast at startup with fix instructions. Database/Redis connection settings belong to your application (see the examples' local configs). See [CONFIGURATION.md](CONFIGURATION.md).
197
+ There are exactly two sources of truth. **Required settings are true required constructor parameters** — `Producer(bootstrap_servers, topic, ...)` and `Consumer(bootstrap_servers, topic, group_id, handler)` fail with a native `TypeError` the moment an argument is missing (they are never silently defaulted or read from the environment). **Optional settings directly hold their built-in defaults** and fold into `ProducerOptions` / `ConsumerOptions` (dedup, backpressure, DLQ, tuning, metrics window) — omit them and the documented defaults apply. The library reads no environment variables; if your host wants env-driven config, resolve it yourself and pass the values explicitly (the examples do exactly that). Database/Redis connection settings belong to your application (see the examples' local configs). See [CONFIGURATION.md](CONFIGURATION.md).
198
+
199
+ ## Logging
200
+
201
+ streamgate logs through the standard-library `logging` module — **the library emits records and never configures logging** (no handlers added, no levels changed at import). Every logger lives under the `streamgate` namespace (`streamgate.ingest.producer`, `streamgate.consumer.loop`, ...), so the whole tree is controlled by the host:
202
+
203
+ ```python
204
+ import logging
205
+
206
+ logging.basicConfig(level=logging.INFO, format="%(asctime)s %(levelname)s %(name)s %(message)s")
207
+ logging.getLogger("streamgate").setLevel(logging.WARNING) # or any subtree: silence INFO noise
208
+ ```
209
+
210
+ Unconfigured hosts see WARNING and above only (stdlib `lastResort` on stderr). Event names (`batch_handle_start`, `poison_message_skipped`, ...) are a stable public contract; structured fields travel as flat `LogRecord` extras. For JSON output, attach any stdlib-compatible JSON formatter of your choice.
196
211
 
197
212
  ## Examples
198
213
 
@@ -200,7 +215,7 @@ See [`examples/`](examples/) for five runnable pipelines plus a `docker-compose.
200
215
 
201
216
  ## Migrating from 0.x
202
217
 
203
- 1.0.0 redesigned **both** sides (breaking): the consumer side replaces `ConsumerWorker` + `ConsumeSpec` + `RecordWriter` with the flat `Consumer` constructor; the producer side replaces `IngestBinding` + `IngestGateway` with the flat `Producer` constructor (`process()` → `push()`, `IngestOutcome` → `PushResult`, `overwrite` → `force`, `admission` → `DedupOptions`, `entity_key`/`slot_key` split into routing `key` + identity `DedupOptions.key`). See the [CHANGELOG](CHANGELOG.md) for the complete old→new mapping (API, result kinds, metrics names, log events, env keys, import paths).
218
+ 1.0.0 redesigned **both** sides (breaking): the consumer side replaces `ConsumerWorker` + `ConsumeSpec` + `RecordWriter` with the flat `Consumer` constructor; the producer side replaces `IngestBinding` + `IngestGateway` with the flat `Producer` constructor (`process()` → `push()`, `IngestOutcome` → `PushResult`, `overwrite` → `force`, `admission` → `DedupOptions`, `entity_key`/`slot_key` split into routing `key` + identity `DedupOptions.key`). 3.0.0 removed the bundled loguru logging (stdlib `logging` instead — `from streamgate import logger` is gone) **and all environment-variable fallbacks**: required settings are now true required constructor parameters and every `KAFKA__*` / `CONSUMER__*` / `METRICS__*` key is gone — options hold their built-in defaults directly. See the [CHANGELOG](CHANGELOG.md) for the complete old→new mapping (API, result kinds, metrics names, log events, import paths) and the per-key migration table.
204
219
 
205
220
  ## Roadmap
206
221
 
@@ -14,7 +14,7 @@
14
14
  判重、背压、自愈 批量缓冲、重试、坏数据隔离
15
15
  ```
16
16
 
17
- 框架核心只装 3 个依赖(`aiokafka` / `loguru` / `pydantic`),核心层**没有一行数据库、Redis、HTTP 客户端代码**。官方 I/O 策略实现收录在 [`streamgate.contrib`](#不想自己写用-contrib)——按 extras 按需安装、装完即可 import;可运行的完整演示在 [`examples/`](examples/)。
17
+ 框架核心只装 2 个依赖(`aiokafka` / `pydantic`),核心层**没有一行数据库、Redis、HTTP 客户端代码**。日志走标准库 `logging`——库只发日志记录、永不配置。官方 I/O 策略实现收录在 [`streamgate.contrib`](#不想自己写用-contrib)——按 extras 按需安装、装完即可 import;可运行的完整演示在 [`examples/`](examples/)。
18
18
 
19
19
  ---
20
20
 
@@ -358,7 +358,7 @@ await consumer.run()
358
358
  ## 安装
359
359
 
360
360
  ```bash
361
- pip install streamgate # 只装 aiokafka + loguru + pydantic
361
+ pip install streamgate # 只装 aiokafka + pydantic
362
362
  pip install "streamgate[redis]" # + Redis 分布式判重
363
363
  pip install "streamgate[sql]" # + SQL 幂等出口(SQLite + MSSQL)
364
364
  pip install "streamgate[http]" # + HTTP 探活背压
@@ -368,7 +368,20 @@ pip install "streamgate[http]" # + HTTP 探活背压
368
368
 
369
369
  ## 配置
370
370
 
371
- `Producer` / `Consumer` 必填项直接传参,未传时回退同名环境变量(`KAFKA__BOOTSTRAP_SERVERS` / `KAFKA__TOPIC` / `CONSUMER__GROUP_ID`);高级项收口在 `ProducerOptions` / `ConsumerOptions`(判重、背压、DLQ、调优、指标窗口等),同样跟随 `KAFKA__*` / `CONSUMER__*` 环境变量回退。必填项缺失即启动失败并附修复指引,绝不带猜测默认值上路。数据库/Redis 连接配置归你的应用管(示例各自带本地配置)。详见 [CONFIGURATION.md](CONFIGURATION.md)。
371
+ 配置只有两个真相来源:**必填项是真必填构造参数**——`Producer(bootstrap_servers, topic, ...)` 与 `Consumer(bootstrap_servers, topic, group_id, handler)` 缺任一参数即原生 `TypeError`(绝不静默默认、绝不读环境变量);**可选项直接持有内置默认值**,收口在 `ProducerOptions` / `ConsumerOptions`(判重、背压、DLQ、调优、指标窗口等),不传即文档默认行为。库代码不读取任何环境变量;宿主若要 env 驱动配置,自行解析后显式传参(示例即如此)。数据库/Redis 连接配置归你的应用管(示例各自带本地配置)。详见 [CONFIGURATION.md](CONFIGURATION.md)。
372
+
373
+ ## 日志
374
+
375
+ streamgate 走标准库 `logging`——**库只发日志记录、永不配置日志**(import 时不增删任何 handler、不改任何级别)。所有 logger 挂在 `streamgate` 命名空间下(`streamgate.ingest.producer`、`streamgate.consumer.loop` 等),宿主一行整树控制:
376
+
377
+ ```python
378
+ import logging
379
+
380
+ logging.basicConfig(level=logging.INFO, format="%(asctime)s %(levelname)s %(name)s %(message)s")
381
+ logging.getLogger("streamgate").setLevel(logging.WARNING) # 或任意子树:关掉 INFO 噪音
382
+ ```
383
+
384
+ 未配置日志的宿主只能看到 WARNING 及以上(stdlib `lastResort` 输出到 stderr)。事件名(`batch_handle_start`、`poison_message_skipped` 等)是稳定的对外契约;结构化字段以 `LogRecord` extra 平铺。需要 JSON 输出时,接任意 stdlib 兼容的 JSON Formatter 即可。
372
385
 
373
386
  ## 架构
374
387
 
@@ -396,7 +409,7 @@ pip install "streamgate[http]" # + HTTP 探活背压
396
409
 
397
410
  ## 从 0.x 迁移
398
411
 
399
- 1.0.0 对**两端**都做了破坏性重设计:消费侧 `ConsumerWorker` + `ConsumeSpec` + `RecordWriter` 由扁平 `Consumer` 构造器替代;生产侧 `IngestBinding` + `IngestGateway` 由扁平 `Producer` 构造器替代(`process()` → `push()`、`IngestOutcome` → `PushResult`、`overwrite` → `force`、`admission` → `DedupOptions`、`entity_key`/`slot_key` 拆分为路由键 `key` + 身份键 `DedupOptions.key`)。完整的新旧对照(API、结果 kind、指标名、日志事件、环境变量、import 路径)见 [CHANGELOG](CHANGELOG.md)。
412
+ 1.0.0 对**两端**都做了破坏性重设计:消费侧 `ConsumerWorker` + `ConsumeSpec` + `RecordWriter` 由扁平 `Consumer` 构造器替代;生产侧 `IngestBinding` + `IngestGateway` 由扁平 `Producer` 构造器替代(`process()` → `push()`、`IngestOutcome` → `PushResult`、`overwrite` → `force`、`admission` → `DedupOptions`、`entity_key`/`slot_key` 拆分为路由键 `key` + 身份键 `DedupOptions.key`)。3.0.0 移除了内置 loguru 日志(改走标准库 `logging`,`from streamgate import logger` 不复存在),**并移除了全部环境变量回退**:必填项成为真必填构造参数,所有 `KAFKA__*` / `CONSUMER__*` / `METRICS__*` 键不复存在——可选项字段直接持有内置默认值。完整的新旧对照(API、结果 kind、指标名、日志事件、import 路径)与逐项迁移表见 [CHANGELOG](CHANGELOG.md)。
400
413
 
401
414
  ## 路线图
402
415
 
@@ -8,6 +8,7 @@
8
8
  """
9
9
 
10
10
  import asyncio
11
+ import logging
11
12
  import os
12
13
 
13
14
  from health_server import serve_consumer_health
@@ -23,6 +24,9 @@ async def print_orders(
23
24
 
24
25
 
25
26
  async def main() -> None:
27
+ logging.basicConfig(
28
+ level=logging.INFO, format="%(asctime)s %(levelname)s %(name)s %(message)s"
29
+ )
26
30
  consumer = Consumer(
27
31
  bootstrap_servers=os.environ.get("KAFKA__BOOTSTRAP_SERVERS", "localhost:29092"),
28
32
  topic=os.environ.get("KAFKA__TOPIC", "orders"),
@@ -5,8 +5,11 @@
5
5
  """
6
6
 
7
7
  import asyncio
8
+ import logging
8
9
 
9
- from streamgate import Consumer, logger
10
+ from streamgate import Consumer
11
+
12
+ logger = logging.getLogger(__name__)
10
13
 
11
14
 
12
15
  async def serve_consumer_health(consumer: Consumer, port: int) -> None:
@@ -36,11 +39,11 @@ async def serve_consumer_health(consumer: Consumer, port: int) -> None:
36
39
  )
37
40
  await writer.drain()
38
41
  except Exception as e:
39
- logger.debug("health_endpoint_error", error=str(e))
42
+ logger.debug("health_endpoint_error: %s", e)
40
43
  finally:
41
44
  writer.close()
42
45
 
43
46
  server = await asyncio.start_server(_handle, "0.0.0.0", port)
44
- logger.info("health_endpoint_listening", port=port)
47
+ logger.info("health_endpoint_listening: port=%d", port)
45
48
  async with server:
46
49
  await server.serve_forever()
@@ -10,6 +10,7 @@ BACKPRESSURE__TRIP_SECONDS 制造积压拒绝;恢复 consume.py 后磁滞放
10
10
  """
11
11
 
12
12
  import asyncio
13
+ import logging
13
14
  import os
14
15
 
15
16
  from models import OrderIn
@@ -37,6 +38,9 @@ def backpressure_config() -> BackpressureConfig:
37
38
 
38
39
 
39
40
  async def main() -> None:
41
+ logging.basicConfig(
42
+ level=logging.INFO, format="%(asctime)s %(levelname)s %(name)s %(message)s"
43
+ )
40
44
  producer = Producer(
41
45
  os.environ.get("KAFKA__BOOTSTRAP_SERVERS", "localhost:29092"),
42
46
  topic=os.environ.get("KAFKA__TOPIC", "orders"),
@@ -10,6 +10,7 @@
10
10
  """
11
11
 
12
12
  import asyncio
13
+ import logging
13
14
  import os
14
15
  from pathlib import Path
15
16
 
@@ -37,6 +38,9 @@ async def init_db() -> None:
37
38
 
38
39
 
39
40
  async def main() -> None:
41
+ logging.basicConfig(
42
+ level=logging.INFO, format="%(asctime)s %(levelname)s %(name)s %(message)s"
43
+ )
40
44
  await init_db()
41
45
  consumer = MssqlConsumer(
42
46
  db=db_conn(),
@@ -5,8 +5,11 @@
5
5
  """
6
6
 
7
7
  import asyncio
8
+ import logging
8
9
 
9
- from streamgate import Consumer, logger
10
+ from streamgate import Consumer
11
+
12
+ logger = logging.getLogger(__name__)
10
13
 
11
14
 
12
15
  async def serve_consumer_health(consumer: Consumer, port: int) -> None:
@@ -36,11 +39,11 @@ async def serve_consumer_health(consumer: Consumer, port: int) -> None:
36
39
  )
37
40
  await writer.drain()
38
41
  except Exception as e:
39
- logger.debug("health_endpoint_error", error=str(e))
42
+ logger.debug("health_endpoint_error: %s", e)
40
43
  finally:
41
44
  writer.close()
42
45
 
43
46
  server = await asyncio.start_server(_handle, "0.0.0.0", port)
44
- logger.info("health_endpoint_listening", port=port)
47
+ logger.info("health_endpoint_listening: port=%d", port)
45
48
  async with server:
46
49
  await server.serve_forever()