youeduc-sdk-messaging 0.0.0 → 0.5.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +218 -0
- package/dist/admin.d.ts +61 -0
- package/dist/admin.js +143 -0
- package/dist/admin.js.map +1 -0
- package/dist/config.d.ts +74 -0
- package/dist/config.js +128 -0
- package/dist/config.js.map +1 -0
- package/dist/consumer-client.d.ts +155 -0
- package/dist/consumer-client.js +299 -0
- package/dist/consumer-client.js.map +1 -0
- package/dist/consumer.d.ts +156 -0
- package/dist/consumer.js +1228 -0
- package/dist/consumer.js.map +1 -0
- package/dist/contracts/index.d.ts +16 -0
- package/dist/contracts/index.js +42 -0
- package/dist/contracts/index.js.map +1 -0
- package/dist/contracts/schemas.d.ts +570 -0
- package/dist/contracts/schemas.js +902 -0
- package/dist/contracts/schemas.js.map +1 -0
- package/dist/contracts/usuarios.d.ts +163 -0
- package/dist/contracts/usuarios.js +5 -0
- package/dist/contracts/usuarios.js.map +1 -0
- package/dist/dedup.d.ts +29 -0
- package/dist/dedup.js +65 -0
- package/dist/dedup.js.map +1 -0
- package/dist/dlq.d.ts +161 -0
- package/dist/dlq.js +341 -0
- package/dist/dlq.js.map +1 -0
- package/dist/errors.d.ts +72 -0
- package/dist/errors.js +95 -0
- package/dist/errors.js.map +1 -0
- package/dist/headers.d.ts +92 -0
- package/dist/headers.js +144 -0
- package/dist/headers.js.map +1 -0
- package/dist/index.d.ts +30 -0
- package/dist/index.js +32 -0
- package/dist/index.js.map +1 -0
- package/dist/integrations/redis.d.ts +118 -0
- package/dist/integrations/redis.js +272 -0
- package/dist/integrations/redis.js.map +1 -0
- package/dist/integrations/runtime.d.ts +232 -0
- package/dist/integrations/runtime.js +554 -0
- package/dist/integrations/runtime.js.map +1 -0
- package/dist/logger.d.ts +37 -0
- package/dist/logger.js +49 -0
- package/dist/logger.js.map +1 -0
- package/dist/messaging.d.ts +145 -0
- package/dist/messaging.js +164 -0
- package/dist/messaging.js.map +1 -0
- package/dist/naming.d.ts +64 -0
- package/dist/naming.js +167 -0
- package/dist/naming.js.map +1 -0
- package/dist/payload.d.ts +8 -0
- package/dist/payload.js +32 -0
- package/dist/payload.js.map +1 -0
- package/dist/publisher.d.ts +50 -0
- package/dist/publisher.js +199 -0
- package/dist/publisher.js.map +1 -0
- package/dist/retry.d.ts +24 -0
- package/dist/retry.js +42 -0
- package/dist/retry.js.map +1 -0
- package/dist/schema.d.ts +83 -0
- package/dist/schema.js +676 -0
- package/dist/schema.js.map +1 -0
- package/dist/telemetry.d.ts +52 -0
- package/dist/telemetry.js +221 -0
- package/dist/telemetry.js.map +1 -0
- package/dist/transport.d.ts +130 -0
- package/dist/transport.js +283 -0
- package/dist/transport.js.map +1 -0
- package/dist/types.d.ts +148 -0
- package/dist/types.js +35 -0
- package/dist/types.js.map +1 -0
- package/dist/util.d.ts +38 -0
- package/dist/util.js +112 -0
- package/dist/util.js.map +1 -0
- package/dist/w3c.d.ts +16 -0
- package/dist/w3c.js +119 -0
- package/dist/w3c.js.map +1 -0
- package/package.json +74 -1
package/dist/consumer.js
ADDED
|
@@ -0,0 +1,1228 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Kafka consumer with retry through the group's retry topics, DLQ, deduplication and manual
|
|
3
|
+
* commits (spec §5). Mirrors Python's `consumer.py` step by step.
|
|
4
|
+
*
|
|
5
|
+
* Topology (per consumer group):
|
|
6
|
+
*
|
|
7
|
+
* {domain}.events.{version} main topic (shared)
|
|
8
|
+
* {main}.{group}.retry.{30s,5m,1h,12h} group retry topics (read back by the group)
|
|
9
|
+
* {main}.{group}.dlq group DLQ
|
|
10
|
+
*
|
|
11
|
+
* With `resourcePrefix` (spec §2.7) the main topic becomes `{prefix}.{domain}.events.{version}` and
|
|
12
|
+
* the group id `{prefix}.{group}`; retry and DLQ derive from the prefixed main topic and the bare
|
|
13
|
+
* group. Telemetry, logs and the getters (`group`, `mainTopic`...) show the full names.
|
|
14
|
+
*
|
|
15
|
+
* The group subscribes by regex (`naming.retryTopicPattern`) to the main topic and to **every**
|
|
16
|
+
* retry topic of the group, including delays that left the `RetryPolicy`. A retry is processed
|
|
17
|
+
* straight from its retry topic when due (`x-retry-due-at`); it is never republished to the main
|
|
18
|
+
* topic. The next attempt always follows the CURRENT policy, by `x-retry-count`.
|
|
19
|
+
*
|
|
20
|
+
* Kafka client: the callback (librdkafka) API of `@confluentinc/kafka-javascript`, wrapped by
|
|
21
|
+
* `ConfluentConsumerClient` (`consumer-client.ts`, which documents why `KafkaJS.Consumer.run()`
|
|
22
|
+
* cannot implement §5 and how rebalances are serialized with `consume()`). Mapping to the spec:
|
|
23
|
+
*
|
|
24
|
+
* - §5.1 config: `enable.auto.commit=false`, `enable.auto.offset.store=false`,
|
|
25
|
+
* `auto.offset.reset=earliest` (always; the user's `latest` is applied on assignment, §5.7),
|
|
26
|
+
* `max.poll.interval.ms` (also librdkafka's rebalance timeout), eager `roundrobin` assignor.
|
|
27
|
+
* - Poll loop = Python's `getmany` loop: `client.poll({maxRecords, timeoutMs})` → batch grouped
|
|
28
|
+
* by partition → processed under the batch lock: partitions concurrently
|
|
29
|
+
* (`Promise.allSettled`), records of a partition sequentially by offset (§5.4).
|
|
30
|
+
* - Each record ends in ADVANCE (final outcome, or retry/DLQ send acknowledged) or HOLD. HOLD =
|
|
31
|
+
* `pause` + `seek(offset)` + resume timer (§5.4); the rest of the partition's batch is skipped.
|
|
32
|
+
* A seek that fails while the partition is still ours is retried before resuming, so a
|
|
33
|
+
* partition never resumes from an unknown position.
|
|
34
|
+
* - End of batch: commit `last contiguous ADVANCE + 1` per partition, only for partitions still
|
|
35
|
+
* assigned (`offset_commit_skipped_unassigned`); failure → `offset_commit_failed` (§5.4).
|
|
36
|
+
* - Not-due retry (§5.5): HOLD for `min(due − now, topic delay)` (10 ms floor), nothing recorded.
|
|
37
|
+
* - Routing failure (§5.6): retriable → pending route cache (target + headers) + HOLD for
|
|
38
|
+
* `routingFailureBackoffMs`, resent without re-running the handler; non-retriable retry send →
|
|
39
|
+
* DLQ `routing_failed` with the handler's error; non-retriable DLQ send → one shrunk attempt
|
|
40
|
+
* (`x-last-error` ≤ 256 code points, no trace context, nothing injected) → then fatal.
|
|
41
|
+
* - Dedup (§8): checked before the payload is decoded, marked only after handler success.
|
|
42
|
+
* - Processing timeout (§7): the handler's `AbortSignal` is aborted at `processingTimeoutMs`,
|
|
43
|
+
* then it gets the same time again; still running → `message_handler_abandoned`, detached.
|
|
44
|
+
* - Lifecycle (§5.7): `stop()` drains at the record boundary, commits and closes; after
|
|
45
|
+
* `drainTimeoutMs` the loop is cancelled (every await of the pipeline is raced against the
|
|
46
|
+
* cancellation; handlers are aborted) and the unfinished offset is not committed. Revoke asks
|
|
47
|
+
* the partitions to stop at the next record boundary and waits for the batch lock (the batch
|
|
48
|
+
* commits what it finished), then drops timers/pending routes of the revoked partitions.
|
|
49
|
+
* `latest`: main-topic partitions without a committed offset move to the end when assigned (the
|
|
50
|
+
* confluent client turns it into the assign start offset); records fetched before that reset
|
|
51
|
+
* are dropped. The reset serial is read right after each poll returns (as in Python, after
|
|
52
|
+
* `getmany`), so a record fetched after the reset is never dropped (§5.7).
|
|
53
|
+
* Fatal poll errors (codes 29, 30, 31, 35, 58, authentication, fatal/stopped client, or any
|
|
54
|
+
* non-Kafka error) stop the consumer and `wait()` rethrows; others log `consumer_poll_failed`
|
|
55
|
+
* and wait `routingFailureBackoffMs`.
|
|
56
|
+
*
|
|
57
|
+
* A handler may run more than once for the same message (crash before commit, failed commit,
|
|
58
|
+
* rebalance): handlers must be idempotent or use a `DuplicateChecker`. A record may take up to
|
|
59
|
+
* `2 × processingTimeoutMs`.
|
|
60
|
+
*/
|
|
61
|
+
import { performance } from 'node:perf_hooks';
|
|
62
|
+
import { createConfluentConsumer, partitionKey, } from './consumer-client.js';
|
|
63
|
+
import { HandlerCancelledError, InvalidHandlerError, InvalidPayloadError, NotStartedError, PermanentProcessingError, PublishError, SchemaNotFoundError, SchemaValidationError, } from './errors.js';
|
|
64
|
+
import * as hdr from './headers.js';
|
|
65
|
+
import { defaultLogger } from './logger.js';
|
|
66
|
+
import * as naming from './naming.js';
|
|
67
|
+
import { decodePayload } from './payload.js';
|
|
68
|
+
import { RetryPolicy } from './retry.js';
|
|
69
|
+
import { NoopTelemetry } from './telemetry.js';
|
|
70
|
+
import { isKafkaClientError } from './transport.js';
|
|
71
|
+
import { createMessageInfo, } from './types.js';
|
|
72
|
+
import { errorFields, errorMessage, errorTypeName, formatSeconds, isPositiveFinite, pyRepr, truncateChars, } from './util.js';
|
|
73
|
+
const ADVANCE = 'advance';
|
|
74
|
+
const HOLD = 'hold';
|
|
75
|
+
const FATAL = 'fatal';
|
|
76
|
+
const DEFAULT_RETRY_POLICY = new RetryPolicy();
|
|
77
|
+
const AUTO_OFFSET_RESET = ['earliest', 'latest'];
|
|
78
|
+
/**
|
|
79
|
+
* Group `session.timeout.ms`, the same as the Python SDK (aiokafka's default). `maxPollIntervalMs`
|
|
80
|
+
* must be >= it (§5.1; librdkafka refuses the opposite).
|
|
81
|
+
*/
|
|
82
|
+
const SESSION_TIMEOUT_MS = 10_000;
|
|
83
|
+
/** Floor of the resume timer: avoids a hot loop when a retry is due in a few ms. */
|
|
84
|
+
const MIN_RESUME_DELAY_MS = 10;
|
|
85
|
+
/** Second DLQ attempt after a non-retriable error: reduced headers. */
|
|
86
|
+
const SHRUNK_ERROR_LENGTH = 256;
|
|
87
|
+
const TRACE_HEADERS = ['traceparent', 'tracestate'];
|
|
88
|
+
/**
|
|
89
|
+
* Poll errors that repeating does not fix (spec §5.7): topic/group/cluster authorization (29, 30,
|
|
90
|
+
* 31), unsupported version (35), SASL authentication (58), local authentication (-169), client
|
|
91
|
+
* stopped/not connected (-172, librdkafka `ERR__STATE`) and librdkafka fatal errors (-150).
|
|
92
|
+
*/
|
|
93
|
+
const FATAL_POLL_CODES = new Set([29, 30, 31, 35, 58, -169, -172, -150]);
|
|
94
|
+
/** Whether a poll error stops the consumer: any non-Kafka error, a fatal code or `isFatal`. */
|
|
95
|
+
export function isFatalPollError(error) {
|
|
96
|
+
if (!isKafkaClientError(error))
|
|
97
|
+
return true;
|
|
98
|
+
if (error.isFatal === true)
|
|
99
|
+
return true;
|
|
100
|
+
return FATAL_POLL_CODES.has(error.code);
|
|
101
|
+
}
|
|
102
|
+
const keyDecoder = new TextDecoder('utf-8', { fatal: false, ignoreBOM: true });
|
|
103
|
+
function isAsyncFunction(value) {
|
|
104
|
+
return typeof value === 'function' && Object.prototype.toString.call(value) === '[object AsyncFunction]';
|
|
105
|
+
}
|
|
106
|
+
function isPromiseLike(value) {
|
|
107
|
+
return ((typeof value === 'object' || typeof value === 'function') &&
|
|
108
|
+
value !== null &&
|
|
109
|
+
typeof value.then === 'function');
|
|
110
|
+
}
|
|
111
|
+
function isAbortError(error) {
|
|
112
|
+
return typeof error === 'object' && error !== null && error.name === 'AbortError';
|
|
113
|
+
}
|
|
114
|
+
function round3(value) {
|
|
115
|
+
return Math.round(value * 1000) / 1000;
|
|
116
|
+
}
|
|
117
|
+
function elapsedMs(started) {
|
|
118
|
+
return Math.max(performance.now() - started, 0);
|
|
119
|
+
}
|
|
120
|
+
function errorInfo(error) {
|
|
121
|
+
return errorFields(error, hdr.MAX_ERROR_LENGTH);
|
|
122
|
+
}
|
|
123
|
+
/** Cancellable sleep (no timer left behind). */
|
|
124
|
+
function sleep(ms) {
|
|
125
|
+
let timer;
|
|
126
|
+
const promise = new Promise((resolve) => {
|
|
127
|
+
timer = setTimeout(resolve, ms);
|
|
128
|
+
});
|
|
129
|
+
return { promise, cancel: () => clearTimeout(timer) };
|
|
130
|
+
}
|
|
131
|
+
const TIMED_OUT = Symbol('timed out');
|
|
132
|
+
async function withTimeout(promise, ms) {
|
|
133
|
+
const timer = sleep(ms);
|
|
134
|
+
try {
|
|
135
|
+
return await Promise.race([promise, timer.promise.then(() => TIMED_OUT)]);
|
|
136
|
+
}
|
|
137
|
+
finally {
|
|
138
|
+
timer.cancel();
|
|
139
|
+
}
|
|
140
|
+
}
|
|
141
|
+
function settle(promise) {
|
|
142
|
+
return promise.then(() => ({ ok: true }), (error) => ({ ok: false, error }));
|
|
143
|
+
}
|
|
144
|
+
/** Raised at every pipeline await once `stop()`'s drain timeout cancelled the loop. */
|
|
145
|
+
class LoopCancelledError extends Error {
|
|
146
|
+
constructor() {
|
|
147
|
+
super('consumer loop cancelled');
|
|
148
|
+
this.name = 'CancelledError';
|
|
149
|
+
}
|
|
150
|
+
}
|
|
151
|
+
/** FIFO async mutex (asyncio.Lock semantics). */
|
|
152
|
+
class Mutex {
|
|
153
|
+
#tail = Promise.resolve();
|
|
154
|
+
run(fn) {
|
|
155
|
+
const result = this.#tail.then(fn);
|
|
156
|
+
this.#tail = result.then(() => undefined, () => undefined);
|
|
157
|
+
return result;
|
|
158
|
+
}
|
|
159
|
+
}
|
|
160
|
+
const INTERNALS = new WeakMap();
|
|
161
|
+
/** @internal */
|
|
162
|
+
export function consumerInternals(consumer) {
|
|
163
|
+
const internals = INTERNALS.get(consumer);
|
|
164
|
+
if (internals === undefined)
|
|
165
|
+
throw new TypeError('not a Consumer');
|
|
166
|
+
return internals;
|
|
167
|
+
}
|
|
168
|
+
/**
|
|
169
|
+
* Consumes the domain's main topic and the group's retry topics.
|
|
170
|
+
*
|
|
171
|
+
* Handler contract: resolving = success; rejecting with `PermanentProcessingError` = DLQ; any
|
|
172
|
+
* other rejection (including the processing timeout) = retry, until `retryPolicy` is exhausted.
|
|
173
|
+
*/
|
|
174
|
+
export class Consumer {
|
|
175
|
+
/** Effective group id (prefixed): Kafka, telemetry and logs. */
|
|
176
|
+
#group;
|
|
177
|
+
/** Logical group (no prefix): builds the retry/DLQ topics and the subscription. */
|
|
178
|
+
#logicalGroup;
|
|
179
|
+
#resourcePrefix;
|
|
180
|
+
#mainTopic;
|
|
181
|
+
#retryTopics;
|
|
182
|
+
#dlqTopic;
|
|
183
|
+
#pattern;
|
|
184
|
+
#config;
|
|
185
|
+
#domain;
|
|
186
|
+
#handlers;
|
|
187
|
+
#validator;
|
|
188
|
+
#sender;
|
|
189
|
+
#retryPolicy;
|
|
190
|
+
#duplicateChecker;
|
|
191
|
+
#telemetry;
|
|
192
|
+
#processingTimeoutMs;
|
|
193
|
+
#maxPollRecords;
|
|
194
|
+
#pollTimeoutMs;
|
|
195
|
+
#maxPollIntervalMs;
|
|
196
|
+
#routingFailureBackoffMs;
|
|
197
|
+
#autoOffsetReset;
|
|
198
|
+
#consumerFactory;
|
|
199
|
+
#clock;
|
|
200
|
+
#logger;
|
|
201
|
+
#client = null;
|
|
202
|
+
#task = null;
|
|
203
|
+
#taskDone = false;
|
|
204
|
+
#closing = null;
|
|
205
|
+
#started = false;
|
|
206
|
+
#stopping = false;
|
|
207
|
+
#revoking = 0;
|
|
208
|
+
#stoppedLogged = false;
|
|
209
|
+
#fatal = null;
|
|
210
|
+
#signalStop = () => undefined;
|
|
211
|
+
#stopped;
|
|
212
|
+
#cancelled = false;
|
|
213
|
+
#rejectCancel = () => undefined;
|
|
214
|
+
#cancellation;
|
|
215
|
+
#batchLock = new Mutex();
|
|
216
|
+
#resumeTimers = new Map();
|
|
217
|
+
/** Routing decision whose send failed (retriable), per partition. */
|
|
218
|
+
#pendingRoutes = new Map();
|
|
219
|
+
/** HOLD seeks that failed while the partition was still ours: re-seek before resuming. */
|
|
220
|
+
#pendingSeeks = new Map();
|
|
221
|
+
/** `latest`: partitions moved to the end, and the number of that reset. */
|
|
222
|
+
#resetSerial = 0;
|
|
223
|
+
#resetAt = new Map();
|
|
224
|
+
/**
|
|
225
|
+
* Partitions handed back by `onPartitionsRevoked` and not assigned again since. The listener
|
|
226
|
+
* returns before the client actually unassigns them, so `client.assignment()` still lists them
|
|
227
|
+
* for a moment: a batch fetched before the revoke (consume() in flight when the rebalance was
|
|
228
|
+
* served) and processed in that window is dropped here instead of being handled against a
|
|
229
|
+
* partition that already belongs to the new owner.
|
|
230
|
+
*/
|
|
231
|
+
#revoked = new Set();
|
|
232
|
+
/** Handlers in flight (aborted on cancellation) and handlers that ignored the abort. */
|
|
233
|
+
#active = new Set();
|
|
234
|
+
#detached = new Set();
|
|
235
|
+
constructor(config, options) {
|
|
236
|
+
const retryPolicy = options.retryPolicy ?? DEFAULT_RETRY_POLICY;
|
|
237
|
+
this.#logicalGroup = naming.consumerGroup(options.group);
|
|
238
|
+
this.#resourcePrefix = naming.resourcePrefix(options.resourcePrefix);
|
|
239
|
+
this.#group = naming.consumerGroup(options.group, { resourcePrefix: this.#resourcePrefix });
|
|
240
|
+
this.#mainTopic = naming.topic(options.domain, options.version, { resourcePrefix: this.#resourcePrefix });
|
|
241
|
+
this.#retryTopics = Object.freeze(retryPolicy.retryTopics(this.#mainTopic, options.group));
|
|
242
|
+
this.#dlqTopic = naming.dlqTopic(this.#mainTopic, options.group);
|
|
243
|
+
this.#pattern = naming.retryTopicPattern(this.#mainTopic, options.group);
|
|
244
|
+
const entries = options.handlers instanceof Map
|
|
245
|
+
? [...options.handlers.entries()]
|
|
246
|
+
: Object.entries((options.handlers ?? {}));
|
|
247
|
+
if (entries.length === 0)
|
|
248
|
+
throw new TypeError("'handlers' must not be empty");
|
|
249
|
+
for (const [eventType, handler] of entries) {
|
|
250
|
+
if (typeof eventType !== 'string') {
|
|
251
|
+
throw new TypeError(`handler key must be a string; got ${String(eventType)}`);
|
|
252
|
+
}
|
|
253
|
+
naming.splitEventType(eventType);
|
|
254
|
+
if (!isAsyncFunction(handler)) {
|
|
255
|
+
throw new TypeError(`handler for ${pyRepr(eventType)} must be an async function`);
|
|
256
|
+
}
|
|
257
|
+
}
|
|
258
|
+
const maxPollRecords = options.maxPollRecords ?? 100;
|
|
259
|
+
const pollTimeoutMs = options.pollTimeoutMs ?? 1000;
|
|
260
|
+
const maxPollIntervalMs = options.maxPollIntervalMs ?? 300_000;
|
|
261
|
+
for (const [name, value] of [
|
|
262
|
+
['maxPollRecords', maxPollRecords],
|
|
263
|
+
['pollTimeoutMs', pollTimeoutMs],
|
|
264
|
+
['maxPollIntervalMs', maxPollIntervalMs],
|
|
265
|
+
]) {
|
|
266
|
+
if (typeof value !== 'number' || !Number.isInteger(value) || value < 1) {
|
|
267
|
+
throw new RangeError(`'${name}' must be an integer >= 1; got ${String(value)}`);
|
|
268
|
+
}
|
|
269
|
+
}
|
|
270
|
+
if (maxPollIntervalMs < SESSION_TIMEOUT_MS) {
|
|
271
|
+
throw new RangeError(`'maxPollIntervalMs' must be >= the session timeout (${SESSION_TIMEOUT_MS} ms); got ${maxPollIntervalMs}`);
|
|
272
|
+
}
|
|
273
|
+
const processingTimeoutMs = options.processingTimeoutMs ?? null;
|
|
274
|
+
if (processingTimeoutMs !== null) {
|
|
275
|
+
if (!isPositiveFinite(processingTimeoutMs)) {
|
|
276
|
+
throw new RangeError(`'processingTimeoutMs' must be a finite number > 0; got ${String(processingTimeoutMs)}`);
|
|
277
|
+
}
|
|
278
|
+
if (2 * processingTimeoutMs >= maxPollIntervalMs) {
|
|
279
|
+
throw new RangeError("'processingTimeoutMs' must be less than half of 'maxPollIntervalMs' (a record may take up" +
|
|
280
|
+
` to 2x the timeout); got ${processingTimeoutMs}ms vs ${maxPollIntervalMs}ms`);
|
|
281
|
+
}
|
|
282
|
+
}
|
|
283
|
+
const routingFailureBackoffMs = options.routingFailureBackoffMs ?? 5000;
|
|
284
|
+
if (!isPositiveFinite(routingFailureBackoffMs)) {
|
|
285
|
+
throw new RangeError(`'routingFailureBackoffMs' must be a finite number > 0; got ${String(routingFailureBackoffMs)}`);
|
|
286
|
+
}
|
|
287
|
+
const autoOffsetReset = options.autoOffsetReset ?? 'earliest';
|
|
288
|
+
if (!AUTO_OFFSET_RESET.includes(autoOffsetReset)) {
|
|
289
|
+
throw new TypeError(`'autoOffsetReset' must be one of ('earliest', 'latest'); got ${pyRepr(String(autoOffsetReset))}`);
|
|
290
|
+
}
|
|
291
|
+
const clock = options.clock ?? Date.now;
|
|
292
|
+
if (typeof clock !== 'function')
|
|
293
|
+
throw new TypeError("'clock' must be a function");
|
|
294
|
+
this.#config = config;
|
|
295
|
+
this.#domain = options.domain;
|
|
296
|
+
this.#handlers = new Map(entries);
|
|
297
|
+
this.#validator = options.validator;
|
|
298
|
+
this.#sender = options.sender;
|
|
299
|
+
this.#retryPolicy = retryPolicy;
|
|
300
|
+
this.#duplicateChecker = options.duplicateChecker ?? null;
|
|
301
|
+
this.#telemetry = options.telemetry ?? new NoopTelemetry();
|
|
302
|
+
this.#processingTimeoutMs = processingTimeoutMs;
|
|
303
|
+
this.#maxPollRecords = maxPollRecords;
|
|
304
|
+
this.#pollTimeoutMs = pollTimeoutMs;
|
|
305
|
+
this.#maxPollIntervalMs = maxPollIntervalMs;
|
|
306
|
+
this.#routingFailureBackoffMs = routingFailureBackoffMs;
|
|
307
|
+
this.#autoOffsetReset = autoOffsetReset;
|
|
308
|
+
this.#consumerFactory = options.consumerFactory ?? createConfluentConsumer;
|
|
309
|
+
this.#clock = clock;
|
|
310
|
+
this.#logger = options.logger;
|
|
311
|
+
this.#stopped = new Promise((resolve) => {
|
|
312
|
+
this.#signalStop = resolve;
|
|
313
|
+
});
|
|
314
|
+
this.#cancellation = new Promise((_resolve, reject) => {
|
|
315
|
+
this.#rejectCancel = reject;
|
|
316
|
+
});
|
|
317
|
+
this.#cancellation.catch(() => undefined);
|
|
318
|
+
INTERNALS.set(this, {
|
|
319
|
+
resumeTimers: () => new Map(this.#resumeTimers),
|
|
320
|
+
pendingRoutes: () => new Map(this.#pendingRoutes),
|
|
321
|
+
resumePartition: (tp) => this.#resumePartition(tp),
|
|
322
|
+
setHandler: (eventType, handler) => {
|
|
323
|
+
this.#handlers.set(eventType, handler);
|
|
324
|
+
},
|
|
325
|
+
});
|
|
326
|
+
}
|
|
327
|
+
// ---------------------------------------------------------------------------- properties
|
|
328
|
+
get mainTopic() {
|
|
329
|
+
return this.#mainTopic;
|
|
330
|
+
}
|
|
331
|
+
/** Retry topics of the current policy (targets of new retries). */
|
|
332
|
+
get retryTopics() {
|
|
333
|
+
return [...this.#retryTopics];
|
|
334
|
+
}
|
|
335
|
+
/** Subscribed regex (source): main topic + every retry topic of the group. */
|
|
336
|
+
get subscriptionPattern() {
|
|
337
|
+
return this.#pattern;
|
|
338
|
+
}
|
|
339
|
+
get dlqTopic() {
|
|
340
|
+
return this.#dlqTopic;
|
|
341
|
+
}
|
|
342
|
+
/** Effective Kafka group id (with the resource prefix, if any). */
|
|
343
|
+
get group() {
|
|
344
|
+
return this.#group;
|
|
345
|
+
}
|
|
346
|
+
get resourcePrefix() {
|
|
347
|
+
return this.#resourcePrefix;
|
|
348
|
+
}
|
|
349
|
+
get running() {
|
|
350
|
+
return this.#task !== null && !this.#taskDone && !this.#stopping;
|
|
351
|
+
}
|
|
352
|
+
get #log() {
|
|
353
|
+
return this.#logger ?? defaultLogger();
|
|
354
|
+
}
|
|
355
|
+
/** Final librdkafka configuration handed to the client factory. */
|
|
356
|
+
consumerConfig() {
|
|
357
|
+
return {
|
|
358
|
+
...this.#config.clientConfig(),
|
|
359
|
+
'group.id': this.#group,
|
|
360
|
+
'enable.auto.commit': false,
|
|
361
|
+
'enable.auto.offset.store': false,
|
|
362
|
+
// Always "earliest": an uncommitted retry is never skipped. The user's "latest" is applied
|
|
363
|
+
// to the main topic only, on assignment.
|
|
364
|
+
'auto.offset.reset': 'earliest',
|
|
365
|
+
'max.poll.interval.ms': this.#maxPollIntervalMs,
|
|
366
|
+
'session.timeout.ms': SESSION_TIMEOUT_MS,
|
|
367
|
+
// Eager protocol: a revoke always covers the whole assignment (see consumer-client.ts).
|
|
368
|
+
'partition.assignment.strategy': 'roundrobin',
|
|
369
|
+
};
|
|
370
|
+
}
|
|
371
|
+
// ---------------------------------------------------------------------------- lifecycle
|
|
372
|
+
async start() {
|
|
373
|
+
if (this.#started)
|
|
374
|
+
throw new Error('Consumer has already been started');
|
|
375
|
+
this.#started = true;
|
|
376
|
+
let client = null;
|
|
377
|
+
try {
|
|
378
|
+
client = await this.#consumerFactory(this.consumerConfig(), {
|
|
379
|
+
onCommitFailed: (error) => this.#log.warn('offset_commit_failed', {
|
|
380
|
+
messaging: { group: this.#group },
|
|
381
|
+
error: errorInfo(error),
|
|
382
|
+
}),
|
|
383
|
+
});
|
|
384
|
+
await client.connect();
|
|
385
|
+
client.subscribe(this.#pattern, this.#listener(client));
|
|
386
|
+
}
|
|
387
|
+
catch (error) {
|
|
388
|
+
if (client !== null) {
|
|
389
|
+
try {
|
|
390
|
+
await client.disconnect(); // release what a failed start may have allocated
|
|
391
|
+
}
|
|
392
|
+
catch {
|
|
393
|
+
// the original error matters
|
|
394
|
+
}
|
|
395
|
+
}
|
|
396
|
+
this.#started = false;
|
|
397
|
+
throw error;
|
|
398
|
+
}
|
|
399
|
+
this.#client = client;
|
|
400
|
+
if (this.#stopping) {
|
|
401
|
+
// stop() called during start(): do not start the loop.
|
|
402
|
+
await this.#closeClient();
|
|
403
|
+
return;
|
|
404
|
+
}
|
|
405
|
+
const task = this.#run(client);
|
|
406
|
+
this.#task = task;
|
|
407
|
+
void task.finally(() => {
|
|
408
|
+
this.#taskDone = true;
|
|
409
|
+
});
|
|
410
|
+
this.#log.info('consumer_started', {
|
|
411
|
+
messaging: {
|
|
412
|
+
group: this.#group,
|
|
413
|
+
subscription_pattern: this.#pattern,
|
|
414
|
+
retry_topics: [...this.#retryTopics],
|
|
415
|
+
dlq_topic: this.#dlqTopic,
|
|
416
|
+
},
|
|
417
|
+
});
|
|
418
|
+
}
|
|
419
|
+
/**
|
|
420
|
+
* Graceful shutdown: the in-flight record of each partition finishes and the progress is
|
|
421
|
+
* committed. After `drainTimeoutMs` the loop is cancelled (the in-flight handler is aborted)
|
|
422
|
+
* and the unfinished offset is not committed: it will be redelivered. Idempotent.
|
|
423
|
+
*/
|
|
424
|
+
async stop(options = {}) {
|
|
425
|
+
const drainTimeoutMs = options.drainTimeoutMs ?? 30_000;
|
|
426
|
+
if (!this.#started)
|
|
427
|
+
return;
|
|
428
|
+
this.#stopping = true;
|
|
429
|
+
this.#signalStop();
|
|
430
|
+
this.#cancelTimers();
|
|
431
|
+
const task = this.#task;
|
|
432
|
+
if (task !== null && !this.#taskDone) {
|
|
433
|
+
const finished = await withTimeout(task, Math.max(drainTimeoutMs, 0));
|
|
434
|
+
if (finished === TIMED_OUT) {
|
|
435
|
+
this.#log.warn('consumer_drain_timeout', {
|
|
436
|
+
messaging: { group: this.#group, drain_timeout_s: drainTimeoutMs / 1000 },
|
|
437
|
+
});
|
|
438
|
+
this.#cancel();
|
|
439
|
+
await task;
|
|
440
|
+
}
|
|
441
|
+
}
|
|
442
|
+
this.#cancelTimers();
|
|
443
|
+
this.#pendingRoutes.clear();
|
|
444
|
+
this.#pendingSeeks.clear();
|
|
445
|
+
await this.#closeClient();
|
|
446
|
+
if (!this.#stoppedLogged) {
|
|
447
|
+
this.#stoppedLogged = true;
|
|
448
|
+
this.#log.info('consumer_stopped', { messaging: { group: this.#group } });
|
|
449
|
+
}
|
|
450
|
+
}
|
|
451
|
+
/** Waits for the loop to end; rethrows the fatal error that stopped it, if any. */
|
|
452
|
+
async wait() {
|
|
453
|
+
const task = this.#task;
|
|
454
|
+
if (task === null)
|
|
455
|
+
throw new NotStartedError('Consumer is not running');
|
|
456
|
+
await task;
|
|
457
|
+
if (this.#fatal !== null)
|
|
458
|
+
throw this.#fatal.error;
|
|
459
|
+
}
|
|
460
|
+
/** Closes the Kafka client exactly once (leaves the group). */
|
|
461
|
+
async #closeClient() {
|
|
462
|
+
const client = this.#client;
|
|
463
|
+
if (client === null)
|
|
464
|
+
return;
|
|
465
|
+
this.#closing ??= (async () => {
|
|
466
|
+
try {
|
|
467
|
+
await client.disconnect();
|
|
468
|
+
}
|
|
469
|
+
catch (error) {
|
|
470
|
+
this.#log.warn('consumer_close_failed', {
|
|
471
|
+
messaging: { group: this.#group },
|
|
472
|
+
error: errorInfo(error),
|
|
473
|
+
});
|
|
474
|
+
}
|
|
475
|
+
})();
|
|
476
|
+
await this.#closing;
|
|
477
|
+
}
|
|
478
|
+
/** Fatal error: ends the loop (`wait()` rethrows). The record is not committed. */
|
|
479
|
+
#fail(error) {
|
|
480
|
+
if (this.#fatal === null) {
|
|
481
|
+
this.#log.error('consumer_crashed', { messaging: { group: this.#group }, error: errorInfo(error) });
|
|
482
|
+
this.#fatal = { error };
|
|
483
|
+
}
|
|
484
|
+
this.#stopping = true;
|
|
485
|
+
this.#signalStop();
|
|
486
|
+
}
|
|
487
|
+
#cancel() {
|
|
488
|
+
if (this.#cancelled)
|
|
489
|
+
return;
|
|
490
|
+
this.#cancelled = true;
|
|
491
|
+
const error = new LoopCancelledError();
|
|
492
|
+
this.#rejectCancel(error);
|
|
493
|
+
for (const controller of this.#active)
|
|
494
|
+
controller.abort(error);
|
|
495
|
+
}
|
|
496
|
+
/** Races `promise` against the loop cancellation (the Python task cancellation points). */
|
|
497
|
+
#abortable(promise) {
|
|
498
|
+
promise.catch(() => undefined); // the loser of the race must not be an unhandled rejection
|
|
499
|
+
if (this.#cancelled)
|
|
500
|
+
return Promise.reject(new LoopCancelledError());
|
|
501
|
+
return Promise.race([promise, this.#cancellation]);
|
|
502
|
+
}
|
|
503
|
+
// ---------------------------------------------------------------------------- poll loop
|
|
504
|
+
async #run(client) {
|
|
505
|
+
try {
|
|
506
|
+
while (!this.#stopping) {
|
|
507
|
+
let records;
|
|
508
|
+
try {
|
|
509
|
+
records = await this.#abortable(client.poll({ maxRecords: this.#maxPollRecords, timeoutMs: this.#pollTimeoutMs }));
|
|
510
|
+
}
|
|
511
|
+
catch (error) {
|
|
512
|
+
if (error instanceof LoopCancelledError)
|
|
513
|
+
throw error;
|
|
514
|
+
if (isFatalPollError(error)) {
|
|
515
|
+
this.#fail(error);
|
|
516
|
+
break;
|
|
517
|
+
}
|
|
518
|
+
this.#log.error('consumer_poll_failed', {
|
|
519
|
+
messaging: { group: this.#group },
|
|
520
|
+
error: errorInfo(error),
|
|
521
|
+
});
|
|
522
|
+
await this.#sleepUnlessStopping(this.#routingFailureBackoffMs);
|
|
523
|
+
continue;
|
|
524
|
+
}
|
|
525
|
+
// Serial read AFTER the poll: only a reset completed after the poll returned proves the
|
|
526
|
+
// records were fetched before it (they are below the new end position: dropping them IS
|
|
527
|
+
// "latest"). A reset completed during the poll (always the case with the confluent client,
|
|
528
|
+
// which applies rebalances before consume()) may have post-reset records: kept, never lost.
|
|
529
|
+
const serial = this.#resetSerial;
|
|
530
|
+
if (records.length === 0)
|
|
531
|
+
continue;
|
|
532
|
+
if (this.#stopping) {
|
|
533
|
+
// Not committed: redelivered to whoever takes the partitions.
|
|
534
|
+
this.#log.debug('consumer_batch_discarded_on_stop', { messaging: { group: this.#group } });
|
|
535
|
+
break;
|
|
536
|
+
}
|
|
537
|
+
await this.#batchLock.run(() => this.#processBatch(client, records, serial));
|
|
538
|
+
}
|
|
539
|
+
}
|
|
540
|
+
catch (error) {
|
|
541
|
+
if (!(error instanceof LoopCancelledError))
|
|
542
|
+
this.#fail(error);
|
|
543
|
+
}
|
|
544
|
+
finally {
|
|
545
|
+
this.#stopping = true;
|
|
546
|
+
this.#signalStop();
|
|
547
|
+
this.#cancelTimers();
|
|
548
|
+
this.#pendingRoutes.clear();
|
|
549
|
+
this.#pendingSeeks.clear();
|
|
550
|
+
await this.#closeClient();
|
|
551
|
+
}
|
|
552
|
+
}
|
|
553
|
+
async #sleepUnlessStopping(ms) {
|
|
554
|
+
const timer = sleep(ms);
|
|
555
|
+
try {
|
|
556
|
+
await this.#abortable(Promise.race([timer.promise, this.#stopped]));
|
|
557
|
+
}
|
|
558
|
+
finally {
|
|
559
|
+
timer.cancel();
|
|
560
|
+
}
|
|
561
|
+
}
|
|
562
|
+
#draining() {
|
|
563
|
+
return this.#stopping || this.#revoking > 0;
|
|
564
|
+
}
|
|
565
|
+
async #processBatch(client, records, serial) {
|
|
566
|
+
const assigned = new Set(client.assignment().map(partitionKey));
|
|
567
|
+
const batch = new Map();
|
|
568
|
+
for (const record of records) {
|
|
569
|
+
const key = partitionKey(record);
|
|
570
|
+
let entry = batch.get(key);
|
|
571
|
+
if (entry === undefined) {
|
|
572
|
+
entry = { tp: { topic: record.topic, partition: record.partition }, records: [] };
|
|
573
|
+
batch.set(key, entry);
|
|
574
|
+
}
|
|
575
|
+
entry.records.push(record);
|
|
576
|
+
}
|
|
577
|
+
// Records fetched before a "latest" reset of the partition are dropped (its position already
|
|
578
|
+
// moved to the end); records of partitions no longer ours (revoked meanwhile, even if the
|
|
579
|
+
// client has not unassigned them yet) too: the new owner reads them from the committed offset.
|
|
580
|
+
// Eager protocol: between a revoke and the next assignment no record of the partition can be
|
|
581
|
+
// fetched, so a revoked partition's records in a batch are always from before the revoke.
|
|
582
|
+
const work = [...batch.entries()].filter(([key]) => (this.#resetAt.get(key) ?? 0) <= serial && assigned.has(key) && !this.#revoked.has(key));
|
|
583
|
+
if (work.length === 0)
|
|
584
|
+
return;
|
|
585
|
+
const results = await Promise.allSettled(work.map(([, entry]) => this.#processPartition(client, entry.tp, entry.records)));
|
|
586
|
+
const offsets = new Map();
|
|
587
|
+
for (const result of results) {
|
|
588
|
+
if (result.status === 'rejected')
|
|
589
|
+
throw result.reason;
|
|
590
|
+
const { tp, next } = result.value;
|
|
591
|
+
if (next !== null)
|
|
592
|
+
offsets.set(partitionKey(tp), { tp, offset: next });
|
|
593
|
+
}
|
|
594
|
+
if (offsets.size === 0)
|
|
595
|
+
return;
|
|
596
|
+
// A partition that stopped being ours (rebalance) is left out: committing it would be
|
|
597
|
+
// rejected and take the other partitions' commit down with it.
|
|
598
|
+
const nowAssigned = new Set(client.assignment().map(partitionKey));
|
|
599
|
+
const committable = [...offsets.entries()].filter(([key]) => nowAssigned.has(key));
|
|
600
|
+
if (committable.length !== offsets.size) {
|
|
601
|
+
this.#log.warn('offset_commit_skipped_unassigned', {
|
|
602
|
+
messaging: {
|
|
603
|
+
group: this.#group,
|
|
604
|
+
partitions: [...offsets.keys()].filter((key) => !nowAssigned.has(key)).sort(),
|
|
605
|
+
},
|
|
606
|
+
});
|
|
607
|
+
}
|
|
608
|
+
if (committable.length === 0)
|
|
609
|
+
return;
|
|
610
|
+
try {
|
|
611
|
+
await this.#abortable(client.commit(committable.map(([, { tp, offset }]) => ({ topic: tp.topic, partition: tp.partition, offset }))));
|
|
612
|
+
}
|
|
613
|
+
catch (error) {
|
|
614
|
+
if (error instanceof LoopCancelledError || !isKafkaClientError(error))
|
|
615
|
+
throw error;
|
|
616
|
+
// Without the commit the message is redelivered; at-least-once + dedup cover it.
|
|
617
|
+
this.#log.warn('offset_commit_failed', { messaging: { group: this.#group }, error: errorInfo(error) });
|
|
618
|
+
}
|
|
619
|
+
}
|
|
620
|
+
/** Records in offset order; returns the next committable offset. */
|
|
621
|
+
async #processPartition(client, tp, records) {
|
|
622
|
+
let next = null;
|
|
623
|
+
for (const record of [...records].sort((a, b) => a.offset - b.offset)) {
|
|
624
|
+
// stop/revoke: finish the in-flight record, start no other. The rest is redone from the
|
|
625
|
+
// committed offset.
|
|
626
|
+
if (this.#draining())
|
|
627
|
+
break;
|
|
628
|
+
const step = await this.#processRecord(client, tp, record);
|
|
629
|
+
if (step !== ADVANCE)
|
|
630
|
+
break;
|
|
631
|
+
next = record.offset + 1;
|
|
632
|
+
}
|
|
633
|
+
return { tp, next };
|
|
634
|
+
}
|
|
635
|
+
// ---------------------------------------------------------------------------- rebalance / timers
|
|
636
|
+
#listener(client) {
|
|
637
|
+
return {
|
|
638
|
+
onPartitionsRevoked: (partitions) => this.#onPartitionsRevoked(partitions),
|
|
639
|
+
onPartitionsAssigned: (partitions) => this.#onPartitionsAssigned(client, partitions),
|
|
640
|
+
};
|
|
641
|
+
}
|
|
642
|
+
async #onPartitionsRevoked(revoked) {
|
|
643
|
+
// Ask the partitions to stop at the next record boundary (the rebalance does not wait for
|
|
644
|
+
// the whole batch) and wait for the commit of what was finished.
|
|
645
|
+
this.#revoking += 1;
|
|
646
|
+
try {
|
|
647
|
+
await this.#batchLock.run(async () => {
|
|
648
|
+
for (const tp of revoked) {
|
|
649
|
+
const key = partitionKey(tp);
|
|
650
|
+
this.#cancelTimer(key);
|
|
651
|
+
this.#pendingRoutes.delete(key);
|
|
652
|
+
this.#pendingSeeks.delete(key);
|
|
653
|
+
this.#resetAt.delete(key);
|
|
654
|
+
this.#revoked.add(key);
|
|
655
|
+
}
|
|
656
|
+
});
|
|
657
|
+
}
|
|
658
|
+
finally {
|
|
659
|
+
this.#revoking -= 1;
|
|
660
|
+
}
|
|
661
|
+
this.#log.info('consumer_partitions_revoked', {
|
|
662
|
+
messaging: { group: this.#group, partitions: revoked.length },
|
|
663
|
+
});
|
|
664
|
+
}
|
|
665
|
+
async #onPartitionsAssigned(client, assigned) {
|
|
666
|
+
// Partitions come back resumed at the committed offset; due dates are re-evaluated. Called
|
|
667
|
+
// before the client assigns them: records fetched from now on belong to this assignment.
|
|
668
|
+
for (const tp of assigned)
|
|
669
|
+
this.#revoked.delete(partitionKey(tp));
|
|
670
|
+
this.#log.info('consumer_partitions_assigned', {
|
|
671
|
+
messaging: { group: this.#group, partitions: assigned.length },
|
|
672
|
+
});
|
|
673
|
+
if (this.#autoOffsetReset !== 'latest')
|
|
674
|
+
return;
|
|
675
|
+
const mainTps = assigned.filter((tp) => tp.topic === this.#mainTopic);
|
|
676
|
+
if (mainTps.length === 0)
|
|
677
|
+
return;
|
|
678
|
+
// Lock before any await: a batch already fetched from the start waits here and is dropped
|
|
679
|
+
// afterwards (serial) instead of being processed.
|
|
680
|
+
await this.#batchLock.run(async () => {
|
|
681
|
+
const toReset = [];
|
|
682
|
+
for (const tp of mainTps) {
|
|
683
|
+
let committed;
|
|
684
|
+
try {
|
|
685
|
+
committed = await client.committed(tp);
|
|
686
|
+
}
|
|
687
|
+
catch (error) {
|
|
688
|
+
// When in doubt stay at "earliest": reprocess, never lose.
|
|
689
|
+
this.#log.warn('consumer_committed_offset_failed', {
|
|
690
|
+
messaging: { topic: tp.topic, partition: tp.partition },
|
|
691
|
+
error: errorInfo(error),
|
|
692
|
+
});
|
|
693
|
+
continue;
|
|
694
|
+
}
|
|
695
|
+
if (committed === null)
|
|
696
|
+
toReset.push(tp);
|
|
697
|
+
}
|
|
698
|
+
if (toReset.length === 0)
|
|
699
|
+
return;
|
|
700
|
+
try {
|
|
701
|
+
await client.seekToEnd(toReset);
|
|
702
|
+
}
|
|
703
|
+
catch (error) {
|
|
704
|
+
// No guarantee the reset happened: drop nothing.
|
|
705
|
+
this.#log.warn('consumer_seek_to_end_failed', {
|
|
706
|
+
messaging: { group: this.#group },
|
|
707
|
+
error: errorInfo(error),
|
|
708
|
+
});
|
|
709
|
+
return;
|
|
710
|
+
}
|
|
711
|
+
this.#resetSerial += 1;
|
|
712
|
+
for (const tp of toReset)
|
|
713
|
+
this.#resetAt.set(partitionKey(tp), this.#resetSerial);
|
|
714
|
+
this.#log.info('consumer_partitions_reset_to_latest', {
|
|
715
|
+
messaging: { group: this.#group, partitions: toReset.length },
|
|
716
|
+
});
|
|
717
|
+
});
|
|
718
|
+
}
|
|
719
|
+
#isAssigned(client, tp) {
|
|
720
|
+
const key = partitionKey(tp);
|
|
721
|
+
return client.assignment().some((assigned) => partitionKey(assigned) === key);
|
|
722
|
+
}
|
|
723
|
+
#skipUnassigned(tp) {
|
|
724
|
+
// Not ours anymore: whoever takes it starts from the committed offset.
|
|
725
|
+
this.#log.debug('partition_hold_skipped_unassigned', {
|
|
726
|
+
messaging: { topic: tp.topic, partition: tp.partition },
|
|
727
|
+
});
|
|
728
|
+
this.#pendingRoutes.delete(partitionKey(tp));
|
|
729
|
+
}
|
|
730
|
+
/** Moves the partition back to `offset`, pauses it and schedules the resume. */
|
|
731
|
+
async #hold(client, tp, offset, delayMs) {
|
|
732
|
+
const key = partitionKey(tp);
|
|
733
|
+
if (!this.#isAssigned(client, tp)) {
|
|
734
|
+
this.#skipUnassigned(tp);
|
|
735
|
+
return;
|
|
736
|
+
}
|
|
737
|
+
try {
|
|
738
|
+
client.pause([tp]);
|
|
739
|
+
await this.#abortable(client.seek(tp, offset));
|
|
740
|
+
this.#pendingSeeks.delete(key);
|
|
741
|
+
}
|
|
742
|
+
catch (error) {
|
|
743
|
+
if (error instanceof LoopCancelledError)
|
|
744
|
+
throw error;
|
|
745
|
+
if (!this.#isAssigned(client, tp)) {
|
|
746
|
+
this.#skipUnassigned(tp);
|
|
747
|
+
return;
|
|
748
|
+
}
|
|
749
|
+
// Still ours but the position is unknown: never resume before the seek succeeds.
|
|
750
|
+
this.#pendingSeeks.set(key, offset);
|
|
751
|
+
}
|
|
752
|
+
this.#scheduleResume(tp, delayMs);
|
|
753
|
+
}
|
|
754
|
+
#scheduleResume(tp, delayMs) {
|
|
755
|
+
const key = partitionKey(tp);
|
|
756
|
+
this.#cancelTimer(key);
|
|
757
|
+
if (this.#stopping)
|
|
758
|
+
return;
|
|
759
|
+
const delay = Math.max(delayMs, MIN_RESUME_DELAY_MS);
|
|
760
|
+
const entry = {
|
|
761
|
+
tp,
|
|
762
|
+
delayMs: delay,
|
|
763
|
+
cancelled: false,
|
|
764
|
+
timer: setTimeout(() => {
|
|
765
|
+
if (this.#resumeTimers.get(key) === entry)
|
|
766
|
+
void this.#resumePartition(tp);
|
|
767
|
+
}, delay),
|
|
768
|
+
};
|
|
769
|
+
this.#resumeTimers.set(key, entry);
|
|
770
|
+
}
|
|
771
|
+
async #resumePartition(tp) {
|
|
772
|
+
const key = partitionKey(tp);
|
|
773
|
+
this.#cancelTimer(key);
|
|
774
|
+
const client = this.#client;
|
|
775
|
+
if (this.#stopping || client === null)
|
|
776
|
+
return;
|
|
777
|
+
try {
|
|
778
|
+
if (!this.#isAssigned(client, tp))
|
|
779
|
+
return;
|
|
780
|
+
const offset = this.#pendingSeeks.get(key);
|
|
781
|
+
if (offset !== undefined) {
|
|
782
|
+
await client.seek(tp, offset);
|
|
783
|
+
this.#pendingSeeks.delete(key);
|
|
784
|
+
}
|
|
785
|
+
if (this.#stopping)
|
|
786
|
+
return;
|
|
787
|
+
client.resume([tp]);
|
|
788
|
+
}
|
|
789
|
+
catch (error) {
|
|
790
|
+
this.#log.warn('partition_resume_failed', {
|
|
791
|
+
messaging: { topic: tp.topic, partition: tp.partition },
|
|
792
|
+
error: errorInfo(error),
|
|
793
|
+
});
|
|
794
|
+
// Try again later: a partition left paused would silently stop consuming.
|
|
795
|
+
if (this.#client !== null && this.#isAssigned(this.#client, tp)) {
|
|
796
|
+
this.#scheduleResume(tp, this.#routingFailureBackoffMs);
|
|
797
|
+
}
|
|
798
|
+
}
|
|
799
|
+
}
|
|
800
|
+
#cancelTimer(key) {
|
|
801
|
+
const entry = this.#resumeTimers.get(key);
|
|
802
|
+
if (entry === undefined)
|
|
803
|
+
return;
|
|
804
|
+
entry.cancelled = true;
|
|
805
|
+
clearTimeout(entry.timer);
|
|
806
|
+
this.#resumeTimers.delete(key);
|
|
807
|
+
}
|
|
808
|
+
#cancelTimers() {
|
|
809
|
+
for (const key of [...this.#resumeTimers.keys()])
|
|
810
|
+
this.#cancelTimer(key);
|
|
811
|
+
}
|
|
812
|
+
// ---------------------------------------------------------------------------- per record
|
|
813
|
+
/** Never throws (except the loop cancellation). */
|
|
814
|
+
async #processRecord(client, tp, record) {
|
|
815
|
+
try {
|
|
816
|
+
return await this.#handleRecord(client, tp, record);
|
|
817
|
+
}
|
|
818
|
+
catch (error) {
|
|
819
|
+
if (error instanceof LoopCancelledError)
|
|
820
|
+
throw error;
|
|
821
|
+
// Unexpected internal error: never risk losing the message.
|
|
822
|
+
this.#log.error('message_processing_failed', {
|
|
823
|
+
messaging: {
|
|
824
|
+
topic: record.topic,
|
|
825
|
+
partition: record.partition,
|
|
826
|
+
offset: record.offset,
|
|
827
|
+
group: this.#group,
|
|
828
|
+
},
|
|
829
|
+
error: errorInfo(error),
|
|
830
|
+
});
|
|
831
|
+
await this.#hold(client, tp, record.offset, this.#routingFailureBackoffMs);
|
|
832
|
+
return HOLD;
|
|
833
|
+
}
|
|
834
|
+
}
|
|
835
|
+
async #handleRecord(client, tp, record) {
|
|
836
|
+
const started = performance.now();
|
|
837
|
+
const hdrs = hdr.decode(record.headers);
|
|
838
|
+
const key = partitionKey(tp);
|
|
839
|
+
const pending = this.#pendingRoutes.get(key);
|
|
840
|
+
this.#pendingRoutes.delete(key);
|
|
841
|
+
if (pending !== undefined && pending.offset === record.offset) {
|
|
842
|
+
// Resend of a decision already taken: the handler does NOT run again.
|
|
843
|
+
return this.#telemetry.consumeSpan(pending.info, hdrs, (span) => this.#deliver({
|
|
844
|
+
client,
|
|
845
|
+
tp,
|
|
846
|
+
record,
|
|
847
|
+
hdrs,
|
|
848
|
+
info: pending.info,
|
|
849
|
+
fields: pending.fields,
|
|
850
|
+
span,
|
|
851
|
+
started,
|
|
852
|
+
dedupKey: pending.dedupKey,
|
|
853
|
+
}, pending));
|
|
854
|
+
}
|
|
855
|
+
const retryDelayMs = naming.retryDelayOf(record.topic, this.#mainTopic, this.#logicalGroup);
|
|
856
|
+
if (retryDelayMs !== null) {
|
|
857
|
+
const dueMs = hdr.retryDueAtMs(hdrs);
|
|
858
|
+
const nowMs = this.#clock();
|
|
859
|
+
if (dueMs > nowMs) {
|
|
860
|
+
// Each retry topic has a fixed delay: due dates are monotonic in the partition, so
|
|
861
|
+
// blocking it at the first not-due record is correct. Capped by the topic delay against
|
|
862
|
+
// an absurd header.
|
|
863
|
+
const waitMs = Math.min(dueMs - nowMs, retryDelayMs);
|
|
864
|
+
this.#log.debug('message_retry_not_due', {
|
|
865
|
+
messaging: {
|
|
866
|
+
topic: record.topic,
|
|
867
|
+
partition: record.partition,
|
|
868
|
+
offset: record.offset,
|
|
869
|
+
wait_s: round3(waitMs / 1000),
|
|
870
|
+
},
|
|
871
|
+
});
|
|
872
|
+
await this.#hold(client, tp, record.offset, waitMs);
|
|
873
|
+
return HOLD;
|
|
874
|
+
}
|
|
875
|
+
}
|
|
876
|
+
const eventType = hdrs[hdr.EVENT_TYPE] ?? '';
|
|
877
|
+
const info = createMessageInfo({
|
|
878
|
+
topic: record.topic,
|
|
879
|
+
eventType,
|
|
880
|
+
messageId: hdrs[hdr.MESSAGE_ID] ?? '',
|
|
881
|
+
schemaVersion: hdrs[hdr.SCHEMA_VERSION] ?? '',
|
|
882
|
+
key: record.key === null ? null : keyDecoder.decode(record.key),
|
|
883
|
+
correlationId: hdrs[hdr.CORRELATION_ID] ?? null,
|
|
884
|
+
// Empty = no tenant, like ConsumedMessage (spec §4.1).
|
|
885
|
+
tenantId: hdrs[hdr.TENANT_ID] || null,
|
|
886
|
+
partition: record.partition,
|
|
887
|
+
offset: record.offset,
|
|
888
|
+
group: this.#group,
|
|
889
|
+
retryCount: hdr.retryCount(hdrs),
|
|
890
|
+
});
|
|
891
|
+
const fields = this.#logFields(info);
|
|
892
|
+
const handler = this.#handlers.get(eventType);
|
|
893
|
+
if (handler === undefined) {
|
|
894
|
+
this.#telemetry.recordConsume(info, { outcome: 'skipped', durationMs: elapsedMs(started) });
|
|
895
|
+
this.#log.debug('message_skipped', { messaging: fields });
|
|
896
|
+
return ADVANCE;
|
|
897
|
+
}
|
|
898
|
+
// idempotency_key first; without any key there is no dedup.
|
|
899
|
+
const dedupKey = hdr.dedupKey(hdrs) ?? '';
|
|
900
|
+
if (this.#duplicateChecker !== null && dedupKey !== '' && (await this.#isProcessed(dedupKey, fields))) {
|
|
901
|
+
this.#telemetry.recordConsume(info, { outcome: 'deduplicated', durationMs: elapsedMs(started) });
|
|
902
|
+
this.#log.info('message_deduplicated', { messaging: fields });
|
|
903
|
+
return ADVANCE;
|
|
904
|
+
}
|
|
905
|
+
return this.#telemetry.consumeSpan(info, hdrs, (span) => this.#processInSpan({ client, tp, record, hdrs, info, fields, span, started, dedupKey }, handler));
|
|
906
|
+
}
|
|
907
|
+
async #processInSpan(ctx, handler) {
|
|
908
|
+
const { record, hdrs, info } = ctx;
|
|
909
|
+
let payload;
|
|
910
|
+
try {
|
|
911
|
+
payload = decodePayload(record.value);
|
|
912
|
+
}
|
|
913
|
+
catch (error) {
|
|
914
|
+
if (!(error instanceof InvalidPayloadError))
|
|
915
|
+
throw error;
|
|
916
|
+
return this.#deliver(ctx, this.#dlqRoute(ctx, error, 'invalid_payload'));
|
|
917
|
+
}
|
|
918
|
+
try {
|
|
919
|
+
this.#validator.validate({
|
|
920
|
+
domain: this.#domain,
|
|
921
|
+
eventType: info.eventType,
|
|
922
|
+
version: info.schemaVersion,
|
|
923
|
+
payload,
|
|
924
|
+
});
|
|
925
|
+
}
|
|
926
|
+
catch (error) {
|
|
927
|
+
if (error instanceof SchemaNotFoundError || error instanceof SchemaValidationError) {
|
|
928
|
+
this.#telemetry.recordSchemaError(info);
|
|
929
|
+
this.#log.warn('message_schema_error', { messaging: ctx.fields, error: errorInfo(error) });
|
|
930
|
+
return this.#deliver(ctx, this.#dlqRoute(ctx, error, 'schema_error'));
|
|
931
|
+
}
|
|
932
|
+
// Validator failure (e.g. registry unavailable), not the payload's.
|
|
933
|
+
this.#log.warn('message_schema_check_failed', { messaging: ctx.fields, error: errorInfo(error) });
|
|
934
|
+
return this.#onFailure(ctx, error);
|
|
935
|
+
}
|
|
936
|
+
const headers = Object.freeze(Object.assign(Object.create(null), hdrs));
|
|
937
|
+
const message = Object.freeze({
|
|
938
|
+
messageId: info.messageId,
|
|
939
|
+
eventType: info.eventType,
|
|
940
|
+
schemaVersion: info.schemaVersion,
|
|
941
|
+
payload,
|
|
942
|
+
key: info.key,
|
|
943
|
+
correlationId: hdrs[hdr.CORRELATION_ID] ?? '',
|
|
944
|
+
idempotencyKey: hdrs[hdr.IDEMPOTENCY_KEY] || null,
|
|
945
|
+
tenantId: hdrs[hdr.TENANT_ID] || null,
|
|
946
|
+
producer: hdrs[hdr.PRODUCER] ?? '',
|
|
947
|
+
publishedAt: hdrs[hdr.PUBLISHED_AT] ?? '',
|
|
948
|
+
retryCount: info.retryCount,
|
|
949
|
+
topic: record.topic,
|
|
950
|
+
partition: record.partition,
|
|
951
|
+
offset: record.offset,
|
|
952
|
+
headers,
|
|
953
|
+
});
|
|
954
|
+
const outcome = await this.#runHandler(handler, message, ctx.fields);
|
|
955
|
+
if (outcome.ok) {
|
|
956
|
+
const durationMs = elapsedMs(ctx.started);
|
|
957
|
+
this.#finish(ctx, 'success');
|
|
958
|
+
this.#log.info('message_consumed', {
|
|
959
|
+
messaging: ctx.fields,
|
|
960
|
+
processing_duration_ms: round3(durationMs),
|
|
961
|
+
});
|
|
962
|
+
// Only after success (retries never mark: they must not become duplicates).
|
|
963
|
+
await this.#markProcessed(ctx);
|
|
964
|
+
return ADVANCE;
|
|
965
|
+
}
|
|
966
|
+
return this.#onFailure(ctx, outcome.error);
|
|
967
|
+
}
|
|
968
|
+
#onFailure(ctx, error) {
|
|
969
|
+
if (error instanceof PermanentProcessingError) {
|
|
970
|
+
return this.#deliver(ctx, this.#dlqRoute(ctx, error, 'permanent_error'));
|
|
971
|
+
}
|
|
972
|
+
const delayMs = this.#retryPolicy.delayFor(ctx.info.retryCount);
|
|
973
|
+
if (delayMs === null)
|
|
974
|
+
return this.#deliver(ctx, this.#dlqRoute(ctx, error, 'max_retries'));
|
|
975
|
+
return this.#deliver(ctx, this.#retryRoute(ctx, error, delayMs));
|
|
976
|
+
}
|
|
977
|
+
/** Runs the handler; resolves with its outcome (transient or not), never rejects (but cancellation). */
|
|
978
|
+
async #runHandler(handler, message, fields) {
|
|
979
|
+
const controller = new AbortController();
|
|
980
|
+
let running;
|
|
981
|
+
try {
|
|
982
|
+
const result = handler(message, controller.signal);
|
|
983
|
+
if (!isPromiseLike(result))
|
|
984
|
+
throw new TypeError('handler did not return a Promise');
|
|
985
|
+
running = Promise.resolve(result);
|
|
986
|
+
}
|
|
987
|
+
catch (error) {
|
|
988
|
+
// Could not be called or did not return a promise: configuration error, repeating does not help.
|
|
989
|
+
return {
|
|
990
|
+
ok: false,
|
|
991
|
+
error: new InvalidHandlerError(`handler could not be started: ${errorTypeName(error)}: ${errorMessage(error)}`),
|
|
992
|
+
};
|
|
993
|
+
}
|
|
994
|
+
this.#active.add(controller);
|
|
995
|
+
const settled = settle(running);
|
|
996
|
+
try {
|
|
997
|
+
const timeoutMs = this.#processingTimeoutMs;
|
|
998
|
+
const outcome = timeoutMs === null
|
|
999
|
+
? await this.#abortable(settled)
|
|
1000
|
+
: await this.#abortable(withTimeout(settled, timeoutMs));
|
|
1001
|
+
if (outcome === TIMED_OUT) {
|
|
1002
|
+
const limit = timeoutMs ?? 0;
|
|
1003
|
+
const error = new Error(`handler did not finish within processing timeout of ${formatSeconds(limit)}s`);
|
|
1004
|
+
error.name = 'TimeoutError';
|
|
1005
|
+
controller.abort(error);
|
|
1006
|
+
if ((await this.#abortable(withTimeout(settled, limit))) === TIMED_OUT) {
|
|
1007
|
+
// Ignored the abort: keeps running detached (side effects may run in parallel with the retry).
|
|
1008
|
+
this.#log.error('message_handler_abandoned', {
|
|
1009
|
+
messaging: fields,
|
|
1010
|
+
processing_timeout_s: limit / 1000,
|
|
1011
|
+
});
|
|
1012
|
+
this.#detach(running);
|
|
1013
|
+
}
|
|
1014
|
+
this.#log.warn('message_handler_timeout', { messaging: fields, error: errorInfo(error) });
|
|
1015
|
+
return { ok: false, error };
|
|
1016
|
+
}
|
|
1017
|
+
if (!outcome.ok && isAbortError(outcome.error) && !controller.signal.aborted) {
|
|
1018
|
+
// The handler ended aborted without the consumer asking for it.
|
|
1019
|
+
return {
|
|
1020
|
+
ok: false,
|
|
1021
|
+
error: new HandlerCancelledError('handler task was cancelled', { cause: outcome.error }),
|
|
1022
|
+
};
|
|
1023
|
+
}
|
|
1024
|
+
return outcome;
|
|
1025
|
+
}
|
|
1026
|
+
catch (error) {
|
|
1027
|
+
if (error instanceof LoopCancelledError) {
|
|
1028
|
+
// The loop itself was cancelled (stop timeout): abort the handler.
|
|
1029
|
+
controller.abort(error);
|
|
1030
|
+
this.#detach(running);
|
|
1031
|
+
}
|
|
1032
|
+
throw error;
|
|
1033
|
+
}
|
|
1034
|
+
finally {
|
|
1035
|
+
this.#active.delete(controller);
|
|
1036
|
+
}
|
|
1037
|
+
}
|
|
1038
|
+
#detach(running) {
|
|
1039
|
+
const tracked = running.then(() => undefined, () => undefined);
|
|
1040
|
+
this.#detached.add(tracked);
|
|
1041
|
+
void tracked.then(() => this.#detached.delete(tracked));
|
|
1042
|
+
}
|
|
1043
|
+
// ---------------------------------------------------------------------------- routing
|
|
1044
|
+
#retryRoute(ctx, error, delayMs) {
|
|
1045
|
+
const { record } = ctx;
|
|
1046
|
+
const nowMs = this.#clock();
|
|
1047
|
+
return {
|
|
1048
|
+
offset: record.offset,
|
|
1049
|
+
kind: 'retry',
|
|
1050
|
+
target: naming.retryTopic(this.#mainTopic, this.#logicalGroup, delayMs),
|
|
1051
|
+
headers: hdr.buildRetryHeaders(ctx.hdrs, error, {
|
|
1052
|
+
dueAtMs: nowMs + delayMs,
|
|
1053
|
+
topic: record.topic,
|
|
1054
|
+
partition: record.partition,
|
|
1055
|
+
offset: record.offset,
|
|
1056
|
+
clock: () => nowMs,
|
|
1057
|
+
}),
|
|
1058
|
+
error,
|
|
1059
|
+
reason: null,
|
|
1060
|
+
attempt: ctx.info.retryCount + 1,
|
|
1061
|
+
info: ctx.info,
|
|
1062
|
+
fields: ctx.fields,
|
|
1063
|
+
dedupKey: ctx.dedupKey,
|
|
1064
|
+
shrunk: false,
|
|
1065
|
+
};
|
|
1066
|
+
}
|
|
1067
|
+
#dlqRoute(ctx, error, reason) {
|
|
1068
|
+
const { record } = ctx;
|
|
1069
|
+
// Same clock as x-retry-due-at (injectable in tests).
|
|
1070
|
+
const nowMs = this.#clock();
|
|
1071
|
+
return {
|
|
1072
|
+
offset: record.offset,
|
|
1073
|
+
kind: 'dlq',
|
|
1074
|
+
target: this.#dlqTopic,
|
|
1075
|
+
headers: hdr.buildDlqHeaders(ctx.hdrs, error, {
|
|
1076
|
+
reason,
|
|
1077
|
+
topic: record.topic,
|
|
1078
|
+
partition: record.partition,
|
|
1079
|
+
offset: record.offset,
|
|
1080
|
+
clock: () => nowMs,
|
|
1081
|
+
}),
|
|
1082
|
+
error,
|
|
1083
|
+
reason,
|
|
1084
|
+
attempt: ctx.info.retryCount,
|
|
1085
|
+
info: ctx.info,
|
|
1086
|
+
fields: ctx.fields,
|
|
1087
|
+
dedupKey: ctx.dedupKey,
|
|
1088
|
+
shrunk: false,
|
|
1089
|
+
};
|
|
1090
|
+
}
|
|
1091
|
+
/** Sends the routing decision; on failure decides between HOLD, fallback and fatal. */
|
|
1092
|
+
async #deliver(ctx, route) {
|
|
1093
|
+
try {
|
|
1094
|
+
await this.#send(ctx, route);
|
|
1095
|
+
}
|
|
1096
|
+
catch (error) {
|
|
1097
|
+
if (error instanceof LoopCancelledError)
|
|
1098
|
+
throw error;
|
|
1099
|
+
return this.#onRouteFailure(ctx, route, error);
|
|
1100
|
+
}
|
|
1101
|
+
const { info } = ctx;
|
|
1102
|
+
if (route.kind === 'retry') {
|
|
1103
|
+
this.#telemetry.recordRetry(info, { attempt: route.attempt, retryTopic: route.target });
|
|
1104
|
+
this.#log.warn('message_retry_scheduled', {
|
|
1105
|
+
messaging: { ...ctx.fields, retry_topic: route.target, retry_attempt: route.attempt },
|
|
1106
|
+
error: errorInfo(route.error),
|
|
1107
|
+
});
|
|
1108
|
+
ctx.span.recordError(route.error);
|
|
1109
|
+
this.#finish(ctx, 'retry');
|
|
1110
|
+
// No dedup mark: the retry must be processed when it comes back.
|
|
1111
|
+
return ADVANCE;
|
|
1112
|
+
}
|
|
1113
|
+
const reason = route.reason ?? 'permanent_error';
|
|
1114
|
+
this.#telemetry.recordDlq(info, { reason, dlqTopic: route.target });
|
|
1115
|
+
this.#log.error('message_sent_to_dlq', {
|
|
1116
|
+
messaging: { ...ctx.fields, dlq_topic: route.target },
|
|
1117
|
+
error: { ...errorInfo(route.error), dlq_reason: reason },
|
|
1118
|
+
});
|
|
1119
|
+
ctx.span.recordError(route.error);
|
|
1120
|
+
this.#finish(ctx, 'dlq');
|
|
1121
|
+
// No dedup mark: the message must be reprocessable from the DLQ (replay). A redelivery
|
|
1122
|
+
// before the commit may duplicate it in the DLQ.
|
|
1123
|
+
return ADVANCE;
|
|
1124
|
+
}
|
|
1125
|
+
/** Sends the original value bytes and key to `route.target`. */
|
|
1126
|
+
async #send(ctx, route) {
|
|
1127
|
+
const { record } = ctx;
|
|
1128
|
+
const value = record.value ?? Buffer.alloc(0);
|
|
1129
|
+
const headers = { ...route.headers };
|
|
1130
|
+
if (route.shrunk) {
|
|
1131
|
+
// No trace context: minimal headers to fit the broker limit.
|
|
1132
|
+
await this.#abortable(this.#sender.send(route.target, value, { key: record.key, headers: hdr.encode(headers) }));
|
|
1133
|
+
return;
|
|
1134
|
+
}
|
|
1135
|
+
const { info } = route;
|
|
1136
|
+
const publishInfo = createMessageInfo({
|
|
1137
|
+
topic: route.target,
|
|
1138
|
+
eventType: info.eventType,
|
|
1139
|
+
messageId: info.messageId,
|
|
1140
|
+
schemaVersion: info.schemaVersion,
|
|
1141
|
+
key: info.key,
|
|
1142
|
+
correlationId: info.correlationId,
|
|
1143
|
+
tenantId: info.tenantId,
|
|
1144
|
+
group: this.#group,
|
|
1145
|
+
retryCount: route.attempt,
|
|
1146
|
+
});
|
|
1147
|
+
// publishSpan injects the W3C context into `headers` before they are encoded.
|
|
1148
|
+
await this.#telemetry.publishSpan(publishInfo, headers, () => this.#abortable(this.#sender.send(route.target, value, { key: record.key, headers: hdr.encode(headers) })));
|
|
1149
|
+
}
|
|
1150
|
+
async #onRouteFailure(ctx, route, error) {
|
|
1151
|
+
const nonRetriable = error instanceof PublishError && !error.retriable;
|
|
1152
|
+
this.#log.error('message_routing_failed', {
|
|
1153
|
+
messaging: { ...ctx.fields, target_topic: route.target, retriable: !nonRetriable },
|
|
1154
|
+
error: errorInfo(error),
|
|
1155
|
+
});
|
|
1156
|
+
ctx.span.recordError(error);
|
|
1157
|
+
if (nonRetriable) {
|
|
1158
|
+
if (route.kind === 'retry') {
|
|
1159
|
+
// The retry will never be accepted: straight to the DLQ.
|
|
1160
|
+
return this.#deliver(ctx, this.#dlqRoute(ctx, route.error, 'routing_failed'));
|
|
1161
|
+
}
|
|
1162
|
+
if (!route.shrunk)
|
|
1163
|
+
return this.#deliver(ctx, shrink(route));
|
|
1164
|
+
// Not even the DLQ accepts it: stopping is the only option that does not lose the message.
|
|
1165
|
+
this.#fail(error);
|
|
1166
|
+
return FATAL;
|
|
1167
|
+
}
|
|
1168
|
+
// Retriable: keep the decision and resend after the backoff, without running the handler.
|
|
1169
|
+
this.#pendingRoutes.set(partitionKey(ctx.tp), route);
|
|
1170
|
+
await this.#hold(ctx.client, ctx.tp, ctx.record.offset, this.#routingFailureBackoffMs);
|
|
1171
|
+
return HOLD;
|
|
1172
|
+
}
|
|
1173
|
+
#finish(ctx, outcome) {
|
|
1174
|
+
ctx.span.setOutcome(outcome);
|
|
1175
|
+
this.#telemetry.recordConsume(ctx.info, { outcome, durationMs: elapsedMs(ctx.started) });
|
|
1176
|
+
}
|
|
1177
|
+
async #isProcessed(key, fields) {
|
|
1178
|
+
const checker = this.#duplicateChecker;
|
|
1179
|
+
if (checker === null)
|
|
1180
|
+
return false;
|
|
1181
|
+
try {
|
|
1182
|
+
return await this.#abortable(checker.isProcessed(key));
|
|
1183
|
+
}
|
|
1184
|
+
catch (error) {
|
|
1185
|
+
if (error instanceof LoopCancelledError)
|
|
1186
|
+
throw error;
|
|
1187
|
+
this.#log.warn('dedup_check_failed', { messaging: fields, error: errorInfo(error) });
|
|
1188
|
+
return false;
|
|
1189
|
+
}
|
|
1190
|
+
}
|
|
1191
|
+
async #markProcessed(ctx) {
|
|
1192
|
+
const checker = this.#duplicateChecker;
|
|
1193
|
+
if (checker === null || ctx.dedupKey === '')
|
|
1194
|
+
return;
|
|
1195
|
+
try {
|
|
1196
|
+
await this.#abortable(checker.markProcessed(ctx.dedupKey));
|
|
1197
|
+
}
|
|
1198
|
+
catch (error) {
|
|
1199
|
+
if (error instanceof LoopCancelledError)
|
|
1200
|
+
throw error;
|
|
1201
|
+
this.#log.warn('dedup_mark_failed', { messaging: ctx.fields, error: errorInfo(error) });
|
|
1202
|
+
}
|
|
1203
|
+
}
|
|
1204
|
+
#logFields(info) {
|
|
1205
|
+
// Never the payload nor the key (may contain PII).
|
|
1206
|
+
return {
|
|
1207
|
+
topic: info.topic,
|
|
1208
|
+
partition: info.partition,
|
|
1209
|
+
offset: info.offset,
|
|
1210
|
+
group: this.#group,
|
|
1211
|
+
event_type: info.eventType,
|
|
1212
|
+
message_id: info.messageId,
|
|
1213
|
+
schema_version: info.schemaVersion,
|
|
1214
|
+
correlation_id: info.correlationId,
|
|
1215
|
+
retry_count: info.retryCount,
|
|
1216
|
+
};
|
|
1217
|
+
}
|
|
1218
|
+
}
|
|
1219
|
+
function shrink(route) {
|
|
1220
|
+
const headers = { ...route.headers };
|
|
1221
|
+
const lastError = headers[hdr.LAST_ERROR];
|
|
1222
|
+
if (lastError !== undefined)
|
|
1223
|
+
headers[hdr.LAST_ERROR] = truncateChars(lastError, SHRUNK_ERROR_LENGTH);
|
|
1224
|
+
for (const name of TRACE_HEADERS)
|
|
1225
|
+
delete headers[name];
|
|
1226
|
+
return { ...route, headers, shrunk: true };
|
|
1227
|
+
}
|
|
1228
|
+
//# sourceMappingURL=consumer.js.map
|