@friggframework/core 2.0.0--canary.656.10f1676.0 → 2.0.0--canary.657.8f335eb.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/CLAUDE.md CHANGED
@@ -381,7 +381,10 @@ in this order:
381
381
  | 4 | The fixed `backOff` ladder (1, 3, 10, 30, 60, 180 s) |
382
382
 
383
383
  - A 429 with no hint from steps 1 to 3 keeps today's ladder: the same calls,
384
- the same delays, then a plain `FetchError`. Nothing budgets it.
384
+ the same delays, then a plain `FetchError`. Nothing budgets it. A response
385
+ that `classify` names as a limit but that has no time follows the same
386
+ ladder, and the last error is a `FetchError` flagged `isRateLimited` with
387
+ the `reason`, so the queue worker does not halt it.
385
388
  - With a hint, the wait is `max(hint, minRetryAfterMs, 1 s)` plus at most 10 %
386
389
  jitter. The total sleep of one request is capped at `maxInProcessWaitMs`
387
390
  (default 5 minutes) and at the time left in the invocation less one request
@@ -543,6 +546,58 @@ The single source of the max receive count is
543
546
  devtools integration builder uses it for the queue's `RedrivePolicy` and sets
544
547
  it as `FRIGG_QUEUE_MAX_RECEIVE_COUNT` on the queue worker function.
545
548
 
549
+ **Rate-limit deferral** (ADR-049): when a queue handler throws an error with
550
+ `isRateLimited` and a `retryAt` (a `RateLimitError` from the Requester, or one a
551
+ module throws), `Worker.run` does not give the message back to SQS for a 30
552
+ minute redelivery that spends one of its three receives. It puts the message
553
+ back so it runs at `retryAt`:
554
+
555
+ | Wait | What the worker does | Record |
556
+ |---|---|---|
557
+ | Up to 900 s | Sends the same body again with `DelaySeconds`, then acknowledges the old message | handled |
558
+ | Over 900 s, and `SCHEDULER_ROLE_ARN` is set | Creates a one-time EventBridge schedule at `retryAt` that sends the body to the queue, then acknowledges | handled |
559
+ | Over 900 s, no scheduler | Extends the message's visibility timeout to `retryAt` (12 h at most). This uses one receive | reported failed |
560
+
561
+ - The new message starts at receive count 1. A deferred body carries
562
+ `_frigg: { deferrals, firstDeferredAt }`; `_frigg` is reserved and the handler
563
+ never sees it.
564
+ - Two caps bound it: `FRIGG_QUEUE_MAX_DEFERRALS` (default 10) and
565
+ `FRIGG_QUEUE_MAX_DEFERRED_MS` (default 24 h, from the first deferral to the
566
+ retry time). Past a cap, with no `retryAt`, with no event source ARN or on a
567
+ FIFO queue, the record fails like any retryable error and `delivery`
568
+ applies.
569
+ - The mock scheduler is never used for a deferral: a schedule kept in memory
570
+ would be lost with the acknowledged message.
571
+ - With a `processId` in the message, the worker writes
572
+ `Process.context.rateLimit = { status: 'WAITING' | 'EXHAUSTED', mechanism,
573
+ retryAt, reason, module, deferrals, updatedAt }`, and `null` after the message
574
+ ran. `EXHAUSTED` means a cap ended the deferrals, or a visibility change came
575
+ on the last delivery.
576
+ - Delivery stays at least once. The new message is sent before the old one is
577
+ acknowledged, so a crash between the two can run the work twice.
578
+ - Handlers check `delivery.isLastAttempt` first, then rethrow. The delay and
579
+ schedule tiers send a new message, so they never reach the last attempt. The
580
+ visibility tier and a cap use a receive: on the last one the worker writes an
581
+ ERROR (`frigg.worker.record_lost_rate_limited`), and the handler has this one
582
+ chance to end the run.
583
+
584
+ ```javascript
585
+ async processBatch({ data, delivery }) {
586
+ try {
587
+ await this.syncPage(data);
588
+ } catch (error) {
589
+ if (delivery?.isLastAttempt) return this.failRun(data.processId, error);
590
+ throw error; // a RateLimitError too: core puts the message back at retryAt
591
+ }
592
+ }
593
+ ```
594
+
595
+ Records: `frigg.worker.record_deferred` (`mechanism` `delay` or `schedule`),
596
+ `record_visibility_extended`, `record_deferral_capped`,
597
+ `record_lost_rate_limited` and `rate_limit_state_failed`. A queue in a stack
598
+ that does not own it (`ownership.queue: 'external'`) needs its own
599
+ `sqs:ChangeMessageVisibility` and scheduler access.
600
+
546
601
  ### 8. Error Handling (`/errors`)
547
602
 
548
603
  **Purpose**: Standardized error types with proper HTTP semantics.
package/core/CLAUDE.md CHANGED
@@ -79,6 +79,25 @@ await worker.send({
79
79
  }, delaySeconds);
80
80
  ```
81
81
 
82
+ **Rate-limit deferral** (ADR-049): `run()` handles an error with `isRateLimited`
83
+ and a `retryAt` through `defer(record, body, error, delivery)`, which returns
84
+ `{ outcome }`:
85
+ - `acked`: the message was sent again with `DelaySeconds` (wait up to 900 s) or
86
+ scheduled with the one-time scheduler (longer wait). The record is handled.
87
+ - `failed`: the visibility timeout was extended to `retryAt`, or a cap
88
+ (`FRIGG_QUEUE_MAX_DEFERRALS`, `FRIGG_QUEUE_MAX_DEFERRED_MS`) ended the
89
+ deferrals. The record is reported failed.
90
+ - `skipped`: no `retryAt`, no `eventSourceARN`, a FIFO queue, or a failed send.
91
+ The record fails as any error does, and `record_failed` carries
92
+ `deferral: { skipped }`.
93
+
94
+ Subclasses override two no-op hooks to keep run state: `recordRateLimitWait(body,
95
+ error, state)` runs after the message was put back, and `clearRateLimitWait(body)`
96
+ runs after a deferred or redelivered message succeeded. A failing hook logs
97
+ `frigg.worker.rate_limit_state_failed` and never changes the outcome. The queue
98
+ worker of an integration (`createQueueWorker`) writes `Process.context.rateLimit`
99
+ in them. See `packages/core/CLAUDE.md` section 7.
100
+
82
101
  ### Delegate Pattern System (`Delegate.js:3-27`)
83
102
 
84
103
  **Purpose**: Observer/delegation pattern for decoupled component communication
package/core/Worker.js CHANGED
@@ -1,14 +1,39 @@
1
- const { SQSClient, GetQueueUrlCommand, SendMessageCommand } = require('@aws-sdk/client-sqs');
1
+ const { randomUUID } = require('node:crypto');
2
+ const {
3
+ SQSClient,
4
+ GetQueueUrlCommand,
5
+ SendMessageCommand,
6
+ ChangeMessageVisibilityCommand,
7
+ } = require('@aws-sdk/client-sqs');
2
8
  const _ = require('lodash');
3
9
  const { RequiredPropertyError } = require('../errors');
4
10
  const { get } = require('../assertions');
5
11
  const { readQueueDelivery } = require('../queues/queue-delivery');
12
+ const { awsConfigOptions } = require('../queues/queuer-util');
13
+ const {
14
+ MAX_DELAY_SECONDS,
15
+ MAX_VISIBILITY_TIMEOUT_SECONDS,
16
+ isDeferralCapped,
17
+ nextDeferral,
18
+ readDeferral,
19
+ readDeferralLimits,
20
+ withDeferral,
21
+ } = require('../queues/queue-deferral');
6
22
  const { runMessageScope } = require('./invocation-scope');
7
- const { getLogger } = require('../logs');
23
+ const { getLogger, serializeError } = require('../logs');
8
24
 
9
- const sqs = new SQSClient({ region: process.env.AWS_REGION });
25
+ const sqs = new SQSClient({
26
+ region: process.env.AWS_REGION,
27
+ ...awsConfigOptions(),
28
+ });
10
29
  const log = getLogger('frigg.worker');
11
30
 
31
+ const validDate = (value) =>
32
+ value instanceof Date && !Number.isNaN(value.getTime()) ? value : null;
33
+ const queueNameOf = (arn) =>
34
+ typeof arn === 'string' && arn ? arn.split(':').pop() : undefined;
35
+ const causeOf = (error) => serializeError(error).message;
36
+
12
37
  class Worker {
13
38
  async getQueueURL(params) {
14
39
  // Passing params in because there will be multiple QueueNames
@@ -33,12 +58,31 @@ class Worker {
33
58
  // messageId, receiveCount and the event come from the scope.
34
59
  log.debug('Record started', { eventName: 'frigg.worker.record_started' });
35
60
 
61
+ const delivery = readQueueDelivery(record);
62
+ let runParams;
36
63
  try {
37
- const runParams = JSON.parse(record.body);
64
+ runParams = JSON.parse(record.body);
38
65
  this._validateParams(runParams);
39
- await this._run(runParams, context, readQueueDelivery(record));
66
+ await this._run(runParams, context, delivery);
67
+ await this._clearDeferredState(runParams, delivery);
40
68
  log.debug('Record succeeded', { eventName: 'frigg.worker.record_succeeded' });
41
69
  } catch (error) {
70
+ let deferral;
71
+ if (error.isRateLimited && runParams) {
72
+ deferral = await this.defer(
73
+ record,
74
+ runParams,
75
+ error,
76
+ delivery
77
+ );
78
+ if (deferral.outcome === 'acked') return;
79
+ if (deferral.outcome === 'failed') {
80
+ batchItemFailures.push({
81
+ itemIdentifier: record.messageId,
82
+ });
83
+ return;
84
+ }
85
+ }
42
86
  if (error.isHaltError) {
43
87
  // HaltError means "discard this message, don't retry".
44
88
  // Treat as success so SQS deletes it from the queue.
@@ -55,6 +99,14 @@ class Worker {
55
99
  log.warn('Record failed, returned for retry', {
56
100
  eventName: 'frigg.worker.record_failed',
57
101
  error,
102
+ ...(deferral && {
103
+ deferral: {
104
+ skipped: deferral.skipped,
105
+ ...(deferral.cause && {
106
+ cause: causeOf(deferral.cause),
107
+ }),
108
+ },
109
+ }),
58
110
  });
59
111
  batchItemFailures.push({ itemIdentifier: record.messageId });
60
112
  }
@@ -75,6 +127,240 @@ class Worker {
75
127
  // parameters
76
128
  }
77
129
 
130
+ /**
131
+ * Puts a rate-limited message back so it runs at `error.retryAt`, without
132
+ * spending a receive. The message is sent again with a delay of up to 900 s,
133
+ * or scheduled with the one-time scheduler, or has its visibility timeout
134
+ * extended. The new message goes out before the old one is acknowledged, so
135
+ * delivery stays at least once.
136
+ *
137
+ * @returns {Promise<{outcome: 'acked'|'failed'|'skipped', skipped?: string, cause?: Error}>}
138
+ * `acked`: the message is handled. `failed`: report the record failed and
139
+ * let SQS redeliver it. `skipped`: nothing was done; fail the record as
140
+ * any other error.
141
+ */
142
+ async defer(record, body, error, delivery) {
143
+ const retryAt = validDate(error.retryAt);
144
+ if (!retryAt) return { outcome: 'skipped', skipped: 'no_retry_at' };
145
+ const queueName = queueNameOf(record.eventSourceARN);
146
+ if (!queueName || queueName.endsWith('.fifo')) {
147
+ return { outcome: 'skipped', skipped: 'no_event_source' };
148
+ }
149
+
150
+ const now = Date.now();
151
+ const waitMs = Math.max(0, retryAt.getTime() - now);
152
+ const deferral = nextDeferral(body, now);
153
+ const details = {
154
+ waitMs,
155
+ retryAt: retryAt.toISOString(),
156
+ deferrals: deferral.deferrals,
157
+ firstDeferredAt: deferral.firstDeferredAt,
158
+ reason: error.reason,
159
+ module: error.module,
160
+ statusCode: error.statusCode,
161
+ };
162
+ const state = (status, mechanism) => ({
163
+ status,
164
+ mechanism,
165
+ deferrals: deferral.deferrals,
166
+ retryAt,
167
+ });
168
+
169
+ const limits = readDeferralLimits();
170
+ if (isDeferralCapped({ ...deferral, retryAt }, limits)) {
171
+ log.warn('Rate-limit deferral capped', {
172
+ eventName: 'frigg.worker.record_deferral_capped',
173
+ ...details,
174
+ maxDeferrals: limits.maxDeferrals,
175
+ maxDeferredMs: limits.maxDeferredMs,
176
+ deferredMs:
177
+ retryAt.getTime() - Date.parse(deferral.firstDeferredAt),
178
+ error,
179
+ });
180
+ await this._recordWait(body, error, state('EXHAUSTED', 'none'));
181
+ this._logIfLost(delivery, error, details);
182
+ return { outcome: 'failed' };
183
+ }
184
+
185
+ const next = withDeferral(body, deferral);
186
+
187
+ if (waitMs <= MAX_DELAY_SECONDS * 1000) {
188
+ const delaySeconds = Math.ceil(waitMs / 1000);
189
+ try {
190
+ const queueUrl = await this._queueUrl(queueName);
191
+ await this.send({ ...next, QueueUrl: queueUrl }, delaySeconds);
192
+ } catch (cause) {
193
+ return { outcome: 'skipped', skipped: 'send_failed', cause };
194
+ }
195
+ await this._recordWait(body, error, state('WAITING', 'delay'));
196
+ log.warn('Record deferred', {
197
+ eventName: 'frigg.worker.record_deferred',
198
+ mechanism: 'delay',
199
+ delaySeconds,
200
+ ...details,
201
+ error,
202
+ });
203
+ return { outcome: 'acked' };
204
+ }
205
+
206
+ let scheduleError;
207
+ const scheduler = this.getSchedulerService();
208
+ if (scheduler) {
209
+ const scheduleName = `frigg-defer-${
210
+ record.messageId || randomUUID()
211
+ }`;
212
+ try {
213
+ await scheduler.scheduleOneTime({
214
+ scheduleName,
215
+ scheduleAt: retryAt,
216
+ queueResourceId: record.eventSourceARN,
217
+ payload: next,
218
+ });
219
+ } catch (cause) {
220
+ if (cause?.name !== 'ConflictException') scheduleError = cause;
221
+ }
222
+ if (!scheduleError) {
223
+ await this._recordWait(
224
+ body,
225
+ error,
226
+ state('WAITING', 'schedule')
227
+ );
228
+ log.warn('Record deferred', {
229
+ eventName: 'frigg.worker.record_deferred',
230
+ mechanism: 'schedule',
231
+ scheduleName,
232
+ ...details,
233
+ error,
234
+ });
235
+ return { outcome: 'acked' };
236
+ }
237
+ }
238
+
239
+ const visibilityTimeout = Math.min(
240
+ Math.ceil(waitMs / 1000),
241
+ MAX_VISIBILITY_TIMEOUT_SECONDS
242
+ );
243
+ try {
244
+ const queueUrl = await this._queueUrl(queueName);
245
+ await sqs.send(
246
+ new ChangeMessageVisibilityCommand({
247
+ QueueUrl: queueUrl,
248
+ ReceiptHandle: record.receiptHandle,
249
+ VisibilityTimeout: visibilityTimeout,
250
+ })
251
+ );
252
+ } catch (cause) {
253
+ return { outcome: 'skipped', skipped: 'visibility_failed', cause };
254
+ }
255
+ await this._recordWait(
256
+ body,
257
+ error,
258
+ state(
259
+ delivery.isLastAttempt ? 'EXHAUSTED' : 'WAITING',
260
+ 'visibility'
261
+ )
262
+ );
263
+ log.warn('Record visibility extended', {
264
+ eventName: 'frigg.worker.record_visibility_extended',
265
+ mechanism: 'visibility',
266
+ visibilityTimeout,
267
+ ...details,
268
+ ...(scheduleError && { scheduleError: causeOf(scheduleError) }),
269
+ error,
270
+ });
271
+ this._logIfLost(delivery, error, details);
272
+ return { outcome: 'failed' };
273
+ }
274
+
275
+ /**
276
+ * The one-time scheduler, or null when the stack has none. The mock
277
+ * scheduler is never used: a schedule kept in memory would be lost with
278
+ * the message.
279
+ */
280
+ getSchedulerService() {
281
+ if (this._schedulerService !== undefined) return this._schedulerService;
282
+ this._schedulerService = null;
283
+ if (
284
+ process.env.SCHEDULER_ROLE_ARN &&
285
+ process.env.SCHEDULER_PROVIDER !== 'mock'
286
+ ) {
287
+ const {
288
+ createSchedulerService,
289
+ SCHEDULER_PROVIDERS,
290
+ } = require('../infrastructure/scheduler/scheduler-service-factory');
291
+ this._schedulerService = createSchedulerService({
292
+ provider: SCHEDULER_PROVIDERS.EVENTBRIDGE,
293
+ });
294
+ }
295
+ return this._schedulerService;
296
+ }
297
+
298
+ /**
299
+ * Hook: the message was put back for a rate limit. Runs after it was sent.
300
+ * @param {Object} body The parsed message body.
301
+ * @param {Error} error The RateLimitError.
302
+ * @param {{status: 'WAITING'|'EXHAUSTED', mechanism: string, deferrals: number, retryAt: Date}} state
303
+ */
304
+ async recordRateLimitWait() {}
305
+
306
+ /**
307
+ * Hook: a message that was put back for a rate limit ran without error.
308
+ * @param {Object} body The parsed message body.
309
+ */
310
+ async clearRateLimitWait() {}
311
+
312
+ _queueUrl(queueName) {
313
+ this._queueUrls ??= new Map();
314
+ if (!this._queueUrls.has(queueName)) {
315
+ const lookup = this.getQueueURL({ QueueName: queueName }).catch(
316
+ (error) => {
317
+ this._queueUrls.delete(queueName);
318
+ throw error;
319
+ }
320
+ );
321
+ this._queueUrls.set(queueName, lookup);
322
+ }
323
+ return this._queueUrls.get(queueName);
324
+ }
325
+
326
+ async _recordWait(body, error, state) {
327
+ await this._runStateHook('record', () =>
328
+ this.recordRateLimitWait(body, error, state)
329
+ );
330
+ }
331
+
332
+ async _clearDeferredState(body, delivery) {
333
+ if (readDeferral(body).deferrals === 0 && !(delivery.receiveCount > 1))
334
+ return;
335
+ await this._runStateHook('clear', () => this.clearRateLimitWait(body));
336
+ }
337
+
338
+ async _runStateHook(operation, hook) {
339
+ try {
340
+ await hook();
341
+ } catch (error) {
342
+ log.warn('Rate-limit run state not written', {
343
+ eventName: 'frigg.worker.rate_limit_state_failed',
344
+ operation,
345
+ error,
346
+ });
347
+ }
348
+ }
349
+
350
+ _logIfLost(delivery, error, details) {
351
+ if (!delivery.isLastAttempt) return;
352
+ log.error(
353
+ 'Rate-limited record lost: the message goes to the dead-letter queue next',
354
+ {
355
+ eventName: 'frigg.worker.record_lost_rate_limited',
356
+ retryAt: details.retryAt,
357
+ reason: error.reason,
358
+ module: error.module,
359
+ error,
360
+ }
361
+ );
362
+ }
363
+
78
364
  // returns the message id
79
365
  async send(params, delay = 0) {
80
366
  this._validateParams(params);
@@ -58,7 +58,7 @@ class FetchError extends BaseError {
58
58
  const provided = options.responseBody ?? options.body;
59
59
  let responseBody = provided;
60
60
  if (
61
- !responseBody &&
61
+ responseBody === undefined &&
62
62
  response &&
63
63
  !response.bodyUsed &&
64
64
  typeof response.text === 'function'
@@ -165,6 +165,32 @@ const createQueueWorker = (integrationClass) => {
165
165
  const integrationName = integrationClass.Definition.name;
166
166
 
167
167
  class QueueWorker extends Worker {
168
+ async recordRateLimitWait(body, error, state) {
169
+ const processId = body.data?.processId;
170
+ if (!processId) return;
171
+ await createProcessRepository().applyProcessUpdate(processId, {
172
+ set: {
173
+ 'context.rateLimit': {
174
+ status: state.status,
175
+ mechanism: state.mechanism,
176
+ retryAt: state.retryAt.toISOString(),
177
+ reason: error.reason,
178
+ module: error.module,
179
+ deferrals: state.deferrals,
180
+ updatedAt: new Date().toISOString(),
181
+ },
182
+ },
183
+ });
184
+ }
185
+
186
+ async clearRateLimitWait(body) {
187
+ const processId = body?.data?.processId;
188
+ if (!processId) return;
189
+ await createProcessRepository().applyProcessUpdate(processId, {
190
+ set: { 'context.rateLimit': null },
191
+ });
192
+ }
193
+
168
194
  async _run(params, context, delivery) {
169
195
  const logCtx = {
170
196
  integration: integrationName,
@@ -398,6 +398,12 @@ class OAuth2Requester extends Requester {
398
398
  );
399
399
  transportError.statusCode = status;
400
400
  transportError.isTokenRefreshTransportFailure = true;
401
+ if (error?.isRateLimited) {
402
+ transportError.isRateLimited = true;
403
+ transportError.retryAt = error.retryAt;
404
+ transportError.waitMs = error.waitMs;
405
+ transportError.reason = error.reason;
406
+ }
401
407
  return transportError;
402
408
  }
403
409
 
@@ -90,7 +90,7 @@ function hintFromWait(waitMs, now, { source = 'header', ...extra } = {}) {
90
90
  * Builds a hint from an absolute time. A time in the past waits 0 ms.
91
91
  */
92
92
  function hintFromRetryAt(retryAt, now, { source = 'header', ...extra } = {}) {
93
- const at = retryAt instanceof Date ? retryAt.getTime() : NaN;
93
+ const at = retryAt instanceof Date ? retryAt.getTime() : Number.NaN;
94
94
  if (Number.isNaN(at)) return null;
95
95
  const waitMs = Math.max(0, at - now);
96
96
  if (waitMs > MAX_HINT_WAIT_MS) return null;
@@ -216,7 +216,7 @@ function pickStructured(items) {
216
216
 
217
217
  function readKeyValue(text, key) {
218
218
  const match = new RegExp(
219
- `(?:^|[,;\\s])${key}\\s*=\\s*([^,;\\s]+)`,
219
+ String.raw`(?:^|[,;\s])${key}\s*=\s*([^,;\s]+)`,
220
220
  'i'
221
221
  ).exec(text);
222
222
  return match ? match[1] : undefined;
@@ -243,8 +243,10 @@ function classifyRateLimit(policy, signal, options = {}) {
243
243
  }
244
244
 
245
245
  /**
246
- * Like classifyRateLimit, but always answers: with no hint it returns the
247
- * step of the fixed backoff ladder, with source "backoff".
246
+ * Like classifyRateLimit, but a throttled response always gets a hint: with
247
+ * none found it gets the step of the fixed backoff ladder, with source
248
+ * "backoff". Null when the response is not throttled: a status other than 429
249
+ * that classify did not recognise.
248
250
  */
249
251
  function resolveRateLimitHint(policy, signal, options = {}) {
250
252
  const {
@@ -258,6 +260,7 @@ function resolveRateLimitHint(policy, signal, options = {}) {
258
260
  onClassifyError,
259
261
  });
260
262
  if (hint) return hint;
263
+ if (signal.status !== 429 && !classified) return null;
261
264
 
262
265
  const waitMs =
263
266
  attempt < backOff.length ? Number(backOff[attempt]) * 1000 || 0 : 0;
@@ -10,7 +10,6 @@ const { redactUrl } = require('../../logs/redact');
10
10
  const { toSanitizedSurrogate } = require('../../logs/serialize');
11
11
  const { getLoggerScope } = require('../../logs/context');
12
12
  const {
13
- classifyRateLimit,
14
13
  computeScopeKey,
15
14
  computeWaitMs,
16
15
  headerValue,
@@ -20,7 +19,18 @@ const {
20
19
  } = require('./rate-limit');
21
20
 
22
21
  const DEFAULT_REQUEST_TIMEOUT_MS = 60_000;
23
- const JSON_CONTENT_TYPE = /^application\/(json|vnd\.api\+json|hal\+json)/;
22
+
23
+ function isJsonMediaType(contentType) {
24
+ const mediaType = String(contentType ?? '')
25
+ .split(';')[0]
26
+ .trim()
27
+ .toLowerCase();
28
+ return (
29
+ mediaType === 'application/json' ||
30
+ mediaType === 'text/json' ||
31
+ mediaType.endsWith('+json')
32
+ );
33
+ }
24
34
 
25
35
  // A node-fetch error message holds the raw URL, and util.inspect prints the
26
36
  // cause chain, so the FetchError keeps only a sanitized copy.
@@ -46,6 +56,7 @@ class Requester extends Delegate {
46
56
  super(params);
47
57
  this.backOff = get(params, 'backOff', [1, 3, 10, 30, 60, 180]);
48
58
  this._rateLimitPolicy = readRateLimitPolicy(this.constructor);
59
+ this._random = params?.random ?? Math.random;
49
60
  this.isRefreshable = false;
50
61
  this.refreshCount = 0;
51
62
  this.authGraceRetryCount = 0;
@@ -294,7 +305,12 @@ class Requester extends Delegate {
294
305
  clearRequestTimer();
295
306
  const delay = this.backOff[attempt] * 1000;
296
307
  await new Promise((resolve) => setTimeout(resolve, delay));
297
- return this._rawRequest(url, options, attempt + 1);
308
+ return this._rawRequest(
309
+ url,
310
+ options,
311
+ attempt + 1,
312
+ waitedMs
313
+ );
298
314
  }
299
315
  const fetchError = await FetchError.create({
300
316
  resource: encodedUrl,
@@ -318,70 +334,27 @@ class Requester extends Delegate {
318
334
  status,
319
335
  attempt
320
336
  );
321
- if (throttle?.throttled) {
322
- const { hint } = throttle;
323
- if (hint.source === 'backoff') {
324
- if (attempt < this.backOff.length) {
325
- clearRequestTimer();
326
- const delay = this.backOff[attempt] * 1000;
327
- await new Promise((resolve) =>
328
- setTimeout(resolve, delay)
329
- );
330
- return this._rawRequest(
331
- url,
332
- options,
333
- attempt + 1,
334
- waitedMs
335
- );
336
- }
337
- } else {
338
- const budgetMs = inProcessBudgetMs({
339
- policy: this._rateLimitPolicy,
340
- requestTimeoutMs: this.requestTimeoutMs,
341
- remainingMs: remainingInvocationMs(),
342
- waitedMs,
343
- });
344
- const { waitMs, fits } = computeWaitMs({
345
- hint,
346
- policy: this._rateLimitPolicy,
347
- budgetMs,
348
- });
349
- this._logRateLimited({
350
- status,
351
- hint,
352
- waitMs,
353
- attempt,
354
- waitedMs,
355
- action: fits ? 'wait' : 'throw',
356
- });
357
- if (fits) {
358
- clearRequestTimer();
359
- await new Promise((resolve) =>
360
- setTimeout(resolve, waitMs)
361
- );
362
- return this._rawRequest(
363
- url,
364
- options,
365
- attempt + 1,
366
- waitedMs + waitMs
367
- );
368
- }
369
- this._logRequestFailed(encodedUrl, options, status);
370
- const rateLimitError = await RateLimitError.create({
371
- resource: encodedUrl,
372
- init: options,
373
- response,
374
- responseBody: throttle.responseBody,
375
- hint,
376
- waitMs,
377
- module: this._telemetryModuleLabel(),
378
- scopeKey: computeScopeKey(this._rateLimitPolicy, this),
379
- });
380
- throw this._maybeFlagTimeoutDuringBodyRead(
381
- rateLimitError,
382
- timeoutMs
383
- );
384
- }
337
+ const throttleRetry = await this._throttleRetry({
338
+ throttle,
339
+ status,
340
+ attempt,
341
+ waitedMs,
342
+ encodedUrl,
343
+ options,
344
+ response,
345
+ timeoutMs,
346
+ });
347
+ if (throttleRetry) {
348
+ clearRequestTimer();
349
+ await new Promise((resolve) =>
350
+ setTimeout(resolve, throttleRetry.delayMs)
351
+ );
352
+ return this._rawRequest(
353
+ url,
354
+ options,
355
+ attempt + 1,
356
+ waitedMs + throttleRetry.hintedMs
357
+ );
385
358
  }
386
359
 
387
360
  // If the status is retriable and there are back off requests left, retry the request
@@ -389,7 +362,7 @@ class Requester extends Delegate {
389
362
  clearRequestTimer();
390
363
  const delay = this.backOff[attempt] * 1000;
391
364
  await new Promise((resolve) => setTimeout(resolve, delay));
392
- return this._rawRequest(url, options, attempt + 1);
365
+ return this._rawRequest(url, options, attempt + 1, waitedMs);
393
366
  }
394
367
 
395
368
  if (status === 401) {
@@ -410,7 +383,12 @@ class Requester extends Delegate {
410
383
  // This request did not try the current token. Retry with
411
384
  // it. Do not spend one more provider-side rotation.
412
385
  clearRequestTimer();
413
- return this._rawRequest(url, options, attempt + 1);
386
+ return this._rawRequest(
387
+ url,
388
+ options,
389
+ attempt + 1,
390
+ waitedMs
391
+ );
414
392
  }
415
393
 
416
394
  if (!this.isRefreshable) {
@@ -427,7 +405,12 @@ class Requester extends Delegate {
427
405
  await new Promise((resolve) =>
428
406
  setTimeout(resolve, delay)
429
407
  );
430
- return this._rawRequest(url, options, attempt + 1);
408
+ return this._rawRequest(
409
+ url,
410
+ options,
411
+ attempt + 1,
412
+ waitedMs
413
+ );
431
414
  }
432
415
 
433
416
  throw await this._invalidateAuth(
@@ -460,7 +443,12 @@ class Requester extends Delegate {
460
443
  const refreshSucceeded = await this._refreshAuthOnce();
461
444
  if (refreshSucceeded) {
462
445
  clearRequestTimer();
463
- return this._rawRequest(url, options, attempt + 1);
446
+ return this._rawRequest(
447
+ url,
448
+ options,
449
+ attempt + 1,
450
+ waitedMs
451
+ );
464
452
  }
465
453
 
466
454
  throw await this._invalidateAuth(encodedUrl, options, response);
@@ -477,6 +465,10 @@ class Requester extends Delegate {
477
465
  response,
478
466
  responseBody: throttle?.responseBody,
479
467
  });
468
+ if (throttle?.throttled && status !== 429) {
469
+ fetchError.isRateLimited = true;
470
+ fetchError.reason = throttle.hint.reason;
471
+ }
480
472
  throw this._maybeFlagTimeoutDuringBodyRead(
481
473
  fetchError,
482
474
  timeoutMs
@@ -505,6 +497,69 @@ class Requester extends Delegate {
505
497
  }
506
498
  }
507
499
 
500
+ /**
501
+ * What to do about a throttled response: the delay before the next
502
+ * attempt, or null when there is no attempt left. A wait that a hint set
503
+ * and that does not fit the budget throws RateLimitError.
504
+ *
505
+ * @returns {Promise<{delayMs: number, hintedMs: number}|null>}
506
+ * `hintedMs` is the part of the delay that a provider or a policy set,
507
+ * which counts against the in-process cap.
508
+ */
509
+ async _throttleRetry({
510
+ throttle,
511
+ status,
512
+ attempt,
513
+ waitedMs,
514
+ encodedUrl,
515
+ options,
516
+ response,
517
+ timeoutMs,
518
+ }) {
519
+ if (!throttle?.throttled) return null;
520
+ const { hint } = throttle;
521
+ if (hint.source === 'backoff') {
522
+ return attempt < this.backOff.length
523
+ ? { delayMs: this.backOff[attempt] * 1000, hintedMs: 0 }
524
+ : null;
525
+ }
526
+
527
+ const budgetMs = inProcessBudgetMs({
528
+ policy: this._rateLimitPolicy,
529
+ requestTimeoutMs: this.requestTimeoutMs,
530
+ remainingMs: remainingInvocationMs(),
531
+ waitedMs,
532
+ });
533
+ const { waitMs, fits } = computeWaitMs({
534
+ hint,
535
+ policy: this._rateLimitPolicy,
536
+ budgetMs,
537
+ random: this._random,
538
+ });
539
+ this._logRateLimited({
540
+ status,
541
+ hint,
542
+ waitMs,
543
+ attempt,
544
+ waitedMs,
545
+ action: fits ? 'wait' : 'throw',
546
+ });
547
+ if (fits) return { delayMs: waitMs, hintedMs: waitMs };
548
+
549
+ this._logRequestFailed(encodedUrl, options, status);
550
+ const rateLimitError = await RateLimitError.create({
551
+ resource: encodedUrl,
552
+ init: options,
553
+ response,
554
+ responseBody: throttle.responseBody,
555
+ hint,
556
+ waitMs,
557
+ module: this._telemetryModuleLabel(),
558
+ scopeKey: computeScopeKey(this._rateLimitPolicy, this),
559
+ });
560
+ throw this._maybeFlagTimeoutDuringBodyRead(rateLimitError, timeoutMs);
561
+ }
562
+
508
563
  /**
509
564
  * Decides whether a response says a limit was hit. A 429 always does. A
510
565
  * 4xx or 5xx other than 401 does only when the module's classify() names
@@ -528,21 +583,17 @@ class Requester extends Delegate {
528
583
  await this._readBodyForClassify(response));
529
584
  }
530
585
 
531
- const signal = { status, headers: response.headers, body };
532
- const options = {
533
- attempt,
534
- backOff: this.backOff,
535
- now: Date.now(),
536
- onClassifyError: (error) => this._logClassifyFailed(status, error),
537
- };
538
- if (status === 429) {
539
- return {
540
- throttled: true,
541
- hint: resolveRateLimitHint(policy, signal, options),
542
- responseBody,
543
- };
544
- }
545
- const hint = classifyRateLimit(policy, signal, options);
586
+ const hint = resolveRateLimitHint(
587
+ policy,
588
+ { status, headers: response.headers, body },
589
+ {
590
+ attempt,
591
+ backOff: this.backOff,
592
+ now: Date.now(),
593
+ onClassifyError: (error) =>
594
+ this._logClassifyFailed(status, error),
595
+ }
596
+ );
546
597
  return { throttled: Boolean(hint), hint, responseBody };
547
598
  }
548
599
 
@@ -551,11 +602,7 @@ class Requester extends Delegate {
551
602
  return {};
552
603
  }
553
604
  const text = await response.text();
554
- if (
555
- !JSON_CONTENT_TYPE.test(
556
- headerValue(response.headers, 'content-type') || ''
557
- )
558
- ) {
605
+ if (!isJsonMediaType(headerValue(response.headers, 'content-type'))) {
559
606
  return { text };
560
607
  }
561
608
  try {
package/package.json CHANGED
@@ -1,12 +1,13 @@
1
1
  {
2
2
  "name": "@friggframework/core",
3
3
  "prettier": "@friggframework/prettier-config",
4
- "version": "2.0.0--canary.656.10f1676.0",
4
+ "version": "2.0.0--canary.657.8f335eb.0",
5
5
  "dependencies": {
6
6
  "@aws-sdk/client-apigatewaymanagementapi": "^3.588.0",
7
7
  "@aws-sdk/client-kms": "^3.588.0",
8
8
  "@aws-sdk/client-lambda": "^3.714.0",
9
9
  "@aws-sdk/client-s3": "^3.588.0",
10
+ "@aws-sdk/client-scheduler": "^3.588.0",
10
11
  "@aws-sdk/client-sqs": "^3.588.0",
11
12
  "@aws-sdk/client-ssm": "^3.588.0",
12
13
  "@aws-sdk/s3-request-presigner": "^3.588.0",
@@ -47,9 +48,9 @@
47
48
  }
48
49
  },
49
50
  "devDependencies": {
50
- "@friggframework/eslint-config": "2.0.0--canary.656.10f1676.0",
51
- "@friggframework/prettier-config": "2.0.0--canary.656.10f1676.0",
52
- "@friggframework/test": "2.0.0--canary.656.10f1676.0",
51
+ "@friggframework/eslint-config": "2.0.0--canary.657.8f335eb.0",
52
+ "@friggframework/prettier-config": "2.0.0--canary.657.8f335eb.0",
53
+ "@friggframework/test": "2.0.0--canary.657.8f335eb.0",
53
54
  "@prisma/client": "^6.19.3",
54
55
  "@types/lodash": "4.17.15",
55
56
  "@typescript-eslint/eslint-plugin": "^8.0.0",
@@ -89,5 +90,5 @@
89
90
  "publishConfig": {
90
91
  "access": "public"
91
92
  },
92
- "gitHead": "10f1676a00a793a52a2f0a6edefb4d08ec677a15"
93
+ "gitHead": "8f335eb16f678038ad9448d9603c734ad663b3dc"
93
94
  }
@@ -0,0 +1,79 @@
1
+ const MAX_DEFERRALS = 10;
2
+ const MAX_DEFERRED_MS = 24 * 60 * 60 * 1000;
3
+ const MAX_DELAY_SECONDS = 900;
4
+ const MAX_VISIBILITY_TIMEOUT_SECONDS = 43_200;
5
+ const MAX_DEFERRALS_ENV = 'FRIGG_QUEUE_MAX_DEFERRALS';
6
+ const MAX_DEFERRED_MS_ENV = 'FRIGG_QUEUE_MAX_DEFERRED_MS';
7
+
8
+ const toPositiveInteger = (value) => {
9
+ const number = Number(value);
10
+ return Number.isInteger(number) && number > 0 ? number : undefined;
11
+ };
12
+
13
+ /**
14
+ * @typedef {Object} DeferralLimits
15
+ * @property {number} maxDeferrals Times one message may be put back for a rate limit.
16
+ * @property {number} maxDeferredMs Longest total time from the first deferral to the last retry time.
17
+ */
18
+
19
+ /**
20
+ * @param {Object} [env=process.env]
21
+ * @returns {DeferralLimits}
22
+ */
23
+ const readDeferralLimits = (env = process.env) => ({
24
+ maxDeferrals: toPositiveInteger(env[MAX_DEFERRALS_ENV]) ?? MAX_DEFERRALS,
25
+ maxDeferredMs:
26
+ toPositiveInteger(env[MAX_DEFERRED_MS_ENV]) ?? MAX_DEFERRED_MS,
27
+ });
28
+
29
+ /**
30
+ * The deferral counters a message body carries in `_frigg`.
31
+ * @returns {{deferrals: number, firstDeferredAt: string|undefined}}
32
+ */
33
+ const readDeferral = (body) => {
34
+ const frigg = body?._frigg;
35
+ const deferrals =
36
+ Number.isInteger(frigg?.deferrals) && frigg.deferrals > 0
37
+ ? frigg.deferrals
38
+ : 0;
39
+ const firstDeferredAt =
40
+ typeof frigg?.firstDeferredAt === 'string' &&
41
+ !Number.isNaN(Date.parse(frigg.firstDeferredAt))
42
+ ? frigg.firstDeferredAt
43
+ : undefined;
44
+ return { deferrals, firstDeferredAt };
45
+ };
46
+
47
+ const nextDeferral = (body, now = Date.now()) => {
48
+ const current = readDeferral(body);
49
+ return {
50
+ deferrals: current.deferrals + 1,
51
+ firstDeferredAt: current.firstDeferredAt ?? new Date(now).toISOString(),
52
+ };
53
+ };
54
+
55
+ const withDeferral = (body, deferral) => ({
56
+ ...body,
57
+ _frigg: { ...body._frigg, ...deferral },
58
+ });
59
+
60
+ const isDeferralCapped = (
61
+ { deferrals, firstDeferredAt, retryAt },
62
+ { maxDeferrals, maxDeferredMs }
63
+ ) =>
64
+ deferrals > maxDeferrals ||
65
+ retryAt.getTime() - Date.parse(firstDeferredAt) > maxDeferredMs;
66
+
67
+ module.exports = {
68
+ MAX_DEFERRALS,
69
+ MAX_DEFERRALS_ENV,
70
+ MAX_DEFERRED_MS,
71
+ MAX_DEFERRED_MS_ENV,
72
+ MAX_DELAY_SECONDS,
73
+ MAX_VISIBILITY_TIMEOUT_SECONDS,
74
+ isDeferralCapped,
75
+ nextDeferral,
76
+ readDeferral,
77
+ readDeferralLimits,
78
+ withDeferral,
79
+ };
@@ -126,4 +126,4 @@ const QueuerUtil = {
126
126
  },
127
127
  };
128
128
 
129
- module.exports = { QueuerUtil };
129
+ module.exports = { QueuerUtil, awsConfigOptions };
@@ -47,8 +47,28 @@ declare module "@friggframework/core" {
47
47
  send(params: object & { QueueUrl: any }, delay?: number): Promise<string>;
48
48
 
49
49
  sendAsyncSQSMessage(params: SendSQSMessageParams): Promise<string>;
50
+
51
+ /**
52
+ * Hook: a message was put back for a rate limit. Runs after the message
53
+ * was sent. A failure is logged and never changes the outcome.
54
+ */
55
+ recordRateLimitWait(
56
+ body: object,
57
+ error: Error,
58
+ state: RateLimitWaitState
59
+ ): Promise<void>;
60
+
61
+ /** Hook: a message that was put back for a rate limit ran without error. */
62
+ clearRateLimitWait(body: object): Promise<void>;
50
63
  }
51
64
 
65
+ export type RateLimitWaitState = {
66
+ status: "WAITING" | "EXHAUSTED";
67
+ mechanism: "delay" | "schedule" | "visibility" | "none";
68
+ deferrals: number;
69
+ retryAt: Date;
70
+ };
71
+
52
72
  interface IWorker {
53
73
  getQueueURL(params: GetQueueURLParams): Promise<string | undefined>;
54
74
  run(params: { Records: any }, context?: object): Promise<BatchItemFailuresResponse>;
@@ -16,6 +16,12 @@ declare module "@friggframework/errors" {
16
16
  readonly body: any;
17
17
  isTimeout?: boolean;
18
18
  timeoutMs?: number;
19
+ /**
20
+ * True when `classify` named the response as a limit but gave no time. A
21
+ * `RateLimitError` always has it.
22
+ */
23
+ isRateLimited?: boolean;
24
+ reason?: RateLimitReason;
19
25
 
20
26
  static create(options?: CreateFetchErrorParams): Promise<FetchError>;
21
27
  }