@ahrowe/mongo 0.1.2 → 0.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +19 -0
- package/README.md +50 -20
- package/dist/db.d.ts +31 -13
- package/dist/db.d.ts.map +1 -1
- package/dist/db.js +177 -156
- package/dist/db.js.map +1 -1
- package/dist/index.d.ts +2 -2
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +1 -1
- package/dist/index.js.map +1 -1
- package/dist/outboxRelay.d.ts +34 -5
- package/dist/outboxRelay.d.ts.map +1 -1
- package/dist/outboxRelay.js +75 -52
- package/dist/outboxRelay.js.map +1 -1
- package/dist/types/outbox.d.ts +14 -2
- package/dist/types/outbox.d.ts.map +1 -1
- package/dist/types/outbox.js +4 -1
- package/dist/types/outbox.js.map +1 -1
- package/docs/CLAUDE.md +116 -86
- package/package.json +5 -2
- package/dist/batchify.d.ts +0 -36
- package/dist/batchify.d.ts.map +0 -1
- package/dist/batchify.js +0 -57
- package/dist/batchify.js.map +0 -1
package/docs/CLAUDE.md
CHANGED
|
@@ -35,8 +35,8 @@ Every `createService<T>(...)` call returns:
|
|
|
35
35
|
(per-call, or as a `createService` option for a service-wide default) to enforce
|
|
36
36
|
one. `findOne` throws if more than one document matches.
|
|
37
37
|
- `insert(doc | doc[], options?)` — generates `_id` and `createdOn` if not supplied;
|
|
38
|
-
runs the schema validator (if any) unless `{ skipValidation: true }`;
|
|
39
|
-
transaction
|
|
38
|
+
runs the schema validator (if any) unless `{ skipValidation: true }`; runs in a
|
|
39
|
+
transaction (see "Transactions" below). Enqueues a `'created'` outbox row
|
|
40
40
|
(in the same transaction as the write) for each inserted document.
|
|
41
41
|
- `upsertOne(query, updateBody, options?)` — create-or-update a single document.
|
|
42
42
|
If the query matches, behaves like `updateOne` (enqueues `'updated'`). If not,
|
|
@@ -49,13 +49,14 @@ Every `createService<T>(...)` call returns:
|
|
|
49
49
|
is rejected**; use `upsertOne` for create-or-update, `insert` for explicit creates. Adds `updatedOn: new Date()` to the `$set`
|
|
50
50
|
unless `addUpdatedOnField: false` was passed to `createService`. Enqueues an
|
|
51
51
|
`'updated'` outbox row with `{ prevDoc, doc, meta }` for each document that
|
|
52
|
-
actually changed. `updateMany
|
|
53
|
-
"Events are delivered via a transactional outbox" below).
|
|
52
|
+
actually changed. For `updateMany`, see "How updateMany/removeMany work" below.
|
|
54
53
|
- `removeOne(query, options?)` and `removeMany(query, options?)` — `removeMany`
|
|
55
54
|
returns removed docs under `results` unless `{ returnRemoved: false }`. Enqueues
|
|
56
55
|
a `'removed'` outbox row with `{ doc, meta }` for each removed document (no row
|
|
57
|
-
if nothing matched). `removeMany
|
|
58
|
-
|
|
56
|
+
if nothing matched). For `removeMany`, see "How updateMany/removeMany work" below.
|
|
57
|
+
- Write methods reject `projection` and `includeResultMetadata` (types omit them,
|
|
58
|
+
and they throw at runtime): validation and outbox events need the full
|
|
59
|
+
document. Read the fields you need with `find`/`findOne` afterwards.
|
|
59
60
|
- `createIndex(fields, options?)` — `fields` is `IndexSpec<T>`: any top-level key of
|
|
60
61
|
`T` or a dotted path into it (`'users._id'`), following arrays into their element
|
|
61
62
|
type, mapped to `IndexDirection` (`1 | -1 | 'text' | 'hashed' | '2dsphere' | '2d'`).
|
|
@@ -65,12 +66,13 @@ Every `createService<T>(...)` call returns:
|
|
|
65
66
|
having to enumerate that far.
|
|
66
67
|
- `count`, `exists`, `aggregate`, `distinct`, `dropIndex`, `bulkWrite`
|
|
67
68
|
(requires `{ allowBulkwrite: true }` — guards against accidental unvalidated writes).
|
|
68
|
-
`bulkWrite`
|
|
69
|
+
`bulkWrite` writes no outbox rows, so it throws on a service that emits them.
|
|
70
|
+
`dropIndex(indexName, options?)` drops
|
|
69
71
|
a single index by name (not by key spec) and does not enqueue outbox rows.
|
|
70
|
-
- `on(event, cb)` / `onPropertiesUpdated(properties, cb)
|
|
71
|
-
|
|
72
|
-
|
|
73
|
-
itself.
|
|
72
|
+
- `on(event, cb)` / `onPropertiesUpdated(properties, cb)`: register a listener
|
|
73
|
+
for `'created' | 'updated' | 'removed'` on this collection. Returns a function
|
|
74
|
+
that removes it. Listeners are called by `startOutboxRelay` (or
|
|
75
|
+
`dispatchEvent`), not by the write call itself.
|
|
74
76
|
- `withSession(cb)` / `startSession(options?)` — exported from the module (not
|
|
75
77
|
per-service) for manual multi-document transactions across several services.
|
|
76
78
|
- `disconnect()` — exported from the module; closes both MongoDB clients.
|
|
@@ -94,8 +96,9 @@ The contract is validator-agnostic — plug in Zod (`schema.safeParse(entity)`),
|
|
|
94
96
|
directly — writing to the data collection can't be suppressed per-call, so for
|
|
95
97
|
each document actually affected, one row is written to an internal `outbox`
|
|
96
98
|
collection **in the same transaction** as the data write. A separate process,
|
|
97
|
-
`startOutboxRelay({ instanceId, ... })`, claims pending rows (atomic
|
|
98
|
-
`findOneAndUpdate
|
|
99
|
+
`startOutboxRelay({ instanceId, ... })`, claims pending rows (an atomic
|
|
100
|
+
`findOneAndUpdate` that pushes `nextAttemptAt` past a lease, safe across
|
|
101
|
+
multiple pods) and dispatches each to whatever's
|
|
99
102
|
registered via `.on(event, cb)` on that collection's service, retrying with
|
|
100
103
|
backoff and dead-lettering after `maxAttempts`. This means delivery is durable —
|
|
101
104
|
it survives a crash between the write committing and a listener running — and
|
|
@@ -108,8 +111,10 @@ const relay = startOutboxRelay({
|
|
|
108
111
|
instanceId: process.env.HOSTNAME ?? 'local', // required — identifies this lease holder
|
|
109
112
|
pollIntervalMs: 500, // idle poll interval when the outbox is empty
|
|
110
113
|
batchSize: 20, // rows claimed+dispatched per tick
|
|
111
|
-
leaseMs: 30_000, // how long a
|
|
114
|
+
leaseMs: 30_000, // how long a claimed row stays hidden from other claims
|
|
112
115
|
maxAttempts: 10, // attempts before a row is dead-lettered to 'failed'
|
|
116
|
+
keepDoneMs: 7 * 24 * 60 * 60 * 1000, // how long 'done' rows are kept; Infinity = forever
|
|
117
|
+
keepFailedMs: 90 * 24 * 60 * 60 * 1000, // how long 'failed' rows are kept for inspection/replay; Infinity = forever
|
|
113
118
|
// backoffMs: (attempt) => ms, // override the default capped-exponential backoff
|
|
114
119
|
// autoStart: false, // don't start the poll loop — drive tick()/drain() manually (useful in tests)
|
|
115
120
|
});
|
|
@@ -119,6 +124,15 @@ const relay = startOutboxRelay({
|
|
|
119
124
|
Run exactly one (or more — it's safe across multiple pods) relay per
|
|
120
125
|
deployment that needs events delivered; nothing dispatches them otherwise.
|
|
121
126
|
|
|
127
|
+
A relay only claims rows for collections that have a service (`createService`)
|
|
128
|
+
in its own process. Rows for any other collection stay `pending` until a relay
|
|
129
|
+
that has that service picks them up, so a relay process without a given
|
|
130
|
+
service can't swallow its events.
|
|
131
|
+
|
|
132
|
+
A listener must finish within `leaseMs`. Past that, another instance may
|
|
133
|
+
reclaim the row and run it again; the late holder's outcome is then discarded
|
|
134
|
+
(with a warning) instead of overwriting the newer claim's.
|
|
135
|
+
|
|
122
136
|
Each listener receives `{ prevDoc, doc, meta, context, eventId }` — not just
|
|
123
137
|
`{ prevDoc, doc, meta }`. `eventId` is the outbox row's `_id`, stable across
|
|
124
138
|
retries of the same row; since delivery is **at-least-once**, use it as an
|
|
@@ -134,6 +148,16 @@ Set `{ emitOutboxEvents: false }` on `createService` for infra collections with
|
|
|
134
148
|
no listeners (sequences, the outbox itself, etc.) — writes then take a leaner
|
|
135
149
|
fast path with no pre-reads and no transaction-wrapped outbox insert.
|
|
136
150
|
|
|
151
|
+
Set `{ toOutboxDoc: (doc) => ... }` on `createService` to decide what a
|
|
152
|
+
collection's outbox rows hold of a document, `doc` and `prevDoc` alike. Rows
|
|
153
|
+
reach every listener, and whatever those store (an audit log, a broker), so a
|
|
154
|
+
secret in a row leaks there. `({ secret, ...hook }) => hook` leaves it out (a
|
|
155
|
+
misspelled field is a type error), and an update that changes only that field
|
|
156
|
+
records no row. To have the change show without the value, return a hash of it
|
|
157
|
+
instead. The function must be pure and must not modify the document it gets. It
|
|
158
|
+
also gets stored documents that were never validated against `T` (older ones,
|
|
159
|
+
`skipValidation` writes), so handle a missing field: a throw fails the write.
|
|
160
|
+
|
|
137
161
|
If a listener needs request/actor context (e.g. an audit log), register
|
|
138
162
|
`setOutboxContextProvider(fn)` once at startup. It's called **synchronously at
|
|
139
163
|
write time**, in the same transaction as the data write, and the captured value
|
|
@@ -154,15 +178,14 @@ const failed = await getOutboxCollection().find({ status: OutboxStatus.Failed })
|
|
|
154
178
|
// inspect failed[i].lastError, fix the underlying issue, then requeue:
|
|
155
179
|
await getOutboxCollection().updateOne(
|
|
156
180
|
{ _id: failed[0]._id },
|
|
157
|
-
{ $set: { status: OutboxStatus.Pending, nextAttemptAt: new Date(), attempts: 0, lastError: null } },
|
|
181
|
+
{ $set: { status: OutboxStatus.Pending, nextAttemptAt: new Date(), attempts: 0, lastError: null, expireAt: null } },
|
|
158
182
|
);
|
|
159
183
|
```
|
|
160
184
|
|
|
161
|
-
`
|
|
162
|
-
|
|
163
|
-
|
|
164
|
-
|
|
165
|
-
service.
|
|
185
|
+
`getRegisteredCollections()` lists the collections with a service in this
|
|
186
|
+
process (the ones the relay claims rows for). Reach for it only if you're
|
|
187
|
+
building custom tooling around the outbox; normal listener registration is just
|
|
188
|
+
`.on(...)` on the service.
|
|
166
189
|
|
|
167
190
|
### Multiple consumers
|
|
168
191
|
|
|
@@ -179,47 +202,59 @@ move is to bridge: run one relay whose only job is to republish each row to a
|
|
|
179
202
|
message broker (Kafka, SNS, SQS, etc.), and let downstream services subscribe
|
|
180
203
|
there. The outbox guarantees the write reaches the broker exactly once; the
|
|
181
204
|
broker handles fan-out from there. That bridging code lives in your app — this
|
|
182
|
-
package has no broker dependency by design.
|
|
205
|
+
package has no broker dependency by design. On the consuming side, hand each
|
|
206
|
+
received event to `dispatchEvent(collectionName, event)`: it calls the
|
|
207
|
+
listeners registered via `.on(...)` exactly like the relay does (awaits all of
|
|
208
|
+
them, rejects with the first failure so your consumer can retry).
|
|
183
209
|
|
|
184
210
|
Until a second independent consumer appears, don't add the broker. Bundle the
|
|
185
211
|
relay and all listeners in the same process.
|
|
186
212
|
|
|
187
213
|
### Outbox row retention
|
|
188
214
|
|
|
189
|
-
|
|
190
|
-
**7 days**
|
|
191
|
-
|
|
192
|
-
|
|
193
|
-
|
|
194
|
-
|
|
195
|
-
|
|
196
|
-
|
|
197
|
-
|
|
198
|
-
|
|
199
|
-
|
|
200
|
-
|
|
201
|
-
|
|
202
|
-
|
|
203
|
-
|
|
204
|
-
|
|
205
|
-
|
|
206
|
-
|
|
207
|
-
|
|
208
|
-
|
|
209
|
-
|
|
210
|
-
(
|
|
211
|
-
|
|
212
|
-
|
|
213
|
-
|
|
214
|
-
|
|
215
|
+
When a row is marked `done` or `failed`, the relay sets its `expireAt` from
|
|
216
|
+
`keepDoneMs` (default **7 days**) or `keepFailedMs` (default **90 days**), and a
|
|
217
|
+
TTL index deletes it after that; `Infinity` keeps rows forever. MongoDB's TTL
|
|
218
|
+
task runs every 60 seconds, so deletion isn't instant. Retention is stored per
|
|
219
|
+
row, so changing it needs no index migration; it applies to rows settled from
|
|
220
|
+
then on. A row no relay processes stays until one does. When requeuing a failed row by hand, clear
|
|
221
|
+
`expireAt` (as in the snippet above), or the TTL index may delete it while it
|
|
222
|
+
waits.
|
|
223
|
+
|
|
224
|
+
## How updateMany/removeMany work
|
|
225
|
+
|
|
226
|
+
Both collect the matched `_id`s, then work through them in chunks of 500 inside a
|
|
227
|
+
single transaction: per chunk, one read of the current documents, one native
|
|
228
|
+
`updateMany`/`deleteMany` by `_id`, and (for `updateMany`) one read of the
|
|
229
|
+
results, plus one outbox insert. A failure in any chunk rolls back every chunk,
|
|
230
|
+
data **and** outbox rows. Every updated document is validated, and each outbox
|
|
231
|
+
row's `{ prevDoc, doc }` pair is exact: once the transaction has read a document,
|
|
232
|
+
any concurrent write to it makes the transaction fail with a transient write
|
|
233
|
+
conflict, and the whole call is retried.
|
|
234
|
+
|
|
235
|
+
With `emitOutboxEvents: false`, `updateMany` still validates every document
|
|
236
|
+
(skipping the `prevDoc` reads and outbox rows); only without a validator, or
|
|
237
|
+
with `{ skipValidation: true }`, does it become one native `updateMany`.
|
|
238
|
+
|
|
239
|
+
There is **no size cap by default**. Pass `{ maxBatchSize }` (per-call, or as a
|
|
240
|
+
`createService` option) to fail fast: a single transaction processing a very
|
|
241
|
+
large matched set risks MongoDB's transaction lifetime limit (default 60s).
|
|
242
|
+
Past the cap it throws before writing anything; it never processes a partial
|
|
243
|
+
batch, so it can't be used to loop. For more than fits one transaction, split
|
|
244
|
+
the work into several calls (see the README for a rerunnable migration loop).
|
|
215
245
|
|
|
216
246
|
## Transactions
|
|
217
247
|
|
|
218
|
-
|
|
219
|
-
|
|
220
|
-
|
|
221
|
-
|
|
222
|
-
|
|
248
|
+
Write methods always run in a transaction, so a data write and its outbox row
|
|
249
|
+
commit together. If the passed `{ session }` has an active transaction, the call
|
|
250
|
+
joins it; otherwise the call runs in its own (on that session, if one was passed).
|
|
251
|
+
To compose several calls, even across different services/collections, into one
|
|
252
|
+
atomic transaction, pass a session **inside** `session.withTransaction(...)`, as in
|
|
253
|
+
the README. A bare session from `startSession()`/`withSession()` gives each call
|
|
254
|
+
its own transaction, not a shared one.
|
|
255
|
+
|
|
256
|
+
Exceptions: `bulkWrite`, and `removeOne`/`removeMany` on a service with
|
|
257
|
+
`emitOutboxEvents: false`, don't use a transaction.
|
|
223
258
|
|
|
224
259
|
## Gotchas (read before integrating)
|
|
225
260
|
|
|
@@ -239,42 +274,23 @@ multiple service calls — even across different `createService(...)` instances/
|
|
|
239
274
|
somewhere in the deployment.
|
|
240
275
|
- **Two MongoDB clients per `connect()` call** (primary + secondary-preferred reader).
|
|
241
276
|
`preferReadFromSecondary: true` only affects `find`/`findCursor`/`count`/`aggregate`.
|
|
242
|
-
- **`bulkWrite` requires `{ allowBulkwrite: true }`**
|
|
243
|
-
|
|
244
|
-
|
|
245
|
-
reliable path, since it reuses the exact same single-document logic
|
|
246
|
-
(`performSingleUpdate` in `db.ts`). To populate `{ prevDoc, doc }` on the outbox
|
|
247
|
-
row, the document is read with a separate `findOne` *before* running the atomic
|
|
248
|
-
`findOneAndUpdate` — a concurrent write landing in that gap means `prevDoc` may
|
|
249
|
-
not reflect the actual immediately-prior state. The atomic update itself is
|
|
250
|
-
unaffected (Mongo still applies it correctly); only the `prevDoc` value in the
|
|
251
|
-
outbox row can be stale. Treat `prevDoc` as best-effort, not authoritative, for
|
|
252
|
-
anything safety-critical.
|
|
253
|
-
- **`emitOutboxEvents: false` (no-outbox) writes only validate one sampled
|
|
254
|
-
document on `updateMany`, not every updated document.** For `$set`-with-literal-values
|
|
255
|
-
updates this is representative (every matched doc lands on the same value for
|
|
256
|
-
the touched fields), but for operators whose result depends on a document's
|
|
257
|
-
prior value (`$inc`, `$mul`, `$min`, `$max`, `$push`/`$addToSet`, `$rename`,
|
|
258
|
-
positional array filters), the same update can be valid for the sampled
|
|
259
|
-
document and invalid for another — undetected on that fast path. The default
|
|
260
|
-
reliable path (`emitOutboxEvents: true`, the default) validates every document
|
|
261
|
-
individually and doesn't have this gap.
|
|
277
|
+
- **`bulkWrite` requires `{ allowBulkwrite: true }`** and a service with
|
|
278
|
+
`emitOutboxEvents: false` — no schema validation runs on bulk writes, and no
|
|
279
|
+
outbox rows are written.
|
|
262
280
|
- **Nothing dispatches outbox events unless a relay is running.** Outbox rows
|
|
263
281
|
accumulate (durably, harmlessly) until some process calls `startOutboxRelay`.
|
|
264
282
|
Make sure at least one instance in your deployment runs it, or `.on(...)`
|
|
265
283
|
listeners will never fire.
|
|
266
|
-
- **`createService` called twice for the same collection name
|
|
267
|
-
call's
|
|
268
|
-
|
|
269
|
-
`EventEmitter` would silently replace the first's, orphaning any listeners
|
|
270
|
-
already registered on it. Reuse means `.on(...)`/`onPropertiesUpdated(...)`
|
|
284
|
+
- **`createService` called twice for the same collection name shares the first
|
|
285
|
+
call's listeners (and logs a warning).** Listeners are registered per
|
|
286
|
+
collection name, process-wide, so `.on(...)`/`onPropertiesUpdated(...)`
|
|
271
287
|
registered on *either* instance still receive dispatched events — but other
|
|
272
288
|
per-instance config is **not** shared, and a mismatch there is easy to miss
|
|
273
|
-
since the shared
|
|
289
|
+
since the shared listeners make everything else look consistent:
|
|
274
290
|
- **`emitOutboxEvents`** — if one instance has it `false`, writes through
|
|
275
|
-
that instance never enqueue an outbox row, so
|
|
276
|
-
silently never fire for those writes specifically (
|
|
277
|
-
|
|
291
|
+
that instance never enqueue an outbox row, so the shared listeners
|
|
292
|
+
silently never fire for those writes specifically (no event is generated
|
|
293
|
+
at all). This can be an intentional choice (e.g. a
|
|
278
294
|
backfill/migration instance skipping outbox writes on purpose), so it's
|
|
279
295
|
not something `createService` should reject — just be deliberate about it.
|
|
280
296
|
- **`validate`** — two instances can enforce different schemas on the same
|
|
@@ -284,10 +300,24 @@ multiple service calls — even across different `createService(...)` instances/
|
|
|
284
300
|
- **`addCreatedOnField`/`addUpdatedOnField`** — documents written through
|
|
285
301
|
one instance may end up with `createdOn`/`updatedOn` and documents
|
|
286
302
|
written through the other may not, for the same collection.
|
|
303
|
+
- **`toOutboxDoc`** — a field one instance keeps out of the outbox reaches
|
|
304
|
+
the shared listeners through the other. Give every instance the same one.
|
|
287
305
|
Prefer creating each service once (e.g. in a shared module) and importing
|
|
288
306
|
that instance everywhere it's used, unless one of the above divergences is
|
|
289
|
-
actually what you want.
|
|
290
|
-
|
|
291
|
-
|
|
292
|
-
|
|
293
|
-
|
|
307
|
+
actually what you want.
|
|
308
|
+
|
|
309
|
+
## Upgrading from 0.1
|
|
310
|
+
|
|
311
|
+
`CHANGELOG.md` (shipped with the package) lists what changed in 0.2.0. One step
|
|
312
|
+
has no place there: once no 0.1 relay runs any more, run this once per database.
|
|
313
|
+
It requeues rows a 0.1 relay left in status `processing` (a relay of this version
|
|
314
|
+
never claims them), and drops the indexes 0.1 created:
|
|
315
|
+
|
|
316
|
+
```js
|
|
317
|
+
db.outbox.updateMany({ status: 'processing' }, [
|
|
318
|
+
{ $set: { status: 'pending', leasedBy: null, nextAttemptAt: { $ifNull: ['$leasedUntil', '$$NOW'] } } },
|
|
319
|
+
{ $unset: 'leasedUntil' },
|
|
320
|
+
]);
|
|
321
|
+
db.outbox.dropIndex('status_1_nextAttemptAt_1_leasedUntil_1');
|
|
322
|
+
db.outbox.dropIndex('processedOn_1'); // only if it still exists
|
|
323
|
+
```
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@ahrowe/mongo",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.2.0",
|
|
4
4
|
"type": "module",
|
|
5
5
|
"license": "MIT",
|
|
6
6
|
"description": "Typed MongoDB CRUD service factory with built-in transactions, event hooks, and optional document validation.",
|
|
@@ -23,7 +23,8 @@
|
|
|
23
23
|
"types": "./dist/index.d.ts",
|
|
24
24
|
"files": [
|
|
25
25
|
"dist",
|
|
26
|
-
"docs/CLAUDE.md"
|
|
26
|
+
"docs/CLAUDE.md",
|
|
27
|
+
"CHANGELOG.md"
|
|
27
28
|
],
|
|
28
29
|
"sideEffects": false,
|
|
29
30
|
"exports": {
|
|
@@ -40,6 +41,7 @@
|
|
|
40
41
|
"mongodb": "^7.0.0"
|
|
41
42
|
},
|
|
42
43
|
"devDependencies": {
|
|
44
|
+
"@changesets/cli": "^3.0.3",
|
|
43
45
|
"@eslint/js": "^10.0.1",
|
|
44
46
|
"@faker-js/faker": "^10.1.0",
|
|
45
47
|
"@types/lodash": "^4.17.21",
|
|
@@ -56,6 +58,7 @@
|
|
|
56
58
|
"build": "tsc",
|
|
57
59
|
"lint": "eslint .",
|
|
58
60
|
"test": "vitest",
|
|
61
|
+
"changeset": "changeset",
|
|
59
62
|
"release": "bash scripts/release.sh"
|
|
60
63
|
}
|
|
61
64
|
}
|
package/dist/batchify.d.ts
DELETED
|
@@ -1,36 +0,0 @@
|
|
|
1
|
-
interface BatchifyOptions<T> {
|
|
2
|
-
/**
|
|
3
|
-
* Returns the next batch of data, or an array containing all data (sliced into
|
|
4
|
-
* batches automatically). When a function: receives the requested batch size
|
|
5
|
-
* and the number of items already processed.
|
|
6
|
-
*/
|
|
7
|
-
dataDelivery?: T[] | ((requestedBatchSize: number, alreadyProcessed: number) => T[] | Promise<T[]>);
|
|
8
|
-
/** Number of items fetched per `dataDelivery` call. Default 1000. */
|
|
9
|
-
batchSize?: number;
|
|
10
|
-
/**
|
|
11
|
-
* Number of items from the current batch processed concurrently. Defaults to
|
|
12
|
-
* `batchSize` (the whole batch at once). Set to `1` for strictly sequential
|
|
13
|
-
* processing — required when items share a MongoDB session/transaction, which
|
|
14
|
-
* doesn't allow concurrent operations.
|
|
15
|
-
*/
|
|
16
|
-
concurrency?: number;
|
|
17
|
-
/** Function to run for each item. Receives the item and its global index. */
|
|
18
|
-
functionToRun?: (data: T, index: number) => void | Promise<void>;
|
|
19
|
-
/** Invoked after each batch completes, with the total number of items processed so far. */
|
|
20
|
-
onProgress?: (processedAmount: number) => void | Promise<void>;
|
|
21
|
-
/** Called when processing an item fails. Receives the failed item and the error. */
|
|
22
|
-
onError?: (entry: T, err: unknown) => void | Promise<void>;
|
|
23
|
-
/** If true, stop fetching further batches once any error has occurred. */
|
|
24
|
-
returnOnError?: boolean;
|
|
25
|
-
/**
|
|
26
|
-
* If true, rethrow the first error out of `batchify()` itself once the batch
|
|
27
|
-
* it occurred in has settled, instead of only invoking `onError`. Use this when
|
|
28
|
-
* the caller needs the failure to actually propagate (e.g. to abort a transaction).
|
|
29
|
-
*/
|
|
30
|
-
throwOnError?: boolean;
|
|
31
|
-
/** Milliseconds to wait after completing each batch before starting the next. */
|
|
32
|
-
pauseAfterBatch?: number;
|
|
33
|
-
}
|
|
34
|
-
export declare function batchify<T>({ dataDelivery, batchSize, concurrency, functionToRun, onProgress, onError, returnOnError, throwOnError, pauseAfterBatch, }?: BatchifyOptions<T>): Promise<void>;
|
|
35
|
-
export {};
|
|
36
|
-
//# sourceMappingURL=batchify.d.ts.map
|
package/dist/batchify.d.ts.map
DELETED
|
@@ -1 +0,0 @@
|
|
|
1
|
-
{"version":3,"file":"batchify.d.ts","sourceRoot":"","sources":["../src/batchify.ts"],"names":[],"mappings":"AAAA,UAAU,eAAe,CAAC,CAAC;IACzB;;;;OAIG;IACH,YAAY,CAAC,EAAE,CAAC,EAAE,GAAG,CAAC,CAAC,kBAAkB,EAAE,MAAM,EAAE,gBAAgB,EAAE,MAAM,KAAK,CAAC,EAAE,GAAG,OAAO,CAAC,CAAC,EAAE,CAAC,CAAC,CAAC;IACpG,qEAAqE;IACrE,SAAS,CAAC,EAAE,MAAM,CAAC;IACnB;;;;;OAKG;IACH,WAAW,CAAC,EAAE,MAAM,CAAC;IACrB,6EAA6E;IAC7E,aAAa,CAAC,EAAE,CAAC,IAAI,EAAE,CAAC,EAAE,KAAK,EAAE,MAAM,KAAK,IAAI,GAAG,OAAO,CAAC,IAAI,CAAC,CAAC;IACjE,2FAA2F;IAC3F,UAAU,CAAC,EAAE,CAAC,eAAe,EAAE,MAAM,KAAK,IAAI,GAAG,OAAO,CAAC,IAAI,CAAC,CAAC;IAC/D,oFAAoF;IACpF,OAAO,CAAC,EAAE,CAAC,KAAK,EAAE,CAAC,EAAE,GAAG,EAAE,OAAO,KAAK,IAAI,GAAG,OAAO,CAAC,IAAI,CAAC,CAAC;IAC3D,0EAA0E;IAC1E,aAAa,CAAC,EAAE,OAAO,CAAC;IACxB;;;;OAIG;IACH,YAAY,CAAC,EAAE,OAAO,CAAC;IACvB,iFAAiF;IACjF,eAAe,CAAC,EAAE,MAAM,CAAC;CAC1B;AAED,wBAAsB,QAAQ,CAAC,CAAC,EAAE,EAChC,YAAuB,EACvB,SAAgB,EAChB,WAAW,EACX,aAA+B,EAC/B,UAA4B,EAC5B,OAAyB,EACzB,aAAqB,EACrB,YAAoB,EACpB,eAAmB,GACpB,GAAE,eAAe,CAAC,CAAC,CAAM,GAAG,OAAO,CAAC,IAAI,CAAC,CAuDzC"}
|
package/dist/batchify.js
DELETED
|
@@ -1,57 +0,0 @@
|
|
|
1
|
-
export async function batchify({ dataDelivery = () => [], batchSize = 1000, concurrency, functionToRun = () => undefined, onProgress = () => undefined, onError = () => undefined, returnOnError = false, throwOnError = false, pauseAfterBatch = 0, } = {}) {
|
|
2
|
-
const effectiveConcurrency = concurrency ?? batchSize;
|
|
3
|
-
let iteration = 0;
|
|
4
|
-
let hasError = false;
|
|
5
|
-
let firstError;
|
|
6
|
-
let currentBatch = [];
|
|
7
|
-
const getNextBatch = async () => {
|
|
8
|
-
if (typeof dataDelivery === 'function') {
|
|
9
|
-
currentBatch = await dataDelivery(batchSize, iteration * batchSize);
|
|
10
|
-
}
|
|
11
|
-
else {
|
|
12
|
-
currentBatch = dataDelivery.slice(batchSize * iteration, batchSize + batchSize * iteration);
|
|
13
|
-
}
|
|
14
|
-
};
|
|
15
|
-
const runItem = async (entry, globalIndex) => {
|
|
16
|
-
try {
|
|
17
|
-
await functionToRun(entry, globalIndex);
|
|
18
|
-
}
|
|
19
|
-
catch (error) {
|
|
20
|
-
hasError = true;
|
|
21
|
-
if (throwOnError && !firstError)
|
|
22
|
-
firstError = { error };
|
|
23
|
-
await onError(entry, error);
|
|
24
|
-
}
|
|
25
|
-
};
|
|
26
|
-
const processCurrentBatch = async () => {
|
|
27
|
-
for (let start = 0; start < currentBatch.length; start += effectiveConcurrency) {
|
|
28
|
-
if (firstError)
|
|
29
|
-
return;
|
|
30
|
-
const chunk = currentBatch.slice(start, start + effectiveConcurrency);
|
|
31
|
-
await Promise.allSettled(chunk.map((entry, i) => runItem(entry, batchSize * iteration + start + i)));
|
|
32
|
-
}
|
|
33
|
-
};
|
|
34
|
-
await getNextBatch();
|
|
35
|
-
await onProgress(0);
|
|
36
|
-
while (currentBatch.length) {
|
|
37
|
-
await processCurrentBatch();
|
|
38
|
-
if (firstError)
|
|
39
|
-
throw firstError.error;
|
|
40
|
-
if (currentBatch.length < batchSize) {
|
|
41
|
-
await onProgress(batchSize * iteration + currentBatch.length);
|
|
42
|
-
return;
|
|
43
|
-
}
|
|
44
|
-
iteration += 1;
|
|
45
|
-
await onProgress(batchSize * iteration);
|
|
46
|
-
if (hasError && returnOnError) {
|
|
47
|
-
return;
|
|
48
|
-
}
|
|
49
|
-
await getNextBatch();
|
|
50
|
-
if (currentBatch.length && pauseAfterBatch) {
|
|
51
|
-
await new Promise((resolve) => {
|
|
52
|
-
setTimeout(resolve, pauseAfterBatch);
|
|
53
|
-
});
|
|
54
|
-
}
|
|
55
|
-
}
|
|
56
|
-
}
|
|
57
|
-
//# sourceMappingURL=batchify.js.map
|
package/dist/batchify.js.map
DELETED
|
@@ -1 +0,0 @@
|
|
|
1
|
-
{"version":3,"file":"batchify.js","sourceRoot":"","sources":["../src/batchify.ts"],"names":[],"mappings":"AAkCA,MAAM,CAAC,KAAK,UAAU,QAAQ,CAAI,EAChC,YAAY,GAAG,GAAG,EAAE,CAAC,EAAE,EACvB,SAAS,GAAG,IAAI,EAChB,WAAW,EACX,aAAa,GAAG,GAAG,EAAE,CAAC,SAAS,EAC/B,UAAU,GAAG,GAAG,EAAE,CAAC,SAAS,EAC5B,OAAO,GAAG,GAAG,EAAE,CAAC,SAAS,EACzB,aAAa,GAAG,KAAK,EACrB,YAAY,GAAG,KAAK,EACpB,eAAe,GAAG,CAAC,MACG,EAAE;IACxB,MAAM,oBAAoB,GAAG,WAAW,IAAI,SAAS,CAAC;IACtD,IAAI,SAAS,GAAG,CAAC,CAAC;IAClB,IAAI,QAAQ,GAAG,KAAK,CAAC;IACrB,IAAI,UAA0C,CAAC;IAE/C,IAAI,YAAY,GAAQ,EAAE,CAAC;IAE3B,MAAM,YAAY,GAAG,KAAK,IAAI,EAAE;QAC9B,IAAI,OAAO,YAAY,KAAK,UAAU,EAAE,CAAC;YACvC,YAAY,GAAG,MAAM,YAAY,CAAC,SAAS,EAAE,SAAS,GAAG,SAAS,CAAC,CAAC;QACtE,CAAC;aAAM,CAAC;YACN,YAAY,GAAG,YAAY,CAAC,KAAK,CAAC,SAAS,GAAG,SAAS,EAAE,SAAS,GAAG,SAAS,GAAG,SAAS,CAAC,CAAC;QAC9F,CAAC;IACH,CAAC,CAAC;IAEF,MAAM,OAAO,GAAG,KAAK,EAAE,KAAQ,EAAE,WAAmB,EAAE,EAAE;QACtD,IAAI,CAAC;YACH,MAAM,aAAa,CAAC,KAAK,EAAE,WAAW,CAAC,CAAC;QAC1C,CAAC;QAAC,OAAO,KAAK,EAAE,CAAC;YACf,QAAQ,GAAG,IAAI,CAAC;YAChB,IAAI,YAAY,IAAI,CAAC,UAAU;gBAAE,UAAU,GAAG,EAAE,KAAK,EAAE,CAAC;YACxD,MAAM,OAAO,CAAC,KAAK,EAAE,KAAK,CAAC,CAAC;QAC9B,CAAC;IACH,CAAC,CAAC;IAEF,MAAM,mBAAmB,GAAG,KAAK,IAAI,EAAE;QACrC,KAAK,IAAI,KAAK,GAAG,CAAC,EAAE,KAAK,GAAG,YAAY,CAAC,MAAM,EAAE,KAAK,IAAI,oBAAoB,EAAE,CAAC;YAC/E,IAAI,UAAU;gBAAE,OAAO;YACvB,MAAM,KAAK,GAAG,YAAY,CAAC,KAAK,CAAC,KAAK,EAAE,KAAK,GAAG,oBAAoB,CAAC,CAAC;YACtE,MAAM,OAAO,CAAC,UAAU,CAAC,KAAK,CAAC,GAAG,CAAC,CAAC,KAAK,EAAE,CAAC,EAAE,EAAE,CAAC,OAAO,CAAC,KAAK,EAAE,SAAS,GAAG,SAAS,GAAG,KAAK,GAAG,CAAC,CAAC,CAAC,CAAC,CAAC;QACvG,CAAC;IACH,CAAC,CAAC;IAEF,MAAM,YAAY,EAAE,CAAC;IACrB,MAAM,UAAU,CAAC,CAAC,CAAC,CAAC;IACpB,OAAO,YAAY,CAAC,MAAM,EAAE,CAAC;QAC3B,MAAM,mBAAmB,EAAE,CAAC;QAC5B,IAAI,UAAU;YAAE,MAAM,UAAU,CAAC,KAAK,CAAC;QACvC,IAAI,YAAY,CAAC,MAAM,GAAG,SAAS,EAAE,CAAC;YACpC,MAAM,UAAU,CAAC,SAAS,GAAG,SAAS,GAAG,YAAY,CAAC,MAAM,CAAC,CAAC;YAC9D,OAAO;QACT,CAAC;QACD,SAAS,IAAI,CAAC,CAAC;QACf,MAAM,UAAU,CAAC,SAAS,GAAG,SAAS,CAAC,CAAC;QACxC,IAAI,QAAQ,IAAI,aAAa,EAAE,CAAC;YAC9B,OAAO;QACT,CAAC;QACD,MAAM,YAAY,EAAE,CAAC;QACrB,IAAI,YAAY,CAAC,MAAM,IAAI,eAAe,EAAE,CAAC;YAC3C,MAAM,IAAI,OAAO,CAAC,CAAC,OAAO,EAAE,EAAE;gBAC5B,UAAU,CAAC,OAAO,EAAE,eAAe,CAAC,CAAC;YACvC,CAAC,CAAC,CAAC;QACL,CAAC;IACH,CAAC;AACH,CAAC"}
|