@cloudbitmaps/tools 0.0.0-stage → 0.20.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/cost.d.ts ADDED
@@ -0,0 +1,353 @@
1
+ /** Pluggable rate card. Rates differ by cloud/region/term and drift over time; the formulas don't. */
2
+ export interface PricingProfile {
3
+ readonly name: string;
4
+ readonly storage: {
5
+ /** Object GET (per **million** requests). A ranged GET bills as a full GET. */
6
+ readonly getPerMillion: number;
7
+ /** Object PUT (per million) — what a load pays, per request. */
8
+ readonly putPerMillion: number;
9
+ readonly storagePerGiBMonth: number;
10
+ /**
11
+ * Requests one pointer read costs: a read of a segment's registry row, whose version comes back with its bytes.
12
+ * Default **1**, and 1 on S3, GCS and Azure Blob, each of which answers it with one GET. Charged for each operand
13
+ * of an intersection, for the pointer reads a load makes, and for each pointer refresh. Set it when a
14
+ * registry of your own takes more than one request to read a row.
15
+ */
16
+ readonly requestsPerPointerRead?: number;
17
+ /**
18
+ * Requests one tail read costs: the read of a segment's index from the end of its generation, which needs the
19
+ * object's size. Default **1**, S3's, whose suffix-range GET returns the size with the bytes, and GCS's, the same.
20
+ * **2** on Azure Blob, which takes no suffix range and reads the properties and then the bytes. Charged for each
21
+ * operand of an intersection. A load reads no index of its own: a row that carries a summary of its current
22
+ * generation gives the load the size it needs, and the checks it makes are one metadata request on every backend.
23
+ * A chunk read knows its range, and is one request everywhere.
24
+ */
25
+ readonly requestsPerSizedRead?: number;
26
+ };
27
+ /**
28
+ * The always-on Redis the verdict compares against: exactly one of the two. `sizedToData`, the default's, prices
29
+ * the cheapest cluster that holds the report's stored bytes; `monthlyUSD` is one cluster you name, whatever the
30
+ * data size. A profile carrying both is refused rather than read one way.
31
+ */
32
+ readonly redis: {
33
+ readonly monthlyUSD: number;
34
+ readonly sizedToData?: never;
35
+ } | {
36
+ readonly sizedToData: RedisSizing;
37
+ readonly monthlyUSD?: never;
38
+ };
39
+ }
40
+ /** One node type a Redis cluster can be built from, as its cloud prices it. */
41
+ export interface RedisNodeType {
42
+ readonly name: string;
43
+ /** Memory, in GiB, as the cloud publishes it for the node type. */
44
+ readonly memoryGiB: number;
45
+ /**
46
+ * SSD, in GiB, on a data-tiering node (ElastiCache's `r6gd`): keys stay in memory, and the values read least
47
+ * recently move to the SSD. Counted in full beside the memory, as AWS counts a data-tiering node's capacity.
48
+ */
49
+ readonly ssdGiB?: number;
50
+ /** Price per node-hour, in USD. */
51
+ readonly hourlyUSD: number;
52
+ /**
53
+ * The most shards a cluster of this type is priced at. The default catalogue prices its burstable nodes as one
54
+ * shard: a cluster sharded across them is not how data anyone queries is held. Default: no limit.
55
+ */
56
+ readonly maxShards?: number;
57
+ }
58
+ /**
59
+ * Redis sized to hold the data: enough shards for the bytes, each a primary and its replicas, on whichever node
60
+ * type in `nodeTypes` makes that cheapest. Prices within a millionth of a dollar of the cheapest are a tie, which
61
+ * fewer nodes win, then the lower price, then the name, so the answer never depends on the order of the rows. The data is held at its stored,
62
+ * compressed size, which is a floor on the memory Redis needs for it: a native Redis bitmap is sized by its highest
63
+ * id, not by how many ids it holds, so sparse ids need more.
64
+ */
65
+ export interface RedisSizing {
66
+ /** Where the prices come from, and when; the report's notes quote it. */
67
+ readonly source: string;
68
+ readonly nodeTypes: readonly RedisNodeType[];
69
+ /** Replicas per shard. AWS's best practice is 2; Multi-AZ needs at least 1. */
70
+ readonly replicasPerShard: number;
71
+ /** The share of each node's memory kept back from data: ElastiCache's `reserved-memory-percent`, 25% by default. */
72
+ readonly reservedMemoryFraction: number;
73
+ }
74
+ /**
75
+ * **ElastiCache for Redis OSS in us-east-1, on-demand**: node prices from AWS's public price list, version
76
+ * 20260914063714 (published 2026-09-14, effective 2026-09-01,
77
+ * https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonElastiCache/20260914063714/us-east-1/index.json), and the
78
+ * memory AWS publishes for each node type. Every shard is a primary and two replicas, AWS's best practice, with
79
+ * ElastiCache's default 25% of each node's memory reserved, and the burstable `t4g` nodes are priced as one shard.
80
+ *
81
+ * It lists only rows that are the cheapest fit for some size of data: the data-tiering `r6gd` rows above about
82
+ * 30 GiB, and the larger in-memory `r6g` rows, `2xlarge` and up, when those are left out. The current `m7g` and `r7g` nodes are not here,
83
+ * because an `m6g` or `r6g` with the same memory costs less. It is the cheapest cluster of THIS kind; Redis bought
84
+ * another way costs less — one replica a shard a third less, ElastiCache for Valkey 20% less a node, reserved nodes
85
+ * less again — so against those the saving the verdict reports is smaller. Pass their prices to compare with them.
86
+ */
87
+ export declare const ELASTICACHE_REDIS_US_EAST_1_ONDEMAND: RedisSizing;
88
+ /**
89
+ * One ElastiCache HA cluster, a primary and two replicas of cache.m7g.large — the m7g family's smallest node —
90
+ * at $0.158 an hour each: 3 × 730 h × $0.158 = $346.02, published to the dollar. The per-request crossovers on the
91
+ * benchmarks page are drawn against this one cluster, whatever the data size; pass it as `pricing.redis` to compare
92
+ * with it. It is not the cheapest cluster for any size of data: three cache.m6g.large hold the same memory for less.
93
+ */
94
+ export declare const ONE_REDIS_HA_CLUSTER: Readonly<{
95
+ monthlyUSD: number;
96
+ }>;
97
+ /**
98
+ * Default profile — **AWS us-east-1, on-demand**, mid-2026, from the fact-checked published pricing rather than
99
+ * copied from a blog post, with Redis sized to the data. Override it for your region, cloud, or committed term.
100
+ * Its prices are as old as the release of this package that ships them: a planning figure, not a quote.
101
+ */
102
+ export declare const AWS_US_EAST_1_ONDEMAND: PricingProfile;
103
+ /** Sustained access pattern. All rates default to 0; unspecified ⇒ that op contributes nothing. */
104
+ export interface Workload {
105
+ /**
106
+ * Point reads (`has`) per second. Each cache miss is one object GET, once the segment's pointer and index are
107
+ * read.
108
+ */
109
+ readonly readsPerSec?: number;
110
+ readonly intersectsPerSec?: number;
111
+ /** Cache hit rate in `[0, 1]` — hits are free; only misses cost. Default 0. */
112
+ readonly cacheHitRate?: number;
113
+ /**
114
+ * Range requests one intersection makes for its chunks, summed over its operands. Default 1. The engine reads each
115
+ * operand's needed chunks (the chunk-skipping survivors) as coalesced ranges: chunks within 256 KiB of each other
116
+ * are read in one request, up to 1 MiB, so this is the requests the chunks need, not the chunks. It follows the
117
+ * bytes the needed chunks span in each object: two segments sharing 100 chunks that sit side by side make one request
118
+ * of each, `2`; the same 100 spread over each object make as many as it takes to cover most of the object, a
119
+ * request per MiB, and read most of its bytes. Count it by running the engine over a layout of your own, as
120
+ * `bench/range-counts.cjs` does, or take it as one request per MiB the shared chunks span, per operand. The model adds each operand's pointer and index reads itself (see
121
+ * {@link Workload.operandsPerIntersect}), so count chunk requests only. One more GET per operand whose index
122
+ * outgrows the reader's tail read (256 KiB by default), which then reads the index whole, belongs here too, and so
123
+ * does a pointer re-read by an intersect slow enough to outlive {@link Workload.genTtlMs}.
124
+ */
125
+ readonly chunksPerIntersect?: number;
126
+ /**
127
+ * Segments each intersection reads, `exclude` operands included; at least 1. Default 2. An intersection is priced
128
+ * **cold**: before its chunks, each operand's pointer is read, then its index, in one read of the object's tail —
129
+ * 2 GETs an operand on S3 and GCS, so a cold intersect that makes `r` range requests of each of two segments is `4 + 2r` GETs there,
130
+ * and 3 on Azure Blob, whose tail read is two requests (see {@link PricingProfile}). `cacheHitRate`
131
+ * does not apply to intersections, so a long-lived reader that answers a repeat from its cache pays less. Other
132
+ * combines read their operands the same way and can be priced here too, with the chunk requests they make.
133
+ */
134
+ readonly operandsPerIntersect?: number;
135
+ /**
136
+ * Generations published per month across the modeled data — the write side of a loaded store. Default **0**
137
+ * ⇒ loads are not modeled and the report *discloses* the omission rather than silently under-reporting.
138
+ */
139
+ readonly loadsPerMonth?: number;
140
+ /**
141
+ * PUT-class requests one load's object write issues, each priced at the PUT rate. Default **1** (a single-object
142
+ * PUT). A multipart write of `P` parts bills `P + 2` (initiate, the parts, complete) — set it when you know your
143
+ * object sizes. The model adds what `store.load()` does around the write, at any `keep` from 1 to 64: the pointer's
144
+ * write, PUT-class on S3, and four GETs: the pointer read twice, one check that the next generation number is
145
+ * free, and one that the current generation's object is there, which tells the collection that deletes by name it
146
+ * may. The current generation's size comes from the row's summary of it, so the load reads no index. Collection
147
+ * deletes the generation the window pushed out by name, so it lists only on every 16th generation, which adds a
148
+ * PUT-class request and two pointer reads there and makes no check that the current generation is there, a
149
+ * sixteenth of each on average. That is a segment with two
150
+ * generations behind it, whose row carries a summary; its first two loads make fewer requests and collect nothing,
151
+ * and the first load onto a row with no summary of its current generation (one a rollback left when it could not
152
+ * open the key, one a registry of your own stores without one, or one this store cannot use) reads that generation's index, one tail read, in
153
+ * place of the check that its object is there. A publish that loses a race to another writer
154
+ * reads the pointer again, and a load whose check finds the number taken (a crashed load's object, or the
155
+ * generations a rollback left above the pointer) lists the segment's objects to number past them and to collect,
156
+ * two PUT-class requests on S3 and two more pointer reads. So does every load whose `keep` is above 64, which lists
157
+ * to collect: one more PUT-class request and two more pointer reads than the model counts. The counts
158
+ * are a cleartext segment's: an encrypted segment's load reads its row once more, after its ids and before it
159
+ * unwraps the key, one more GET ($0.40 per million at the default prices) that the model leaves out, beside the
160
+ * key-management calls it does not price either.
161
+ */
162
+ readonly requestsPerLoad?: number;
163
+ /**
164
+ * Segments each long-lived reader process keeps reading. A reader re-reads a segment's pointer when it reads the
165
+ * segment after {@link Workload.genTtlMs} has passed, so each costs at most one GET per `genTtlMs` in each process
166
+ * that reads it — 1,314,000 GETs a month at the default 2 s — and the whole term at most one GET per point read
167
+ * (`readsPerSec`, across the fleet). Intersections are priced cold and pay their own pointer reads. Default **0**
168
+ * ⇒ the refresh is not modeled, and the report *discloses* the omission when there are reads to refresh for.
169
+ *
170
+ * It assumes each hot segment stays open in the reader's cache (`cache.readerMax`, 1,024 segments by default,
171
+ * and `cache.readerMaxBytes`, 64 MiB of parsed index). A read of a segment the cache evicted that needs its index or
172
+ * a chunk the chunk cache does not hold opens it again, a tail read, which is not priced here; with a timed refresh the store keeps the segment's resolution apart from its
173
+ * reader, so the eviction reads its pointer again only for a segment whose resolution was let go too, and with none
174
+ * every reopen reads it. Neither is the re-open every reader makes after each load priced, for the new generation's
175
+ * index. Size those caches to keep the hot set open; the report says so when `hotSegments` is more than a store
176
+ * keeps open by default.
177
+ */
178
+ readonly hotSegments?: number;
179
+ /**
180
+ * Long-lived reader processes, each keeping its own {@link Workload.hotSegments} open and refreshing their
181
+ * pointers on its own: ten processes reading the same hundred segments pay ten times one process's refresh.
182
+ * At least 1. Default 1.
183
+ */
184
+ readonly readerProcesses?: number;
185
+ /**
186
+ * How long the reader trusts a pointer, in ms: the store's `cache.genTtlMs`. Default 2000, the store's own
187
+ * default; pass the store's when it sets one. `0` turns the timed refresh off, and the model bills none: a
188
+ * reader then re-reads a pointer only when its cache evicts the segment, a read finds its generation swept, or a
189
+ * write invalidates it, none of which this term prices.
190
+ */
191
+ readonly genTtlMs?: number;
192
+ /**
193
+ * Segments the retention sweep (`retireExpired`) retires a month. Each costs 7 registry reads, 2 writes and a
194
+ * delete with {@link Workload.conditionalDelete} on, and 6 reads and 2 writes with it off, priced at the GET and PUT
195
+ * rates, a delete unbilled, as S3 leaves it. Default
196
+ * **0**.
197
+ */
198
+ readonly retirementsPerMonth?: number;
199
+ /**
200
+ * Tombstones the sweep purges a month: at a steady state, as many as it retires a month, a `tombstoneGraceMs` later.
201
+ * A purge costs 4 reads and 2 deletes with {@link Workload.conditionalDelete} on, and 3 reads and a write with it
202
+ * off. Default **0**.
203
+ */
204
+ readonly purgesPerMonth?: number;
205
+ /**
206
+ * Whether the registry's deletes remove a row for good (`RegCaps.conditionalDelete`), which the shipped registries
207
+ * do where the backend applies a delete precondition. Default **true**. `false` prices the sweep of a registry that
208
+ * only tombstones: a purge rewrites the row instead of deleting it, and every later full sweep reads each tombstone
209
+ * left (two reads per purged segment), which this term does not price.
210
+ */
211
+ readonly conditionalDelete?: boolean;
212
+ }
213
+ /** One (group of) segment(s) for planning. `count` = how many like this (default 1). */
214
+ export interface SegmentSizing {
215
+ readonly sizeBytes?: number;
216
+ readonly cardinality?: number;
217
+ /** Number of segments with these characteristics. Default 1. */
218
+ readonly count?: number;
219
+ }
220
+ export interface EstimateInput {
221
+ readonly segments: readonly SegmentSizing[];
222
+ readonly workload?: Workload;
223
+ readonly pricing?: PricingProfile;
224
+ }
225
+ export interface CostReport {
226
+ readonly monthlyUSD: {
227
+ readonly byOp: {
228
+ readonly reads: number;
229
+ readonly intersects: number;
230
+ readonly storage: number;
231
+ /**
232
+ * Loads: each its object's `requestsPerLoad` PUT-class requests plus what `store.load()` adds around them.
233
+ * 0 unless `loadsPerMonth` is set.
234
+ */
235
+ readonly loads: number;
236
+ /** The pointer refresh of {@link Workload.hotSegments}. 0 unless `hotSegments` is set. */
237
+ readonly pointerRefresh: number;
238
+ /**
239
+ * The retention sweep's retirements and purges: each's registry reads at the GET rate and its writes at the PUT
240
+ * rate. 0 unless `retirementsPerMonth` or `purgesPerMonth` is set.
241
+ */
242
+ readonly retention: number;
243
+ };
244
+ readonly total: number;
245
+ };
246
+ /**
247
+ * The Redis the verdict compares against. Sized, the default, it is the cheapest cluster in
248
+ * `pricing.redis.sizedToData` that holds this report's stored bytes; fixed, it is `pricing.redis.monthlyUSD` as
249
+ * given.
250
+ *
251
+ * **It is not additive across reports.** Each report sizes Redis to its own bytes, so a per-segment report
252
+ * (`groundedReport` of one segment's size) compares with a cluster holding that one segment alone, and the baselines of a store's
253
+ * segments do not sum to the store's. To judge a store, price all its segments in one `estimateCost`. To alarm on
254
+ * one, sum `monthlyUSD.total` over its segments and compare the sum with the Redis you would run for it: a
255
+ * per-segment verdict against that whole price fires only when one segment alone costs more than all of it.
256
+ */
257
+ readonly redisBaseline: {
258
+ readonly basis: 'fixed';
259
+ readonly monthlyUSD: number;
260
+ } | {
261
+ readonly basis: 'sized-to-data';
262
+ readonly monthlyUSD: number;
263
+ /** The node type, the shards, the nodes (every shard's primary and replicas), and whether they tier to SSD. */
264
+ readonly cluster: {
265
+ readonly nodeType: string;
266
+ readonly shards: number;
267
+ readonly nodes: number;
268
+ readonly dataTiering: boolean;
269
+ };
270
+ };
271
+ /**
272
+ * Sustained read rate at which the pay-per-use model's cost passes {@link CostReport.redisBaseline}, **evaluated
273
+ * at this report's `cacheHitRate`** — so a higher cache-hit rate raises it (cache hits are free) — with the other
274
+ * request axes at 0, and with what keeping the data readable costs taken out of the baseline first: storage, and
275
+ * this report's pointer refresh. Loads, the write side, are left out of it. `Infinity` means it never crosses (a
276
+ * 100% cache-hit rate). The published anchor (~329 reads/s) is against {@link ONE_REDIS_HA_CLUSTER}, at
277
+ * `cacheHitRate: 0` with no refresh modeled.
278
+ */
279
+ readonly redisCrossover: {
280
+ readonly readsPerSec: number;
281
+ };
282
+ readonly verdict: 'win-big' | 'win' | 'lose-zone';
283
+ readonly rationale: string;
284
+ readonly assumptions: {
285
+ readonly cacheHitRate: number;
286
+ readonly pricingName: string;
287
+ /**
288
+ * True when the size was measured (a `groundedReport` given a byte count), false for a pure `estimateCost` and
289
+ * for a `groundedReport` given `storageBytes: null`.
290
+ */
291
+ readonly grounded: boolean;
292
+ /** What the model assumed and what it left out, in words. The last says how the Redis was priced. */
293
+ readonly notes: readonly string[];
294
+ };
295
+ }
296
+ /**
297
+ * What the retention sweep makes per segment, counted with a store that counts its requests
298
+ * (`tests/core/retention-hard-purge.test.ts` holds each figure to the engine, and `tests/tools/cost.test.ts` holds the
299
+ * estimator to these). With the registry's `conditionalDelete` on, a retirement files a due-index pointer to the
300
+ * tombstone and a purge removes the row and every pointer: a later sweep reads nothing of a purged segment. With it
301
+ * off, a retirement files no pointer, a purge rewrites the row as a tombstone, and every later full sweep reads what
302
+ * is left, two objects per purged segment.
303
+ */
304
+ export declare const RETENTION_SWEEP_REQUESTS: Readonly<{
305
+ conditionalDelete: {
306
+ retirement: {
307
+ reads: number;
308
+ writes: number;
309
+ deletes: number;
310
+ };
311
+ purge: {
312
+ reads: number;
313
+ writes: number;
314
+ deletes: number;
315
+ };
316
+ };
317
+ tombstoning: {
318
+ retirement: {
319
+ reads: number;
320
+ writes: number;
321
+ deletes: number;
322
+ };
323
+ purge: {
324
+ reads: number;
325
+ writes: number;
326
+ deletes: number;
327
+ };
328
+ };
329
+ }>;
330
+ /**
331
+ * **Planning** cost estimate — pure, no instance or live data needed (sizing, sales, what-if). Segment sizes
332
+ * are taken as given (or roughly derived from cardinality); use {@link groundedReport} with a segment's measured
333
+ * size for exact, real sizes. See {@link CostReport}.
334
+ */
335
+ export declare function estimateCost(input: EstimateInput): CostReport;
336
+ /**
337
+ * **Grounded** report from a measured byte total + a supplied workload. A segment handle's `stat()` reports the
338
+ * size of its current generation as `size`, read from the `.crbm` footer and index with no payload reads:
339
+ *
340
+ * ```ts
341
+ * const report = groundedReport({ storageBytes: (await seg.stat()).sizeBytes, workload: { readsPerSec: 50 } });
342
+ * ```
343
+ *
344
+ * Sum the sizes of several segments to price them together; see {@link CostReport.redisBaseline} for why one report
345
+ * of the sum is not the sum of the reports. `storageBytes: null`, which `stat()` answers for a segment with no
346
+ * generation and on a store whose source cannot report a size, prices storage at $0 and says so in the notes, with
347
+ * `assumptions.grounded` false: nothing was measured, so the report does not claim a confident $0.
348
+ */
349
+ export declare function groundedReport(input: {
350
+ readonly storageBytes: number | null;
351
+ readonly workload?: Workload;
352
+ readonly pricing?: PricingProfile;
353
+ }): CostReport;
@@ -0,0 +1,14 @@
1
+ /**
2
+ * `@cloudbitmaps/tools` — offline tools for CloudBitmaps that need nothing internal from a store.
3
+ *
4
+ * The cost model: {@link estimateCost} prices a store you plan, from the segment sizes and the workload you give it,
5
+ * and {@link groundedReport} one you run, from the bytes it stores, which a segment handle's `stat()` reports as
6
+ * `size`. Each report compares the store's monthly cost on object storage with an always-on Redis, and says where
7
+ * Redis wins. It is a planning tool: the price lists here are as old as the release that ships them, so pass your
8
+ * own for your region, your cloud or a committed term.
9
+ *
10
+ * Nothing here reads or writes a store. It takes `ValidationError` from `@cloudbitmaps/core`'s public entry, so a
11
+ * caller classifies the model's refusals as it does the library's.
12
+ */
13
+ export { estimateCost, groundedReport, AWS_US_EAST_1_ONDEMAND, ELASTICACHE_REDIS_US_EAST_1_ONDEMAND, ONE_REDIS_HA_CLUSTER, } from './cost.js';
14
+ export type { PricingProfile, RedisNodeType, RedisSizing, CostReport, Workload, SegmentSizing, EstimateInput, } from './cost.js';