cursedbelt-server 1.1.0 → 2.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/server/bench/assert.d.ts +61 -0
- package/dist/server/bench/assert.js +117 -0
- package/dist/server/bench/budget.d.ts +130 -0
- package/dist/server/bench/budget.js +131 -0
- package/dist/server/bench/cpuBudget.d.ts +45 -0
- package/dist/server/bench/cpuBudget.js +34 -0
- package/dist/server/bench/cpuClock.d.ts +65 -0
- package/dist/server/bench/cpuClock.js +100 -0
- package/dist/server/bench/index.d.ts +40 -0
- package/dist/server/bench/index.js +40 -0
- package/dist/server/bench/recorder.d.ts +70 -0
- package/dist/server/bench/recorder.js +95 -0
- package/dist/server/bench/runBench.d.ts +61 -0
- package/dist/server/bench/runBench.js +61 -0
- package/dist/server/d1/backup.d.ts +110 -0
- package/dist/server/d1/backup.js +128 -0
- package/dist/server/d1/fakeD1.d.ts +41 -0
- package/dist/server/d1/fakeD1.js +185 -0
- package/dist/server/d1/index.d.ts +24 -0
- package/dist/server/d1/index.js +24 -0
- package/dist/server/d1/kysely.d.ts +56 -0
- package/dist/server/d1/kysely.js +138 -0
- package/dist/server/d1/limits.d.ts +56 -0
- package/dist/server/d1/limits.js +96 -0
- package/dist/server/d1/local.d.ts +31 -0
- package/dist/server/d1/local.js +135 -0
- package/dist/server/d1/remote.d.ts +59 -0
- package/dist/server/d1/remote.js +124 -0
- package/dist/server/d1/scheduling.d.ts +113 -0
- package/dist/server/d1/scheduling.js +164 -0
- package/dist/server/d1/types.d.ts +143 -0
- package/dist/server/d1/types.js +80 -0
- package/dist/server/d1/values.d.ts +50 -0
- package/dist/server/d1/values.js +124 -0
- package/dist/server/sync/http.d.ts +20 -3
- package/dist/server/sync/http.js +20 -14
- package/dist/server/sync/index.d.ts +10 -2
- package/dist/server/sync/index.js +9 -1
- package/dist/server/sync/planner.d.ts +38 -8
- package/dist/server/sync/planner.js +32 -8
- package/dist/server/sync/signal.d.ts +161 -0
- package/dist/server/sync/signal.js +348 -0
- package/dist/server/sync/timer.d.ts +63 -19
- package/dist/server/sync/timer.js +104 -45
- package/dist/server/sync/types.d.ts +0 -2
- package/package.json +21 -3
- package/src/leafSubpathsImportNothing.spec.ts +15 -3
- package/src/noTimerDialsAPeer.spec.ts +469 -0
- package/src/server/bench/assert.ts +192 -0
- package/src/server/bench/budget.spec.ts +126 -0
- package/src/server/bench/budget.ts +207 -0
- package/src/server/bench/cpuBudget.spec.ts +302 -0
- package/src/server/bench/cpuBudget.ts +81 -0
- package/src/server/bench/cpuClock.ts +119 -0
- package/src/server/bench/index.ts +81 -0
- package/src/server/bench/recorder.ts +163 -0
- package/src/server/bench/runBench.ts +110 -0
- package/src/server/d1/backup.spec.ts +121 -0
- package/src/server/d1/backup.ts +186 -0
- package/src/server/d1/fakeD1.ts +193 -0
- package/src/server/d1/index.ts +62 -0
- package/src/server/d1/kysely.spec.ts +145 -0
- package/src/server/d1/kysely.ts +169 -0
- package/src/server/d1/limits.spec.ts +90 -0
- package/src/server/d1/limits.ts +123 -0
- package/src/server/d1/local.ts +173 -0
- package/src/server/d1/remote.ts +182 -0
- package/src/server/d1/sameShape.spec.ts +279 -0
- package/src/server/d1/scheduling.spec.ts +120 -0
- package/src/server/d1/scheduling.ts +210 -0
- package/src/server/d1/types.ts +163 -0
- package/src/server/d1/values.ts +138 -0
- package/src/server/sync/http.ts +31 -16
- package/src/server/sync/index.ts +23 -1
- package/src/server/sync/planner.spec.ts +33 -16
- package/src/server/sync/planner.ts +48 -11
- package/src/server/sync/signal.spec.ts +306 -0
- package/src/server/sync/signal.ts +422 -0
- package/src/server/sync/timer.spec.ts +97 -16
- package/src/server/sync/timer.ts +124 -47
- package/src/server/sync/types.ts +0 -2
|
@@ -0,0 +1,81 @@
|
|
|
1
|
+
import type { MiddlewareHandler } from 'hono';
|
|
2
|
+
import type { CursedbeltEnv } from '../context';
|
|
3
|
+
import { normalizeRoute } from '../metrics/normalizeRoute';
|
|
4
|
+
import type { CpuBudgetConfig } from './budget';
|
|
5
|
+
import { type CpuClock, processCpuClock } from './cpuClock';
|
|
6
|
+
import { type CpuRecorder, createCpuRecorder } from './recorder';
|
|
7
|
+
|
|
8
|
+
/**
|
|
9
|
+
* Per-route CPU attribution, mounted like any other cursedbelt middleware.
|
|
10
|
+
*
|
|
11
|
+
* ```ts
|
|
12
|
+
* const cpu = createCpuRecorder({ clock: processCpuClock(), config: BUDGETS })
|
|
13
|
+
* app.use('*', cpuBudget({ recorder: cpu }))
|
|
14
|
+
* ```
|
|
15
|
+
*
|
|
16
|
+
* 🔴 **It records CPU, never wall clock.** `requestLogger` beside it records duration,
|
|
17
|
+
* and duration is the quantity Cloudflare explicitly does not bill. The two numbers
|
|
18
|
+
* diverge by exactly the amount a handler spends awaiting a subrequest, which on this
|
|
19
|
+
* fleet is most of it — so a route can be slow and free, or fast and expensive, and only
|
|
20
|
+
* this middleware can tell you which.
|
|
21
|
+
*
|
|
22
|
+
* Like `requestLogger`, it records even when a downstream handler THROWS: the CPU was
|
|
23
|
+
* spent either way, and an endpoint that is expensive only on its error path is exactly
|
|
24
|
+
* the sort of thing a budget is supposed to catch. The error is re-thrown unchanged.
|
|
25
|
+
*/
|
|
26
|
+
export interface CpuBudgetOpts {
|
|
27
|
+
/** The recorder to write through. Provide one so the gate can read its report. */
|
|
28
|
+
recorder?: CpuRecorder;
|
|
29
|
+
/** Built into a recorder when `recorder` is omitted. */
|
|
30
|
+
config?: CpuBudgetConfig;
|
|
31
|
+
/** Defaults to {@link processCpuClock} — the local PROXY. */
|
|
32
|
+
clock?: CpuClock;
|
|
33
|
+
/**
|
|
34
|
+
* Called after each request with the sample. For shipping the number somewhere (a
|
|
35
|
+
* telemetry sink, a log line) without coupling this module to a destination.
|
|
36
|
+
*/
|
|
37
|
+
onSample?: (sample: {
|
|
38
|
+
method: string;
|
|
39
|
+
route: string;
|
|
40
|
+
cpuMs: number;
|
|
41
|
+
proxy: boolean;
|
|
42
|
+
status: number;
|
|
43
|
+
}) => void;
|
|
44
|
+
}
|
|
45
|
+
|
|
46
|
+
export function cpuBudget(
|
|
47
|
+
opts: CpuBudgetOpts = {},
|
|
48
|
+
): MiddlewareHandler<CursedbeltEnv> & { recorder: CpuRecorder } {
|
|
49
|
+
const recorder =
|
|
50
|
+
opts.recorder ??
|
|
51
|
+
createCpuRecorder({ clock: opts.clock ?? processCpuClock(), config: opts.config });
|
|
52
|
+
|
|
53
|
+
const handler: MiddlewareHandler<CursedbeltEnv> = async (c, next) => {
|
|
54
|
+
const end = recorder.clock.start();
|
|
55
|
+
let thrown: unknown;
|
|
56
|
+
let didThrow = false;
|
|
57
|
+
try {
|
|
58
|
+
await next();
|
|
59
|
+
} catch (err) {
|
|
60
|
+
thrown = err;
|
|
61
|
+
didThrow = true;
|
|
62
|
+
}
|
|
63
|
+
const cpuMs = end();
|
|
64
|
+
|
|
65
|
+
// `normalizeRoute` must run AFTER `next()` — before it, Hono has not matched a route
|
|
66
|
+
// pattern yet and every request would aggregate under its raw path.
|
|
67
|
+
const { method, route } = normalizeRoute(c);
|
|
68
|
+
recorder.record(method, route, cpuMs);
|
|
69
|
+
opts.onSample?.({
|
|
70
|
+
method,
|
|
71
|
+
route,
|
|
72
|
+
cpuMs,
|
|
73
|
+
proxy: recorder.clock.proxy,
|
|
74
|
+
status: didThrow ? 500 : c.res.status,
|
|
75
|
+
});
|
|
76
|
+
|
|
77
|
+
if (didThrow) throw thrown;
|
|
78
|
+
};
|
|
79
|
+
|
|
80
|
+
return Object.assign(handler, { recorder });
|
|
81
|
+
}
|
|
@@ -0,0 +1,119 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Where a CPU-millisecond comes from, and how honest it is.
|
|
3
|
+
*
|
|
4
|
+
* 🔴 **Every reading carries `proxy`, and it is not decoration.** The task that asked for
|
|
5
|
+
* this module asked for the label in the same breath as the measurement: *"a local number
|
|
6
|
+
* that is called a CPU-ms measurement will be quoted as one."* A number printed without
|
|
7
|
+
* its provenance becomes a fact about Cloudflare's billing the first time somebody pastes
|
|
8
|
+
* it into a summary.
|
|
9
|
+
*
|
|
10
|
+
* Two sources, and they are not interchangeable:
|
|
11
|
+
*
|
|
12
|
+
* · {@link processCpuClock} — `process.cpuUsage()` deltas on Bun/Node. `proxy: true`.
|
|
13
|
+
* · {@link workerCpuClock} — a reader the Worker runtime supplies. `proxy: false`.
|
|
14
|
+
*
|
|
15
|
+
* There is deliberately no built-in Cloudflare reader here. The isolate does not hand a
|
|
16
|
+
* handler its own CPU time; `cpuTime` arrives on the trace event a **tail worker** sees.
|
|
17
|
+
* Inventing an API that does not exist would produce a clock that reads zero for ever and
|
|
18
|
+
* a gate that can never go red — so the platform passes its reader in, or there is none.
|
|
19
|
+
*/
|
|
20
|
+
|
|
21
|
+
/** A started measurement. Call it to end the measurement and get CPU-ms. */
|
|
22
|
+
export type CpuSpan = () => number;
|
|
23
|
+
|
|
24
|
+
export interface CpuClock {
|
|
25
|
+
/** Human-readable provenance, e.g. `'process.cpuUsage'`. Printed in every report. */
|
|
26
|
+
readonly source: string;
|
|
27
|
+
/**
|
|
28
|
+
* 🔴 True when this number is a PROXY for Worker CPU-ms rather than a reading of it.
|
|
29
|
+
* The report prints it and {@link CpuBudgetReport} carries it downstream.
|
|
30
|
+
*/
|
|
31
|
+
readonly proxy: boolean;
|
|
32
|
+
/** True when the clock can actually measure. A false one records nothing. */
|
|
33
|
+
readonly available: boolean;
|
|
34
|
+
start(): CpuSpan;
|
|
35
|
+
}
|
|
36
|
+
|
|
37
|
+
/**
|
|
38
|
+
* `process.cpuUsage()` deltas — user + system, converted from µs to ms.
|
|
39
|
+
*
|
|
40
|
+
* 🔴 **The delta is PROCESS-WIDE, not per-request.** If two requests are in flight the
|
|
41
|
+
* reading for each includes the other's CPU, and any background work (a metrics flush, a
|
|
42
|
+
* GC pause attributed to the interval) lands on whichever handler happened to be open.
|
|
43
|
+
* That is why {@link runCpuBench} drives its cases SERIALLY — a serial driver is what
|
|
44
|
+
* makes this proxy sound enough to gate on. Mounted in production it is advisory: useful
|
|
45
|
+
* for ranking routes, not for a billing claim.
|
|
46
|
+
*/
|
|
47
|
+
export function processCpuClock(): CpuClock {
|
|
48
|
+
const usable = typeof process !== 'undefined' && typeof process.cpuUsage === 'function';
|
|
49
|
+
if (!usable) return unavailableCpuClock('process.cpuUsage (absent)');
|
|
50
|
+
return {
|
|
51
|
+
source: 'process.cpuUsage',
|
|
52
|
+
proxy: true,
|
|
53
|
+
available: true,
|
|
54
|
+
start(): CpuSpan {
|
|
55
|
+
const before = process.cpuUsage();
|
|
56
|
+
return () => {
|
|
57
|
+
const d = process.cpuUsage(before);
|
|
58
|
+
return (d.user + d.system) / 1000;
|
|
59
|
+
};
|
|
60
|
+
},
|
|
61
|
+
};
|
|
62
|
+
}
|
|
63
|
+
|
|
64
|
+
/**
|
|
65
|
+
* A clock over a CPU-ms reader the runtime supplies. `proxy: false` — this is the real
|
|
66
|
+
* quantity Cloudflare bills, so a report built on it may be quoted as one.
|
|
67
|
+
*
|
|
68
|
+
* @param readCpuMs Returns monotonically non-decreasing CPU-ms for the current invocation.
|
|
69
|
+
*/
|
|
70
|
+
export function workerCpuClock(readCpuMs: () => number, source = 'worker-runtime'): CpuClock {
|
|
71
|
+
return {
|
|
72
|
+
source,
|
|
73
|
+
proxy: false,
|
|
74
|
+
available: true,
|
|
75
|
+
start(): CpuSpan {
|
|
76
|
+
const before = readCpuMs();
|
|
77
|
+
return () => Math.max(0, readCpuMs() - before);
|
|
78
|
+
},
|
|
79
|
+
};
|
|
80
|
+
}
|
|
81
|
+
|
|
82
|
+
/**
|
|
83
|
+
* A clock that cannot measure. Recording through it is a no-op, and the report says so
|
|
84
|
+
* rather than reporting a confident zero.
|
|
85
|
+
*/
|
|
86
|
+
export function unavailableCpuClock(source = 'unavailable'): CpuClock {
|
|
87
|
+
return {
|
|
88
|
+
source,
|
|
89
|
+
proxy: true,
|
|
90
|
+
available: false,
|
|
91
|
+
start: () => () => Number.NaN,
|
|
92
|
+
};
|
|
93
|
+
}
|
|
94
|
+
|
|
95
|
+
/**
|
|
96
|
+
* A deterministic clock for tests: each `start()` consumes the next value in `values`,
|
|
97
|
+
* repeating the last one once exhausted. Lets the assertion be proven in both directions
|
|
98
|
+
* without burning real CPU, which is the difference between a test that is fast and
|
|
99
|
+
* reliable and one that is neither.
|
|
100
|
+
*/
|
|
101
|
+
export function fixedCpuClock(values: number[], source = 'fixed'): CpuClock {
|
|
102
|
+
if (values.length === 0) throw new Error('fixedCpuClock: needs at least one value');
|
|
103
|
+
let i = 0;
|
|
104
|
+
return {
|
|
105
|
+
source,
|
|
106
|
+
proxy: true,
|
|
107
|
+
available: true,
|
|
108
|
+
start(): CpuSpan {
|
|
109
|
+
const v = values[Math.min(i, values.length - 1)];
|
|
110
|
+
i += 1;
|
|
111
|
+
return () => v;
|
|
112
|
+
},
|
|
113
|
+
};
|
|
114
|
+
}
|
|
115
|
+
|
|
116
|
+
/** `workerCpuClock` when the platform supplied a reader, else the local proxy. */
|
|
117
|
+
export function resolveCpuClock(readCpuMs?: (() => number) | null): CpuClock {
|
|
118
|
+
return readCpuMs ? workerCpuClock(readCpuMs) : processCpuClock();
|
|
119
|
+
}
|
|
@@ -0,0 +1,81 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* `cursedbelt-server/bench` — per-route CPU attribution for a Hono app, a budget the app
|
|
3
|
+
* declares, and an assertion that reddens a build when a route outgrows it.
|
|
4
|
+
*
|
|
5
|
+
* ## The shape of a use
|
|
6
|
+
*
|
|
7
|
+
* ```ts
|
|
8
|
+
* // routes.budgets.ts — the app declares what each route may cost
|
|
9
|
+
* export const BUDGETS: CpuBudgetConfig = {
|
|
10
|
+
* routes: {
|
|
11
|
+
* 'GET /api/notes/:id': 4,
|
|
12
|
+
* 'POST /api/unlock': { exempt: true, reason: 'argon2id KDF — 332 CPU-ms, measured' },
|
|
13
|
+
* },
|
|
14
|
+
* }
|
|
15
|
+
*
|
|
16
|
+
* // app.ts
|
|
17
|
+
* export const cpu = createCpuRecorder({ clock: processCpuClock(), config: BUDGETS })
|
|
18
|
+
* app.use('*', cpuBudget({ recorder: cpu }))
|
|
19
|
+
*
|
|
20
|
+
* // cpuBudget.spec.ts — the gate
|
|
21
|
+
* const report = await runCpuBench({ app, recorder: cpu, cases: CASES })
|
|
22
|
+
* assertCpuBudgets(report, { requireDeclared: true })
|
|
23
|
+
* ```
|
|
24
|
+
*
|
|
25
|
+
* ## 🔴 Two things to carry away
|
|
26
|
+
*
|
|
27
|
+
* **Wall clock is never billed.** Cloudflare's own pricing page says *"No charge or limit
|
|
28
|
+
* for duration"*. `requestLogger` records duration; this records CPU; they are different
|
|
29
|
+
* numbers and only the second one costs money.
|
|
30
|
+
*
|
|
31
|
+
* **A local reading is a PROXY.** `process.cpuUsage()` on this Mac is not Worker CPU-ms.
|
|
32
|
+
* Every report carries `proxy: true` and every formatted report says so, because the
|
|
33
|
+
* alternative is a number that gets quoted as a billing fact.
|
|
34
|
+
*/
|
|
35
|
+
|
|
36
|
+
export {
|
|
37
|
+
type AssertCpuBudgetsOpts,
|
|
38
|
+
assertCpuBudgets,
|
|
39
|
+
checkCpuBudgets,
|
|
40
|
+
type CpuViolation,
|
|
41
|
+
type CpuViolationKind,
|
|
42
|
+
formatCpuBudgetReport,
|
|
43
|
+
} from './assert';
|
|
44
|
+
export {
|
|
45
|
+
type CpuBudgetConfig,
|
|
46
|
+
type CpuBudgetExemption,
|
|
47
|
+
type CpuBudgetLimit,
|
|
48
|
+
DAYS_PER_MONTH,
|
|
49
|
+
DEFAULT_ROUTE_CPU_BUDGET_MS,
|
|
50
|
+
deriveCpuBudgetMs,
|
|
51
|
+
isExemption,
|
|
52
|
+
monthlyOverageUsd,
|
|
53
|
+
OWNER_PROJECTED_REQUESTS_PER_DAY,
|
|
54
|
+
type ResolvedBudget,
|
|
55
|
+
resolveBudget,
|
|
56
|
+
type RouteBudget,
|
|
57
|
+
validateCpuBudgetConfig,
|
|
58
|
+
WORKERS_PAID_INCLUDED_CPU_MS,
|
|
59
|
+
WORKERS_PAID_INCLUDED_REQUESTS,
|
|
60
|
+
WORKERS_PAID_USD_PER_MILLION_CPU_MS,
|
|
61
|
+
WORKERS_PAID_USD_PER_MILLION_REQUESTS,
|
|
62
|
+
} from './budget';
|
|
63
|
+
export { type CpuBudgetOpts, cpuBudget } from './cpuBudget';
|
|
64
|
+
export {
|
|
65
|
+
type CpuClock,
|
|
66
|
+
type CpuSpan,
|
|
67
|
+
fixedCpuClock,
|
|
68
|
+
processCpuClock,
|
|
69
|
+
resolveCpuClock,
|
|
70
|
+
unavailableCpuClock,
|
|
71
|
+
workerCpuClock,
|
|
72
|
+
} from './cpuClock';
|
|
73
|
+
export {
|
|
74
|
+
type CpuBudgetReport,
|
|
75
|
+
type CpuRecorder,
|
|
76
|
+
type CpuRecorderOpts,
|
|
77
|
+
createCpuRecorder,
|
|
78
|
+
percentile,
|
|
79
|
+
type RouteCpuStats,
|
|
80
|
+
} from './recorder';
|
|
81
|
+
export { BENCH_ORIGIN, type BenchCase, runCpuBench, type RunCpuBenchOpts } from './runBench';
|
|
@@ -0,0 +1,163 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Per-route CPU samples, and the percentiles read off them.
|
|
3
|
+
*
|
|
4
|
+
* 🔴 **p99, not a mean.** A mean over a route that is fast 99 times and 400 ms once
|
|
5
|
+
* reports ~4 ms and looks healthy. The number that decides whether a route is safe to
|
|
6
|
+
* make common is its tail, so the mean is reported beside p99 and never instead of it.
|
|
7
|
+
*/
|
|
8
|
+
|
|
9
|
+
import {
|
|
10
|
+
type CpuBudgetConfig,
|
|
11
|
+
type ResolvedBudget,
|
|
12
|
+
resolveBudget,
|
|
13
|
+
validateCpuBudgetConfig,
|
|
14
|
+
} from './budget';
|
|
15
|
+
import type { CpuClock } from './cpuClock';
|
|
16
|
+
|
|
17
|
+
/**
|
|
18
|
+
* Nearest-rank percentile over an ASCENDING-sorted array: the smallest value at or below
|
|
19
|
+
* which at least `p`% of samples fall. No interpolation — an interpolated p99 reports a
|
|
20
|
+
* CPU cost no request ever actually paid, which is the wrong direction to be wrong in
|
|
21
|
+
* when the number is a ceiling.
|
|
22
|
+
*/
|
|
23
|
+
export function percentile(sortedAsc: readonly number[], p: number): number {
|
|
24
|
+
if (sortedAsc.length === 0) return Number.NaN;
|
|
25
|
+
if (p <= 0) return sortedAsc[0];
|
|
26
|
+
if (p >= 100) return sortedAsc[sortedAsc.length - 1];
|
|
27
|
+
const rank = Math.ceil((p / 100) * sortedAsc.length);
|
|
28
|
+
return sortedAsc[Math.min(sortedAsc.length, Math.max(1, rank)) - 1];
|
|
29
|
+
}
|
|
30
|
+
|
|
31
|
+
export interface RouteCpuStats {
|
|
32
|
+
method: string;
|
|
33
|
+
route: string;
|
|
34
|
+
/** `'GET /api/notes/:id'` — how the route is named in a violation message. */
|
|
35
|
+
key: string;
|
|
36
|
+
samples: number;
|
|
37
|
+
/** Samples actually retained (≤ `samples` once the reservoir is full). */
|
|
38
|
+
retained: number;
|
|
39
|
+
mean: number;
|
|
40
|
+
p50: number;
|
|
41
|
+
p95: number;
|
|
42
|
+
p99: number;
|
|
43
|
+
max: number;
|
|
44
|
+
/** Total CPU-ms observed across every sample — what the route costs in aggregate. */
|
|
45
|
+
totalCpuMs: number;
|
|
46
|
+
/** The ceiling in CPU-ms, or `null` when exempt. */
|
|
47
|
+
budgetMs: number | null;
|
|
48
|
+
exemptReason?: string;
|
|
49
|
+
noticeAboveMs?: number;
|
|
50
|
+
/** False when no explicit entry matched and the default was applied. */
|
|
51
|
+
declared: boolean;
|
|
52
|
+
}
|
|
53
|
+
|
|
54
|
+
export interface CpuBudgetReport {
|
|
55
|
+
/** Provenance of every number below — e.g. `'process.cpuUsage'`. */
|
|
56
|
+
source: string;
|
|
57
|
+
/** 🔴 True ⇒ these are a PROXY for Worker CPU-ms, not a reading of them. */
|
|
58
|
+
proxy: boolean;
|
|
59
|
+
/** False ⇒ nothing could be measured; every stat is NaN and nothing may be concluded. */
|
|
60
|
+
available: boolean;
|
|
61
|
+
routes: RouteCpuStats[];
|
|
62
|
+
}
|
|
63
|
+
|
|
64
|
+
export interface CpuRecorderOpts {
|
|
65
|
+
clock: CpuClock;
|
|
66
|
+
config?: CpuBudgetConfig;
|
|
67
|
+
/**
|
|
68
|
+
* Samples retained per route. Beyond this, reservoir sampling keeps the retained set
|
|
69
|
+
* representative of the WHOLE run rather than of its first N requests — truncating
|
|
70
|
+
* would quietly turn a long-lived mount into a measurement of its own warm-up.
|
|
71
|
+
*/
|
|
72
|
+
maxSamplesPerRoute?: number;
|
|
73
|
+
/**
|
|
74
|
+
* Deterministic source of randomness for the reservoir, for tests. Defaults to
|
|
75
|
+
* `Math.random`.
|
|
76
|
+
*/
|
|
77
|
+
random?: () => number;
|
|
78
|
+
}
|
|
79
|
+
|
|
80
|
+
export interface CpuRecorder {
|
|
81
|
+
readonly clock: CpuClock;
|
|
82
|
+
readonly config: CpuBudgetConfig;
|
|
83
|
+
record(method: string, route: string, cpuMs: number): void;
|
|
84
|
+
report(): CpuBudgetReport;
|
|
85
|
+
reset(): void;
|
|
86
|
+
}
|
|
87
|
+
|
|
88
|
+
interface Bucket {
|
|
89
|
+
method: string;
|
|
90
|
+
route: string;
|
|
91
|
+
seen: number;
|
|
92
|
+
samples: number[];
|
|
93
|
+
total: number;
|
|
94
|
+
}
|
|
95
|
+
|
|
96
|
+
export function createCpuRecorder(opts: CpuRecorderOpts): CpuRecorder {
|
|
97
|
+
const config = opts.config ?? {};
|
|
98
|
+
validateCpuBudgetConfig(config);
|
|
99
|
+
const cap = opts.maxSamplesPerRoute ?? 10_000;
|
|
100
|
+
if (!(cap > 0)) throw new Error('createCpuRecorder: maxSamplesPerRoute must be > 0');
|
|
101
|
+
const random = opts.random ?? Math.random;
|
|
102
|
+
let buckets = new Map<string, Bucket>();
|
|
103
|
+
|
|
104
|
+
return {
|
|
105
|
+
clock: opts.clock,
|
|
106
|
+
config,
|
|
107
|
+
record(method, route, cpuMs) {
|
|
108
|
+
// A NaN reading means the clock could not measure. Recording it would poison every
|
|
109
|
+
// percentile on the route, so an unmeasurable request is not a sample.
|
|
110
|
+
if (!Number.isFinite(cpuMs)) return;
|
|
111
|
+
const key = `${method} ${route}`;
|
|
112
|
+
let b = buckets.get(key);
|
|
113
|
+
if (!b) {
|
|
114
|
+
b = { method, route, seen: 0, samples: [], total: 0 };
|
|
115
|
+
buckets.set(key, b);
|
|
116
|
+
}
|
|
117
|
+
b.seen += 1;
|
|
118
|
+
b.total += cpuMs;
|
|
119
|
+
if (b.samples.length < cap) {
|
|
120
|
+
b.samples.push(cpuMs);
|
|
121
|
+
} else {
|
|
122
|
+
// Algorithm R: keep each observed sample with equal probability.
|
|
123
|
+
const j = Math.floor(random() * b.seen);
|
|
124
|
+
if (j < cap) b.samples[j] = cpuMs;
|
|
125
|
+
}
|
|
126
|
+
},
|
|
127
|
+
report(): CpuBudgetReport {
|
|
128
|
+
const routes: RouteCpuStats[] = [];
|
|
129
|
+
for (const [key, b] of buckets) {
|
|
130
|
+
const sorted = [...b.samples].sort((x, y) => x - y);
|
|
131
|
+
const resolved: ResolvedBudget = resolveBudget(config, b.method, b.route);
|
|
132
|
+
routes.push({
|
|
133
|
+
method: b.method,
|
|
134
|
+
route: b.route,
|
|
135
|
+
key,
|
|
136
|
+
samples: b.seen,
|
|
137
|
+
retained: sorted.length,
|
|
138
|
+
mean: b.seen > 0 ? b.total / b.seen : Number.NaN,
|
|
139
|
+
p50: percentile(sorted, 50),
|
|
140
|
+
p95: percentile(sorted, 95),
|
|
141
|
+
p99: percentile(sorted, 99),
|
|
142
|
+
max: percentile(sorted, 100),
|
|
143
|
+
totalCpuMs: b.total,
|
|
144
|
+
budgetMs: resolved.cpuMs,
|
|
145
|
+
exemptReason: resolved.exemptReason,
|
|
146
|
+
noticeAboveMs: resolved.noticeAboveMs,
|
|
147
|
+
declared: resolved.declared,
|
|
148
|
+
});
|
|
149
|
+
}
|
|
150
|
+
// Worst offender first — a report is read from the top.
|
|
151
|
+
routes.sort((a, z) => (z.p99 || 0) - (a.p99 || 0));
|
|
152
|
+
return {
|
|
153
|
+
source: opts.clock.source,
|
|
154
|
+
proxy: opts.clock.proxy,
|
|
155
|
+
available: opts.clock.available,
|
|
156
|
+
routes,
|
|
157
|
+
};
|
|
158
|
+
},
|
|
159
|
+
reset() {
|
|
160
|
+
buckets = new Map();
|
|
161
|
+
},
|
|
162
|
+
};
|
|
163
|
+
}
|
|
@@ -0,0 +1,110 @@
|
|
|
1
|
+
import type { CpuRecorder } from './recorder';
|
|
2
|
+
import type { CpuBudgetReport } from './recorder';
|
|
3
|
+
|
|
4
|
+
/**
|
|
5
|
+
* Drive a Hono app's routes and measure what each costs in CPU.
|
|
6
|
+
*
|
|
7
|
+
* 🔴 **Serial on purpose, and this is the load-bearing decision.** `process.cpuUsage()`
|
|
8
|
+
* deltas are PROCESS-wide (see `cpuClock.ts`), so two requests in flight each absorb the
|
|
9
|
+
* other's CPU and every number is inflated by an amount nobody can subtract afterwards.
|
|
10
|
+
* Running one request at a time is what makes the local proxy sound enough to gate on.
|
|
11
|
+
* There is deliberately no `concurrency` option — it would produce numbers that look
|
|
12
|
+
* finer-grained and are strictly less true.
|
|
13
|
+
*
|
|
14
|
+
* Wall-clock cost of the bench itself is irrelevant; it is a gate, not a load test.
|
|
15
|
+
*/
|
|
16
|
+
|
|
17
|
+
/** RFC 2606 reserved TLD — a bench request must never be able to leave the process. */
|
|
18
|
+
export const BENCH_ORIGIN = 'http://cpu-bench.invalid';
|
|
19
|
+
|
|
20
|
+
export interface BenchCase {
|
|
21
|
+
/** Default `'GET'`. */
|
|
22
|
+
method?: string;
|
|
23
|
+
/** Path with a leading slash, e.g. `'/api/notes/abc'`. */
|
|
24
|
+
path: string;
|
|
25
|
+
/** Extra request init — headers, body. `method` above wins over `init.method`. */
|
|
26
|
+
init?: RequestInit;
|
|
27
|
+
/** Overrides the run-wide `iterations` for this case. */
|
|
28
|
+
iterations?: number;
|
|
29
|
+
/**
|
|
30
|
+
* Statuses this case is expected to return. Default: anything `< 400`.
|
|
31
|
+
*
|
|
32
|
+
* 🔴 This guard is the point. A route that 404s or 500s costs almost no CPU, so a
|
|
33
|
+
* bench that does not check the response reports a beautiful number for a handler
|
|
34
|
+
* that never ran — a green budget over a broken route.
|
|
35
|
+
*/
|
|
36
|
+
expectStatus?: number | number[] | ((status: number) => boolean);
|
|
37
|
+
}
|
|
38
|
+
|
|
39
|
+
export interface RunCpuBenchOpts {
|
|
40
|
+
/** Anything with Hono's `fetch` shape. */
|
|
41
|
+
app: { fetch: (req: Request, env?: unknown, ctx?: unknown) => Response | Promise<Response> };
|
|
42
|
+
/** The recorder the app's `cpuBudget()` middleware writes through. */
|
|
43
|
+
recorder: CpuRecorder;
|
|
44
|
+
cases: BenchCase[];
|
|
45
|
+
/** Measured iterations per case. Default 30 — above `assertCpuBudgets`' 20-sample floor. */
|
|
46
|
+
iterations?: number;
|
|
47
|
+
/**
|
|
48
|
+
* Unmeasured iterations per case, run first and then DISCARDED.
|
|
49
|
+
*
|
|
50
|
+
* 🔴 Without this the bench measures JIT warm-up. A first call through a cold handler
|
|
51
|
+
* can cost an order of magnitude more than its steady state, and with 30 samples one
|
|
52
|
+
* cold call lands squarely in the p99 — so the route that fails is whichever one the
|
|
53
|
+
* bench happened to touch first. Default 5.
|
|
54
|
+
*/
|
|
55
|
+
warmup?: number;
|
|
56
|
+
/** Origin for the synthesized requests. Default {@link BENCH_ORIGIN}. */
|
|
57
|
+
origin?: string;
|
|
58
|
+
/** Passed through to `app.fetch` as the Worker `env`. */
|
|
59
|
+
env?: unknown;
|
|
60
|
+
/** Passed through to `app.fetch` as the Worker execution context. */
|
|
61
|
+
ctx?: unknown;
|
|
62
|
+
}
|
|
63
|
+
|
|
64
|
+
function statusAllowed(c: BenchCase, status: number): boolean {
|
|
65
|
+
const expect = c.expectStatus;
|
|
66
|
+
if (expect === undefined) return status < 400;
|
|
67
|
+
if (typeof expect === 'function') return expect(status);
|
|
68
|
+
if (Array.isArray(expect)) return expect.includes(status);
|
|
69
|
+
return expect === status;
|
|
70
|
+
}
|
|
71
|
+
|
|
72
|
+
function describe(c: BenchCase): string {
|
|
73
|
+
return `${(c.method ?? 'GET').toUpperCase()} ${c.path}`;
|
|
74
|
+
}
|
|
75
|
+
|
|
76
|
+
async function fire(opts: RunCpuBenchOpts, c: BenchCase): Promise<void> {
|
|
77
|
+
const method = (c.method ?? c.init?.method ?? 'GET').toUpperCase();
|
|
78
|
+
const req = new Request(`${opts.origin ?? BENCH_ORIGIN}${c.path}`, { ...c.init, method });
|
|
79
|
+
const res = await opts.app.fetch(req, opts.env, opts.ctx);
|
|
80
|
+
if (!statusAllowed(c, res.status)) {
|
|
81
|
+
const body = await res.text().catch(() => '');
|
|
82
|
+
throw new Error(
|
|
83
|
+
`cpu-bench: ${describe(c)} returned ${res.status}, which this case does not expect. ` +
|
|
84
|
+
'A route that errors costs no CPU, so measuring it would report a budget it never ' +
|
|
85
|
+
`met. Fix the case or set expectStatus.${body ? ` Body: ${body.slice(0, 200)}` : ''}`,
|
|
86
|
+
);
|
|
87
|
+
}
|
|
88
|
+
// Drain the body so a streamed response's work is actually done before the span ends.
|
|
89
|
+
if (res.body && !res.bodyUsed) await res.arrayBuffer().catch(() => undefined);
|
|
90
|
+
}
|
|
91
|
+
|
|
92
|
+
export async function runCpuBench(opts: RunCpuBenchOpts): Promise<CpuBudgetReport> {
|
|
93
|
+
if (opts.cases.length === 0) throw new Error('runCpuBench: no cases given');
|
|
94
|
+
const iterations = opts.iterations ?? 30;
|
|
95
|
+
const warmup = opts.warmup ?? 5;
|
|
96
|
+
if (!(iterations > 0)) throw new Error('runCpuBench: iterations must be > 0');
|
|
97
|
+
|
|
98
|
+
for (const c of opts.cases) {
|
|
99
|
+
for (let i = 0; i < warmup; i += 1) await fire(opts, c);
|
|
100
|
+
}
|
|
101
|
+
// 🔴 Everything above was warm-up. Discard it — measuring it is the bug this guards.
|
|
102
|
+
opts.recorder.reset();
|
|
103
|
+
|
|
104
|
+
for (const c of opts.cases) {
|
|
105
|
+
const n = c.iterations ?? iterations;
|
|
106
|
+
for (let i = 0; i < n; i += 1) await fire(opts, c);
|
|
107
|
+
}
|
|
108
|
+
|
|
109
|
+
return opts.recorder.report();
|
|
110
|
+
}
|
|
@@ -0,0 +1,121 @@
|
|
|
1
|
+
import { Database } from 'bun:sqlite';
|
|
2
|
+
import { afterEach, describe, expect, test } from 'bun:test';
|
|
3
|
+
import { existsSync, mkdtempSync, rmSync, statSync } from 'node:fs';
|
|
4
|
+
import { tmpdir } from 'node:os';
|
|
5
|
+
import { join } from 'node:path';
|
|
6
|
+
import { backupFor, checkpointWal, createLocalBackup, createTimeTravelBackup } from './backup';
|
|
7
|
+
|
|
8
|
+
const dirs: string[] = [];
|
|
9
|
+
const scratch = (): string => {
|
|
10
|
+
const d = mkdtempSync(join(tmpdir(), 'd1-backup-'));
|
|
11
|
+
dirs.push(d);
|
|
12
|
+
return d;
|
|
13
|
+
};
|
|
14
|
+
afterEach(() => {
|
|
15
|
+
for (const d of dirs.splice(0)) rmSync(d, { recursive: true, force: true });
|
|
16
|
+
});
|
|
17
|
+
|
|
18
|
+
/** A WAL-mode database on disk with enough rows to produce a real WAL file. */
|
|
19
|
+
function seeded(dir: string): { path: string; db: Database } {
|
|
20
|
+
const path = join(dir, 'app.sqlite');
|
|
21
|
+
const db = new Database(path, { create: true });
|
|
22
|
+
db.exec('PRAGMA journal_mode = WAL');
|
|
23
|
+
db.run('CREATE TABLE t (id INTEGER PRIMARY KEY, v TEXT)');
|
|
24
|
+
const insert = db.prepare('INSERT INTO t (v) VALUES (?)');
|
|
25
|
+
for (let i = 0; i < 2_000; i++) insert.run(`row-${i}`);
|
|
26
|
+
return { path, db };
|
|
27
|
+
}
|
|
28
|
+
|
|
29
|
+
describe('the local half keeps checkpointing the WAL', () => {
|
|
30
|
+
test('🔴 wal_checkpoint(TRUNCATE) actually drains the WAL file', () => {
|
|
31
|
+
// The measurement this exists for: `family.sqlite` was once 2.2 MB of principal
|
|
32
|
+
// against a 2.1 MB WAL. An un-checkpointed copy is half a database.
|
|
33
|
+
const dir = scratch();
|
|
34
|
+
const { path, db } = seeded(dir);
|
|
35
|
+
const walPath = `${path}-wal`;
|
|
36
|
+
|
|
37
|
+
expect(existsSync(walPath)).toBe(true);
|
|
38
|
+
expect(statSync(walPath).size).toBeGreaterThan(0);
|
|
39
|
+
|
|
40
|
+
const result = checkpointWal(db);
|
|
41
|
+
expect(result.busy).toBe(false);
|
|
42
|
+
// 🔴 The post-condition is the FILE SIZE, not the pragma's counters. Measured
|
|
43
|
+
// 2026-09-16: under TRUNCATE, SQLite reports the counters from after the
|
|
44
|
+
// truncation, so a fully successful checkpoint returns `checkpointed: 0`. A test
|
|
45
|
+
// asserting `checkpointed > 0` would be asserting a structurally impossible value
|
|
46
|
+
// — it fails on a working checkpoint, which is how a real check gets deleted.
|
|
47
|
+
expect(result.checkpointed).toBe(0);
|
|
48
|
+
// TRUNCATE, not PASSIVE: the file is emptied, not merely merged.
|
|
49
|
+
expect(statSync(walPath).size).toBe(0);
|
|
50
|
+
db.close();
|
|
51
|
+
});
|
|
52
|
+
|
|
53
|
+
test('capture() checkpoints and writes a readable snapshot containing every row', async () => {
|
|
54
|
+
const dir = scratch();
|
|
55
|
+
const { path, db } = seeded(dir);
|
|
56
|
+
const destDir = join(dir, 'backups');
|
|
57
|
+
|
|
58
|
+
const point = await createLocalBackup({ db, sourcePath: path, destDir }).capture();
|
|
59
|
+
|
|
60
|
+
expect(point.kind).toBe('file');
|
|
61
|
+
expect(existsSync(point.ref)).toBe(true);
|
|
62
|
+
expect(point.bytes).toBeGreaterThan(0);
|
|
63
|
+
expect(point.exportCommand).toBeNull();
|
|
64
|
+
|
|
65
|
+
// The snapshot is a real database with the full row count — which is what would
|
|
66
|
+
// have failed had the WAL not been merged in.
|
|
67
|
+
const restored = new Database(point.ref, { readonly: true });
|
|
68
|
+
const count = restored.query('SELECT COUNT(*) AS n FROM t').get() as { n: number };
|
|
69
|
+
expect(count.n).toBe(2_000);
|
|
70
|
+
restored.close();
|
|
71
|
+
db.close();
|
|
72
|
+
});
|
|
73
|
+
|
|
74
|
+
test('the local half reports that it DOES need a periodic capture', () => {
|
|
75
|
+
const dir = scratch();
|
|
76
|
+
const { path, db } = seeded(dir);
|
|
77
|
+
expect(createLocalBackup({ db, sourcePath: path, destDir: dir }).needsPeriodicCapture).toBe(true);
|
|
78
|
+
db.close();
|
|
79
|
+
});
|
|
80
|
+
});
|
|
81
|
+
|
|
82
|
+
describe('the D1 half records a coordinate instead of copying bytes', () => {
|
|
83
|
+
const at = new Date('2026-09-16T04:30:00.000Z');
|
|
84
|
+
|
|
85
|
+
test('capture() returns a restorable timestamp and the two real commands', async () => {
|
|
86
|
+
const point = await createTimeTravelBackup({ databaseName: 'patterns', now: () => at }).capture();
|
|
87
|
+
|
|
88
|
+
expect(point.kind).toBe('time-travel');
|
|
89
|
+
expect(point.ref).toBe('2026-09-16T04:30:00.000Z');
|
|
90
|
+
expect(point.restoreCommand).toBe(
|
|
91
|
+
'wrangler d1 time-travel restore patterns --timestamp=2026-09-16T04:30:00.000Z',
|
|
92
|
+
);
|
|
93
|
+
expect(point.exportCommand).toContain('wrangler d1 export patterns --remote');
|
|
94
|
+
});
|
|
95
|
+
|
|
96
|
+
test('🔴 it reports needsPeriodicCapture: false — Time Travel is automatic', () => {
|
|
97
|
+
// A nightly no-op job logging "backup complete" is worse than no job, because it
|
|
98
|
+
// reads as evidence. A scheduler must honour this flag.
|
|
99
|
+
expect(createTimeTravelBackup({ databaseName: 'patterns' }).needsPeriodicCapture).toBe(false);
|
|
100
|
+
});
|
|
101
|
+
|
|
102
|
+
test('bytes is null rather than a fabricated zero', () => {
|
|
103
|
+
// Cloudflare reports no size for a restore point; a 0 would read as a measurement.
|
|
104
|
+
return createTimeTravelBackup({ databaseName: 'x', now: () => at })
|
|
105
|
+
.capture()
|
|
106
|
+
.then((p) => expect(p.bytes).toBeNull());
|
|
107
|
+
});
|
|
108
|
+
});
|
|
109
|
+
|
|
110
|
+
describe('backupFor picks the implementation from the driver flavor', () => {
|
|
111
|
+
test('local flavor checkpoints; d1 flavor does not need to', async () => {
|
|
112
|
+
const dir = scratch();
|
|
113
|
+
const { path, db } = seeded(dir);
|
|
114
|
+
const local = () => ({ db, sourcePath: path, destDir: join(dir, 'b') });
|
|
115
|
+
const remote = () => ({ databaseName: 'patterns' });
|
|
116
|
+
|
|
117
|
+
expect(backupFor('local', local, remote).needsPeriodicCapture).toBe(true);
|
|
118
|
+
expect(backupFor('d1', local, remote).needsPeriodicCapture).toBe(false);
|
|
119
|
+
db.close();
|
|
120
|
+
});
|
|
121
|
+
});
|