@pithy-sh/leaderboard 0.1.5 → 0.1.6
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/rank/lock.d.ts +42 -1
- package/dist/rank/lock.d.ts.map +1 -1
- package/dist/rank/lock.js +53 -3
- package/dist/rank/worker.entry.d.ts.map +1 -1
- package/dist/rank/worker.entry.js +2 -2
- package/dist/version.generated.d.ts +1 -1
- package/dist/version.generated.js +1 -1
- package/package.json +2 -2
- package/src/rank/lock.ts +60 -2
- package/src/rank/worker.entry.ts +7 -2
- package/src/rank/wrangler.jsonc +3 -1
- package/src/version.generated.ts +1 -1
package/dist/rank/lock.d.ts
CHANGED
|
@@ -21,9 +21,50 @@ export declare const REFRESH_LOCK = "rank-refresh";
|
|
|
21
21
|
* A refresh that finishes releases the lock immediately, so this only matters when an instance dies
|
|
22
22
|
* mid-pass. It must be comfortably longer than any real refresh so a slow-but-alive instance is never
|
|
23
23
|
* stolen from; one hour is far past the ~5-minute worst case at the ~1M-player shard boundary. Override
|
|
24
|
-
* with `LEADERBOARD_LOCK_STALE_MS` if a deployment refreshes boards larger than that
|
|
24
|
+
* with `LEADERBOARD_LOCK_STALE_MS` if a deployment refreshes boards larger than that, within the bounds
|
|
25
|
+
* {@link requireLockStaleMs} enforces.
|
|
25
26
|
*/
|
|
26
27
|
export declare const DEFAULT_LOCK_STALE_MS: number;
|
|
28
|
+
/**
|
|
29
|
+
* The longest stale horizon this lock accepts — a day, and it is refused rather than clamped.
|
|
30
|
+
*
|
|
31
|
+
* The ceiling is not decoration on the finiteness check, for the reason `MAX_SECRETS_CACHE_TTL_SECONDS`
|
|
32
|
+
* records: `Infinity` is the loud spelling of *never reclaim*, and a number typed in the wrong unit is
|
|
33
|
+
* the quiet one. `3_600_000` is the default written in milliseconds, which is right; the same number
|
|
34
|
+
* typed as seconds-since-somebody-thought-it-was-seconds is `3_600_000_000` — forty-one days, during
|
|
35
|
+
* which a crashed instance's lock is never reclaimed, every cron fire stands down, and the ranks a
|
|
36
|
+
* board serves quietly stop moving. There is no error in that, and no line in the log; the boards just
|
|
37
|
+
* stop. A day is far past the ~5-minute worst case at the ~1M-player shard boundary and still short
|
|
38
|
+
* enough that a wedged lock is a today problem.
|
|
39
|
+
*/
|
|
40
|
+
export declare const MAX_LOCK_STALE_MS: number;
|
|
41
|
+
/**
|
|
42
|
+
* Refuse a stale horizon that is not a duration — checked before anything is compared to it.
|
|
43
|
+
*
|
|
44
|
+
* `Number(env.LEADERBOARD_LOCK_STALE_MS)` is `NaN` for `"1h"`, `"3600 ms"` or a typo, and `Infinity`
|
|
45
|
+
* for `"1e999"`. Neither widens the horizon, and neither is caught by the `??` default one frame up,
|
|
46
|
+
* which answers `undefined` and nothing else. What each does instead is the whole reason this exists,
|
|
47
|
+
* and the two directions are different failures:
|
|
48
|
+
*
|
|
49
|
+
* - **Fails open — a negative or zero horizon.** `staleBefore` lands at or after `now`, so the takeover
|
|
50
|
+
* `WHERE acquiredAt < staleBefore` is true of *every* row, fresh ones included. Each cron fire steals
|
|
51
|
+
* the lock from the instance still holding it, both keep writing, and their chunked rank writes
|
|
52
|
+
* interleave into the duplicate-and-gapped rank set the lock exists to prevent. At-most-one is gone,
|
|
53
|
+
* silently, and every board reads as ranked.
|
|
54
|
+
* - **Fails obscurely — `NaN`, `Infinity`, or past the JS date range.** `new Date(now - NaN)` is an
|
|
55
|
+
* Invalid Date, which `SQLiteDate`'s encode side refuses, so a `ZodError` — *"expected date, received
|
|
56
|
+
* Date"* — leaves `acquireRefreshLock` outside any Workflow step and kills the run. It names nothing
|
|
57
|
+
* an operator can act on, and it does it on every fire.
|
|
58
|
+
*
|
|
59
|
+
* There is no safe number to clamp a typo to, so it is named and refused, exactly as
|
|
60
|
+
* `@pithy-sh/email`'s `assertBatchSize` refuses `SCHEDULER_BATCH_SIZE` (#250) — this repository's own
|
|
61
|
+
* precedent, and the production bug that taught it. `core/internal`, because the number is ours to fix
|
|
62
|
+
* and the operator reading our logs is who can fix it (#521).
|
|
63
|
+
*
|
|
64
|
+
* It returns the value it checked so the env boundary reads as one expression: the coercion cannot be
|
|
65
|
+
* written without the check beside it, which is what `ci/environmentNumbers.test.ts` gates repo-wide.
|
|
66
|
+
*/
|
|
67
|
+
export declare function requireLockStaleMs(staleMs: number): number;
|
|
27
68
|
/**
|
|
28
69
|
* Try to take the refresh lock for `holder`, returning whether it was acquired.
|
|
29
70
|
*
|
package/dist/rank/lock.d.ts.map
CHANGED
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"lock.d.ts","sourceRoot":"","sources":["../../src/rank/lock.ts"],"names":[],"mappings":"
|
|
1
|
+
{"version":3,"file":"lock.d.ts","sourceRoot":"","sources":["../../src/rank/lock.ts"],"names":[],"mappings":"AAKA,OAAO,EAA2B,KAAK,mBAAmB,EAAE,MAAM,gBAAgB,CAAC;AAEnF;;;;;;;;;;;GAWG;AAEH,4FAA4F;AAC5F,eAAO,MAAM,YAAY,iBAAiB,CAAC;AAE3C;;;;;;;;GAQG;AACH,eAAO,MAAM,qBAAqB,QAAiB,CAAC;AAEpD;;;;;;;;;;;GAWG;AACH,eAAO,MAAM,iBAAiB,QAAsB,CAAC;AAErD;;;;;;;;;;;;;;;;;;;;;;;;;GAyBG;AACH,wBAAgB,kBAAkB,CAAC,OAAO,EAAE,MAAM,GAAG,MAAM,CAS1D;AAED;;;;;;;;GAQG;AACH,wBAAsB,kBAAkB,CACtC,EAAE,EAAE,mBAAmB,EACvB,MAAM,EAAE,MAAM,EACd,GAAG,EAAE,IAAI,EACT,OAAO,GAAE,MAA8B,GACtC,OAAO,CAAC,OAAO,CAAC,CA4BlB;AAED,oGAAoG;AACpG,wBAAsB,kBAAkB,CAAC,EAAE,EAAE,mBAAmB,EAAE,MAAM,EAAE,MAAM,GAAG,OAAO,CAAC,IAAI,CAAC,CAE/F"}
|
package/dist/rank/lock.js
CHANGED
|
@@ -2,6 +2,7 @@
|
|
|
2
2
|
// SPDX-License-Identifier: MIT
|
|
3
3
|
import { LeaderboardLock } from "../data/lock.js";
|
|
4
4
|
import { LEADERBOARD_LOCKS_TABLE } from "../data/tables.js";
|
|
5
|
+
import { InternalError } from "@pithy-sh/core/src/error/pithyError";
|
|
5
6
|
//#region src/rank/lock.ts
|
|
6
7
|
/**
|
|
7
8
|
* The rank-refresh advisory lock: at most one refresh runs at a time.
|
|
@@ -23,10 +24,58 @@ const REFRESH_LOCK = "rank-refresh";
|
|
|
23
24
|
* A refresh that finishes releases the lock immediately, so this only matters when an instance dies
|
|
24
25
|
* mid-pass. It must be comfortably longer than any real refresh so a slow-but-alive instance is never
|
|
25
26
|
* stolen from; one hour is far past the ~5-minute worst case at the ~1M-player shard boundary. Override
|
|
26
|
-
* with `LEADERBOARD_LOCK_STALE_MS` if a deployment refreshes boards larger than that
|
|
27
|
+
* with `LEADERBOARD_LOCK_STALE_MS` if a deployment refreshes boards larger than that, within the bounds
|
|
28
|
+
* {@link requireLockStaleMs} enforces.
|
|
27
29
|
*/
|
|
28
30
|
const DEFAULT_LOCK_STALE_MS = 36e5;
|
|
29
31
|
/**
|
|
32
|
+
* The longest stale horizon this lock accepts — a day, and it is refused rather than clamped.
|
|
33
|
+
*
|
|
34
|
+
* The ceiling is not decoration on the finiteness check, for the reason `MAX_SECRETS_CACHE_TTL_SECONDS`
|
|
35
|
+
* records: `Infinity` is the loud spelling of *never reclaim*, and a number typed in the wrong unit is
|
|
36
|
+
* the quiet one. `3_600_000` is the default written in milliseconds, which is right; the same number
|
|
37
|
+
* typed as seconds-since-somebody-thought-it-was-seconds is `3_600_000_000` — forty-one days, during
|
|
38
|
+
* which a crashed instance's lock is never reclaimed, every cron fire stands down, and the ranks a
|
|
39
|
+
* board serves quietly stop moving. There is no error in that, and no line in the log; the boards just
|
|
40
|
+
* stop. A day is far past the ~5-minute worst case at the ~1M-player shard boundary and still short
|
|
41
|
+
* enough that a wedged lock is a today problem.
|
|
42
|
+
*/
|
|
43
|
+
const MAX_LOCK_STALE_MS = 864e5;
|
|
44
|
+
/**
|
|
45
|
+
* Refuse a stale horizon that is not a duration — checked before anything is compared to it.
|
|
46
|
+
*
|
|
47
|
+
* `Number(env.LEADERBOARD_LOCK_STALE_MS)` is `NaN` for `"1h"`, `"3600 ms"` or a typo, and `Infinity`
|
|
48
|
+
* for `"1e999"`. Neither widens the horizon, and neither is caught by the `??` default one frame up,
|
|
49
|
+
* which answers `undefined` and nothing else. What each does instead is the whole reason this exists,
|
|
50
|
+
* and the two directions are different failures:
|
|
51
|
+
*
|
|
52
|
+
* - **Fails open — a negative or zero horizon.** `staleBefore` lands at or after `now`, so the takeover
|
|
53
|
+
* `WHERE acquiredAt < staleBefore` is true of *every* row, fresh ones included. Each cron fire steals
|
|
54
|
+
* the lock from the instance still holding it, both keep writing, and their chunked rank writes
|
|
55
|
+
* interleave into the duplicate-and-gapped rank set the lock exists to prevent. At-most-one is gone,
|
|
56
|
+
* silently, and every board reads as ranked.
|
|
57
|
+
* - **Fails obscurely — `NaN`, `Infinity`, or past the JS date range.** `new Date(now - NaN)` is an
|
|
58
|
+
* Invalid Date, which `SQLiteDate`'s encode side refuses, so a `ZodError` — *"expected date, received
|
|
59
|
+
* Date"* — leaves `acquireRefreshLock` outside any Workflow step and kills the run. It names nothing
|
|
60
|
+
* an operator can act on, and it does it on every fire.
|
|
61
|
+
*
|
|
62
|
+
* There is no safe number to clamp a typo to, so it is named and refused, exactly as
|
|
63
|
+
* `@pithy-sh/email`'s `assertBatchSize` refuses `SCHEDULER_BATCH_SIZE` (#250) — this repository's own
|
|
64
|
+
* precedent, and the production bug that taught it. `core/internal`, because the number is ours to fix
|
|
65
|
+
* and the operator reading our logs is who can fix it (#521).
|
|
66
|
+
*
|
|
67
|
+
* It returns the value it checked so the env boundary reads as one expression: the coercion cannot be
|
|
68
|
+
* written without the check beside it, which is what `ci/environmentNumbers.test.ts` gates repo-wide.
|
|
69
|
+
*/
|
|
70
|
+
function requireLockStaleMs(staleMs) {
|
|
71
|
+
if (!Number.isInteger(staleMs) || staleMs < 1 || staleMs > 864e5) throw new InternalError({
|
|
72
|
+
message: "The leaderboard rank refresh is misconfigured.",
|
|
73
|
+
action: `Set LEADERBOARD_LOCK_STALE_MS to a whole number of milliseconds from 1 to ${MAX_LOCK_STALE_MS}, or unset it for the default of ${DEFAULT_LOCK_STALE_MS}.`,
|
|
74
|
+
detail: `LEADERBOARD_LOCK_STALE_MS resolved to ${String(staleMs)}; a stale horizon must be a whole number of milliseconds from 1 to ${MAX_LOCK_STALE_MS}. At or below zero every cron fire steals a live holder's lock and two refreshes interleave their rank writes; above the range the horizon is not a date at all and the refresh cannot run.`
|
|
75
|
+
});
|
|
76
|
+
return staleMs;
|
|
77
|
+
}
|
|
78
|
+
/**
|
|
30
79
|
* Try to take the refresh lock for `holder`, returning whether it was acquired.
|
|
31
80
|
*
|
|
32
81
|
* Atomic on D1's single thread: the upsert either inserts the row (no holder yet) or, on conflict, takes
|
|
@@ -36,7 +85,8 @@ const DEFAULT_LOCK_STALE_MS = 36e5;
|
|
|
36
85
|
* to stand down.
|
|
37
86
|
*/
|
|
38
87
|
async function acquireRefreshLock(db, holder, now, staleMs = DEFAULT_LOCK_STALE_MS) {
|
|
39
|
-
const
|
|
88
|
+
const horizon = requireLockStaleMs(staleMs);
|
|
89
|
+
const staleBefore = LeaderboardLock.shape.acquiredAt.encode(new Date(now.getTime() - horizon));
|
|
40
90
|
const row = LeaderboardLock.encode({
|
|
41
91
|
name: REFRESH_LOCK,
|
|
42
92
|
holder,
|
|
@@ -53,4 +103,4 @@ async function releaseRefreshLock(db, holder) {
|
|
|
53
103
|
await db.deleteFrom(LEADERBOARD_LOCKS_TABLE).where("name", "=", REFRESH_LOCK).where("holder", "=", holder).execute();
|
|
54
104
|
}
|
|
55
105
|
//#endregion
|
|
56
|
-
export { DEFAULT_LOCK_STALE_MS, REFRESH_LOCK, acquireRefreshLock, releaseRefreshLock };
|
|
106
|
+
export { DEFAULT_LOCK_STALE_MS, MAX_LOCK_STALE_MS, REFRESH_LOCK, acquireRefreshLock, releaseRefreshLock, requireLockStaleMs };
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"worker.entry.d.ts","sourceRoot":"","sources":["../../src/rank/worker.entry.ts"],"names":[],"mappings":"AAGA,OAAO,EAAE,kBAAkB,EAAE,KAAK,aAAa,EAAE,KAAK,YAAY,EAAE,MAAM,oBAAoB,CAAC;AAE/F,OAAO,KAAK,EAAE,UAAU,EAAE,MAAM,2BAA2B,CAAC;AAW5D;;;;;;;;;;;;;;;;;;;;;;;;;;GA0BG;AAEH,UAAU,aAAa;IACrB,EAAE,EAAE,UAAU,CAAC;IACf,yGAAyG;IACzG,kBAAkB,EAAE,MAAM,CAAC;IAC3B,2GAA2G;IAC3G,yBAAyB,CAAC,EAAE,MAAM,CAAC;IACnC,kGAAkG;IAClG,YAAY,EAAE;QAAE,MAAM,CAAC,OAAO,CAAC,EAAE;YAAE,EAAE,CAAC,EAAE,MAAM,CAAA;SAAE,GAAG,OAAO,CAAC,OAAO,CAAC,CAAA;KAAE,CAAC;CACvE;AAED,qBAAa,mBAAoB,SAAQ,kBAAkB,CAAC,aAAa,EAAE,OAAO,CAAC;IAClE,GAAG,CAAC,MAAM,EAAE,aAAa,CAAC,OAAO,CAAC,EAAE,IAAI,EAAE,YAAY,GAAG,OAAO,CAAC,IAAI,CAAC,
|
|
1
|
+
{"version":3,"file":"worker.entry.d.ts","sourceRoot":"","sources":["../../src/rank/worker.entry.ts"],"names":[],"mappings":"AAGA,OAAO,EAAE,kBAAkB,EAAE,KAAK,aAAa,EAAE,KAAK,YAAY,EAAE,MAAM,oBAAoB,CAAC;AAE/F,OAAO,KAAK,EAAE,UAAU,EAAE,MAAM,2BAA2B,CAAC;AAW5D;;;;;;;;;;;;;;;;;;;;;;;;;;GA0BG;AAEH,UAAU,aAAa;IACrB,EAAE,EAAE,UAAU,CAAC;IACf,yGAAyG;IACzG,kBAAkB,EAAE,MAAM,CAAC;IAC3B,2GAA2G;IAC3G,yBAAyB,CAAC,EAAE,MAAM,CAAC;IACnC,kGAAkG;IAClG,YAAY,EAAE;QAAE,MAAM,CAAC,OAAO,CAAC,EAAE;YAAE,EAAE,CAAC,EAAE,MAAM,CAAA;SAAE,GAAG,OAAO,CAAC,OAAO,CAAC,CAAA;KAAE,CAAC;CACvE;AAED,qBAAa,mBAAoB,SAAQ,kBAAkB,CAAC,aAAa,EAAE,OAAO,CAAC;IAClE,GAAG,CAAC,MAAM,EAAE,aAAa,CAAC,OAAO,CAAC,EAAE,IAAI,EAAE,YAAY,GAAG,OAAO,CAAC,IAAI,CAAC,CAoEpF;CACF;;IAGC;;;;;;OAMG;IACG,SAAS,cAAc,OAAO,OAAO,aAAa,GAAG,OAAO,CAAC,IAAI,CAAC"}
|
|
@@ -3,7 +3,7 @@
|
|
|
3
3
|
import { windowKeyAt } from "../window/schedule.js";
|
|
4
4
|
import { LeaderboardConfig } from "../config/config.js";
|
|
5
5
|
import { leaderboardDatabase } from "../data/tables.js";
|
|
6
|
-
import { acquireRefreshLock, releaseRefreshLock } from "./lock.js";
|
|
6
|
+
import { acquireRefreshLock, releaseRefreshLock, requireLockStaleMs } from "./lock.js";
|
|
7
7
|
import { REFRESH_BATCH_CHUNKS, refreshWindowRanks } from "./materialize.js";
|
|
8
8
|
import { leaderboardWorkflowRetry } from "./retryPolicy.js";
|
|
9
9
|
import { pruneBoards } from "../retention/prune.js";
|
|
@@ -17,7 +17,7 @@ var RankRefreshWorkflow = class extends WorkflowEntrypoint {
|
|
|
17
17
|
const steps = classifiedSteps(step, leaderboardWorkflowRetry, NonRetryableError);
|
|
18
18
|
const config = LeaderboardConfig.parse(JSON.parse(this.env.LEADERBOARD_CONFIG));
|
|
19
19
|
const db = leaderboardDatabase(this.env.DB);
|
|
20
|
-
const staleMs = this.env.LEADERBOARD_LOCK_STALE_MS ? Number(this.env.LEADERBOARD_LOCK_STALE_MS) : void 0;
|
|
20
|
+
const staleMs = this.env.LEADERBOARD_LOCK_STALE_MS ? requireLockStaleMs(Number(this.env.LEADERBOARD_LOCK_STALE_MS)) : void 0;
|
|
21
21
|
const ctx = await steps.do("refresh-context", async () => ({
|
|
22
22
|
holder: crypto.randomUUID(),
|
|
23
23
|
nowMs: Date.now()
|
|
@@ -3,5 +3,5 @@
|
|
|
3
3
|
/** This package's npm name — the join key against a release feed. */
|
|
4
4
|
export declare const PACKAGE_NAME = "@pithy-sh/leaderboard";
|
|
5
5
|
/** This package's version, stamped from its own package.json at generation time. */
|
|
6
|
-
export declare const PACKAGE_VERSION = "0.1.
|
|
6
|
+
export declare const PACKAGE_VERSION = "0.1.6";
|
|
7
7
|
//# sourceMappingURL=version.generated.d.ts.map
|
|
@@ -4,6 +4,6 @@
|
|
|
4
4
|
/** This package's npm name — the join key against a release feed. */
|
|
5
5
|
const PACKAGE_NAME = "@pithy-sh/leaderboard";
|
|
6
6
|
/** This package's version, stamped from its own package.json at generation time. */
|
|
7
|
-
const PACKAGE_VERSION = "0.1.
|
|
7
|
+
const PACKAGE_VERSION = "0.1.6";
|
|
8
8
|
//#endregion
|
|
9
9
|
export { PACKAGE_NAME, PACKAGE_VERSION };
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@pithy-sh/leaderboard",
|
|
3
|
-
"version": "0.1.
|
|
3
|
+
"version": "0.1.6",
|
|
4
4
|
"license": "MIT",
|
|
5
5
|
"repository": {
|
|
6
6
|
"type": "git",
|
|
@@ -38,7 +38,7 @@
|
|
|
38
38
|
"dependencies": {
|
|
39
39
|
"@cloudflare/workers-types": "^5.20260729.1",
|
|
40
40
|
"@hono/zod-validator": "^0.9.0",
|
|
41
|
-
"@pithy-sh/core": "^0.
|
|
41
|
+
"@pithy-sh/core": "^0.3.0",
|
|
42
42
|
"croner": "^10.0.1"
|
|
43
43
|
},
|
|
44
44
|
"devDependencies": {
|
package/src/rank/lock.ts
CHANGED
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
// SPDX-FileCopyrightText: 2026 Pithy
|
|
2
2
|
// SPDX-License-Identifier: MIT
|
|
3
3
|
|
|
4
|
+
import { InternalError } from "@pithy-sh/core/src/error/pithyError";
|
|
4
5
|
import { LeaderboardLock } from "../data/lock";
|
|
5
6
|
import { LEADERBOARD_LOCKS_TABLE, type LeaderboardDatabase } from "../data/tables";
|
|
6
7
|
|
|
@@ -26,10 +27,62 @@ export const REFRESH_LOCK = "rank-refresh";
|
|
|
26
27
|
* A refresh that finishes releases the lock immediately, so this only matters when an instance dies
|
|
27
28
|
* mid-pass. It must be comfortably longer than any real refresh so a slow-but-alive instance is never
|
|
28
29
|
* stolen from; one hour is far past the ~5-minute worst case at the ~1M-player shard boundary. Override
|
|
29
|
-
* with `LEADERBOARD_LOCK_STALE_MS` if a deployment refreshes boards larger than that
|
|
30
|
+
* with `LEADERBOARD_LOCK_STALE_MS` if a deployment refreshes boards larger than that, within the bounds
|
|
31
|
+
* {@link requireLockStaleMs} enforces.
|
|
30
32
|
*/
|
|
31
33
|
export const DEFAULT_LOCK_STALE_MS = 60 * 60 * 1000;
|
|
32
34
|
|
|
35
|
+
/**
|
|
36
|
+
* The longest stale horizon this lock accepts — a day, and it is refused rather than clamped.
|
|
37
|
+
*
|
|
38
|
+
* The ceiling is not decoration on the finiteness check, for the reason `MAX_SECRETS_CACHE_TTL_SECONDS`
|
|
39
|
+
* records: `Infinity` is the loud spelling of *never reclaim*, and a number typed in the wrong unit is
|
|
40
|
+
* the quiet one. `3_600_000` is the default written in milliseconds, which is right; the same number
|
|
41
|
+
* typed as seconds-since-somebody-thought-it-was-seconds is `3_600_000_000` — forty-one days, during
|
|
42
|
+
* which a crashed instance's lock is never reclaimed, every cron fire stands down, and the ranks a
|
|
43
|
+
* board serves quietly stop moving. There is no error in that, and no line in the log; the boards just
|
|
44
|
+
* stop. A day is far past the ~5-minute worst case at the ~1M-player shard boundary and still short
|
|
45
|
+
* enough that a wedged lock is a today problem.
|
|
46
|
+
*/
|
|
47
|
+
export const MAX_LOCK_STALE_MS = 24 * 60 * 60 * 1000;
|
|
48
|
+
|
|
49
|
+
/**
|
|
50
|
+
* Refuse a stale horizon that is not a duration — checked before anything is compared to it.
|
|
51
|
+
*
|
|
52
|
+
* `Number(env.LEADERBOARD_LOCK_STALE_MS)` is `NaN` for `"1h"`, `"3600 ms"` or a typo, and `Infinity`
|
|
53
|
+
* for `"1e999"`. Neither widens the horizon, and neither is caught by the `??` default one frame up,
|
|
54
|
+
* which answers `undefined` and nothing else. What each does instead is the whole reason this exists,
|
|
55
|
+
* and the two directions are different failures:
|
|
56
|
+
*
|
|
57
|
+
* - **Fails open — a negative or zero horizon.** `staleBefore` lands at or after `now`, so the takeover
|
|
58
|
+
* `WHERE acquiredAt < staleBefore` is true of *every* row, fresh ones included. Each cron fire steals
|
|
59
|
+
* the lock from the instance still holding it, both keep writing, and their chunked rank writes
|
|
60
|
+
* interleave into the duplicate-and-gapped rank set the lock exists to prevent. At-most-one is gone,
|
|
61
|
+
* silently, and every board reads as ranked.
|
|
62
|
+
* - **Fails obscurely — `NaN`, `Infinity`, or past the JS date range.** `new Date(now - NaN)` is an
|
|
63
|
+
* Invalid Date, which `SQLiteDate`'s encode side refuses, so a `ZodError` — *"expected date, received
|
|
64
|
+
* Date"* — leaves `acquireRefreshLock` outside any Workflow step and kills the run. It names nothing
|
|
65
|
+
* an operator can act on, and it does it on every fire.
|
|
66
|
+
*
|
|
67
|
+
* There is no safe number to clamp a typo to, so it is named and refused, exactly as
|
|
68
|
+
* `@pithy-sh/email`'s `assertBatchSize` refuses `SCHEDULER_BATCH_SIZE` (#250) — this repository's own
|
|
69
|
+
* precedent, and the production bug that taught it. `core/internal`, because the number is ours to fix
|
|
70
|
+
* and the operator reading our logs is who can fix it (#521).
|
|
71
|
+
*
|
|
72
|
+
* It returns the value it checked so the env boundary reads as one expression: the coercion cannot be
|
|
73
|
+
* written without the check beside it, which is what `ci/environmentNumbers.test.ts` gates repo-wide.
|
|
74
|
+
*/
|
|
75
|
+
export function requireLockStaleMs(staleMs: number): number {
|
|
76
|
+
if (!Number.isInteger(staleMs) || staleMs < 1 || staleMs > MAX_LOCK_STALE_MS) {
|
|
77
|
+
throw new InternalError({
|
|
78
|
+
message: "The leaderboard rank refresh is misconfigured.",
|
|
79
|
+
action: `Set LEADERBOARD_LOCK_STALE_MS to a whole number of milliseconds from 1 to ${MAX_LOCK_STALE_MS}, or unset it for the default of ${DEFAULT_LOCK_STALE_MS}.`,
|
|
80
|
+
detail: `LEADERBOARD_LOCK_STALE_MS resolved to ${String(staleMs)}; a stale horizon must be a whole number of milliseconds from 1 to ${MAX_LOCK_STALE_MS}. At or below zero every cron fire steals a live holder's lock and two refreshes interleave their rank writes; above the range the horizon is not a date at all and the refresh cannot run.`,
|
|
81
|
+
});
|
|
82
|
+
}
|
|
83
|
+
return staleMs;
|
|
84
|
+
}
|
|
85
|
+
|
|
33
86
|
/**
|
|
34
87
|
* Try to take the refresh lock for `holder`, returning whether it was acquired.
|
|
35
88
|
*
|
|
@@ -45,7 +98,12 @@ export async function acquireRefreshLock(
|
|
|
45
98
|
now: Date,
|
|
46
99
|
staleMs: number = DEFAULT_LOCK_STALE_MS,
|
|
47
100
|
): Promise<boolean> {
|
|
48
|
-
|
|
101
|
+
// Before the upsert, and on the resolved value rather than on the caller's argument, so the check
|
|
102
|
+
// covers every route in: this worker's env var, an adopter who merged the Workflow into their own
|
|
103
|
+
// worker and passed a number of their own, and the default. The comparison two lines down is the only
|
|
104
|
+
// thing that reads this number, and it cannot be reached around.
|
|
105
|
+
const horizon = requireLockStaleMs(staleMs);
|
|
106
|
+
const staleBefore = LeaderboardLock.shape.acquiredAt.encode(new Date(now.getTime() - horizon));
|
|
49
107
|
const row = LeaderboardLock.encode({ name: REFRESH_LOCK, holder, acquiredAt: now });
|
|
50
108
|
|
|
51
109
|
await db
|
package/src/rank/worker.entry.ts
CHANGED
|
@@ -9,7 +9,7 @@ import { LeaderboardConfig } from "../config/config";
|
|
|
9
9
|
import { leaderboardDatabase } from "../data/tables";
|
|
10
10
|
import { pruneBoards } from "../retention/prune";
|
|
11
11
|
import { windowKeyAt } from "../window/schedule";
|
|
12
|
-
import { acquireRefreshLock, releaseRefreshLock } from "./lock";
|
|
12
|
+
import { acquireRefreshLock, releaseRefreshLock, requireLockStaleMs } from "./lock";
|
|
13
13
|
import { type Keyset, REFRESH_BATCH_CHUNKS, type RefreshResult, refreshWindowRanks } from "./materialize";
|
|
14
14
|
import { leaderboardWorkflowRetry } from "./retryPolicy";
|
|
15
15
|
import { materializedBoards } from "./worker";
|
|
@@ -62,7 +62,12 @@ export class RankRefreshWorkflow extends WorkflowEntrypoint<RankWorkerEnv, unkno
|
|
|
62
62
|
const steps = classifiedSteps(step, leaderboardWorkflowRetry, NonRetryableError);
|
|
63
63
|
const config = LeaderboardConfig.parse(JSON.parse(this.env.LEADERBOARD_CONFIG));
|
|
64
64
|
const db = leaderboardDatabase(this.env.DB);
|
|
65
|
-
|
|
65
|
+
// Checked here as well as inside `acquireRefreshLock`, and deliberately: this is where the string an
|
|
66
|
+
// operator typed becomes a number, so this is where the refusal can name the var. The coercion and the
|
|
67
|
+
// check are one expression because a bare `Number(env.…)` is what #521 keeps finding.
|
|
68
|
+
const staleMs = this.env.LEADERBOARD_LOCK_STALE_MS
|
|
69
|
+
? requireLockStaleMs(Number(this.env.LEADERBOARD_LOCK_STALE_MS))
|
|
70
|
+
: undefined;
|
|
66
71
|
|
|
67
72
|
// Mint this instance's identity and its clock ONCE, in a memoized step, and read them from the step's
|
|
68
73
|
// return value. A replay does not re-run the step body, so it reuses the same holder token and the
|
package/src/rank/wrangler.jsonc
CHANGED
|
@@ -46,7 +46,9 @@
|
|
|
46
46
|
"LEADERBOARD_CONFIG": "<filled-at-provision>",
|
|
47
47
|
// Optional. How long a held refresh lock stays valid before a crashed instance's lock is reclaimed,
|
|
48
48
|
// in ms. Defaults to one hour — raise it only if a deployment refreshes boards that take longer than
|
|
49
|
-
// that (well past the ~1M-player shard boundary).
|
|
49
|
+
// that (well past the ~1M-player shard boundary). A whole number of milliseconds from 1 to 86400000;
|
|
50
|
+
// anything else is refused by name on the first fire rather than quietly removing the lock. See
|
|
51
|
+
// `rank/lock.ts`.
|
|
50
52
|
// "LEADERBOARD_LOCK_STALE_MS": "3600000",
|
|
51
53
|
"ENVIRONMENT": "<filled-at-provision>"
|
|
52
54
|
}
|
package/src/version.generated.ts
CHANGED