@prosopo/user-access-policy 3.9.0 → 3.10.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.turbo/turbo-build$colon$cjs.log +35 -35
- package/.turbo/turbo-build$colon$tsc.log +14 -14
- package/.turbo/turbo-build.log +4 -4
- package/CHANGELOG.md +21 -0
- package/dist/cjs/redis/reader/redisAggregate.cjs +22 -4
- package/dist/cjs/redis/reader/redisRulesReader.cjs +133 -35
- package/dist/redis/reader/redisAggregate.d.ts +1 -1
- package/dist/redis/reader/redisAggregate.d.ts.map +1 -1
- package/dist/redis/reader/redisAggregate.js +22 -4
- package/dist/redis/reader/redisAggregate.js.map +1 -1
- package/dist/redis/reader/redisRulesReader.d.ts +4 -1
- package/dist/redis/reader/redisRulesReader.d.ts.map +1 -1
- package/dist/redis/reader/redisRulesReader.js +132 -34
- package/dist/redis/reader/redisRulesReader.js.map +1 -1
- package/dist/tests/redis/redisRulesReaderRank.benchmark.integration.test.d.ts +2 -0
- package/dist/tests/redis/redisRulesReaderRank.benchmark.integration.test.d.ts.map +1 -0
- package/dist/tests/redis/redisRulesReaderRank.benchmark.integration.test.js +189 -0
- package/dist/tests/redis/redisRulesReaderRank.benchmark.integration.test.js.map +1 -0
- package/dist/tests/redis/redisRulesStorage.integration.test.js +31 -0
- package/dist/tests/redis/redisRulesStorage.integration.test.js.map +1 -1
- package/package.json +3 -3
- package/src/redis/reader/redisAggregate.ts +27 -1
- package/src/redis/reader/redisRulesReader.ts +197 -37
- package/src/tests/redis/redisRulesReaderRank.benchmark.integration.test.ts +338 -0
- package/src/tests/redis/redisRulesStorage.integration.test.ts +52 -0
- package/tsconfig.tsbuildinfo +1 -1
- package/vite.test.config.ts +12 -1
|
@@ -29,6 +29,7 @@ export const aggregateRedisKeys = async (
|
|
|
29
29
|
query: string,
|
|
30
30
|
logger: Logger,
|
|
31
31
|
batchHandler?: (keys: string[]) => Promise<void>,
|
|
32
|
+
maxKeys?: number,
|
|
32
33
|
): Promise<string[]> => {
|
|
33
34
|
const keyField = "__key";
|
|
34
35
|
|
|
@@ -38,8 +39,13 @@ export const aggregateRedisKeys = async (
|
|
|
38
39
|
});
|
|
39
40
|
|
|
40
41
|
const foundKeys: string[] = [];
|
|
42
|
+
let stopRequested = false;
|
|
41
43
|
|
|
42
44
|
const addRecordKeys = async (records: object[]) => {
|
|
45
|
+
if (stopRequested) {
|
|
46
|
+
return;
|
|
47
|
+
}
|
|
48
|
+
|
|
43
49
|
const parsedRecords = parseRedisRecords(records, recordSchema, logger);
|
|
44
50
|
|
|
45
51
|
const recordKeys = parsedRecords.map((record) => record[keyField]);
|
|
@@ -47,6 +53,24 @@ export const aggregateRedisKeys = async (
|
|
|
47
53
|
if (batchHandler) {
|
|
48
54
|
await batchHandler(recordKeys);
|
|
49
55
|
} else {
|
|
56
|
+
if (
|
|
57
|
+
maxKeys !== undefined &&
|
|
58
|
+
foundKeys.length + recordKeys.length > maxKeys
|
|
59
|
+
) {
|
|
60
|
+
const remaining = Math.max(0, maxKeys - foundKeys.length);
|
|
61
|
+
foundKeys.push(...recordKeys.slice(0, remaining));
|
|
62
|
+
stopRequested = true;
|
|
63
|
+
|
|
64
|
+
logger.warn(() => ({
|
|
65
|
+
msg: "Redis aggregation candidate cap hit; truncating result set. This can suppress less-frequent rules and should be investigated.",
|
|
66
|
+
data: {
|
|
67
|
+
maxKeys,
|
|
68
|
+
query,
|
|
69
|
+
},
|
|
70
|
+
}));
|
|
71
|
+
return;
|
|
72
|
+
}
|
|
73
|
+
|
|
50
74
|
foundKeys.push(...recordKeys);
|
|
51
75
|
|
|
52
76
|
logger.debug(() => ({
|
|
@@ -68,6 +92,7 @@ export const aggregateRedisKeys = async (
|
|
|
68
92
|
LOAD: `@${keyField}`,
|
|
69
93
|
},
|
|
70
94
|
addRecordKeys,
|
|
95
|
+
() => stopRequested,
|
|
71
96
|
);
|
|
72
97
|
|
|
73
98
|
return foundKeys;
|
|
@@ -78,6 +103,7 @@ const executeAggregation = async (
|
|
|
78
103
|
query: string,
|
|
79
104
|
aggregateOptions: FtAggregateWithCursorOptions,
|
|
80
105
|
handleBatch: (records: object[]) => Promise<void>,
|
|
106
|
+
shouldStop?: () => boolean,
|
|
81
107
|
): Promise<void> => {
|
|
82
108
|
const initialReply = await client.ft.aggregateWithCursor(
|
|
83
109
|
ACCESS_RULES_REDIS_INDEX_NAME,
|
|
@@ -89,7 +115,7 @@ const executeAggregation = async (
|
|
|
89
115
|
|
|
90
116
|
let cursor = initialReply.cursor;
|
|
91
117
|
|
|
92
|
-
while (0 !== cursor) {
|
|
118
|
+
while (0 !== cursor && !shouldStop?.()) {
|
|
93
119
|
const batchReply = await client.ft.cursorRead(
|
|
94
120
|
ACCESS_RULES_REDIS_INDEX_NAME,
|
|
95
121
|
cursor,
|
|
@@ -30,7 +30,7 @@ import {
|
|
|
30
30
|
ACCESS_RULES_REDIS_INDEX_NAME,
|
|
31
31
|
ACCESS_RULE_REDIS_KEY_PREFIX,
|
|
32
32
|
} from "#policy/redis/redisRuleIndex.js";
|
|
33
|
-
import type
|
|
33
|
+
import { AccessPolicyType, type AccessRule } from "#policy/rule.js";
|
|
34
34
|
import { accessRuleInput } from "#policy/ruleInput/ruleInput.js";
|
|
35
35
|
import type {
|
|
36
36
|
AccessRuleEntry,
|
|
@@ -39,6 +39,91 @@ import type {
|
|
|
39
39
|
} from "#policy/rulesStorage.js";
|
|
40
40
|
import { aggregateRedisKeys } from "./redisAggregate.js";
|
|
41
41
|
|
|
42
|
+
// Server-side specificity ranking config.
|
|
43
|
+
//
|
|
44
|
+
// Strict-match findRules (matchingFieldsOnly=true) used to pull every
|
|
45
|
+
// matching rule's hash into Node and sort in JS. With the greedy regression
|
|
46
|
+
// in #2689 this hauled ~1190 hashes per request and pegged provider CPU.
|
|
47
|
+
//
|
|
48
|
+
// The new path lets RediSearch compute specificity itself: one
|
|
49
|
+
// FT.AGGREGATE that filters via the strict AND-of-disjunctions query,
|
|
50
|
+
// counts populated rule fields with APPLY+exists(), sorts by that
|
|
51
|
+
// derived score DESC, and returns only the top-N candidates. Node
|
|
52
|
+
// receives at most TOP_N small records, no HGETALL fanout, no JS rank.
|
|
53
|
+
//
|
|
54
|
+
// TOP_N >> 1 leaves headroom for the deferToVerify filter that runs
|
|
55
|
+
// post-aggregate (a Block rule with deferToVerify is skipped in the
|
|
56
|
+
// blockMiddleware path, so we need a few alternates ready). 20 covers
|
|
57
|
+
// the worst observed concentration of deferToVerify rules per scope by
|
|
58
|
+
// comfortable margin while keeping payload tiny.
|
|
59
|
+
export const SERVER_SIDE_RANK_TOP_N = 20;
|
|
60
|
+
|
|
61
|
+
// Safety cap for the greedy (admin/internal) path. Generous because
|
|
62
|
+
// admin tooling depends on the full result set; not on the hot
|
|
63
|
+
// per-request path.
|
|
64
|
+
const GREEDY_MAX_CANDIDATES = REDIS_BATCH_SIZE * 10;
|
|
65
|
+
|
|
66
|
+
// Fields that contribute one specificity point each. Mirrors
|
|
67
|
+
// SCALAR_USER_SCOPE_FIELDS + clientId + ip-constraint in
|
|
68
|
+
// blacklistRequestInspector.ruleSpecificity. numericIp and the
|
|
69
|
+
// numericIpMaskMin range are mutually exclusive on a rule (see
|
|
70
|
+
// ruleHasIpConstraint), so `exists(@numericIp) + exists(@numericIpMaskMin)`
|
|
71
|
+
// contributes 1 in practice — never 2 — when applied to writer-produced
|
|
72
|
+
// rules.
|
|
73
|
+
const SPECIFICITY_EXPR = [
|
|
74
|
+
"exists(@clientId)",
|
|
75
|
+
"exists(@userId)",
|
|
76
|
+
"exists(@ja4Hash)",
|
|
77
|
+
"exists(@headersHash)",
|
|
78
|
+
"exists(@userAgentHash)",
|
|
79
|
+
"exists(@headHash)",
|
|
80
|
+
"exists(@coords)",
|
|
81
|
+
"exists(@countryCode)",
|
|
82
|
+
"exists(@asn)",
|
|
83
|
+
"exists(@numericIp)",
|
|
84
|
+
"exists(@numericIpMaskMin)",
|
|
85
|
+
].join(" + ");
|
|
86
|
+
|
|
87
|
+
// Block outranks Restrict on equal specificity (defensive: a request
|
|
88
|
+
// that would match a Block rule must never be downgraded). Note the
|
|
89
|
+
// string literal is lowercase because AccessPolicyType.Block = "block"
|
|
90
|
+
// and getRedisRuleValue serialises via String().
|
|
91
|
+
const SEVERITY_EXPR = `(@type == "${AccessPolicyType.Block}")`;
|
|
92
|
+
|
|
93
|
+
// Final rank: specificity weighted, severity as tiebreaker on equal
|
|
94
|
+
// specificity. spec*2 + sev fits both into a single sortable float.
|
|
95
|
+
const RANK_EXPR = "(@_spec * 2) + @_sev";
|
|
96
|
+
|
|
97
|
+
// Every field referenced by SPECIFICITY_EXPR/SEVERITY_EXPR, plus the
|
|
98
|
+
// fields needed to reconstruct an AccessRule downstream. APPLY can only
|
|
99
|
+
// see fields explicitly loaded into the aggregate pipeline, so this
|
|
100
|
+
// list must include every rule attribute the parser cares about.
|
|
101
|
+
const RULE_LOAD_FIELDS = [
|
|
102
|
+
"@__key",
|
|
103
|
+
"@type",
|
|
104
|
+
"@captchaType",
|
|
105
|
+
"@description",
|
|
106
|
+
"@solvedImagesCount",
|
|
107
|
+
"@imageThreshold",
|
|
108
|
+
"@powDifficulty",
|
|
109
|
+
"@unsolvedImagesCount",
|
|
110
|
+
"@frictionlessScore",
|
|
111
|
+
"@deferToVerify",
|
|
112
|
+
"@clientId",
|
|
113
|
+
"@groupId",
|
|
114
|
+
"@userId",
|
|
115
|
+
"@ja4Hash",
|
|
116
|
+
"@headersHash",
|
|
117
|
+
"@userAgentHash",
|
|
118
|
+
"@headHash",
|
|
119
|
+
"@coords",
|
|
120
|
+
"@countryCode",
|
|
121
|
+
"@asn",
|
|
122
|
+
"@numericIp",
|
|
123
|
+
"@numericIpMaskMin",
|
|
124
|
+
"@numericIpMaskMax",
|
|
125
|
+
] as const;
|
|
126
|
+
|
|
42
127
|
export class RedisRulesReader implements AccessRulesReader {
|
|
43
128
|
constructor(
|
|
44
129
|
private readonly client: RedisClientType,
|
|
@@ -84,47 +169,130 @@ export class RedisRulesReader implements AccessRulesReader {
|
|
|
84
169
|
return [];
|
|
85
170
|
}
|
|
86
171
|
|
|
172
|
+
// Hot path: strict-match callers (blockMiddleware /
|
|
173
|
+
// checkForHardBlock) get server-side specificity ranking via
|
|
174
|
+
// FT.AGGREGATE — Redis returns the top N candidates already
|
|
175
|
+
// sorted, no HGETALL fanout. See SPECIFICITY_EXPR / RANK_EXPR
|
|
176
|
+
// above for the score definition.
|
|
177
|
+
if (matchingFieldsOnly) {
|
|
178
|
+
return this.findRulesRanked(filter, query);
|
|
179
|
+
}
|
|
180
|
+
|
|
181
|
+
// Admin / internal callers (greedy mode): keep the FT.SEARCH +
|
|
182
|
+
// HGETALL fanout path. Cap at REDIS_BATCH_SIZE; truncation past
|
|
183
|
+
// that point is the known liability tracked in #2689.
|
|
184
|
+
return this.findRulesGreedy(filter, query);
|
|
185
|
+
}
|
|
186
|
+
|
|
187
|
+
private async findRulesRanked(
|
|
188
|
+
filter: AccessRulesFilter,
|
|
189
|
+
query: string,
|
|
190
|
+
): Promise<AccessRule[]> {
|
|
87
191
|
try {
|
|
88
|
-
|
|
89
|
-
// This avoids a bug in @redis/search where ft.search crashes with
|
|
90
|
-
// "Cannot read properties of null (reading 'length')" when a document
|
|
91
|
-
// is deleted/expired between the index scan and data retrieval.
|
|
92
|
-
const searchReply = await this.client.ft.searchNoContent(
|
|
192
|
+
const reply = await this.client.ft.aggregate(
|
|
93
193
|
ACCESS_RULES_REDIS_INDEX_NAME,
|
|
94
194
|
query,
|
|
95
195
|
{
|
|
96
196
|
DIALECT: REDIS_QUERY_DIALECT,
|
|
97
|
-
|
|
98
|
-
|
|
99
|
-
|
|
100
|
-
|
|
101
|
-
|
|
197
|
+
LOAD: [...RULE_LOAD_FIELDS],
|
|
198
|
+
STEPS: [
|
|
199
|
+
{ type: "APPLY", expression: SPECIFICITY_EXPR, AS: "_spec" },
|
|
200
|
+
{ type: "APPLY", expression: SEVERITY_EXPR, AS: "_sev" },
|
|
201
|
+
{ type: "APPLY", expression: RANK_EXPR, AS: "_rank" },
|
|
202
|
+
{
|
|
203
|
+
type: "SORTBY",
|
|
204
|
+
BY: [{ BY: "@_rank", DIRECTION: "DESC" }],
|
|
205
|
+
// SORTBY MAX would let RediSearch keep only the
|
|
206
|
+
// top-N in a bounded heap rather than fully
|
|
207
|
+
// sorting, but the @redis/search client
|
|
208
|
+
// serialises MAX *before* DESC and RediSearch
|
|
209
|
+
// rejects that order ("MISSING ASC or DESC
|
|
210
|
+
// after sort field (MAX)"). LIMIT below still
|
|
211
|
+
// trims; SORTBY does a full sort. Acceptable
|
|
212
|
+
// because the strict-match filter already
|
|
213
|
+
// keeps the candidate set small in practice.
|
|
214
|
+
},
|
|
215
|
+
{ type: "LIMIT", from: 0, size: SERVER_SIDE_RANK_TOP_N },
|
|
216
|
+
],
|
|
102
217
|
},
|
|
103
218
|
);
|
|
104
219
|
|
|
105
|
-
if (
|
|
106
|
-
|
|
107
|
-
msg: "Executed search query",
|
|
108
|
-
data: {
|
|
109
|
-
inspect: util.inspect(
|
|
110
|
-
{
|
|
111
|
-
filter: filter,
|
|
112
|
-
searchReply: searchReply,
|
|
113
|
-
query: query,
|
|
114
|
-
},
|
|
115
|
-
{ depth: null },
|
|
116
|
-
),
|
|
117
|
-
},
|
|
118
|
-
}));
|
|
220
|
+
if (reply.results.length === 0) {
|
|
221
|
+
return [];
|
|
119
222
|
}
|
|
120
223
|
|
|
121
|
-
|
|
224
|
+
this.logger.debug(() => ({
|
|
225
|
+
msg: "Executed ranked search query",
|
|
226
|
+
data: {
|
|
227
|
+
inspect: util.inspect(
|
|
228
|
+
{
|
|
229
|
+
filter: filter,
|
|
230
|
+
foundCount: reply.results.length,
|
|
231
|
+
query: query,
|
|
232
|
+
},
|
|
233
|
+
{ depth: null },
|
|
234
|
+
),
|
|
235
|
+
},
|
|
236
|
+
}));
|
|
237
|
+
|
|
238
|
+
// FT.AGGREGATE results include the derived _spec/_sev/_rank
|
|
239
|
+
// fields and the loaded rule fields. parseRedisRecords runs
|
|
240
|
+
// through zod which ignores the derived fields. __key is
|
|
241
|
+
// also passed through but unused downstream.
|
|
242
|
+
return parseRedisRecords(reply.results, accessRuleInput, this.logger);
|
|
243
|
+
} catch (e) {
|
|
244
|
+
this.logger.error(() => ({
|
|
245
|
+
err: e,
|
|
246
|
+
data: {
|
|
247
|
+
inspect: util.inspect({ query, filter }, { depth: null }),
|
|
248
|
+
},
|
|
249
|
+
msg: "failed to execute ranked search query",
|
|
250
|
+
}));
|
|
251
|
+
|
|
252
|
+
return [];
|
|
253
|
+
}
|
|
254
|
+
}
|
|
255
|
+
|
|
256
|
+
// Used only by admin / internal callers (matchingFieldsOnly=false).
|
|
257
|
+
// Not on the per-request blockMiddleware path any more, so the CPU
|
|
258
|
+
// cost of the HGETALL fanout is acceptable. Same shape as #2689's
|
|
259
|
+
// aggregate-with-cursor: returns every candidate up to
|
|
260
|
+
// GREEDY_MAX_CANDIDATES so we don't silently drop matches that
|
|
261
|
+
// admin tools depend on (e.g. ASN-wide rule listings).
|
|
262
|
+
private async findRulesGreedy(
|
|
263
|
+
filter: AccessRulesFilter,
|
|
264
|
+
query: string,
|
|
265
|
+
): Promise<AccessRule[]> {
|
|
266
|
+
try {
|
|
267
|
+
const ruleKeys = await aggregateRedisKeys(
|
|
268
|
+
this.client,
|
|
269
|
+
query,
|
|
270
|
+
this.logger,
|
|
271
|
+
undefined,
|
|
272
|
+
GREEDY_MAX_CANDIDATES,
|
|
273
|
+
);
|
|
274
|
+
|
|
275
|
+
if (ruleKeys.length === 0) {
|
|
122
276
|
return [];
|
|
123
277
|
}
|
|
124
278
|
|
|
279
|
+
this.logger.debug(() => ({
|
|
280
|
+
msg: "Executed greedy search query",
|
|
281
|
+
data: {
|
|
282
|
+
inspect: util.inspect(
|
|
283
|
+
{
|
|
284
|
+
filter: filter,
|
|
285
|
+
foundCount: ruleKeys.length,
|
|
286
|
+
query: query,
|
|
287
|
+
},
|
|
288
|
+
{ depth: null },
|
|
289
|
+
),
|
|
290
|
+
},
|
|
291
|
+
}));
|
|
292
|
+
|
|
125
293
|
const { records } = await fetchRedisHashRecords(
|
|
126
294
|
this.client,
|
|
127
|
-
|
|
295
|
+
ruleKeys,
|
|
128
296
|
this.logger,
|
|
129
297
|
);
|
|
130
298
|
|
|
@@ -137,17 +305,9 @@ export class RedisRulesReader implements AccessRulesReader {
|
|
|
137
305
|
this.logger.error(() => ({
|
|
138
306
|
err: e,
|
|
139
307
|
data: {
|
|
140
|
-
inspect: util.inspect(
|
|
141
|
-
{
|
|
142
|
-
query: query,
|
|
143
|
-
filter: filter,
|
|
144
|
-
},
|
|
145
|
-
{
|
|
146
|
-
depth: null,
|
|
147
|
-
},
|
|
148
|
-
),
|
|
308
|
+
inspect: util.inspect({ query, filter }, { depth: null }),
|
|
149
309
|
},
|
|
150
|
-
msg: "failed to execute search query",
|
|
310
|
+
msg: "failed to execute greedy search query",
|
|
151
311
|
}));
|
|
152
312
|
|
|
153
313
|
return [];
|
|
@@ -0,0 +1,338 @@
|
|
|
1
|
+
// Copyright 2021-2026 Prosopo (UK) Ltd.
|
|
2
|
+
//
|
|
3
|
+
// Licensed under the Apache License, Version 2.0 (the "License");
|
|
4
|
+
// you may not use this file except in compliance with the License.
|
|
5
|
+
// You may obtain a copy of the License at
|
|
6
|
+
//
|
|
7
|
+
// http://www.apache.org/licenses/LICENSE-2.0
|
|
8
|
+
//
|
|
9
|
+
// Unless required by applicable law or agreed to in writing, software
|
|
10
|
+
// distributed under the License is distributed on an "AS IS" BASIS,
|
|
11
|
+
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
12
|
+
// See the License for the specific language governing permissions and
|
|
13
|
+
// limitations under the License.
|
|
14
|
+
|
|
15
|
+
import { chunkIntoBatches, executeBatchesSequentially } from "@prosopo/common";
|
|
16
|
+
import type { Logger } from "@prosopo/logger";
|
|
17
|
+
import {
|
|
18
|
+
type RedisConnection,
|
|
19
|
+
createTestRedisConnection,
|
|
20
|
+
setupRedisIndex,
|
|
21
|
+
} from "@prosopo/redis-client";
|
|
22
|
+
import type { RedisClientType } from "redis";
|
|
23
|
+
import { afterAll, beforeAll, describe, expect, test } from "vitest";
|
|
24
|
+
import { RedisRulesReader } from "#policy/redis/reader/redisRulesReader.js";
|
|
25
|
+
import {
|
|
26
|
+
ACCESS_RULES_REDIS_INDEX_NAME,
|
|
27
|
+
accessRulesRedisIndex,
|
|
28
|
+
getAccessRuleRedisKey,
|
|
29
|
+
} from "#policy/redis/redisRuleIndex.js";
|
|
30
|
+
import { getRedisRuleValue } from "#policy/redis/redisRulesWriter.js";
|
|
31
|
+
import { AccessPolicyType, type AccessRule } from "#policy/rule.js";
|
|
32
|
+
import { FilterScopeMatch } from "#policy/rulesStorage.js";
|
|
33
|
+
|
|
34
|
+
// Production observation on pronode10 (2026-06-15):
|
|
35
|
+
// - ~7,500 rules in the access-rule index
|
|
36
|
+
// - greedy + FT.AGGREGATE pulled ~1,190 hashes per request and pegged
|
|
37
|
+
// provider1 Node thread at ~125% CPU.
|
|
38
|
+
// - Strict-match + server-side rank pulls at most SERVER_SIDE_RANK_TOP_N
|
|
39
|
+
// hashes per request, so the cost is independent of total rule count.
|
|
40
|
+
//
|
|
41
|
+
// This benchmark seeds 10k rules with a realistic specificity distribution,
|
|
42
|
+
// runs N findRules calls in matchingFieldsOnly=true mode, and asserts
|
|
43
|
+
// per-call latency stays inside generous CI thresholds. The numbers below
|
|
44
|
+
// are intentionally loose; healthy local runs typically come in 5-10x
|
|
45
|
+
// under cap. The point is to catch regressions where someone re-introduces
|
|
46
|
+
// HGETALL fanout into the hot path.
|
|
47
|
+
|
|
48
|
+
const RULE_COUNT = 10_000;
|
|
49
|
+
const QUERY_COUNT = 200;
|
|
50
|
+
|
|
51
|
+
// CI thresholds. Local development on a warm redis container is steady
|
|
52
|
+
// around p50=20ms / p99=25ms over 10k rules; thresholds set ~3-5x above
|
|
53
|
+
// that to absorb noisy shared CI runners. If these start failing
|
|
54
|
+
// regularly, look for changes that re-introduce per-rule HGETALL fanout
|
|
55
|
+
// or break the SORTBY MAX short-circuit on the Redis side.
|
|
56
|
+
const P50_LATENCY_MS = 80;
|
|
57
|
+
const P99_LATENCY_MS = 250;
|
|
58
|
+
|
|
59
|
+
// Build a deterministic distribution of rules that mirrors production
|
|
60
|
+
// roughly:
|
|
61
|
+
// ~70% low-specificity (1 field, e.g. ja4-only or country-only)
|
|
62
|
+
// ~20% medium (2-3 fields, typical anomaly detector emission)
|
|
63
|
+
// ~10% high (4+ fields, manual portal rules with full scope)
|
|
64
|
+
const buildBenchmarkRule = (i: number): AccessRule => {
|
|
65
|
+
const bucket = i % 10;
|
|
66
|
+
const ja4 = `t13d_bench_${(i % 50).toString().padStart(3, "0")}`;
|
|
67
|
+
const asn = 1000 + (i % 500);
|
|
68
|
+
// description is included in the content hash that generates the
|
|
69
|
+
// rule's Redis key, so a unique description ensures every synthetic
|
|
70
|
+
// rule lands on its own key. It doesn't affect specificity scoring
|
|
71
|
+
// or the strict-match filter.
|
|
72
|
+
const description = `bench-rule-${i}`;
|
|
73
|
+
|
|
74
|
+
if (bucket < 7) {
|
|
75
|
+
// low specificity — anomaly detector style
|
|
76
|
+
const variant = i % 3;
|
|
77
|
+
if (variant === 0)
|
|
78
|
+
return { type: AccessPolicyType.Block, description, ja4Hash: ja4 };
|
|
79
|
+
if (variant === 1)
|
|
80
|
+
return {
|
|
81
|
+
type: AccessPolicyType.Block,
|
|
82
|
+
description,
|
|
83
|
+
asn,
|
|
84
|
+
};
|
|
85
|
+
return {
|
|
86
|
+
type: AccessPolicyType.Restrict,
|
|
87
|
+
description,
|
|
88
|
+
countryCode: `C${i % 30}`,
|
|
89
|
+
};
|
|
90
|
+
}
|
|
91
|
+
|
|
92
|
+
if (bucket < 9) {
|
|
93
|
+
// medium specificity — client-scoped + 1-2 fields
|
|
94
|
+
return {
|
|
95
|
+
type: AccessPolicyType.Block,
|
|
96
|
+
description,
|
|
97
|
+
clientId: `client${i % 20}`,
|
|
98
|
+
ja4Hash: ja4,
|
|
99
|
+
...(i % 2 === 0 && { userAgentHash: `ua${i % 100}` }),
|
|
100
|
+
};
|
|
101
|
+
}
|
|
102
|
+
|
|
103
|
+
// high specificity — full scope portal rules
|
|
104
|
+
return {
|
|
105
|
+
type: AccessPolicyType.Block,
|
|
106
|
+
description,
|
|
107
|
+
clientId: `client${i % 20}`,
|
|
108
|
+
ja4Hash: ja4,
|
|
109
|
+
userAgentHash: `ua${i % 100}`,
|
|
110
|
+
headersHash: `hh${i % 200}`,
|
|
111
|
+
countryCode: `C${i % 30}`,
|
|
112
|
+
};
|
|
113
|
+
};
|
|
114
|
+
|
|
115
|
+
// Deterministic request-scope generator. Every request populates the
|
|
116
|
+
// full set of user-scope fields the rules use, so under strict-match
|
|
117
|
+
// semantics rules whose populated fields are a subset of the request
|
|
118
|
+
// scope can actually apply. Mirrors a fully-instrumented production
|
|
119
|
+
// request (ja4 + UA + headers + country + asn all known).
|
|
120
|
+
const buildBenchmarkRequest = (i: number) => {
|
|
121
|
+
return {
|
|
122
|
+
clientId: `client${i % 20}`,
|
|
123
|
+
scope: {
|
|
124
|
+
ja4Hash: `t13d_bench_${(i % 50).toString().padStart(3, "0")}`,
|
|
125
|
+
userAgentHash: `ua${i % 100}`,
|
|
126
|
+
headersHash: `hh${i % 200}`,
|
|
127
|
+
countryCode: `C${i % 30}`,
|
|
128
|
+
asn: 1000 + (i % 500),
|
|
129
|
+
},
|
|
130
|
+
};
|
|
131
|
+
};
|
|
132
|
+
|
|
133
|
+
const percentile = (sortedMs: number[], q: number): number => {
|
|
134
|
+
if (sortedMs.length === 0) return 0;
|
|
135
|
+
const idx = Math.min(
|
|
136
|
+
sortedMs.length - 1,
|
|
137
|
+
Math.floor((sortedMs.length - 1) * q),
|
|
138
|
+
);
|
|
139
|
+
const v = sortedMs[idx];
|
|
140
|
+
return v ?? 0;
|
|
141
|
+
};
|
|
142
|
+
|
|
143
|
+
describe("redisRulesReader ranked-path benchmark", () => {
|
|
144
|
+
let redisConnection: RedisConnection;
|
|
145
|
+
let redisClient: RedisClientType;
|
|
146
|
+
let reader: RedisRulesReader;
|
|
147
|
+
|
|
148
|
+
const mockLogger = new Proxy(
|
|
149
|
+
{},
|
|
150
|
+
{
|
|
151
|
+
get: () => () => {},
|
|
152
|
+
},
|
|
153
|
+
) as unknown as Logger;
|
|
154
|
+
|
|
155
|
+
// Track every key we insert so cleanup deletes only our rules,
|
|
156
|
+
// leaving any other test suite's data and the shared index intact.
|
|
157
|
+
const seededKeys: string[] = [];
|
|
158
|
+
|
|
159
|
+
beforeAll(async () => {
|
|
160
|
+
redisConnection = createTestRedisConnection();
|
|
161
|
+
redisClient = await setupRedisIndex(
|
|
162
|
+
redisConnection,
|
|
163
|
+
accessRulesRedisIndex,
|
|
164
|
+
mockLogger,
|
|
165
|
+
).getClient();
|
|
166
|
+
reader = new RedisRulesReader(redisClient, mockLogger);
|
|
167
|
+
|
|
168
|
+
// Seed RULE_COUNT rules in batches of 1000 — same chunk size the
|
|
169
|
+
// production writer uses, mirrors realistic insert behaviour.
|
|
170
|
+
const rules: AccessRule[] = [];
|
|
171
|
+
for (let i = 0; i < RULE_COUNT; i++) {
|
|
172
|
+
rules.push(buildBenchmarkRule(i));
|
|
173
|
+
}
|
|
174
|
+
await executeBatchesSequentially(
|
|
175
|
+
chunkIntoBatches(rules, 1000),
|
|
176
|
+
async (batch) => {
|
|
177
|
+
const multi = redisClient.multi();
|
|
178
|
+
for (const rule of batch) {
|
|
179
|
+
const key = getAccessRuleRedisKey(rule);
|
|
180
|
+
seededKeys.push(key);
|
|
181
|
+
multi.hSet(key, getRedisRuleValue(rule));
|
|
182
|
+
}
|
|
183
|
+
await multi.exec();
|
|
184
|
+
},
|
|
185
|
+
);
|
|
186
|
+
|
|
187
|
+
const indexInfo = await redisClient.ft.info(ACCESS_RULES_REDIS_INDEX_NAME);
|
|
188
|
+
// Description is unique per rule so every rule should land on
|
|
189
|
+
// its own key; allow a 10% margin for any incidental dedupe.
|
|
190
|
+
expect(indexInfo.num_docs).toBeGreaterThan(RULE_COUNT * 0.9);
|
|
191
|
+
}, 120_000);
|
|
192
|
+
|
|
193
|
+
afterAll(async () => {
|
|
194
|
+
// Delete only the keys this suite inserted; do not flushAll
|
|
195
|
+
// because we share the redis instance (and the index) with the
|
|
196
|
+
// other integration test file.
|
|
197
|
+
await executeBatchesSequentially(
|
|
198
|
+
chunkIntoBatches(seededKeys, 1000),
|
|
199
|
+
async (batch) => {
|
|
200
|
+
const multi = redisClient.multi();
|
|
201
|
+
for (const key of batch) {
|
|
202
|
+
multi.del(key);
|
|
203
|
+
}
|
|
204
|
+
await multi.exec();
|
|
205
|
+
},
|
|
206
|
+
);
|
|
207
|
+
});
|
|
208
|
+
|
|
209
|
+
test(`ranked findRules p50/p99 over ${QUERY_COUNT} queries against ${RULE_COUNT} rules`, async () => {
|
|
210
|
+
// warm up — first call pays for index page-in
|
|
211
|
+
await reader.findRules(
|
|
212
|
+
{
|
|
213
|
+
policyScope: { clientId: "warmup" },
|
|
214
|
+
policyScopeMatch: FilterScopeMatch.Greedy,
|
|
215
|
+
userScope: { ja4Hash: "warmup" },
|
|
216
|
+
userScopeMatch: FilterScopeMatch.Greedy,
|
|
217
|
+
},
|
|
218
|
+
true,
|
|
219
|
+
);
|
|
220
|
+
|
|
221
|
+
const samples: number[] = [];
|
|
222
|
+
for (let i = 0; i < QUERY_COUNT; i++) {
|
|
223
|
+
const { clientId, scope } = buildBenchmarkRequest(i);
|
|
224
|
+
const start = performance.now();
|
|
225
|
+
await reader.findRules(
|
|
226
|
+
{
|
|
227
|
+
policyScope: { clientId },
|
|
228
|
+
policyScopeMatch: FilterScopeMatch.Greedy,
|
|
229
|
+
userScope: scope,
|
|
230
|
+
userScopeMatch: FilterScopeMatch.Greedy,
|
|
231
|
+
},
|
|
232
|
+
true,
|
|
233
|
+
);
|
|
234
|
+
samples.push(performance.now() - start);
|
|
235
|
+
}
|
|
236
|
+
|
|
237
|
+
samples.sort((a, b) => a - b);
|
|
238
|
+
const p50 = percentile(samples, 0.5);
|
|
239
|
+
const p99 = percentile(samples, 0.99);
|
|
240
|
+
const avg = samples.reduce((a, b) => a + b, 0) / samples.length;
|
|
241
|
+
|
|
242
|
+
console.log(
|
|
243
|
+
`ranked findRules: avg=${avg.toFixed(2)}ms p50=${p50.toFixed(2)}ms p99=${p99.toFixed(2)}ms n=${samples.length}`,
|
|
244
|
+
);
|
|
245
|
+
|
|
246
|
+
expect(p50).toBeLessThan(P50_LATENCY_MS);
|
|
247
|
+
expect(p99).toBeLessThan(P99_LATENCY_MS);
|
|
248
|
+
}, 120_000);
|
|
249
|
+
|
|
250
|
+
test("ranked findRules result size is bounded by SERVER_SIDE_RANK_TOP_N", async () => {
|
|
251
|
+
// Fully-populated request that should apply to many seeded rules
|
|
252
|
+
// — the new path must cap at SERVER_SIDE_RANK_TOP_N (20).
|
|
253
|
+
const { clientId, scope } = buildBenchmarkRequest(0);
|
|
254
|
+
const results = await reader.findRules(
|
|
255
|
+
{
|
|
256
|
+
policyScope: { clientId },
|
|
257
|
+
policyScopeMatch: FilterScopeMatch.Greedy,
|
|
258
|
+
userScope: scope,
|
|
259
|
+
userScopeMatch: FilterScopeMatch.Greedy,
|
|
260
|
+
},
|
|
261
|
+
true,
|
|
262
|
+
);
|
|
263
|
+
expect(results.length).toBeLessThanOrEqual(20);
|
|
264
|
+
expect(results.length).toBeGreaterThan(0);
|
|
265
|
+
}, 60_000);
|
|
266
|
+
|
|
267
|
+
test("HGETALL-free: ranked path returns full rules without per-rule fanout", async () => {
|
|
268
|
+
// The benefit of the ranked path in production is *not* raw
|
|
269
|
+
// localhost latency (the strict-match index walk is heavier
|
|
270
|
+
// than a greedy FT.SEARCH on a quiet box). It's that the
|
|
271
|
+
// ranked path returns full rule bodies inside the single
|
|
272
|
+
// FT.AGGREGATE reply, with no follow-up HGETALL fanout. In
|
|
273
|
+
// production each HGETALL costs ~0.5ms of network RTT over
|
|
274
|
+
// the Docker bridge, so cutting hundreds of round trips per
|
|
275
|
+
// request is the win that doesn't appear in a localhost
|
|
276
|
+
// timing test.
|
|
277
|
+
//
|
|
278
|
+
// This test stands in for that: confirm the ranked path
|
|
279
|
+
// produces fully-populated AccessRule objects from one
|
|
280
|
+
// aggregate call (no separate hash fetches needed).
|
|
281
|
+
const { clientId, scope } = buildBenchmarkRequest(0);
|
|
282
|
+
const ranked = await reader.findRules(
|
|
283
|
+
{
|
|
284
|
+
policyScope: { clientId },
|
|
285
|
+
policyScopeMatch: FilterScopeMatch.Greedy,
|
|
286
|
+
userScope: scope,
|
|
287
|
+
userScopeMatch: FilterScopeMatch.Greedy,
|
|
288
|
+
},
|
|
289
|
+
true,
|
|
290
|
+
);
|
|
291
|
+
|
|
292
|
+
expect(ranked.length).toBeGreaterThan(0);
|
|
293
|
+
// Every returned record must have the policy fields populated —
|
|
294
|
+
// proves the LOAD pipeline includes them rather than relying
|
|
295
|
+
// on a follow-up HGETALL.
|
|
296
|
+
for (const rule of ranked) {
|
|
297
|
+
expect(rule.type).toBeDefined();
|
|
298
|
+
}
|
|
299
|
+
}, 60_000);
|
|
300
|
+
|
|
301
|
+
test("ranked findRules returns rules in specificity-descending order", async () => {
|
|
302
|
+
// A request that matches the high-specificity bucket should
|
|
303
|
+
// surface the rule with the most populated fields first.
|
|
304
|
+
const { clientId, scope } = buildBenchmarkRequest(9);
|
|
305
|
+
const results = await reader.findRules(
|
|
306
|
+
{
|
|
307
|
+
policyScope: { clientId },
|
|
308
|
+
policyScopeMatch: FilterScopeMatch.Greedy,
|
|
309
|
+
userScope: scope,
|
|
310
|
+
userScopeMatch: FilterScopeMatch.Greedy,
|
|
311
|
+
},
|
|
312
|
+
true,
|
|
313
|
+
);
|
|
314
|
+
|
|
315
|
+
if (results.length === 0) return;
|
|
316
|
+
const top = results[0];
|
|
317
|
+
if (!top) return;
|
|
318
|
+
|
|
319
|
+
// The top rule must have at least as many populated user-scope
|
|
320
|
+
// fields as any other returned rule.
|
|
321
|
+
const countFields = (r: AccessRule): number =>
|
|
322
|
+
(r.userId ? 1 : 0) +
|
|
323
|
+
(r.ja4Hash ? 1 : 0) +
|
|
324
|
+
(r.headersHash ? 1 : 0) +
|
|
325
|
+
(r.userAgentHash ? 1 : 0) +
|
|
326
|
+
(r.headHash ? 1 : 0) +
|
|
327
|
+
(r.coords ? 1 : 0) +
|
|
328
|
+
(r.countryCode ? 1 : 0) +
|
|
329
|
+
(r.asn !== undefined ? 1 : 0) +
|
|
330
|
+
(r.clientId ? 1 : 0) +
|
|
331
|
+
(r.numericIp !== undefined || r.numericIpMaskMin !== undefined ? 1 : 0);
|
|
332
|
+
|
|
333
|
+
const topScore = countFields(top);
|
|
334
|
+
for (const r of results) {
|
|
335
|
+
expect(countFields(r)).toBeLessThanOrEqual(topScore);
|
|
336
|
+
}
|
|
337
|
+
}, 60_000);
|
|
338
|
+
});
|