@resq-systems/security 1.0.4 → 2.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +236 -33
- package/lib/controls/address.d.mts +142 -0
- package/lib/controls/address.d.mts.map +1 -0
- package/lib/controls/address.mjs +533 -0
- package/lib/controls/address.mjs.map +1 -0
- package/lib/controls/csrf.d.mts +91 -0
- package/lib/controls/csrf.d.mts.map +1 -0
- package/lib/controls/csrf.mjs +200 -0
- package/lib/controls/csrf.mjs.map +1 -0
- package/lib/controls/index.d.mts +8 -0
- package/lib/controls/index.mjs +8 -0
- package/lib/controls/origin.d.mts +95 -0
- package/lib/controls/origin.d.mts.map +1 -0
- package/lib/controls/origin.mjs +156 -0
- package/lib/controls/origin.mjs.map +1 -0
- package/lib/controls/payload.d.mts +84 -0
- package/lib/controls/payload.d.mts.map +1 -0
- package/lib/controls/payload.mjs +147 -0
- package/lib/controls/payload.mjs.map +1 -0
- package/lib/controls/query.d.mts +157 -0
- package/lib/controls/query.d.mts.map +1 -0
- package/lib/controls/query.mjs +368 -0
- package/lib/controls/query.mjs.map +1 -0
- package/lib/controls/redirect.d.mts +92 -0
- package/lib/controls/redirect.d.mts.map +1 -0
- package/lib/controls/redirect.mjs +110 -0
- package/lib/controls/redirect.mjs.map +1 -0
- package/lib/controls/upload.d.mts +108 -0
- package/lib/controls/upload.d.mts.map +1 -0
- package/lib/controls/upload.mjs +374 -0
- package/lib/controls/upload.mjs.map +1 -0
- package/lib/crypto.d.mts +18 -5
- package/lib/crypto.d.mts.map +1 -1
- package/lib/crypto.mjs +35 -24
- package/lib/crypto.mjs.map +1 -1
- package/lib/hash.d.mts +83 -0
- package/lib/hash.d.mts.map +1 -0
- package/lib/hash.mjs +111 -0
- package/lib/hash.mjs.map +1 -0
- package/lib/index.d.mts +18 -2
- package/lib/index.mjs +20 -2
- package/lib/paths.d.mts +92 -0
- package/lib/paths.d.mts.map +1 -0
- package/lib/paths.mjs +140 -0
- package/lib/paths.mjs.map +1 -0
- package/lib/sanitize.d.mts +137 -35
- package/lib/sanitize.d.mts.map +1 -1
- package/lib/sanitize.mjs +170 -46
- package/lib/sanitize.mjs.map +1 -1
- package/lib/threats/capec.generated.d.mts +59 -0
- package/lib/threats/capec.generated.d.mts.map +1 -0
- package/lib/threats/capec.generated.mjs +644 -0
- package/lib/threats/capec.generated.mjs.map +1 -0
- package/lib/threats/engine.d.mts +94 -0
- package/lib/threats/engine.d.mts.map +1 -0
- package/lib/threats/engine.mjs +167 -0
- package/lib/threats/engine.mjs.map +1 -0
- package/lib/threats/index.d.mts +11 -0
- package/lib/threats/index.mjs +11 -0
- package/lib/threats/rules/datastore.d.mts +13 -0
- package/lib/threats/rules/datastore.d.mts.map +1 -0
- package/lib/threats/rules/datastore.mjs +366 -0
- package/lib/threats/rules/datastore.mjs.map +1 -0
- package/lib/threats/rules/index.d.mts +54 -0
- package/lib/threats/rules/index.d.mts.map +1 -0
- package/lib/threats/rules/index.mjs +121 -0
- package/lib/threats/rules/index.mjs.map +1 -0
- package/lib/threats/rules/markup.d.mts +28 -0
- package/lib/threats/rules/markup.d.mts.map +1 -0
- package/lib/threats/rules/markup.mjs +373 -0
- package/lib/threats/rules/markup.mjs.map +1 -0
- package/lib/threats/rules/protocol.d.mts +49 -0
- package/lib/threats/rules/protocol.d.mts.map +1 -0
- package/lib/threats/rules/protocol.mjs +175 -0
- package/lib/threats/rules/protocol.mjs.map +1 -0
- package/lib/threats/rules/system.d.mts +19 -0
- package/lib/threats/rules/system.d.mts.map +1 -0
- package/lib/threats/rules/system.mjs +455 -0
- package/lib/threats/rules/system.mjs.map +1 -0
- package/lib/threats/rules/web.d.mts +26 -0
- package/lib/threats/rules/web.d.mts.map +1 -0
- package/lib/threats/rules/web.mjs +412 -0
- package/lib/threats/rules/web.mjs.map +1 -0
- package/lib/threats/scoring.d.mts +59 -0
- package/lib/threats/scoring.d.mts.map +1 -0
- package/lib/threats/scoring.mjs +111 -0
- package/lib/threats/scoring.mjs.map +1 -0
- package/lib/threats/types.d.mts +245 -0
- package/lib/threats/types.d.mts.map +1 -0
- package/lib/threats/types.mjs +52 -0
- package/lib/threats/types.mjs.map +1 -0
- package/lib/threats/variants.d.mts +57 -0
- package/lib/threats/variants.d.mts.map +1 -0
- package/lib/threats/variants.mjs +144 -0
- package/lib/threats/variants.mjs.map +1 -0
- package/lib/unicode/confusables.d.mts +82 -0
- package/lib/unicode/confusables.d.mts.map +1 -0
- package/lib/unicode/confusables.mjs +954 -0
- package/lib/unicode/confusables.mjs.map +1 -0
- package/lib/unicode/index.d.mts +126 -0
- package/lib/unicode/index.d.mts.map +1 -0
- package/lib/unicode/index.mjs +288 -0
- package/lib/unicode/index.mjs.map +1 -0
- package/lib/validators.d.mts +341 -164
- package/lib/validators.d.mts.map +1 -1
- package/lib/validators.mjs +519 -338
- package/lib/validators.mjs.map +1 -1
- package/package.json +35 -8
package/lib/validators.mjs
CHANGED
|
@@ -1,3 +1,6 @@
|
|
|
1
|
+
import { MAX_SCAN_LENGTH, scanForThreats } from "./threats/engine.mjs";
|
|
2
|
+
import { foldConfusables } from "./unicode/confusables.mjs";
|
|
3
|
+
import { analyzeIdentifier, containsBidiControls } from "./unicode/index.mjs";
|
|
1
4
|
import { assertNever } from "@resq-systems/types";
|
|
2
5
|
//#region src/validators.ts
|
|
3
6
|
/**
|
|
@@ -16,333 +19,198 @@ import { assertNever } from "@resq-systems/types";
|
|
|
16
19
|
* limitations under the License.
|
|
17
20
|
*/
|
|
18
21
|
/**
|
|
19
|
-
*
|
|
20
|
-
*
|
|
21
|
-
|
|
22
|
-
|
|
23
|
-
|
|
24
|
-
|
|
25
|
-
|
|
26
|
-
|
|
27
|
-
|
|
28
|
-
|
|
29
|
-
|
|
30
|
-
|
|
31
|
-
|
|
32
|
-
|
|
33
|
-
|
|
34
|
-
|
|
35
|
-
|
|
36
|
-
|
|
37
|
-
|
|
38
|
-
|
|
39
|
-
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
|
|
43
|
-
|
|
44
|
-
*
|
|
45
|
-
|
|
46
|
-
|
|
47
|
-
|
|
48
|
-
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
/--\s*$/gm,
|
|
52
|
-
/\/\*[\s\S]*?\*\//g,
|
|
53
|
-
/'\s*OR\s+'[\d\w]+'\s*=\s*'[\d\w]+/gi,
|
|
54
|
-
/'\s*OR\s+\d+\s*=\s*\d+/gi,
|
|
55
|
-
/"\s*OR\s+"[\d\w]+"\s*=\s*"[\d\w]+/gi,
|
|
56
|
-
/1\s*=\s*1/g,
|
|
57
|
-
/;\s*(SELECT|INSERT|UPDATE|DELETE|DROP|EXEC|UNION)/gi,
|
|
58
|
-
/SLEEP\s*\(\s*\d+\s*\)/gi,
|
|
59
|
-
/WAITFOR\s+DELAY/gi,
|
|
60
|
-
/BENCHMARK\s*\(/gi,
|
|
61
|
-
/INFORMATION_SCHEMA/gi,
|
|
62
|
-
/0x[0-9a-f]+/gi,
|
|
63
|
-
/\bEXEC(UTE)?\s*\(/gi,
|
|
64
|
-
/\bxp_\w+/gi
|
|
65
|
-
];
|
|
66
|
-
/**
|
|
67
|
-
* NoSQL injection patterns
|
|
68
|
-
* Detects MongoDB and other NoSQL attack vectors
|
|
69
|
-
*/
|
|
70
|
-
const NOSQL_INJECTION_PATTERNS = [
|
|
71
|
-
/\$(?:gt|gte|lt|lte|ne|eq|in|nin|and|or|not|nor|exists|type|mod|regex|text|where|all|elemMatch|size|slice|expr|jsonSchema|meta)\b/gi,
|
|
72
|
-
/\$where\s*:/gi,
|
|
73
|
-
/\$function\s*:/gi,
|
|
74
|
-
/\{\s*\$[a-z]+\s*:/gi,
|
|
75
|
-
/\[\s*\$[a-z]+\s*\]/gi
|
|
76
|
-
];
|
|
77
|
-
/**
|
|
78
|
-
* Path traversal patterns
|
|
79
|
-
* Detects directory traversal attacks
|
|
80
|
-
*/
|
|
81
|
-
const PATH_TRAVERSAL_PATTERNS = [
|
|
82
|
-
/\.\.[/\\]/g,
|
|
83
|
-
/%2e%2e[%2f%5c]/gi,
|
|
84
|
-
/%252e%252e%252f/gi,
|
|
85
|
-
/\.\.%2f/gi,
|
|
86
|
-
/\.\.%5c/gi,
|
|
87
|
-
/%00/g,
|
|
88
|
-
/\/etc\/passwd/gi,
|
|
89
|
-
/\/etc\/shadow/gi,
|
|
90
|
-
/\/proc\/self/gi,
|
|
91
|
-
/C:\\Windows/gi,
|
|
92
|
-
/C:\\System32/gi
|
|
93
|
-
];
|
|
94
|
-
/**
|
|
95
|
-
* Homoglyph patterns
|
|
96
|
-
* Detects Unicode characters that look like ASCII but aren't
|
|
97
|
-
* Used in phishing and IDN homograph attacks
|
|
98
|
-
*/
|
|
99
|
-
const HOMOGLYPH_MAP = {
|
|
100
|
-
a: [
|
|
101
|
-
"а",
|
|
102
|
-
"ɑ",
|
|
103
|
-
"α",
|
|
104
|
-
"а"
|
|
105
|
-
],
|
|
106
|
-
c: [
|
|
107
|
-
"с",
|
|
108
|
-
"ϲ",
|
|
109
|
-
"ⅽ"
|
|
110
|
-
],
|
|
111
|
-
e: [
|
|
112
|
-
"е",
|
|
113
|
-
"ε",
|
|
114
|
-
"ė"
|
|
115
|
-
],
|
|
116
|
-
o: [
|
|
117
|
-
"о",
|
|
118
|
-
"ο",
|
|
119
|
-
"ᴏ",
|
|
120
|
-
"०"
|
|
121
|
-
],
|
|
122
|
-
p: ["р", "ρ"],
|
|
123
|
-
s: ["ѕ", "ꜱ"],
|
|
124
|
-
x: ["х", "χ"],
|
|
125
|
-
y: ["у", "γ"],
|
|
126
|
-
B: ["В", "Β"],
|
|
127
|
-
H: ["Н", "Η"],
|
|
128
|
-
K: ["К", "Κ"],
|
|
129
|
-
M: ["М", "Μ"],
|
|
130
|
-
P: ["Р", "Ρ"],
|
|
131
|
-
T: ["Т", "Τ"]
|
|
132
|
-
};
|
|
22
|
+
* @fileoverview Field-level validators, output encoders, and the compatibility
|
|
23
|
+
* surface over the context-aware rule engine in `@resq-systems/security/threats`.
|
|
24
|
+
*
|
|
25
|
+
* The pattern arrays that used to live here are gone. Every detector below delegates
|
|
26
|
+
* to {@link scanForThreats} with the context matching its sink, which is what stops a
|
|
27
|
+
* detector meant for file paths from rejecting a biography. New code should call
|
|
28
|
+
* `scanForThreats` directly and declare its own contexts; the `contains*` helpers
|
|
29
|
+
* remain for callers written against the previous API.
|
|
30
|
+
*
|
|
31
|
+
* Detection is defense-in-depth. Output encoding, parameterized queries, path
|
|
32
|
+
* containment, and argv-array process spawning are the controls.
|
|
33
|
+
*
|
|
34
|
+
* @module @resq-systems/security/validators
|
|
35
|
+
*/
|
|
36
|
+
/** Translate the legacy toggles into engine contexts. */
|
|
37
|
+
function contextsFor(config) {
|
|
38
|
+
const contexts = ["general_text"];
|
|
39
|
+
if (config.checkXSS !== false) contexts.push("html");
|
|
40
|
+
if (config.checkSQLInjection !== false) contexts.push("sql");
|
|
41
|
+
if (config.checkNoSQLInjection !== false) contexts.push("nosql");
|
|
42
|
+
if (config.checkCommandInjection === true) contexts.push("shell");
|
|
43
|
+
if (config.checkPathTraversal !== false) contexts.push("filesystem");
|
|
44
|
+
return contexts;
|
|
45
|
+
}
|
|
46
|
+
/**
|
|
47
|
+
* Run one context's rules and keep at most one finding, preserving the
|
|
48
|
+
* one-finding-per-detector contract the `contains*` helpers have always had.
|
|
49
|
+
*/
|
|
50
|
+
function firstFindingOfType(input, contexts, type) {
|
|
51
|
+
const finding = scanForThreats(input, { contexts }).findings.find((candidate) => candidate.type === type);
|
|
52
|
+
return finding ? [finding] : [];
|
|
53
|
+
}
|
|
133
54
|
/**
|
|
134
|
-
* Detect XSS
|
|
135
|
-
*
|
|
55
|
+
* Detect XSS payloads — script tags, inline event handlers, dangerous URI schemes,
|
|
56
|
+
* markup sinks — in a value bound for an HTML context.
|
|
136
57
|
*
|
|
137
|
-
*
|
|
138
|
-
*
|
|
139
|
-
* payloads. Returns at most one finding — the regex catalog is
|
|
140
|
-
* exhaustive enough that the first hit is sufficient for a
|
|
141
|
-
* reject-or-sanitize decision.
|
|
58
|
+
* @param input - String to scan. Truncated at 100 000 characters.
|
|
59
|
+
* @returns Empty array, or a single finding of type `"xss"`.
|
|
142
60
|
*
|
|
143
|
-
* @
|
|
144
|
-
*
|
|
145
|
-
*
|
|
61
|
+
* @remarks
|
|
62
|
+
* Prototype-pollution patterns (`__proto__`, `constructor[`) no longer surface here.
|
|
63
|
+
* They are a distinct weakness class with distinct controls and now report as
|
|
64
|
+
* `prototype_pollution` — see {@link containsPrototypePollution}.
|
|
146
65
|
*
|
|
147
66
|
* @example
|
|
148
67
|
* ```ts
|
|
149
68
|
* containsXSSPatterns(`<img src=x onerror="alert(1)">`);
|
|
150
|
-
* // → [{
|
|
69
|
+
* // → [{ ruleId: "XSS-EVENT-HANDLER-001", type: "xss", severity: "high", … }]
|
|
151
70
|
* ```
|
|
152
71
|
*/
|
|
153
72
|
function containsXSSPatterns(input) {
|
|
154
|
-
|
|
155
|
-
const bounded = input.length > 1e5 ? input.slice(0, 1e5) : input;
|
|
156
|
-
for (const pattern of XSS_PATTERNS) {
|
|
157
|
-
const match = bounded.match(pattern);
|
|
158
|
-
if (match) {
|
|
159
|
-
findings.push({
|
|
160
|
-
type: "xss",
|
|
161
|
-
description: "Potential cross-site scripting (XSS) detected",
|
|
162
|
-
matchedPattern: match[0].slice(0, 50)
|
|
163
|
-
});
|
|
164
|
-
break;
|
|
165
|
-
}
|
|
166
|
-
}
|
|
167
|
-
return findings;
|
|
73
|
+
return firstFindingOfType(input, ["html"], "xss");
|
|
168
74
|
}
|
|
169
75
|
/**
|
|
170
|
-
* Detect
|
|
171
|
-
*
|
|
172
|
-
*
|
|
76
|
+
* Detect prototype-pollution payloads — `__proto__`, `constructor.prototype`, and the
|
|
77
|
+
* nested-object forms that arrive through a JSON body or query-string expansion.
|
|
78
|
+
*
|
|
79
|
+
* **Not the control.** Reject unknown keys with schema validation, build lookup
|
|
80
|
+
* objects with `Object.create(null)`, and use a merge that skips `__proto__`,
|
|
81
|
+
* `constructor`, and `prototype`.
|
|
173
82
|
*
|
|
174
|
-
*
|
|
175
|
-
*
|
|
176
|
-
|
|
83
|
+
* @param input - String to scan.
|
|
84
|
+
* @returns Empty array, or a single finding of type `"prototype_pollution"`.
|
|
85
|
+
*/
|
|
86
|
+
function containsPrototypePollution(input) {
|
|
87
|
+
return firstFindingOfType(input, ["object_merge"], "prototype_pollution");
|
|
88
|
+
}
|
|
89
|
+
/**
|
|
90
|
+
* Detect SQL-injection patterns in a value bound for a query.
|
|
91
|
+
*
|
|
92
|
+
* **Not a replacement for parameterized queries.** A bound parameter is safe whatever
|
|
93
|
+
* keywords it contains; an interpolated one is unsafe however many signatures it
|
|
94
|
+
* dodges. Use this for telemetry alongside binding, never instead of it.
|
|
177
95
|
*
|
|
178
96
|
* @param input - String to scan. Truncated at 100 000 characters.
|
|
179
97
|
* @returns Empty array, or one finding of type `"sql_injection"`.
|
|
180
98
|
*/
|
|
181
99
|
function containsSQLInjection(input) {
|
|
182
|
-
|
|
183
|
-
const bounded = input.length > 1e5 ? input.slice(0, 1e5) : input;
|
|
184
|
-
for (const pattern of SQL_INJECTION_PATTERNS) {
|
|
185
|
-
const match = bounded.match(pattern);
|
|
186
|
-
if (match) {
|
|
187
|
-
findings.push({
|
|
188
|
-
type: "sql_injection",
|
|
189
|
-
description: "Potential SQL injection detected",
|
|
190
|
-
matchedPattern: match[0].slice(0, 50)
|
|
191
|
-
});
|
|
192
|
-
break;
|
|
193
|
-
}
|
|
194
|
-
}
|
|
195
|
-
return findings;
|
|
100
|
+
return firstFindingOfType(input, ["sql"], "sql_injection");
|
|
196
101
|
}
|
|
197
102
|
/**
|
|
198
|
-
* Detect NoSQL
|
|
199
|
-
*
|
|
200
|
-
* structural manipulators that can bypass auth filters in document
|
|
201
|
-
* stores.
|
|
103
|
+
* Detect NoSQL operator injection — `$where`, `$ne`, `$regex`, and the object and
|
|
104
|
+
* array forms that bypass authentication filters in document stores.
|
|
202
105
|
*
|
|
203
106
|
* @param input - String to scan.
|
|
204
107
|
* @returns Empty array, or one finding of type `"nosql_injection"`.
|
|
205
108
|
*/
|
|
206
109
|
function containsNoSQLInjection(input) {
|
|
207
|
-
|
|
208
|
-
for (const pattern of NOSQL_INJECTION_PATTERNS) {
|
|
209
|
-
const match = input.match(pattern);
|
|
210
|
-
if (match) {
|
|
211
|
-
findings.push({
|
|
212
|
-
type: "nosql_injection",
|
|
213
|
-
description: "Potential NoSQL injection detected",
|
|
214
|
-
matchedPattern: match[0].slice(0, 50)
|
|
215
|
-
});
|
|
216
|
-
break;
|
|
217
|
-
}
|
|
218
|
-
}
|
|
219
|
-
return findings;
|
|
110
|
+
return firstFindingOfType(input, ["nosql"], "nosql_injection");
|
|
220
111
|
}
|
|
221
112
|
/**
|
|
222
|
-
* Detect shell command-injection patterns
|
|
223
|
-
*
|
|
224
|
-
* …) and shell-piped exec (`| sh`, `| bash`).
|
|
113
|
+
* Detect shell command-injection patterns — command substitution, chained commands,
|
|
114
|
+
* pipes into an interpreter.
|
|
225
115
|
*
|
|
226
|
-
* **Off by default in {@link detectThreatPatterns}
|
|
227
|
-
*
|
|
228
|
-
*
|
|
229
|
-
* process or shell.
|
|
116
|
+
* **Off by default in {@link detectThreatPatterns}**, because these patterns fire on
|
|
117
|
+
* ordinary prose. Enable only when the value reaches a child process, and prefer
|
|
118
|
+
* spawning with an argv array and `shell: false`, which makes the category moot.
|
|
230
119
|
*
|
|
231
120
|
* @param input - String to scan. Truncated at 100 000 characters.
|
|
232
121
|
* @returns Empty array, or one finding of type `"command_injection"`.
|
|
233
122
|
*/
|
|
234
123
|
function containsCommandInjection(input) {
|
|
235
|
-
|
|
236
|
-
const dangerousPatterns = [
|
|
237
|
-
/\$\([^)]{1,200}\)/g,
|
|
238
|
-
/`[^`]{1,200}`/g,
|
|
239
|
-
/;\s*(rm|del|cat|wget|curl|nc)\b/gi,
|
|
240
|
-
/\|\s*(sh|bash|cmd)\b/gi
|
|
241
|
-
];
|
|
242
|
-
const bounded = input.length > 1e5 ? input.slice(0, 1e5) : input;
|
|
243
|
-
for (const pattern of dangerousPatterns) {
|
|
244
|
-
const match = bounded.match(pattern);
|
|
245
|
-
if (match) {
|
|
246
|
-
findings.push({
|
|
247
|
-
type: "command_injection",
|
|
248
|
-
description: "Potential command injection detected",
|
|
249
|
-
matchedPattern: match[0].slice(0, 50)
|
|
250
|
-
});
|
|
251
|
-
break;
|
|
252
|
-
}
|
|
253
|
-
}
|
|
254
|
-
return findings;
|
|
124
|
+
return firstFindingOfType(input, ["shell"], "command_injection");
|
|
255
125
|
}
|
|
256
126
|
/**
|
|
257
|
-
* Detect path-traversal payloads — `../`, encoded
|
|
258
|
-
*
|
|
259
|
-
*
|
|
260
|
-
*
|
|
127
|
+
* Detect path-traversal payloads — `../`, its percent-encoded and double-encoded
|
|
128
|
+
* forms, NUL truncation, and references to sensitive system paths.
|
|
129
|
+
*
|
|
130
|
+
* **Not the control.** Use `resolveContainedPath` from
|
|
131
|
+
* `@resq-systems/security/paths`, which resolves the candidate against a base
|
|
132
|
+
* directory and verifies containment — a check that also catches absolute paths and
|
|
133
|
+
* separator tricks no signature enumerates.
|
|
261
134
|
*
|
|
262
135
|
* @param input - String to scan.
|
|
263
136
|
* @returns Empty array, or one finding of type `"path_traversal"`.
|
|
264
137
|
*/
|
|
265
138
|
function containsPathTraversal(input) {
|
|
266
|
-
|
|
267
|
-
for (const pattern of PATH_TRAVERSAL_PATTERNS) {
|
|
268
|
-
const match = input.match(pattern);
|
|
269
|
-
if (match) {
|
|
270
|
-
findings.push({
|
|
271
|
-
type: "path_traversal",
|
|
272
|
-
description: "Potential path traversal attack detected",
|
|
273
|
-
matchedPattern: match[0].slice(0, 50)
|
|
274
|
-
});
|
|
275
|
-
break;
|
|
276
|
-
}
|
|
277
|
-
}
|
|
278
|
-
return findings;
|
|
139
|
+
return firstFindingOfType(input, ["filesystem"], "path_traversal");
|
|
279
140
|
}
|
|
280
141
|
/**
|
|
281
|
-
*
|
|
282
|
-
*
|
|
283
|
-
|
|
284
|
-
|
|
142
|
+
* Base metadata for the synthetic finding {@link containsHomoglyphs} produces, shaped
|
|
143
|
+
* like a catalog entry so downstream consumers see one consistent record.
|
|
144
|
+
*/
|
|
145
|
+
const MIXED_SCRIPT_FINDING = {
|
|
146
|
+
ruleId: "UNICODE-MIXED-SCRIPT-001",
|
|
147
|
+
type: "homoglyph",
|
|
148
|
+
severity: "high",
|
|
149
|
+
confidence: "medium",
|
|
150
|
+
description: "Identifier mixes scripts in a combination used for visual spoofing",
|
|
151
|
+
cwe: 1007,
|
|
152
|
+
primaryControl: "Compare UTS #39 skeletons at registration time and enforce an identifier restriction level",
|
|
153
|
+
variant: "nfc"
|
|
154
|
+
};
|
|
155
|
+
/** Overrides applied when the identifier carries a bidirectional control. */
|
|
156
|
+
const BIDI_FINDING_OVERRIDE = {
|
|
157
|
+
ruleId: "UNICODE-BIDI-OVERRIDE-001",
|
|
158
|
+
severity: "critical",
|
|
159
|
+
confidence: "high",
|
|
160
|
+
description: "Bidirectional override character in an identifier",
|
|
161
|
+
cwe: 451
|
|
162
|
+
};
|
|
163
|
+
/**
|
|
164
|
+
* Detect visually confusable characters in a **protected identifier**.
|
|
285
165
|
*
|
|
286
|
-
*
|
|
287
|
-
*
|
|
166
|
+
* Backed by UTS #39 script analysis rather than a hand-written lookalike table, so it
|
|
167
|
+
* reports the actual signal — a Latin/Cyrillic mix in `pаypal` — instead of flagging
|
|
168
|
+
* every non-ASCII character. Single-script values are not confusable with anything, so
|
|
169
|
+
* `Ольга Иванова` and `東京タワー` pass where the previous implementation rejected both.
|
|
288
170
|
*
|
|
289
|
-
*
|
|
290
|
-
*
|
|
291
|
-
*
|
|
171
|
+
* Scope this to usernames, domains, org names, and package names. Do **not** run it on
|
|
172
|
+
* prose or on people's names — see {@link validatePersonName}.
|
|
173
|
+
*
|
|
174
|
+
* @param input - Identifier to scan.
|
|
175
|
+
* @returns Empty array, or a single finding of type `"homoglyph"`.
|
|
292
176
|
*/
|
|
293
177
|
function containsHomoglyphs(input) {
|
|
294
|
-
|
|
295
|
-
|
|
296
|
-
|
|
297
|
-
|
|
298
|
-
|
|
299
|
-
|
|
300
|
-
|
|
301
|
-
|
|
302
|
-
}
|
|
303
|
-
return findings;
|
|
178
|
+
if (!input || typeof input !== "string") return [];
|
|
179
|
+
const analysis = analyzeIdentifier(input.length > 1e5 ? input.slice(0, MAX_SCAN_LENGTH) : input);
|
|
180
|
+
if (!analysis.isMixedScript && !analysis.hasBidiControls) return [];
|
|
181
|
+
return [{
|
|
182
|
+
...MIXED_SCRIPT_FINDING,
|
|
183
|
+
...analysis.hasBidiControls ? BIDI_FINDING_OVERRIDE : {},
|
|
184
|
+
matchedPattern: analysis.scripts.join("+").slice(0, 50)
|
|
185
|
+
}];
|
|
304
186
|
}
|
|
305
|
-
const DEFAULT_CONFIG = {
|
|
306
|
-
checkXSS: true,
|
|
307
|
-
checkSQLInjection: true,
|
|
308
|
-
checkNoSQLInjection: true,
|
|
309
|
-
checkCommandInjection: false,
|
|
310
|
-
checkPathTraversal: true,
|
|
311
|
-
checkHomoglyphs: true
|
|
312
|
-
};
|
|
313
187
|
/**
|
|
314
|
-
* Run
|
|
188
|
+
* Run the enabled detectors against `input` and aggregate findings.
|
|
315
189
|
*
|
|
316
|
-
*
|
|
317
|
-
*
|
|
318
|
-
*
|
|
190
|
+
* @deprecated Prefer {@link scanForThreats}, which takes explicit contexts and returns
|
|
191
|
+
* a score and verdict rather than one boolean. This wrapper maps the legacy toggles
|
|
192
|
+
* onto contexts and keeps the one-finding-per-category shape.
|
|
319
193
|
*
|
|
320
|
-
* Non-string
|
|
321
|
-
*
|
|
322
|
-
* reject non-strings.
|
|
194
|
+
* Non-string input (`null`, `undefined`, a number) is reported safe — wrap your own
|
|
195
|
+
* type validation around this if you need to reject those.
|
|
323
196
|
*
|
|
324
197
|
* @param input - The candidate string.
|
|
325
|
-
* @param config - Detector toggles.
|
|
326
|
-
* except command-injection.
|
|
198
|
+
* @param config - Detector toggles. Everything except command injection defaults on.
|
|
327
199
|
* @returns `{ isSafe, threats }`.
|
|
328
|
-
*
|
|
329
|
-
* @example
|
|
330
|
-
* ```ts
|
|
331
|
-
* const result = detectThreatPatterns(req.body.query);
|
|
332
|
-
* if (!result.isSafe) return new Response(getThreatErrorMessage(result), { status: 400 });
|
|
333
|
-
* ```
|
|
334
200
|
*/
|
|
335
|
-
function detectThreatPatterns(input, config =
|
|
201
|
+
function detectThreatPatterns(input, config = {}) {
|
|
336
202
|
if (!input || typeof input !== "string") return {
|
|
337
203
|
isSafe: true,
|
|
338
204
|
threats: []
|
|
339
205
|
};
|
|
206
|
+
const result = scanForThreats(input, { contexts: contextsFor(config) });
|
|
340
207
|
const threats = [];
|
|
341
|
-
|
|
342
|
-
|
|
343
|
-
|
|
344
|
-
|
|
345
|
-
|
|
208
|
+
const seen = /* @__PURE__ */ new Set();
|
|
209
|
+
for (const finding of result.findings) {
|
|
210
|
+
if (seen.has(finding.type)) continue;
|
|
211
|
+
seen.add(finding.type);
|
|
212
|
+
threats.push(finding);
|
|
213
|
+
}
|
|
346
214
|
if (config.checkHomoglyphs !== false) threats.push(...containsHomoglyphs(input));
|
|
347
215
|
return {
|
|
348
216
|
isSafe: threats.length === 0,
|
|
@@ -350,8 +218,7 @@ function detectThreatPatterns(input, config = DEFAULT_CONFIG) {
|
|
|
350
218
|
};
|
|
351
219
|
}
|
|
352
220
|
/**
|
|
353
|
-
* Boolean shortcut over {@link detectThreatPatterns}
|
|
354
|
-
* findings list when you only need a yes/no decision.
|
|
221
|
+
* Boolean shortcut over {@link detectThreatPatterns}.
|
|
355
222
|
*
|
|
356
223
|
* @param input - String to test.
|
|
357
224
|
* @param config - Optional detector toggles.
|
|
@@ -361,111 +228,409 @@ function isSafeInput(input, config) {
|
|
|
361
228
|
return detectThreatPatterns(input, config).isSafe;
|
|
362
229
|
}
|
|
363
230
|
/**
|
|
364
|
-
* HTML-entity
|
|
365
|
-
* insertion into HTML text and attribute contexts.
|
|
231
|
+
* HTML-entity-escape a value being inserted as **element text**.
|
|
366
232
|
*
|
|
367
|
-
*
|
|
368
|
-
*
|
|
369
|
-
* rendering. For rich-text use a vetted sanitizer (DOMPurify on the
|
|
370
|
-
* client, sanitize-html or similar on the server).
|
|
233
|
+
* Escapes `&`, `<`, `>`, `"`, `'`, and `/`, which covers text nodes and fully quoted
|
|
234
|
+
* attribute values.
|
|
371
235
|
*
|
|
372
|
-
*
|
|
236
|
+
* **Output encoding is context-dependent.** HTML text, quoted attributes, unquoted
|
|
237
|
+
* attributes, URLs, JavaScript string literals, and CSS each have different rules, and
|
|
238
|
+
* no single function is correct for all of them. This one is correct for text; use
|
|
239
|
+
* {@link escapeHtmlAttribute} for attribute values, `sanitizeUrl` for URLs, and
|
|
240
|
+
* `sanitizeHtml` (DOMPurify) when the value is meant to *be* markup.
|
|
373
241
|
*
|
|
374
|
-
*
|
|
375
|
-
*
|
|
242
|
+
* There is deliberately no CSS-context escaper here, and no general JavaScript-string
|
|
243
|
+
* escaper — hand-rolled versions of those are reliably wrong, and the fix is to stop
|
|
244
|
+
* interpolating untrusted values into style and script *source*. Embedding untrusted
|
|
245
|
+
* *data* in a script element is the one tractable case, because `JSON.stringify` fixes
|
|
246
|
+
* the string boundaries first; {@link encodeJsonForScript} covers that and nothing else.
|
|
247
|
+
*
|
|
248
|
+
* @param input - Untrusted string. Non-string or empty input yields `""`.
|
|
249
|
+
* @returns Entity-escaped output safe to interpolate into HTML text.
|
|
250
|
+
*
|
|
251
|
+
* @example
|
|
252
|
+
* ```ts
|
|
253
|
+
* escapeHtmlText('<script>alert("xss")<\/script>');
|
|
254
|
+
* // "<script>alert("xss")</script>"
|
|
255
|
+
* ```
|
|
376
256
|
*/
|
|
377
|
-
function
|
|
257
|
+
function escapeHtmlText(input) {
|
|
378
258
|
if (!input || typeof input !== "string") return "";
|
|
379
259
|
return input.replace(/&/g, "&").replace(/</g, "<").replace(/>/g, ">").replace(/"/g, """).replace(/'/g, "'").replace(/\//g, "/");
|
|
380
260
|
}
|
|
381
261
|
/**
|
|
382
|
-
*
|
|
262
|
+
* Control characters escaped in attribute position.
|
|
263
|
+
*
|
|
264
|
+
* The C0 and C1 ranges plus the two Unicode line terminators. The set is the point:
|
|
265
|
+
* HTML's unquoted-attribute state ends at space, tab, LF, FF or CR, and this used to
|
|
266
|
+
* escape tab, LF and CR but not **form feed**. It also escaped CR, which the input
|
|
267
|
+
* stream preprocessor normalises to LF before the tokenizer runs — so three of the four
|
|
268
|
+
* real terminators were covered, plus the one that cannot matter.
|
|
269
|
+
*
|
|
270
|
+
* @see https://html.spec.whatwg.org/multipage/parsing.html
|
|
271
|
+
*/
|
|
272
|
+
const ATTRIBUTE_CONTROL_CHARS = /[\u0000-\u001f\u007f-\u009f\u2028\u2029]/g;
|
|
273
|
+
/**
|
|
274
|
+
* HTML-entity-escape a value being inserted as an **attribute value**.
|
|
383
275
|
*
|
|
384
|
-
*
|
|
385
|
-
*
|
|
386
|
-
*
|
|
387
|
-
* counterparts.
|
|
388
|
-
* 2. Replace known homoglyphs (Cyrillic `А`, Greek `Ε`, …) with their
|
|
389
|
-
* ASCII equivalents (`A`, `E`, …).
|
|
276
|
+
* Everything {@link escapeHtmlText} escapes, plus backtick, equals, and whitespace —
|
|
277
|
+
* the characters that let a payload break out of an *unquoted* attribute. That case is
|
|
278
|
+
* precisely what generic "escape for display" helpers get wrong.
|
|
390
279
|
*
|
|
391
|
-
*
|
|
392
|
-
*
|
|
280
|
+
* The ceiling on the unquoted case is injection of a valueless boolean attribute —
|
|
281
|
+
* `autofocus`, `disabled`, `formnovalidate` — not script execution: an injected
|
|
282
|
+
* `onmouseover=…` arrives with its `=` already escaped, so it lands as an attribute
|
|
283
|
+
* whose *name* is the escaped text, with no handler bound.
|
|
393
284
|
*
|
|
394
|
-
*
|
|
285
|
+
* Quote your attributes anyway. This makes an unquoted attribute survivable; it does
|
|
286
|
+
* not make it correct.
|
|
395
287
|
*
|
|
396
|
-
* @param input -
|
|
397
|
-
* @returns
|
|
288
|
+
* @param input - Untrusted string. Non-string or empty input yields `""`.
|
|
289
|
+
* @returns Output safe to interpolate into a quoted or unquoted attribute value.
|
|
398
290
|
*/
|
|
399
|
-
function
|
|
291
|
+
function escapeHtmlAttribute(input) {
|
|
400
292
|
if (!input || typeof input !== "string") return "";
|
|
401
|
-
|
|
402
|
-
|
|
403
|
-
|
|
293
|
+
return escapeHtmlText(input).replace(/`/g, "`").replace(/=/g, "=").replace(/ /g, " ").replace(ATTRIBUTE_CONTROL_CHARS, (character) => {
|
|
294
|
+
return `&#x${(character.codePointAt(0) ?? 0).toString(16).toUpperCase().padStart(2, "0")};`;
|
|
295
|
+
});
|
|
296
|
+
}
|
|
297
|
+
/**
|
|
298
|
+
* HTML-entity-escape a value for display.
|
|
299
|
+
*
|
|
300
|
+
* @deprecated Renamed to {@link escapeHtmlText}, which says what it actually does. The
|
|
301
|
+
* old name suggested a general-purpose "make this safe to display" operation, and
|
|
302
|
+
* callers reasonably read it as attribute-safe — which entity escaping alone is not,
|
|
303
|
+
* for *unquoted* attributes. Behaviour is unchanged; only the name is.
|
|
304
|
+
*
|
|
305
|
+
* @param input - Untrusted string.
|
|
306
|
+
* @returns Entity-escaped output.
|
|
307
|
+
*/
|
|
308
|
+
function sanitizeForDisplay(input) {
|
|
309
|
+
return escapeHtmlText(input);
|
|
404
310
|
}
|
|
311
|
+
/** Cap on the input a log value is read from, before escaping expands it. */
|
|
312
|
+
const DEFAULT_LOG_VALUE_LENGTH = 2048;
|
|
313
|
+
/**
|
|
314
|
+
* Characters that must not reach a log sink as themselves.
|
|
315
|
+
*
|
|
316
|
+
* C0 and C1, the zero-width and bidirectional formatting ranges, and the byte-order
|
|
317
|
+
* mark. ESC lives inside C0, which is why no separate ANSI sequence matching is needed:
|
|
318
|
+
* escaping the introducer alone neutralises every terminal sequence *losslessly*,
|
|
319
|
+
* whereas deleting whole sequences would discard the payload a reader is investigating.
|
|
320
|
+
* The bidi range matters for the same reason `UNICODE-BIDI-OVERRIDE-001` exists — a
|
|
321
|
+
* right-to-left override reorders how a log line renders without changing its bytes.
|
|
322
|
+
*/
|
|
323
|
+
const LOG_UNSAFE_CHARS = /[\u0000-\u001f\u007f-\u009f\u200b-\u200f\u2028-\u202e\u2060-\u2064\u2066-\u2069\ufeff]/g;
|
|
324
|
+
/** Readable forms for the three characters a reader expects to recognise. */
|
|
325
|
+
const LOG_SHORTHAND = {
|
|
326
|
+
" ": "\\t",
|
|
327
|
+
"\n": "\\n",
|
|
328
|
+
"\r": "\\r"
|
|
329
|
+
};
|
|
405
330
|
/**
|
|
406
|
-
*
|
|
407
|
-
*
|
|
408
|
-
*
|
|
331
|
+
* Escape a value for inclusion in a log record.
|
|
332
|
+
*
|
|
333
|
+
* This is the control named by the log-injection rules. A log line is a *sink*: a value
|
|
334
|
+
* carrying a newline forges an entry (CWE-117), one carrying a terminal escape rewrites
|
|
335
|
+
* what an operator sees, and one carrying a bidirectional override reorders the line
|
|
336
|
+
* without altering a byte of it.
|
|
337
|
+
*
|
|
338
|
+
* Escaping rather than stripping is deliberate. The record is evidence, so the encoded
|
|
339
|
+
* form is reversible and nothing is silently discarded — contrast `stripAnsi`, which
|
|
340
|
+
* deletes. Structured logging is still the better answer, because it removes the
|
|
341
|
+
* ambiguity this function can only make visible; use both.
|
|
342
|
+
*
|
|
343
|
+
* @param value - Untrusted field value. Non-string or empty input yields `""`.
|
|
344
|
+
* @param options - Optional bounds.
|
|
345
|
+
* @param options.maxLength - Characters read from `value`. Defaults to 2048. Truncation
|
|
346
|
+
* is announced in the output rather than applied silently, and the returned string may
|
|
347
|
+
* exceed this length, because escaping expands.
|
|
348
|
+
* @returns A single-line, control-free rendering of `value`.
|
|
349
|
+
*
|
|
350
|
+
* @example
|
|
351
|
+
* ```ts
|
|
352
|
+
* encodeLogValue("alice\nINFO user promoted to admin");
|
|
353
|
+
* // "alice\\nINFO user promoted to admin" — one line, no forged entry
|
|
354
|
+
* ```
|
|
355
|
+
*/
|
|
356
|
+
function encodeLogValue(value, options = {}) {
|
|
357
|
+
if (!value || typeof value !== "string") return "";
|
|
358
|
+
const { maxLength = DEFAULT_LOG_VALUE_LENGTH } = options;
|
|
359
|
+
const limit = Number.isInteger(maxLength) && maxLength > 0 ? maxLength : DEFAULT_LOG_VALUE_LENGTH;
|
|
360
|
+
const dropped = value.length - limit;
|
|
361
|
+
const encoded = (dropped > 0 ? value.slice(0, limit) : value).replace(LOG_UNSAFE_CHARS, (character) => {
|
|
362
|
+
const shorthand = LOG_SHORTHAND[character];
|
|
363
|
+
if (shorthand !== void 0) return shorthand;
|
|
364
|
+
return `\\u${(character.codePointAt(0) ?? 0).toString(16).padStart(4, "0")}`;
|
|
365
|
+
});
|
|
366
|
+
return dropped > 0 ? `${encoded}[truncated ${dropped} chars]` : encoded;
|
|
367
|
+
}
|
|
368
|
+
/**
|
|
369
|
+
* A leading formula trigger, tolerating the whitespace and quotes a reader strips first.
|
|
370
|
+
*
|
|
371
|
+
* Mirrors `CSV-FORMULA-LEAD-001`, deliberately: the rule sees through leading quotes and
|
|
372
|
+
* spaces because spreadsheet importers do, so an encoder that only looked at index 0
|
|
373
|
+
* would leave ` =cmd|'/c calc'!A1` live.
|
|
374
|
+
*/
|
|
375
|
+
const CSV_FORMULA_LEAD = /^[\s'"]{0,8}[=+\-@\t\r]/;
|
|
376
|
+
/** Fields containing any of these must be quoted per RFC 4180 sections 2.6 and 2.7. */
|
|
377
|
+
const CSV_QUOTE_REQUIRED = /["\r\n]/;
|
|
378
|
+
/**
|
|
379
|
+
* Escape one cell for CSV export.
|
|
380
|
+
*
|
|
381
|
+
* This is the control named by the formula-injection rules. A CSV file is not inert: a
|
|
382
|
+
* cell beginning `=`, `+`, `-`, `@`, tab or CR is evaluated as a formula by Excel,
|
|
383
|
+
* Sheets and LibreOffice when the recipient opens it, so the payload executes on *their*
|
|
384
|
+
* machine, outside the exporting application entirely (CWE-1236).
|
|
385
|
+
*
|
|
386
|
+
* Two separate jobs, in order: neutralise the formula trigger with a leading apostrophe,
|
|
387
|
+
* then apply RFC 4180 quoting so the field cannot break the row.
|
|
388
|
+
*
|
|
389
|
+
* **Only strings are prefixed.** A `number` or `boolean` came from the application's own
|
|
390
|
+
* types and cannot carry a formula, so `-1234` exports as a negative number while
|
|
391
|
+
* `"-1234"` exports as text. Pass numeric columns as numbers, or every negative value in
|
|
392
|
+
* the sheet becomes a string.
|
|
393
|
+
*
|
|
394
|
+
* Three things worth knowing before relying on it:
|
|
395
|
+
* - The leading apostrophe is an Excel convention, **not** an RFC 4180 construct. Readers
|
|
396
|
+
* that do not implement it surface it as a literal character in the data.
|
|
397
|
+
* - NUL is removed rather than escaped, so it does not round-trip.
|
|
398
|
+
* - Scanning the output with `scanForThreats` still reports a finding, by design:
|
|
399
|
+
* `CSV-FORMULA-LEAD-001` sees through the apostrophe and `CSV-DDE-001` is
|
|
400
|
+
* position-independent. The rules describe the *value*; this function protects the
|
|
401
|
+
* *file*. A clean scan is the wrong acceptance test.
|
|
402
|
+
*
|
|
403
|
+
* @param value - Cell value. `null` and `undefined` become `""`.
|
|
404
|
+
* @param options - Optional dialect settings.
|
|
405
|
+
* @param options.delimiter - Field separator the row will be joined with. Defaults to `","`.
|
|
406
|
+
* @returns The escaped field, ready to join into a row.
|
|
407
|
+
*
|
|
408
|
+
* @example
|
|
409
|
+
* ```ts
|
|
410
|
+
* escapeCsvField("=WEBSERVICE(\"https://evil.example\")");
|
|
411
|
+
* // quoted, and inert on open
|
|
412
|
+
* escapeCsvField(-1234); // "-1234" — a number, not a formula
|
|
413
|
+
* ```
|
|
414
|
+
*/
|
|
415
|
+
function escapeCsvField(value, options = {}) {
|
|
416
|
+
if (value === null || value === void 0) return "";
|
|
417
|
+
const delimiter = options.delimiter ?? ",";
|
|
418
|
+
const isUntrustedText = typeof value === "string";
|
|
419
|
+
const cleaned = (isUntrustedText ? value : String(value)).replace(/\u0000/g, "");
|
|
420
|
+
const neutralised = isUntrustedText && CSV_FORMULA_LEAD.test(cleaned) ? `'${cleaned}` : cleaned;
|
|
421
|
+
return CSV_QUOTE_REQUIRED.test(neutralised) || neutralised.includes(delimiter) ? `"${neutralised.replaceAll("\"", "\"\"")}"` : neutralised;
|
|
422
|
+
}
|
|
423
|
+
/**
|
|
424
|
+
* Escape and join one row for CSV export.
|
|
425
|
+
*
|
|
426
|
+
* @param values - Cell values, in column order.
|
|
427
|
+
* @param options - Optional dialect settings.
|
|
428
|
+
* @param options.delimiter - Field separator. Defaults to `","`.
|
|
429
|
+
* @returns The joined row, without a line terminator.
|
|
430
|
+
*
|
|
431
|
+
* @example
|
|
432
|
+
* ```ts
|
|
433
|
+
* toCsvRow(["Ada Lovelace", "=1+1", 42]);
|
|
434
|
+
* ```
|
|
435
|
+
*/
|
|
436
|
+
function toCsvRow(values, options = {}) {
|
|
437
|
+
if (!Array.isArray(values)) return "";
|
|
438
|
+
const delimiter = options.delimiter ?? ",";
|
|
439
|
+
return values.map((value) => escapeCsvField(value, { delimiter })).join(delimiter);
|
|
440
|
+
}
|
|
441
|
+
/**
|
|
442
|
+
* The five characters that must not survive into a script element verbatim.
|
|
443
|
+
*
|
|
444
|
+
* None is a JSON structural character, so each can only ever occur inside a string
|
|
445
|
+
* literal, where a unicode escape is legal and semantically identical. That is what makes
|
|
446
|
+
* this transformation safe to apply to `JSON.stringify` output without reparsing it.
|
|
447
|
+
*
|
|
448
|
+
* `<` and `>` close the element; `&` matters when a caller relocates the payload into a
|
|
449
|
+
* context that *is* entity-decoded; U+2028 and U+2029 terminate a line in JavaScript
|
|
450
|
+
* source, which JSON permits raw inside strings.
|
|
451
|
+
*/
|
|
452
|
+
const SCRIPT_UNSAFE_JSON = /[<>&\u2028\u2029]/g;
|
|
453
|
+
/** Escapes for {@link SCRIPT_UNSAFE_JSON}, all valid inside a JSON string literal. */
|
|
454
|
+
const SCRIPT_JSON_ESCAPES = {
|
|
455
|
+
"<": "\\u003c",
|
|
456
|
+
">": "\\u003e",
|
|
457
|
+
"&": "\\u0026",
|
|
458
|
+
"\u2028": "\\u2028",
|
|
459
|
+
"\u2029": "\\u2029"
|
|
460
|
+
};
|
|
461
|
+
/**
|
|
462
|
+
* Serialise a value for embedding inside a `<script>` element.
|
|
463
|
+
*
|
|
464
|
+
* `JSON.stringify` alone is not safe here. Its output may contain `<\/script>`, which
|
|
465
|
+
* closes the element from *inside a string literal* — the HTML tokenizer never looks at
|
|
466
|
+
* JavaScript syntax — so the remainder of the payload becomes markup.
|
|
467
|
+
*
|
|
468
|
+
* **Script element content only.** The output contains unescaped `"`, so it must never be
|
|
469
|
+
* placed in an attribute; use {@link escapeHtmlAttribute} there. It is also not a general
|
|
470
|
+
* JavaScript-string escaper — it is safe precisely because `JSON.stringify` has already
|
|
471
|
+
* decided where the string boundaries are.
|
|
472
|
+
*
|
|
473
|
+
* Using `<script type="application/json">` with `JSON.parse(el.textContent)` does **not**
|
|
474
|
+
* remove the need for this: a raw `<\/script>` in the data closes that element too.
|
|
475
|
+
*
|
|
476
|
+
* @param value - Any JSON-serialisable value.
|
|
477
|
+
* @returns JSON text safe to place between `<script>` tags.
|
|
478
|
+
* @throws {TypeError} If `value` cannot be represented as JSON — `undefined`, a function
|
|
479
|
+
* or a symbol at the top level (for which `JSON.stringify` returns `undefined` rather
|
|
480
|
+
* than a string), a circular structure, or a `BigInt`. Failing loudly is deliberate: a
|
|
481
|
+
* sentinel string would emit a syntax error into the page instead.
|
|
482
|
+
*
|
|
483
|
+
* @example
|
|
484
|
+
* ```ts
|
|
485
|
+
* const json = encodeJsonForScript({ name: userName });
|
|
486
|
+
* const html = "<script>window.__DATA__ = " + json + ";<\/script>";
|
|
487
|
+
* ```
|
|
488
|
+
*/
|
|
489
|
+
function encodeJsonForScript(value) {
|
|
490
|
+
let serialised;
|
|
491
|
+
try {
|
|
492
|
+
serialised = JSON.stringify(value);
|
|
493
|
+
} catch (cause) {
|
|
494
|
+
throw new TypeError("encodeJsonForScript: value is not JSON-serialisable", { cause });
|
|
495
|
+
}
|
|
496
|
+
if (typeof serialised !== "string") throw new TypeError(`encodeJsonForScript: ${typeof value} has no JSON representation at the top level`);
|
|
497
|
+
return serialised.replace(SCRIPT_UNSAFE_JSON, (character) => SCRIPT_JSON_ESCAPES[character] ?? character);
|
|
498
|
+
}
|
|
499
|
+
/**
|
|
500
|
+
* Fold non-ASCII lookalike characters onto ASCII and compose to NFC.
|
|
501
|
+
*
|
|
502
|
+
* @deprecated Prefer `getSkeleton` and `analyzeIdentifier` from
|
|
503
|
+
* `@resq-systems/security/unicode`. Rewriting a user's identifier into a different
|
|
504
|
+
* string loses information and only *looks* safe — the durable pattern is to store
|
|
505
|
+
* what they typed, index its skeleton, and compare skeletons for collisions.
|
|
506
|
+
*
|
|
507
|
+
* Now backed by the UTS #39 confusable tables rather than the previous 14-entry map,
|
|
508
|
+
* so coverage is far wider. Combining marks are preserved (`e` + U+0301 still composes
|
|
509
|
+
* to `é`) and ASCII characters are never rewritten.
|
|
510
|
+
*
|
|
511
|
+
* @param input - Raw string from an untrusted source. Non-string input yields `""`.
|
|
512
|
+
* @returns NFC-composed string with non-ASCII confusables folded to ASCII.
|
|
513
|
+
*/
|
|
514
|
+
function normalizeUnicode(input) {
|
|
515
|
+
return foldConfusables(input);
|
|
516
|
+
}
|
|
517
|
+
/**
|
|
518
|
+
* Generic user-facing fallback message. Render verbatim when a detector fires and you
|
|
519
|
+
* do not want to reveal which one.
|
|
409
520
|
*/
|
|
410
521
|
const THREAT_DETECTED_MESSAGE = "Input contains potentially unsafe content";
|
|
411
522
|
/**
|
|
412
|
-
*
|
|
413
|
-
*
|
|
523
|
+
* Refinement helper for `zod.string().refine(...)`, `effect/Schema.filter(...)`, or
|
|
524
|
+
* any predicate-based validator. Equivalent to {@link isSafeInput} with defaults.
|
|
414
525
|
*
|
|
415
|
-
*
|
|
526
|
+
* @param input - String to test.
|
|
527
|
+
* @returns `true` when no detector fires.
|
|
416
528
|
*/
|
|
417
529
|
function validateSafeText(input) {
|
|
418
530
|
return isSafeInput(input);
|
|
419
531
|
}
|
|
420
532
|
/**
|
|
421
|
-
*
|
|
422
|
-
*
|
|
423
|
-
*
|
|
424
|
-
*
|
|
533
|
+
* Letters, marks, apostrophes, hyphens, periods, spaces, and the two joiners — nothing
|
|
534
|
+
* else.
|
|
535
|
+
*
|
|
536
|
+
* U+200C (ZWNJ) and U+200D (ZWJ) are part of the spelling, not decoration. Persian and
|
|
537
|
+
* Hindi names need them to be written correctly — a ZWNJ is what keeps the two halves
|
|
538
|
+
* of `میروم` from joining — so a pattern without them rejects the name its owner
|
|
539
|
+
* actually has. They carry no injection risk here: everything a payload needs (`<`,
|
|
540
|
+
* `(`, `;`, `$`, `=`, digits) stays excluded. Written as escapes, not literals — an
|
|
541
|
+
* invisible character pasted into a character class is unreviewable in a diff.
|
|
542
|
+
*/
|
|
543
|
+
const PERSON_NAME_PATTERN = /^[\p{L}\p{M}'’.\-\s\u{200C}\u{200D}]+$/u;
|
|
544
|
+
/** Shortest accepted name. Mononyms and single-letter names exist. */
|
|
545
|
+
const MIN_NAME_LENGTH = 1;
|
|
546
|
+
/** Longest accepted name. */
|
|
547
|
+
const MAX_NAME_LENGTH = 200;
|
|
548
|
+
/**
|
|
549
|
+
* Validate a human name field.
|
|
550
|
+
*
|
|
551
|
+
* The policy is an allowlist of what a name is made of — letters in any script,
|
|
552
|
+
* combining marks, apostrophes, hyphens, periods, spaces — plus a length bound and a
|
|
553
|
+
* bidirectional-control check. Nothing that passes it can carry an injection payload,
|
|
554
|
+
* because `<`, `(`, `;`, `$`, `=`, and every digit are already excluded.
|
|
555
|
+
*
|
|
556
|
+
* It deliberately does **not** run SQL, path-traversal, or confusable detectors. A
|
|
557
|
+
* name is not a query, a path, or a protected identifier, and subjecting one to those
|
|
558
|
+
* checks rejects real people: the previous implementation ran the homoglyph detector
|
|
559
|
+
* here, which failed any name containing а, е, о, р, с, or х — that is, most Russian,
|
|
560
|
+
* Ukrainian, Bulgarian, Serbian, and Greek names.
|
|
561
|
+
*
|
|
562
|
+
* Encode the value at whatever sink it eventually reaches. That is what makes it safe;
|
|
563
|
+
* this function only establishes that it is a name.
|
|
564
|
+
*
|
|
565
|
+
* @param input - Candidate name.
|
|
566
|
+
* @returns `true` when the value is a plausible name.
|
|
567
|
+
*
|
|
568
|
+
* @example
|
|
569
|
+
* ```ts
|
|
570
|
+
* validatePersonName("O'Brien"); // true
|
|
571
|
+
* validatePersonName("José García"); // true
|
|
572
|
+
* validatePersonName("Ольга Иванова"); // true
|
|
573
|
+
* validatePersonName("John123"); // false
|
|
574
|
+
* validatePersonName("<script>x<\/script>"); // false
|
|
575
|
+
* ```
|
|
576
|
+
*/
|
|
577
|
+
function validatePersonName(input) {
|
|
578
|
+
if (typeof input !== "string") return false;
|
|
579
|
+
const normalized = input.normalize("NFC");
|
|
580
|
+
if (normalized.length < MIN_NAME_LENGTH || normalized.length > MAX_NAME_LENGTH) return false;
|
|
581
|
+
if (containsBidiControls(normalized)) return false;
|
|
582
|
+
return PERSON_NAME_PATTERN.test(normalized);
|
|
583
|
+
}
|
|
584
|
+
/**
|
|
585
|
+
* Validate a human name field.
|
|
425
586
|
*
|
|
426
|
-
*
|
|
587
|
+
* @deprecated Renamed to {@link validatePersonName}. The old name implied a general
|
|
588
|
+
* "safe name" check and was implemented as one, running injection and homoglyph
|
|
589
|
+
* detectors against people's names. Behaviour now matches
|
|
590
|
+
* {@link validatePersonName}.
|
|
427
591
|
*
|
|
428
|
-
* @
|
|
429
|
-
*
|
|
592
|
+
* @param input - Candidate name.
|
|
593
|
+
* @returns `true` when the value is a plausible name.
|
|
430
594
|
*/
|
|
431
595
|
function validateSafeName(input) {
|
|
432
|
-
|
|
433
|
-
if (!isSafeInput(normalized, { checkCommandInjection: false })) return false;
|
|
434
|
-
return /^[\p{L}\p{M}'\-\s.]+$/u.test(normalized);
|
|
596
|
+
return validatePersonName(input);
|
|
435
597
|
}
|
|
598
|
+
/** Longest address accepted, per RFC 5321 §4.5.3.1.3. Also bounds regex cost. */
|
|
599
|
+
const MAX_EMAIL_LENGTH = 254;
|
|
600
|
+
/** RFC-shaped address check. Length is bounded before this runs. */
|
|
601
|
+
const EMAIL_PATTERN = /^[a-zA-Z0-9.!#$%&'*+/=?^_`{|}~-]+@[a-zA-Z0-9](?:[a-zA-Z0-9-]{0,61}[a-zA-Z0-9])?(?:\.[a-zA-Z0-9](?:[a-zA-Z0-9-]{0,61}[a-zA-Z0-9])?)*$/;
|
|
436
602
|
/**
|
|
437
|
-
*
|
|
603
|
+
* Validate an email address.
|
|
438
604
|
*
|
|
439
|
-
*
|
|
440
|
-
*
|
|
441
|
-
*
|
|
442
|
-
*
|
|
443
|
-
* operators.
|
|
605
|
+
* Two checks: an RFC-shaped format match (length-bounded first, so the pattern never
|
|
606
|
+
* sees an unbounded string), and UTS #39 identifier analysis of the **domain**, where
|
|
607
|
+
* a mixed-script host is the IDN homograph attack — `аpple.com` with a Cyrillic `а`
|
|
608
|
+
* resolves somewhere else entirely.
|
|
444
609
|
*
|
|
445
|
-
*
|
|
610
|
+
* The local part is not confusable-checked: it is not a routable identifier, and
|
|
611
|
+
* flagging it would reject legitimate internationalized mailboxes.
|
|
612
|
+
*
|
|
613
|
+
* @param input - Candidate address.
|
|
614
|
+
* @returns `true` when the format is valid and the domain is not a script mix.
|
|
446
615
|
*/
|
|
447
616
|
function validateSafeEmail(input) {
|
|
448
|
-
if (input
|
|
449
|
-
if (
|
|
450
|
-
|
|
451
|
-
|
|
452
|
-
|
|
453
|
-
checkNoSQLInjection: true,
|
|
454
|
-
checkCommandInjection: false,
|
|
455
|
-
checkPathTraversal: false,
|
|
456
|
-
checkHomoglyphs: true
|
|
457
|
-
}).isSafe;
|
|
617
|
+
if (typeof input !== "string") return false;
|
|
618
|
+
if (input.length > MAX_EMAIL_LENGTH) return false;
|
|
619
|
+
if (!EMAIL_PATTERN.test(input)) return false;
|
|
620
|
+
const analysis = analyzeIdentifier(input.slice(input.lastIndexOf("@") + 1));
|
|
621
|
+
return !analysis.isMixedScript && !analysis.hasBidiControls;
|
|
458
622
|
}
|
|
459
623
|
/**
|
|
460
|
-
*
|
|
461
|
-
*
|
|
462
|
-
*
|
|
463
|
-
*
|
|
624
|
+
* Render a user-facing error message for a detection result.
|
|
625
|
+
*
|
|
626
|
+
* Uses only the **first** finding: enumerating every category that fired leaks the
|
|
627
|
+
* shape of the rule set to whoever is probing it. Log `result.threats` server-side for
|
|
628
|
+
* diagnostics and return this to the client.
|
|
464
629
|
*
|
|
465
|
-
*
|
|
466
|
-
*
|
|
467
|
-
*
|
|
468
|
-
*
|
|
630
|
+
* @param result - A {@link ThreatDetectionResult}, a `ThreatScanResult`-shaped object,
|
|
631
|
+
* or any `{ isSafe, threats }` pair.
|
|
632
|
+
* @returns A message, or `""` when the result is safe — so `message || undefined`
|
|
633
|
+
* works at a call site.
|
|
469
634
|
*/
|
|
470
635
|
function getThreatErrorMessage(result) {
|
|
471
636
|
if (result.isSafe) return "";
|
|
@@ -477,11 +642,27 @@ function getThreatErrorMessage(result) {
|
|
|
477
642
|
case "nosql_injection": return "Input contains potentially malicious query operators";
|
|
478
643
|
case "command_injection": return "Input contains potentially malicious system commands";
|
|
479
644
|
case "path_traversal": return "Input contains potentially malicious file path characters";
|
|
645
|
+
case "prototype_pollution": return "Input contains potentially malicious object property names";
|
|
480
646
|
case "homoglyph": return "Input contains suspicious lookalike characters";
|
|
647
|
+
case "header_injection": return "Input contains line breaks that are not allowed in this field";
|
|
648
|
+
case "ldap_injection": return "Input contains potentially malicious directory query characters";
|
|
649
|
+
case "xpath_injection": return "Input contains potentially malicious query expressions";
|
|
650
|
+
case "xml_injection": return "Input contains potentially malicious document declarations";
|
|
651
|
+
case "template_injection": return "Input contains potentially malicious template expressions";
|
|
652
|
+
case "file_inclusion": return "Input contains potentially malicious resource references";
|
|
653
|
+
case "ssrf": return "Input contains a network address that is not allowed";
|
|
654
|
+
case "formula_injection": return "Input contains spreadsheet formula characters";
|
|
655
|
+
case "log_injection": return "Input contains characters that are not allowed in this field";
|
|
656
|
+
case "prompt_injection": return "Input contains instructions that are not allowed in this field";
|
|
657
|
+
case "parameter_pollution": return "Input contains additional query parameters that are not allowed";
|
|
658
|
+
case "credential_exposure": return "Request contains credential material that must not be sent or stored here";
|
|
659
|
+
case "jwt_tampering": return "Token is not signed with an accepted algorithm";
|
|
660
|
+
case "double_encoding": return "Input contains characters that are encoded more than once";
|
|
661
|
+
case "resource_abuse": return "Input is too large or too repetitive to process";
|
|
481
662
|
default: return assertNever(threat.type);
|
|
482
663
|
}
|
|
483
664
|
}
|
|
484
665
|
//#endregion
|
|
485
|
-
export { THREAT_DETECTED_MESSAGE, containsCommandInjection, containsHomoglyphs, containsNoSQLInjection, containsPathTraversal, containsSQLInjection, containsXSSPatterns, detectThreatPatterns, getThreatErrorMessage, isSafeInput, normalizeUnicode, sanitizeForDisplay, validateSafeEmail, validateSafeName, validateSafeText };
|
|
666
|
+
export { THREAT_DETECTED_MESSAGE, containsCommandInjection, containsHomoglyphs, containsNoSQLInjection, containsPathTraversal, containsPrototypePollution, containsSQLInjection, containsXSSPatterns, detectThreatPatterns, encodeJsonForScript, encodeLogValue, escapeCsvField, escapeHtmlAttribute, escapeHtmlText, getThreatErrorMessage, isSafeInput, normalizeUnicode, sanitizeForDisplay, toCsvRow, validatePersonName, validateSafeEmail, validateSafeName, validateSafeText };
|
|
486
667
|
|
|
487
668
|
//# sourceMappingURL=validators.mjs.map
|