@resq-systems/security 1.0.4 → 2.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (108) hide show
  1. package/README.md +236 -33
  2. package/lib/controls/address.d.mts +142 -0
  3. package/lib/controls/address.d.mts.map +1 -0
  4. package/lib/controls/address.mjs +533 -0
  5. package/lib/controls/address.mjs.map +1 -0
  6. package/lib/controls/csrf.d.mts +91 -0
  7. package/lib/controls/csrf.d.mts.map +1 -0
  8. package/lib/controls/csrf.mjs +200 -0
  9. package/lib/controls/csrf.mjs.map +1 -0
  10. package/lib/controls/index.d.mts +8 -0
  11. package/lib/controls/index.mjs +8 -0
  12. package/lib/controls/origin.d.mts +95 -0
  13. package/lib/controls/origin.d.mts.map +1 -0
  14. package/lib/controls/origin.mjs +156 -0
  15. package/lib/controls/origin.mjs.map +1 -0
  16. package/lib/controls/payload.d.mts +84 -0
  17. package/lib/controls/payload.d.mts.map +1 -0
  18. package/lib/controls/payload.mjs +147 -0
  19. package/lib/controls/payload.mjs.map +1 -0
  20. package/lib/controls/query.d.mts +157 -0
  21. package/lib/controls/query.d.mts.map +1 -0
  22. package/lib/controls/query.mjs +368 -0
  23. package/lib/controls/query.mjs.map +1 -0
  24. package/lib/controls/redirect.d.mts +92 -0
  25. package/lib/controls/redirect.d.mts.map +1 -0
  26. package/lib/controls/redirect.mjs +110 -0
  27. package/lib/controls/redirect.mjs.map +1 -0
  28. package/lib/controls/upload.d.mts +108 -0
  29. package/lib/controls/upload.d.mts.map +1 -0
  30. package/lib/controls/upload.mjs +374 -0
  31. package/lib/controls/upload.mjs.map +1 -0
  32. package/lib/crypto.d.mts +18 -5
  33. package/lib/crypto.d.mts.map +1 -1
  34. package/lib/crypto.mjs +35 -24
  35. package/lib/crypto.mjs.map +1 -1
  36. package/lib/hash.d.mts +83 -0
  37. package/lib/hash.d.mts.map +1 -0
  38. package/lib/hash.mjs +111 -0
  39. package/lib/hash.mjs.map +1 -0
  40. package/lib/index.d.mts +18 -2
  41. package/lib/index.mjs +20 -2
  42. package/lib/paths.d.mts +92 -0
  43. package/lib/paths.d.mts.map +1 -0
  44. package/lib/paths.mjs +140 -0
  45. package/lib/paths.mjs.map +1 -0
  46. package/lib/sanitize.d.mts +137 -35
  47. package/lib/sanitize.d.mts.map +1 -1
  48. package/lib/sanitize.mjs +170 -46
  49. package/lib/sanitize.mjs.map +1 -1
  50. package/lib/threats/capec.generated.d.mts +59 -0
  51. package/lib/threats/capec.generated.d.mts.map +1 -0
  52. package/lib/threats/capec.generated.mjs +644 -0
  53. package/lib/threats/capec.generated.mjs.map +1 -0
  54. package/lib/threats/engine.d.mts +94 -0
  55. package/lib/threats/engine.d.mts.map +1 -0
  56. package/lib/threats/engine.mjs +167 -0
  57. package/lib/threats/engine.mjs.map +1 -0
  58. package/lib/threats/index.d.mts +11 -0
  59. package/lib/threats/index.mjs +11 -0
  60. package/lib/threats/rules/datastore.d.mts +13 -0
  61. package/lib/threats/rules/datastore.d.mts.map +1 -0
  62. package/lib/threats/rules/datastore.mjs +366 -0
  63. package/lib/threats/rules/datastore.mjs.map +1 -0
  64. package/lib/threats/rules/index.d.mts +54 -0
  65. package/lib/threats/rules/index.d.mts.map +1 -0
  66. package/lib/threats/rules/index.mjs +121 -0
  67. package/lib/threats/rules/index.mjs.map +1 -0
  68. package/lib/threats/rules/markup.d.mts +28 -0
  69. package/lib/threats/rules/markup.d.mts.map +1 -0
  70. package/lib/threats/rules/markup.mjs +373 -0
  71. package/lib/threats/rules/markup.mjs.map +1 -0
  72. package/lib/threats/rules/protocol.d.mts +49 -0
  73. package/lib/threats/rules/protocol.d.mts.map +1 -0
  74. package/lib/threats/rules/protocol.mjs +175 -0
  75. package/lib/threats/rules/protocol.mjs.map +1 -0
  76. package/lib/threats/rules/system.d.mts +19 -0
  77. package/lib/threats/rules/system.d.mts.map +1 -0
  78. package/lib/threats/rules/system.mjs +455 -0
  79. package/lib/threats/rules/system.mjs.map +1 -0
  80. package/lib/threats/rules/web.d.mts +26 -0
  81. package/lib/threats/rules/web.d.mts.map +1 -0
  82. package/lib/threats/rules/web.mjs +412 -0
  83. package/lib/threats/rules/web.mjs.map +1 -0
  84. package/lib/threats/scoring.d.mts +59 -0
  85. package/lib/threats/scoring.d.mts.map +1 -0
  86. package/lib/threats/scoring.mjs +111 -0
  87. package/lib/threats/scoring.mjs.map +1 -0
  88. package/lib/threats/types.d.mts +245 -0
  89. package/lib/threats/types.d.mts.map +1 -0
  90. package/lib/threats/types.mjs +52 -0
  91. package/lib/threats/types.mjs.map +1 -0
  92. package/lib/threats/variants.d.mts +57 -0
  93. package/lib/threats/variants.d.mts.map +1 -0
  94. package/lib/threats/variants.mjs +144 -0
  95. package/lib/threats/variants.mjs.map +1 -0
  96. package/lib/unicode/confusables.d.mts +82 -0
  97. package/lib/unicode/confusables.d.mts.map +1 -0
  98. package/lib/unicode/confusables.mjs +954 -0
  99. package/lib/unicode/confusables.mjs.map +1 -0
  100. package/lib/unicode/index.d.mts +126 -0
  101. package/lib/unicode/index.d.mts.map +1 -0
  102. package/lib/unicode/index.mjs +288 -0
  103. package/lib/unicode/index.mjs.map +1 -0
  104. package/lib/validators.d.mts +341 -164
  105. package/lib/validators.d.mts.map +1 -1
  106. package/lib/validators.mjs +519 -338
  107. package/lib/validators.mjs.map +1 -1
  108. package/package.json +35 -8
@@ -1,3 +1,6 @@
1
+ import { MAX_SCAN_LENGTH, scanForThreats } from "./threats/engine.mjs";
2
+ import { foldConfusables } from "./unicode/confusables.mjs";
3
+ import { analyzeIdentifier, containsBidiControls } from "./unicode/index.mjs";
1
4
  import { assertNever } from "@resq-systems/types";
2
5
  //#region src/validators.ts
3
6
  /**
@@ -16,333 +19,198 @@ import { assertNever } from "@resq-systems/types";
16
19
  * limitations under the License.
17
20
  */
18
21
  /**
19
- * XSS attack patterns
20
- * Detects script injection, event handlers, and dangerous URIs
21
- */
22
- const XSS_PATTERNS = [
23
- /<script\b/gi,
24
- /\bon\w+\s*=/gi,
25
- /javascript\s*:/gi,
26
- /data\s*:\s*text\/html/gi,
27
- /data\s*:\s*application\/javascript/gi,
28
- /expression\s*\(/gi,
29
- /vbscript\s*:/gi,
30
- /<iframe\b/gi,
31
- /<object\b/gi,
32
- /<embed\b/gi,
33
- /<style\b/gi,
34
- /document\s*\.\s*(cookie|domain|write|location)/gi,
35
- /window\s*\.\s*(location|open|eval)/gi,
36
- /\beval\s*\(/gi,
37
- /\bnew\s+Function\s*\(/gi,
38
- /\.innerHTML\s*=/gi,
39
- /__proto__/gi,
40
- /constructor\s*\[/gi
41
- ];
42
- /**
43
- * SQL injection patterns
44
- * Detects common SQL attack vectors
45
- */
46
- const SQL_INJECTION_PATTERNS = [
47
- /\bUNION\s+(ALL\s+)?SELECT\b/gi,
48
- /\bDROP\s+(TABLE|DATABASE|INDEX|VIEW)\b/gi,
49
- /\bDELETE\s+FROM\b/gi,
50
- /\bTRUNCATE\s+TABLE\b/gi,
51
- /--\s*$/gm,
52
- /\/\*[\s\S]*?\*\//g,
53
- /'\s*OR\s+'[\d\w]+'\s*=\s*'[\d\w]+/gi,
54
- /'\s*OR\s+\d+\s*=\s*\d+/gi,
55
- /"\s*OR\s+"[\d\w]+"\s*=\s*"[\d\w]+/gi,
56
- /1\s*=\s*1/g,
57
- /;\s*(SELECT|INSERT|UPDATE|DELETE|DROP|EXEC|UNION)/gi,
58
- /SLEEP\s*\(\s*\d+\s*\)/gi,
59
- /WAITFOR\s+DELAY/gi,
60
- /BENCHMARK\s*\(/gi,
61
- /INFORMATION_SCHEMA/gi,
62
- /0x[0-9a-f]+/gi,
63
- /\bEXEC(UTE)?\s*\(/gi,
64
- /\bxp_\w+/gi
65
- ];
66
- /**
67
- * NoSQL injection patterns
68
- * Detects MongoDB and other NoSQL attack vectors
69
- */
70
- const NOSQL_INJECTION_PATTERNS = [
71
- /\$(?:gt|gte|lt|lte|ne|eq|in|nin|and|or|not|nor|exists|type|mod|regex|text|where|all|elemMatch|size|slice|expr|jsonSchema|meta)\b/gi,
72
- /\$where\s*:/gi,
73
- /\$function\s*:/gi,
74
- /\{\s*\$[a-z]+\s*:/gi,
75
- /\[\s*\$[a-z]+\s*\]/gi
76
- ];
77
- /**
78
- * Path traversal patterns
79
- * Detects directory traversal attacks
80
- */
81
- const PATH_TRAVERSAL_PATTERNS = [
82
- /\.\.[/\\]/g,
83
- /%2e%2e[%2f%5c]/gi,
84
- /%252e%252e%252f/gi,
85
- /\.\.%2f/gi,
86
- /\.\.%5c/gi,
87
- /%00/g,
88
- /\/etc\/passwd/gi,
89
- /\/etc\/shadow/gi,
90
- /\/proc\/self/gi,
91
- /C:\\Windows/gi,
92
- /C:\\System32/gi
93
- ];
94
- /**
95
- * Homoglyph patterns
96
- * Detects Unicode characters that look like ASCII but aren't
97
- * Used in phishing and IDN homograph attacks
98
- */
99
- const HOMOGLYPH_MAP = {
100
- a: [
101
- "а",
102
- "ɑ",
103
- "α",
104
- "а"
105
- ],
106
- c: [
107
- "с",
108
- "ϲ",
109
- "ⅽ"
110
- ],
111
- e: [
112
- "е",
113
- "ε",
114
- "ė"
115
- ],
116
- o: [
117
- "о",
118
- "ο",
119
- "ᴏ",
120
- "०"
121
- ],
122
- p: ["р", "ρ"],
123
- s: ["ѕ", "ꜱ"],
124
- x: ["х", "χ"],
125
- y: ["у", "γ"],
126
- B: ["В", "Β"],
127
- H: ["Н", "Η"],
128
- K: ["К", "Κ"],
129
- M: ["М", "Μ"],
130
- P: ["Р", "Ρ"],
131
- T: ["Т", "Τ"]
132
- };
22
+ * @fileoverview Field-level validators, output encoders, and the compatibility
23
+ * surface over the context-aware rule engine in `@resq-systems/security/threats`.
24
+ *
25
+ * The pattern arrays that used to live here are gone. Every detector below delegates
26
+ * to {@link scanForThreats} with the context matching its sink, which is what stops a
27
+ * detector meant for file paths from rejecting a biography. New code should call
28
+ * `scanForThreats` directly and declare its own contexts; the `contains*` helpers
29
+ * remain for callers written against the previous API.
30
+ *
31
+ * Detection is defense-in-depth. Output encoding, parameterized queries, path
32
+ * containment, and argv-array process spawning are the controls.
33
+ *
34
+ * @module @resq-systems/security/validators
35
+ */
36
+ /** Translate the legacy toggles into engine contexts. */
37
+ function contextsFor(config) {
38
+ const contexts = ["general_text"];
39
+ if (config.checkXSS !== false) contexts.push("html");
40
+ if (config.checkSQLInjection !== false) contexts.push("sql");
41
+ if (config.checkNoSQLInjection !== false) contexts.push("nosql");
42
+ if (config.checkCommandInjection === true) contexts.push("shell");
43
+ if (config.checkPathTraversal !== false) contexts.push("filesystem");
44
+ return contexts;
45
+ }
46
+ /**
47
+ * Run one context's rules and keep at most one finding, preserving the
48
+ * one-finding-per-detector contract the `contains*` helpers have always had.
49
+ */
50
+ function firstFindingOfType(input, contexts, type) {
51
+ const finding = scanForThreats(input, { contexts }).findings.find((candidate) => candidate.type === type);
52
+ return finding ? [finding] : [];
53
+ }
133
54
  /**
134
- * Detect XSS-style payloads (script tags, event handlers, dangerous
135
- * URI schemes, prototype pollution, …) in a UTF-8 input.
55
+ * Detect XSS payloads — script tags, inline event handlers, dangerous URI schemes,
56
+ * markup sinks — in a value bound for an HTML context.
136
57
  *
137
- * Inputs longer than 100 000 characters are truncated before scanning
138
- * to bound regex evaluation cost and prevent ReDoS on crafted
139
- * payloads. Returns at most one finding — the regex catalog is
140
- * exhaustive enough that the first hit is sufficient for a
141
- * reject-or-sanitize decision.
58
+ * @param input - String to scan. Truncated at 100 000 characters.
59
+ * @returns Empty array, or a single finding of type `"xss"`.
142
60
  *
143
- * @param input - String to scan.
144
- * @returns Empty array when nothing matches, or a single
145
- * {@link ThreatFinding} of type `"xss"`.
61
+ * @remarks
62
+ * Prototype-pollution patterns (`__proto__`, `constructor[`) no longer surface here.
63
+ * They are a distinct weakness class with distinct controls and now report as
64
+ * `prototype_pollution` — see {@link containsPrototypePollution}.
146
65
  *
147
66
  * @example
148
67
  * ```ts
149
68
  * containsXSSPatterns(`<img src=x onerror="alert(1)">`);
150
- * // → [{ type: "xss", description: "...", matchedPattern: "onerror=" }]
69
+ * // → [{ ruleId: "XSS-EVENT-HANDLER-001", type: "xss", severity: "high", … }]
151
70
  * ```
152
71
  */
153
72
  function containsXSSPatterns(input) {
154
- const findings = [];
155
- const bounded = input.length > 1e5 ? input.slice(0, 1e5) : input;
156
- for (const pattern of XSS_PATTERNS) {
157
- const match = bounded.match(pattern);
158
- if (match) {
159
- findings.push({
160
- type: "xss",
161
- description: "Potential cross-site scripting (XSS) detected",
162
- matchedPattern: match[0].slice(0, 50)
163
- });
164
- break;
165
- }
166
- }
167
- return findings;
73
+ return firstFindingOfType(input, ["html"], "xss");
168
74
  }
169
75
  /**
170
- * Detect SQL-injection patterns (UNION SELECT, DROP TABLE,
171
- * comment-based bypasses, always-true tautologies, stacked queries)
172
- * in input.
76
+ * Detect prototype-pollution payloads — `__proto__`, `constructor.prototype`, and the
77
+ * nested-object forms that arrive through a JSON body or query-string expansion.
78
+ *
79
+ * **Not the control.** Reject unknown keys with schema validation, build lookup
80
+ * objects with `Object.create(null)`, and use a merge that skips `__proto__`,
81
+ * `constructor`, and `prototype`.
173
82
  *
174
- * **Not a replacement for parameterised queries.** Use this as a
175
- * defense-in-depth signal in addition to a properly bound prepared
176
- * statement, never as the only barrier.
83
+ * @param input - String to scan.
84
+ * @returns Empty array, or a single finding of type `"prototype_pollution"`.
85
+ */
86
+ function containsPrototypePollution(input) {
87
+ return firstFindingOfType(input, ["object_merge"], "prototype_pollution");
88
+ }
89
+ /**
90
+ * Detect SQL-injection patterns in a value bound for a query.
91
+ *
92
+ * **Not a replacement for parameterized queries.** A bound parameter is safe whatever
93
+ * keywords it contains; an interpolated one is unsafe however many signatures it
94
+ * dodges. Use this for telemetry alongside binding, never instead of it.
177
95
  *
178
96
  * @param input - String to scan. Truncated at 100 000 characters.
179
97
  * @returns Empty array, or one finding of type `"sql_injection"`.
180
98
  */
181
99
  function containsSQLInjection(input) {
182
- const findings = [];
183
- const bounded = input.length > 1e5 ? input.slice(0, 1e5) : input;
184
- for (const pattern of SQL_INJECTION_PATTERNS) {
185
- const match = bounded.match(pattern);
186
- if (match) {
187
- findings.push({
188
- type: "sql_injection",
189
- description: "Potential SQL injection detected",
190
- matchedPattern: match[0].slice(0, 50)
191
- });
192
- break;
193
- }
194
- }
195
- return findings;
100
+ return firstFindingOfType(input, ["sql"], "sql_injection");
196
101
  }
197
102
  /**
198
- * Detect NoSQL-injection patterns — Mongo-style operator injection
199
- * (`$where`, `$ne`, `$regex`), JavaScript-in-query payloads, and
200
- * structural manipulators that can bypass auth filters in document
201
- * stores.
103
+ * Detect NoSQL operator injection — `$where`, `$ne`, `$regex`, and the object and
104
+ * array forms that bypass authentication filters in document stores.
202
105
  *
203
106
  * @param input - String to scan.
204
107
  * @returns Empty array, or one finding of type `"nosql_injection"`.
205
108
  */
206
109
  function containsNoSQLInjection(input) {
207
- const findings = [];
208
- for (const pattern of NOSQL_INJECTION_PATTERNS) {
209
- const match = input.match(pattern);
210
- if (match) {
211
- findings.push({
212
- type: "nosql_injection",
213
- description: "Potential NoSQL injection detected",
214
- matchedPattern: match[0].slice(0, 50)
215
- });
216
- break;
217
- }
218
- }
219
- return findings;
110
+ return firstFindingOfType(input, ["nosql"], "nosql_injection");
220
111
  }
221
112
  /**
222
- * Detect shell command-injection patterns: command substitution
223
- * (`$(...)`, backticks), chained dangerous commands (`; rm`, `; curl`,
224
- * …) and shell-piped exec (`| sh`, `| bash`).
113
+ * Detect shell command-injection patterns — command substitution, chained commands,
114
+ * pipes into an interpreter.
225
115
  *
226
- * **Off by default in {@link detectThreatPatterns}** — these patterns
227
- * occasionally fire on legitimate user content. Enable explicitly
228
- * (`checkCommandInjection: true`) only when input flows into a child
229
- * process or shell.
116
+ * **Off by default in {@link detectThreatPatterns}**, because these patterns fire on
117
+ * ordinary prose. Enable only when the value reaches a child process, and prefer
118
+ * spawning with an argv array and `shell: false`, which makes the category moot.
230
119
  *
231
120
  * @param input - String to scan. Truncated at 100 000 characters.
232
121
  * @returns Empty array, or one finding of type `"command_injection"`.
233
122
  */
234
123
  function containsCommandInjection(input) {
235
- const findings = [];
236
- const dangerousPatterns = [
237
- /\$\([^)]{1,200}\)/g,
238
- /`[^`]{1,200}`/g,
239
- /;\s*(rm|del|cat|wget|curl|nc)\b/gi,
240
- /\|\s*(sh|bash|cmd)\b/gi
241
- ];
242
- const bounded = input.length > 1e5 ? input.slice(0, 1e5) : input;
243
- for (const pattern of dangerousPatterns) {
244
- const match = bounded.match(pattern);
245
- if (match) {
246
- findings.push({
247
- type: "command_injection",
248
- description: "Potential command injection detected",
249
- matchedPattern: match[0].slice(0, 50)
250
- });
251
- break;
252
- }
253
- }
254
- return findings;
124
+ return firstFindingOfType(input, ["shell"], "command_injection");
255
125
  }
256
126
  /**
257
- * Detect path-traversal payloads — `../`, encoded dots, raw absolute
258
- * paths trying to escape a base directory. Pair with `path.resolve()`
259
- * + a `startsWith()` containment check on the canonicalised path
260
- * before reading or writing the file.
127
+ * Detect path-traversal payloads — `../`, its percent-encoded and double-encoded
128
+ * forms, NUL truncation, and references to sensitive system paths.
129
+ *
130
+ * **Not the control.** Use `resolveContainedPath` from
131
+ * `@resq-systems/security/paths`, which resolves the candidate against a base
132
+ * directory and verifies containment — a check that also catches absolute paths and
133
+ * separator tricks no signature enumerates.
261
134
  *
262
135
  * @param input - String to scan.
263
136
  * @returns Empty array, or one finding of type `"path_traversal"`.
264
137
  */
265
138
  function containsPathTraversal(input) {
266
- const findings = [];
267
- for (const pattern of PATH_TRAVERSAL_PATTERNS) {
268
- const match = input.match(pattern);
269
- if (match) {
270
- findings.push({
271
- type: "path_traversal",
272
- description: "Potential path traversal attack detected",
273
- matchedPattern: match[0].slice(0, 50)
274
- });
275
- break;
276
- }
277
- }
278
- return findings;
139
+ return firstFindingOfType(input, ["filesystem"], "path_traversal");
279
140
  }
280
141
  /**
281
- * Detect lookalike Unicode characters (Cyrillic / Greek glyphs that
282
- * render identically to common ASCII letters). The classic phishing
283
- * trick is `paypaӏ.com` (`ӏ` instead of `l`); this detector catches
284
- * the building blocks.
142
+ * Base metadata for the synthetic finding {@link containsHomoglyphs} produces, shaped
143
+ * like a catalog entry so downstream consumers see one consistent record.
144
+ */
145
+ const MIXED_SCRIPT_FINDING = {
146
+ ruleId: "UNICODE-MIXED-SCRIPT-001",
147
+ type: "homoglyph",
148
+ severity: "high",
149
+ confidence: "medium",
150
+ description: "Identifier mixes scripts in a combination used for visual spoofing",
151
+ cwe: 1007,
152
+ primaryControl: "Compare UTS #39 skeletons at registration time and enforce an identifier restriction level",
153
+ variant: "nfc"
154
+ };
155
+ /** Overrides applied when the identifier carries a bidirectional control. */
156
+ const BIDI_FINDING_OVERRIDE = {
157
+ ruleId: "UNICODE-BIDI-OVERRIDE-001",
158
+ severity: "critical",
159
+ confidence: "high",
160
+ description: "Bidirectional override character in an identifier",
161
+ cwe: 451
162
+ };
163
+ /**
164
+ * Detect visually confusable characters in a **protected identifier**.
285
165
  *
286
- * Use {@link normalizeUnicode} to *replace* homoglyphs with their
287
- * ASCII equivalents — this function only flags their presence.
166
+ * Backed by UTS #39 script analysis rather than a hand-written lookalike table, so it
167
+ * reports the actual signal — a Latin/Cyrillic mix in `pаypal` — instead of flagging
168
+ * every non-ASCII character. Single-script values are not confusable with anything, so
169
+ * `Ольга Иванова` and `東京タワー` pass where the previous implementation rejected both.
288
170
  *
289
- * @param input - String to scan.
290
- * @returns Empty array, or one finding of type `"homoglyph"` (the
291
- * first matched lookalike).
171
+ * Scope this to usernames, domains, org names, and package names. Do **not** run it on
172
+ * prose or on people's names — see {@link validatePersonName}.
173
+ *
174
+ * @param input - Identifier to scan.
175
+ * @returns Empty array, or a single finding of type `"homoglyph"`.
292
176
  */
293
177
  function containsHomoglyphs(input) {
294
- const findings = [];
295
- for (const [, homoglyphs] of Object.entries(HOMOGLYPH_MAP)) for (const homoglyph of homoglyphs) if (input.includes(homoglyph)) {
296
- findings.push({
297
- type: "homoglyph",
298
- description: "Suspicious lookalike Unicode character detected",
299
- matchedPattern: homoglyph
300
- });
301
- return findings;
302
- }
303
- return findings;
178
+ if (!input || typeof input !== "string") return [];
179
+ const analysis = analyzeIdentifier(input.length > 1e5 ? input.slice(0, MAX_SCAN_LENGTH) : input);
180
+ if (!analysis.isMixedScript && !analysis.hasBidiControls) return [];
181
+ return [{
182
+ ...MIXED_SCRIPT_FINDING,
183
+ ...analysis.hasBidiControls ? BIDI_FINDING_OVERRIDE : {},
184
+ matchedPattern: analysis.scripts.join("+").slice(0, 50)
185
+ }];
304
186
  }
305
- const DEFAULT_CONFIG = {
306
- checkXSS: true,
307
- checkSQLInjection: true,
308
- checkNoSQLInjection: true,
309
- checkCommandInjection: false,
310
- checkPathTraversal: true,
311
- checkHomoglyphs: true
312
- };
313
187
  /**
314
- * Run every enabled detector against `input` and aggregate findings.
188
+ * Run the enabled detectors against `input` and aggregate findings.
315
189
  *
316
- * Returns early-but-not-immediately: each individual detector still
317
- * runs to completion, but each detector returns at most one finding,
318
- * so the aggregate threats array is small (≤ 6 entries).
190
+ * @deprecated Prefer {@link scanForThreats}, which takes explicit contexts and returns
191
+ * a score and verdict rather than one boolean. This wrapper maps the legacy toggles
192
+ * onto contexts and keeps the one-finding-per-category shape.
319
193
  *
320
- * Non-string inputs (`null`, `undefined`, numbers, …) are treated as
321
- * safe — wrap caller-side validation around this if you want to
322
- * reject non-strings.
194
+ * Non-string input (`null`, `undefined`, a number) is reported safe — wrap your own
195
+ * type validation around this if you need to reject those.
323
196
  *
324
197
  * @param input - The candidate string.
325
- * @param config - Detector toggles. Defaults turn on everything
326
- * except command-injection.
198
+ * @param config - Detector toggles. Everything except command injection defaults on.
327
199
  * @returns `{ isSafe, threats }`.
328
- *
329
- * @example
330
- * ```ts
331
- * const result = detectThreatPatterns(req.body.query);
332
- * if (!result.isSafe) return new Response(getThreatErrorMessage(result), { status: 400 });
333
- * ```
334
200
  */
335
- function detectThreatPatterns(input, config = DEFAULT_CONFIG) {
201
+ function detectThreatPatterns(input, config = {}) {
336
202
  if (!input || typeof input !== "string") return {
337
203
  isSafe: true,
338
204
  threats: []
339
205
  };
206
+ const result = scanForThreats(input, { contexts: contextsFor(config) });
340
207
  const threats = [];
341
- if (config.checkXSS !== false) threats.push(...containsXSSPatterns(input));
342
- if (config.checkSQLInjection !== false) threats.push(...containsSQLInjection(input));
343
- if (config.checkNoSQLInjection !== false) threats.push(...containsNoSQLInjection(input));
344
- if (config.checkCommandInjection) threats.push(...containsCommandInjection(input));
345
- if (config.checkPathTraversal !== false) threats.push(...containsPathTraversal(input));
208
+ const seen = /* @__PURE__ */ new Set();
209
+ for (const finding of result.findings) {
210
+ if (seen.has(finding.type)) continue;
211
+ seen.add(finding.type);
212
+ threats.push(finding);
213
+ }
346
214
  if (config.checkHomoglyphs !== false) threats.push(...containsHomoglyphs(input));
347
215
  return {
348
216
  isSafe: threats.length === 0,
@@ -350,8 +218,7 @@ function detectThreatPatterns(input, config = DEFAULT_CONFIG) {
350
218
  };
351
219
  }
352
220
  /**
353
- * Boolean shortcut over {@link detectThreatPatterns} — discards the
354
- * findings list when you only need a yes/no decision.
221
+ * Boolean shortcut over {@link detectThreatPatterns}.
355
222
  *
356
223
  * @param input - String to test.
357
224
  * @param config - Optional detector toggles.
@@ -361,111 +228,409 @@ function isSafeInput(input, config) {
361
228
  return detectThreatPatterns(input, config).isSafe;
362
229
  }
363
230
  /**
364
- * HTML-entity escape `&`, `<`, `>`, `"`, `'`, and `/` for safe
365
- * insertion into HTML text and attribute contexts.
231
+ * HTML-entity-escape a value being inserted as **element text**.
366
232
  *
367
- * **Limited scope.** This is appropriate for plain text destined for
368
- * `textContent` or attribute values, not for unfiltered HTML
369
- * rendering. For rich-text use a vetted sanitizer (DOMPurify on the
370
- * client, sanitize-html or similar on the server).
233
+ * Escapes `&`, `<`, `>`, `"`, `'`, and `/`, which covers text nodes and fully quoted
234
+ * attribute values.
371
235
  *
372
- * Returns `""` for non-string or empty input.
236
+ * **Output encoding is context-dependent.** HTML text, quoted attributes, unquoted
237
+ * attributes, URLs, JavaScript string literals, and CSS each have different rules, and
238
+ * no single function is correct for all of them. This one is correct for text; use
239
+ * {@link escapeHtmlAttribute} for attribute values, `sanitizeUrl` for URLs, and
240
+ * `sanitizeHtml` (DOMPurify) when the value is meant to *be* markup.
373
241
  *
374
- * @param input - Untrusted string.
375
- * @returns Entity-escaped output safe to interpolate into HTML.
242
+ * There is deliberately no CSS-context escaper here, and no general JavaScript-string
243
+ * escaper — hand-rolled versions of those are reliably wrong, and the fix is to stop
244
+ * interpolating untrusted values into style and script *source*. Embedding untrusted
245
+ * *data* in a script element is the one tractable case, because `JSON.stringify` fixes
246
+ * the string boundaries first; {@link encodeJsonForScript} covers that and nothing else.
247
+ *
248
+ * @param input - Untrusted string. Non-string or empty input yields `""`.
249
+ * @returns Entity-escaped output safe to interpolate into HTML text.
250
+ *
251
+ * @example
252
+ * ```ts
253
+ * escapeHtmlText('<script>alert("xss")<\/script>');
254
+ * // "&lt;script&gt;alert(&quot;xss&quot;)&lt;&#x2F;script&gt;"
255
+ * ```
376
256
  */
377
- function sanitizeForDisplay(input) {
257
+ function escapeHtmlText(input) {
378
258
  if (!input || typeof input !== "string") return "";
379
259
  return input.replace(/&/g, "&amp;").replace(/</g, "&lt;").replace(/>/g, "&gt;").replace(/"/g, "&quot;").replace(/'/g, "&#x27;").replace(/\//g, "&#x2F;");
380
260
  }
381
261
  /**
382
- * Canonicalise a string for safe equality checks against ASCII.
262
+ * Control characters escaped in attribute position.
263
+ *
264
+ * The C0 and C1 ranges plus the two Unicode line terminators. The set is the point:
265
+ * HTML's unquoted-attribute state ends at space, tab, LF, FF or CR, and this used to
266
+ * escape tab, LF and CR but not **form feed**. It also escaped CR, which the input
267
+ * stream preprocessor normalises to LF before the tokenizer runs — so three of the four
268
+ * real terminators were covered, plus the one that cannot matter.
269
+ *
270
+ * @see https://html.spec.whatwg.org/multipage/parsing.html
271
+ */
272
+ const ATTRIBUTE_CONTROL_CHARS = /[\u0000-\u001f\u007f-\u009f\u2028\u2029]/g;
273
+ /**
274
+ * HTML-entity-escape a value being inserted as an **attribute value**.
383
275
  *
384
- * Two-pass:
385
- * 1. Normalize to NFC (composed form) so combining-character
386
- * sequences don't compare differently from their pre-composed
387
- * counterparts.
388
- * 2. Replace known homoglyphs (Cyrillic `А`, Greek `Ε`, …) with their
389
- * ASCII equivalents (`A`, `E`, …).
276
+ * Everything {@link escapeHtmlText} escapes, plus backtick, equals, and whitespace —
277
+ * the characters that let a payload break out of an *unquoted* attribute. That case is
278
+ * precisely what generic "escape for display" helpers get wrong.
390
279
  *
391
- * Use before storing user-controlled identifiers (usernames, domain
392
- * names) and before comparing them to a denylist or to each other.
280
+ * The ceiling on the unquoted case is injection of a valueless boolean attribute —
281
+ * `autofocus`, `disabled`, `formnovalidate` — not script execution: an injected
282
+ * `onmouseover=…` arrives with its `=` already escaped, so it lands as an attribute
283
+ * whose *name* is the escaped text, with no handler bound.
393
284
  *
394
- * Returns `""` for non-string or empty input.
285
+ * Quote your attributes anyway. This makes an unquoted attribute survivable; it does
286
+ * not make it correct.
395
287
  *
396
- * @param input - Raw string from an untrusted source.
397
- * @returns ASCII-normalized, NFC-composed string.
288
+ * @param input - Untrusted string. Non-string or empty input yields `""`.
289
+ * @returns Output safe to interpolate into a quoted or unquoted attribute value.
398
290
  */
399
- function normalizeUnicode(input) {
291
+ function escapeHtmlAttribute(input) {
400
292
  if (!input || typeof input !== "string") return "";
401
- let normalized = input.normalize("NFC");
402
- for (const [ascii, homoglyphs] of Object.entries(HOMOGLYPH_MAP)) for (const homoglyph of homoglyphs) normalized = normalized.replace(new RegExp(homoglyph, "g"), ascii);
403
- return normalized;
293
+ return escapeHtmlText(input).replace(/`/g, "&#x60;").replace(/=/g, "&#x3D;").replace(/ /g, "&#x20;").replace(ATTRIBUTE_CONTROL_CHARS, (character) => {
294
+ return `&#x${(character.codePointAt(0) ?? 0).toString(16).toUpperCase().padStart(2, "0")};`;
295
+ });
296
+ }
297
+ /**
298
+ * HTML-entity-escape a value for display.
299
+ *
300
+ * @deprecated Renamed to {@link escapeHtmlText}, which says what it actually does. The
301
+ * old name suggested a general-purpose "make this safe to display" operation, and
302
+ * callers reasonably read it as attribute-safe — which entity escaping alone is not,
303
+ * for *unquoted* attributes. Behaviour is unchanged; only the name is.
304
+ *
305
+ * @param input - Untrusted string.
306
+ * @returns Entity-escaped output.
307
+ */
308
+ function sanitizeForDisplay(input) {
309
+ return escapeHtmlText(input);
404
310
  }
311
+ /** Cap on the input a log value is read from, before escaping expands it. */
312
+ const DEFAULT_LOG_VALUE_LENGTH = 2048;
313
+ /**
314
+ * Characters that must not reach a log sink as themselves.
315
+ *
316
+ * C0 and C1, the zero-width and bidirectional formatting ranges, and the byte-order
317
+ * mark. ESC lives inside C0, which is why no separate ANSI sequence matching is needed:
318
+ * escaping the introducer alone neutralises every terminal sequence *losslessly*,
319
+ * whereas deleting whole sequences would discard the payload a reader is investigating.
320
+ * The bidi range matters for the same reason `UNICODE-BIDI-OVERRIDE-001` exists — a
321
+ * right-to-left override reorders how a log line renders without changing its bytes.
322
+ */
323
+ const LOG_UNSAFE_CHARS = /[\u0000-\u001f\u007f-\u009f\u200b-\u200f\u2028-\u202e\u2060-\u2064\u2066-\u2069\ufeff]/g;
324
+ /** Readable forms for the three characters a reader expects to recognise. */
325
+ const LOG_SHORTHAND = {
326
+ " ": "\\t",
327
+ "\n": "\\n",
328
+ "\r": "\\r"
329
+ };
405
330
  /**
406
- * Generic user-facing fallback message. Render this verbatim when a
407
- * detector fires but you don't want to expose which one. Prefer
408
- * {@link getThreatErrorMessage} for category-specific messages.
331
+ * Escape a value for inclusion in a log record.
332
+ *
333
+ * This is the control named by the log-injection rules. A log line is a *sink*: a value
334
+ * carrying a newline forges an entry (CWE-117), one carrying a terminal escape rewrites
335
+ * what an operator sees, and one carrying a bidirectional override reorders the line
336
+ * without altering a byte of it.
337
+ *
338
+ * Escaping rather than stripping is deliberate. The record is evidence, so the encoded
339
+ * form is reversible and nothing is silently discarded — contrast `stripAnsi`, which
340
+ * deletes. Structured logging is still the better answer, because it removes the
341
+ * ambiguity this function can only make visible; use both.
342
+ *
343
+ * @param value - Untrusted field value. Non-string or empty input yields `""`.
344
+ * @param options - Optional bounds.
345
+ * @param options.maxLength - Characters read from `value`. Defaults to 2048. Truncation
346
+ * is announced in the output rather than applied silently, and the returned string may
347
+ * exceed this length, because escaping expands.
348
+ * @returns A single-line, control-free rendering of `value`.
349
+ *
350
+ * @example
351
+ * ```ts
352
+ * encodeLogValue("alice\nINFO user promoted to admin");
353
+ * // "alice\\nINFO user promoted to admin" — one line, no forged entry
354
+ * ```
355
+ */
356
+ function encodeLogValue(value, options = {}) {
357
+ if (!value || typeof value !== "string") return "";
358
+ const { maxLength = DEFAULT_LOG_VALUE_LENGTH } = options;
359
+ const limit = Number.isInteger(maxLength) && maxLength > 0 ? maxLength : DEFAULT_LOG_VALUE_LENGTH;
360
+ const dropped = value.length - limit;
361
+ const encoded = (dropped > 0 ? value.slice(0, limit) : value).replace(LOG_UNSAFE_CHARS, (character) => {
362
+ const shorthand = LOG_SHORTHAND[character];
363
+ if (shorthand !== void 0) return shorthand;
364
+ return `\\u${(character.codePointAt(0) ?? 0).toString(16).padStart(4, "0")}`;
365
+ });
366
+ return dropped > 0 ? `${encoded}[truncated ${dropped} chars]` : encoded;
367
+ }
368
+ /**
369
+ * A leading formula trigger, tolerating the whitespace and quotes a reader strips first.
370
+ *
371
+ * Mirrors `CSV-FORMULA-LEAD-001`, deliberately: the rule sees through leading quotes and
372
+ * spaces because spreadsheet importers do, so an encoder that only looked at index 0
373
+ * would leave ` =cmd|'/c calc'!A1` live.
374
+ */
375
+ const CSV_FORMULA_LEAD = /^[\s'"]{0,8}[=+\-@\t\r]/;
376
+ /** Fields containing any of these must be quoted per RFC 4180 sections 2.6 and 2.7. */
377
+ const CSV_QUOTE_REQUIRED = /["\r\n]/;
378
+ /**
379
+ * Escape one cell for CSV export.
380
+ *
381
+ * This is the control named by the formula-injection rules. A CSV file is not inert: a
382
+ * cell beginning `=`, `+`, `-`, `@`, tab or CR is evaluated as a formula by Excel,
383
+ * Sheets and LibreOffice when the recipient opens it, so the payload executes on *their*
384
+ * machine, outside the exporting application entirely (CWE-1236).
385
+ *
386
+ * Two separate jobs, in order: neutralise the formula trigger with a leading apostrophe,
387
+ * then apply RFC 4180 quoting so the field cannot break the row.
388
+ *
389
+ * **Only strings are prefixed.** A `number` or `boolean` came from the application's own
390
+ * types and cannot carry a formula, so `-1234` exports as a negative number while
391
+ * `"-1234"` exports as text. Pass numeric columns as numbers, or every negative value in
392
+ * the sheet becomes a string.
393
+ *
394
+ * Three things worth knowing before relying on it:
395
+ * - The leading apostrophe is an Excel convention, **not** an RFC 4180 construct. Readers
396
+ * that do not implement it surface it as a literal character in the data.
397
+ * - NUL is removed rather than escaped, so it does not round-trip.
398
+ * - Scanning the output with `scanForThreats` still reports a finding, by design:
399
+ * `CSV-FORMULA-LEAD-001` sees through the apostrophe and `CSV-DDE-001` is
400
+ * position-independent. The rules describe the *value*; this function protects the
401
+ * *file*. A clean scan is the wrong acceptance test.
402
+ *
403
+ * @param value - Cell value. `null` and `undefined` become `""`.
404
+ * @param options - Optional dialect settings.
405
+ * @param options.delimiter - Field separator the row will be joined with. Defaults to `","`.
406
+ * @returns The escaped field, ready to join into a row.
407
+ *
408
+ * @example
409
+ * ```ts
410
+ * escapeCsvField("=WEBSERVICE(\"https://evil.example\")");
411
+ * // quoted, and inert on open
412
+ * escapeCsvField(-1234); // "-1234" — a number, not a formula
413
+ * ```
414
+ */
415
+ function escapeCsvField(value, options = {}) {
416
+ if (value === null || value === void 0) return "";
417
+ const delimiter = options.delimiter ?? ",";
418
+ const isUntrustedText = typeof value === "string";
419
+ const cleaned = (isUntrustedText ? value : String(value)).replace(/\u0000/g, "");
420
+ const neutralised = isUntrustedText && CSV_FORMULA_LEAD.test(cleaned) ? `'${cleaned}` : cleaned;
421
+ return CSV_QUOTE_REQUIRED.test(neutralised) || neutralised.includes(delimiter) ? `"${neutralised.replaceAll("\"", "\"\"")}"` : neutralised;
422
+ }
423
+ /**
424
+ * Escape and join one row for CSV export.
425
+ *
426
+ * @param values - Cell values, in column order.
427
+ * @param options - Optional dialect settings.
428
+ * @param options.delimiter - Field separator. Defaults to `","`.
429
+ * @returns The joined row, without a line terminator.
430
+ *
431
+ * @example
432
+ * ```ts
433
+ * toCsvRow(["Ada Lovelace", "=1+1", 42]);
434
+ * ```
435
+ */
436
+ function toCsvRow(values, options = {}) {
437
+ if (!Array.isArray(values)) return "";
438
+ const delimiter = options.delimiter ?? ",";
439
+ return values.map((value) => escapeCsvField(value, { delimiter })).join(delimiter);
440
+ }
441
+ /**
442
+ * The five characters that must not survive into a script element verbatim.
443
+ *
444
+ * None is a JSON structural character, so each can only ever occur inside a string
445
+ * literal, where a unicode escape is legal and semantically identical. That is what makes
446
+ * this transformation safe to apply to `JSON.stringify` output without reparsing it.
447
+ *
448
+ * `<` and `>` close the element; `&` matters when a caller relocates the payload into a
449
+ * context that *is* entity-decoded; U+2028 and U+2029 terminate a line in JavaScript
450
+ * source, which JSON permits raw inside strings.
451
+ */
452
+ const SCRIPT_UNSAFE_JSON = /[<>&\u2028\u2029]/g;
453
+ /** Escapes for {@link SCRIPT_UNSAFE_JSON}, all valid inside a JSON string literal. */
454
+ const SCRIPT_JSON_ESCAPES = {
455
+ "<": "\\u003c",
456
+ ">": "\\u003e",
457
+ "&": "\\u0026",
458
+ "\u2028": "\\u2028",
459
+ "\u2029": "\\u2029"
460
+ };
461
+ /**
462
+ * Serialise a value for embedding inside a `<script>` element.
463
+ *
464
+ * `JSON.stringify` alone is not safe here. Its output may contain `<\/script>`, which
465
+ * closes the element from *inside a string literal* — the HTML tokenizer never looks at
466
+ * JavaScript syntax — so the remainder of the payload becomes markup.
467
+ *
468
+ * **Script element content only.** The output contains unescaped `"`, so it must never be
469
+ * placed in an attribute; use {@link escapeHtmlAttribute} there. It is also not a general
470
+ * JavaScript-string escaper — it is safe precisely because `JSON.stringify` has already
471
+ * decided where the string boundaries are.
472
+ *
473
+ * Using `<script type="application/json">` with `JSON.parse(el.textContent)` does **not**
474
+ * remove the need for this: a raw `<\/script>` in the data closes that element too.
475
+ *
476
+ * @param value - Any JSON-serialisable value.
477
+ * @returns JSON text safe to place between `<script>` tags.
478
+ * @throws {TypeError} If `value` cannot be represented as JSON — `undefined`, a function
479
+ * or a symbol at the top level (for which `JSON.stringify` returns `undefined` rather
480
+ * than a string), a circular structure, or a `BigInt`. Failing loudly is deliberate: a
481
+ * sentinel string would emit a syntax error into the page instead.
482
+ *
483
+ * @example
484
+ * ```ts
485
+ * const json = encodeJsonForScript({ name: userName });
486
+ * const html = "<script>window.__DATA__ = " + json + ";<\/script>";
487
+ * ```
488
+ */
489
+ function encodeJsonForScript(value) {
490
+ let serialised;
491
+ try {
492
+ serialised = JSON.stringify(value);
493
+ } catch (cause) {
494
+ throw new TypeError("encodeJsonForScript: value is not JSON-serialisable", { cause });
495
+ }
496
+ if (typeof serialised !== "string") throw new TypeError(`encodeJsonForScript: ${typeof value} has no JSON representation at the top level`);
497
+ return serialised.replace(SCRIPT_UNSAFE_JSON, (character) => SCRIPT_JSON_ESCAPES[character] ?? character);
498
+ }
499
+ /**
500
+ * Fold non-ASCII lookalike characters onto ASCII and compose to NFC.
501
+ *
502
+ * @deprecated Prefer `getSkeleton` and `analyzeIdentifier` from
503
+ * `@resq-systems/security/unicode`. Rewriting a user's identifier into a different
504
+ * string loses information and only *looks* safe — the durable pattern is to store
505
+ * what they typed, index its skeleton, and compare skeletons for collisions.
506
+ *
507
+ * Now backed by the UTS #39 confusable tables rather than the previous 14-entry map,
508
+ * so coverage is far wider. Combining marks are preserved (`e` + U+0301 still composes
509
+ * to `é`) and ASCII characters are never rewritten.
510
+ *
511
+ * @param input - Raw string from an untrusted source. Non-string input yields `""`.
512
+ * @returns NFC-composed string with non-ASCII confusables folded to ASCII.
513
+ */
514
+ function normalizeUnicode(input) {
515
+ return foldConfusables(input);
516
+ }
517
+ /**
518
+ * Generic user-facing fallback message. Render verbatim when a detector fires and you
519
+ * do not want to reveal which one.
409
520
  */
410
521
  const THREAT_DETECTED_MESSAGE = "Input contains potentially unsafe content";
411
522
  /**
412
- * Boolean refinement helper for use with `zod.string().refine(...)`,
413
- * `effect/Schema.filter(...)`, or any predicate-based validator.
523
+ * Refinement helper for `zod.string().refine(...)`, `effect/Schema.filter(...)`, or
524
+ * any predicate-based validator. Equivalent to {@link isSafeInput} with defaults.
414
525
  *
415
- * Equivalent to `isSafeInput(input)` with default config.
526
+ * @param input - String to test.
527
+ * @returns `true` when no detector fires.
416
528
  */
417
529
  function validateSafeText(input) {
418
530
  return isSafeInput(input);
419
531
  }
420
532
  /**
421
- * Refinement for human name fields. More permissive than
422
- * {@link validateSafeText} — allows international letters,
423
- * combining marks, hyphens, apostrophes, and spaces — but still
424
- * rejects HTML/SQL/NoSQL injection patterns and homoglyph forgeries.
533
+ * Letters, marks, apostrophes, hyphens, periods, spaces, and the two joiners — nothing
534
+ * else.
535
+ *
536
+ * U+200C (ZWNJ) and U+200D (ZWJ) are part of the spelling, not decoration. Persian and
537
+ * Hindi names need them to be written correctly — a ZWNJ is what keeps the two halves
538
+ * of `می‌روم` from joining — so a pattern without them rejects the name its owner
539
+ * actually has. They carry no injection risk here: everything a payload needs (`<`,
540
+ * `(`, `;`, `$`, `=`, digits) stays excluded. Written as escapes, not literals — an
541
+ * invisible character pasted into a character class is unreviewable in a diff.
542
+ */
543
+ const PERSON_NAME_PATTERN = /^[\p{L}\p{M}'’.\-\s\u{200C}\u{200D}]+$/u;
544
+ /** Shortest accepted name. Mononyms and single-letter names exist. */
545
+ const MIN_NAME_LENGTH = 1;
546
+ /** Longest accepted name. */
547
+ const MAX_NAME_LENGTH = 200;
548
+ /**
549
+ * Validate a human name field.
550
+ *
551
+ * The policy is an allowlist of what a name is made of — letters in any script,
552
+ * combining marks, apostrophes, hyphens, periods, spaces — plus a length bound and a
553
+ * bidirectional-control check. Nothing that passes it can carry an injection payload,
554
+ * because `<`, `(`, `;`, `$`, `=`, and every digit are already excluded.
555
+ *
556
+ * It deliberately does **not** run SQL, path-traversal, or confusable detectors. A
557
+ * name is not a query, a path, or a protected identifier, and subjecting one to those
558
+ * checks rejects real people: the previous implementation ran the homoglyph detector
559
+ * here, which failed any name containing а, е, о, р, с, or х — that is, most Russian,
560
+ * Ukrainian, Bulgarian, Serbian, and Greek names.
561
+ *
562
+ * Encode the value at whatever sink it eventually reaches. That is what makes it safe;
563
+ * this function only establishes that it is a name.
564
+ *
565
+ * @param input - Candidate name.
566
+ * @returns `true` when the value is a plausible name.
567
+ *
568
+ * @example
569
+ * ```ts
570
+ * validatePersonName("O'Brien"); // true
571
+ * validatePersonName("José García"); // true
572
+ * validatePersonName("Ольга Иванова"); // true
573
+ * validatePersonName("John123"); // false
574
+ * validatePersonName("<script>x<\/script>"); // false
575
+ * ```
576
+ */
577
+ function validatePersonName(input) {
578
+ if (typeof input !== "string") return false;
579
+ const normalized = input.normalize("NFC");
580
+ if (normalized.length < MIN_NAME_LENGTH || normalized.length > MAX_NAME_LENGTH) return false;
581
+ if (containsBidiControls(normalized)) return false;
582
+ return PERSON_NAME_PATTERN.test(normalized);
583
+ }
584
+ /**
585
+ * Validate a human name field.
425
586
  *
426
- * Suitable for first/last/full-name inputs in registration forms.
587
+ * @deprecated Renamed to {@link validatePersonName}. The old name implied a general
588
+ * "safe name" check and was implemented as one, running injection and homoglyph
589
+ * detectors against people's names. Behaviour now matches
590
+ * {@link validatePersonName}.
427
591
  *
428
- * @returns `true` when the name passes both the threat detectors and
429
- * the name-shape regex.
592
+ * @param input - Candidate name.
593
+ * @returns `true` when the value is a plausible name.
430
594
  */
431
595
  function validateSafeName(input) {
432
- const normalized = input.normalize("NFC");
433
- if (!isSafeInput(normalized, { checkCommandInjection: false })) return false;
434
- return /^[\p{L}\p{M}'\-\s.]+$/u.test(normalized);
596
+ return validatePersonName(input);
435
597
  }
598
+ /** Longest address accepted, per RFC 5321 §4.5.3.1.3. Also bounds regex cost. */
599
+ const MAX_EMAIL_LENGTH = 254;
600
+ /** RFC-shaped address check. Length is bounded before this runs. */
601
+ const EMAIL_PATTERN = /^[a-zA-Z0-9.!#$%&'*+/=?^_`{|}~-]+@[a-zA-Z0-9](?:[a-zA-Z0-9-]{0,61}[a-zA-Z0-9])?(?:\.[a-zA-Z0-9](?:[a-zA-Z0-9-]{0,61}[a-zA-Z0-9])?)*$/;
436
602
  /**
437
- * Refinement for email fields. Combines:
603
+ * Validate an email address.
438
604
  *
439
- * 1. RFC-style format check (length-bounded to ≤ 254 chars to
440
- * prevent ReDoS).
441
- * 2. XSS / SQL / NoSQL / homoglyph detectors — emails are extremely
442
- * constrained and should never legitimately contain HTML or query
443
- * operators.
605
+ * Two checks: an RFC-shaped format match (length-bounded first, so the pattern never
606
+ * sees an unbounded string), and UTS #39 identifier analysis of the **domain**, where
607
+ * a mixed-script host is the IDN homograph attack — `аpple.com` with a Cyrillic `а`
608
+ * resolves somewhere else entirely.
444
609
  *
445
- * @returns `true` when both checks pass.
610
+ * The local part is not confusable-checked: it is not a routable identifier, and
611
+ * flagging it would reject legitimate internationalized mailboxes.
612
+ *
613
+ * @param input - Candidate address.
614
+ * @returns `true` when the format is valid and the domain is not a script mix.
446
615
  */
447
616
  function validateSafeEmail(input) {
448
- if (input.length > 254) return false;
449
- if (!/^[a-zA-Z0-9.!#$%&'*+/=?^_`{|}~-]+@[a-zA-Z0-9](?:[a-zA-Z0-9-]{0,61}[a-zA-Z0-9])?(?:\.[a-zA-Z0-9](?:[a-zA-Z0-9-]{0,61}[a-zA-Z0-9])?)*$/.test(input)) return false;
450
- return detectThreatPatterns(input, {
451
- checkXSS: true,
452
- checkSQLInjection: true,
453
- checkNoSQLInjection: true,
454
- checkCommandInjection: false,
455
- checkPathTraversal: false,
456
- checkHomoglyphs: true
457
- }).isSafe;
617
+ if (typeof input !== "string") return false;
618
+ if (input.length > MAX_EMAIL_LENGTH) return false;
619
+ if (!EMAIL_PATTERN.test(input)) return false;
620
+ const analysis = analyzeIdentifier(input.slice(input.lastIndexOf("@") + 1));
621
+ return !analysis.isMixedScript && !analysis.hasBidiControls;
458
622
  }
459
623
  /**
460
- * Map a {@link ThreatDetectionResult} into a user-facing error
461
- * message string suitable for an HTTP 400 response or form
462
- * validation error. Returns `""` when the result is safe (so
463
- * `error || undefined` works).
624
+ * Render a user-facing error message for a detection result.
625
+ *
626
+ * Uses only the **first** finding: enumerating every category that fired leaks the
627
+ * shape of the rule set to whoever is probing it. Log `result.threats` server-side for
628
+ * diagnostics and return this to the client.
464
629
  *
465
- * Uses only the **first** finding for the message — exposing every
466
- * threat type to the user can leak information about the detection
467
- * rules. For full diagnostics, log `result.threats` server-side
468
- * rather than returning them.
630
+ * @param result - A {@link ThreatDetectionResult}, a `ThreatScanResult`-shaped object,
631
+ * or any `{ isSafe, threats }` pair.
632
+ * @returns A message, or `""` when the result is safe — so `message || undefined`
633
+ * works at a call site.
469
634
  */
470
635
  function getThreatErrorMessage(result) {
471
636
  if (result.isSafe) return "";
@@ -477,11 +642,27 @@ function getThreatErrorMessage(result) {
477
642
  case "nosql_injection": return "Input contains potentially malicious query operators";
478
643
  case "command_injection": return "Input contains potentially malicious system commands";
479
644
  case "path_traversal": return "Input contains potentially malicious file path characters";
645
+ case "prototype_pollution": return "Input contains potentially malicious object property names";
480
646
  case "homoglyph": return "Input contains suspicious lookalike characters";
647
+ case "header_injection": return "Input contains line breaks that are not allowed in this field";
648
+ case "ldap_injection": return "Input contains potentially malicious directory query characters";
649
+ case "xpath_injection": return "Input contains potentially malicious query expressions";
650
+ case "xml_injection": return "Input contains potentially malicious document declarations";
651
+ case "template_injection": return "Input contains potentially malicious template expressions";
652
+ case "file_inclusion": return "Input contains potentially malicious resource references";
653
+ case "ssrf": return "Input contains a network address that is not allowed";
654
+ case "formula_injection": return "Input contains spreadsheet formula characters";
655
+ case "log_injection": return "Input contains characters that are not allowed in this field";
656
+ case "prompt_injection": return "Input contains instructions that are not allowed in this field";
657
+ case "parameter_pollution": return "Input contains additional query parameters that are not allowed";
658
+ case "credential_exposure": return "Request contains credential material that must not be sent or stored here";
659
+ case "jwt_tampering": return "Token is not signed with an accepted algorithm";
660
+ case "double_encoding": return "Input contains characters that are encoded more than once";
661
+ case "resource_abuse": return "Input is too large or too repetitive to process";
481
662
  default: return assertNever(threat.type);
482
663
  }
483
664
  }
484
665
  //#endregion
485
- export { THREAT_DETECTED_MESSAGE, containsCommandInjection, containsHomoglyphs, containsNoSQLInjection, containsPathTraversal, containsSQLInjection, containsXSSPatterns, detectThreatPatterns, getThreatErrorMessage, isSafeInput, normalizeUnicode, sanitizeForDisplay, validateSafeEmail, validateSafeName, validateSafeText };
666
+ export { THREAT_DETECTED_MESSAGE, containsCommandInjection, containsHomoglyphs, containsNoSQLInjection, containsPathTraversal, containsPrototypePollution, containsSQLInjection, containsXSSPatterns, detectThreatPatterns, encodeJsonForScript, encodeLogValue, escapeCsvField, escapeHtmlAttribute, escapeHtmlText, getThreatErrorMessage, isSafeInput, normalizeUnicode, sanitizeForDisplay, toCsvRow, validatePersonName, validateSafeEmail, validateSafeName, validateSafeText };
486
667
 
487
668
  //# sourceMappingURL=validators.mjs.map