@resq-systems/security 1.0.5 → 2.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (108) hide show
  1. package/README.md +236 -33
  2. package/lib/controls/address.d.mts +142 -0
  3. package/lib/controls/address.d.mts.map +1 -0
  4. package/lib/controls/address.mjs +533 -0
  5. package/lib/controls/address.mjs.map +1 -0
  6. package/lib/controls/csrf.d.mts +91 -0
  7. package/lib/controls/csrf.d.mts.map +1 -0
  8. package/lib/controls/csrf.mjs +200 -0
  9. package/lib/controls/csrf.mjs.map +1 -0
  10. package/lib/controls/index.d.mts +8 -0
  11. package/lib/controls/index.mjs +8 -0
  12. package/lib/controls/origin.d.mts +95 -0
  13. package/lib/controls/origin.d.mts.map +1 -0
  14. package/lib/controls/origin.mjs +156 -0
  15. package/lib/controls/origin.mjs.map +1 -0
  16. package/lib/controls/payload.d.mts +84 -0
  17. package/lib/controls/payload.d.mts.map +1 -0
  18. package/lib/controls/payload.mjs +147 -0
  19. package/lib/controls/payload.mjs.map +1 -0
  20. package/lib/controls/query.d.mts +157 -0
  21. package/lib/controls/query.d.mts.map +1 -0
  22. package/lib/controls/query.mjs +368 -0
  23. package/lib/controls/query.mjs.map +1 -0
  24. package/lib/controls/redirect.d.mts +92 -0
  25. package/lib/controls/redirect.d.mts.map +1 -0
  26. package/lib/controls/redirect.mjs +110 -0
  27. package/lib/controls/redirect.mjs.map +1 -0
  28. package/lib/controls/upload.d.mts +108 -0
  29. package/lib/controls/upload.d.mts.map +1 -0
  30. package/lib/controls/upload.mjs +374 -0
  31. package/lib/controls/upload.mjs.map +1 -0
  32. package/lib/crypto.d.mts +18 -5
  33. package/lib/crypto.d.mts.map +1 -1
  34. package/lib/crypto.mjs +35 -24
  35. package/lib/crypto.mjs.map +1 -1
  36. package/lib/hash.d.mts +51 -6
  37. package/lib/hash.d.mts.map +1 -1
  38. package/lib/hash.mjs +51 -6
  39. package/lib/hash.mjs.map +1 -1
  40. package/lib/index.d.mts +17 -2
  41. package/lib/index.mjs +19 -2
  42. package/lib/paths.d.mts +92 -0
  43. package/lib/paths.d.mts.map +1 -0
  44. package/lib/paths.mjs +140 -0
  45. package/lib/paths.mjs.map +1 -0
  46. package/lib/sanitize.d.mts +137 -35
  47. package/lib/sanitize.d.mts.map +1 -1
  48. package/lib/sanitize.mjs +170 -46
  49. package/lib/sanitize.mjs.map +1 -1
  50. package/lib/threats/capec.generated.d.mts +59 -0
  51. package/lib/threats/capec.generated.d.mts.map +1 -0
  52. package/lib/threats/capec.generated.mjs +644 -0
  53. package/lib/threats/capec.generated.mjs.map +1 -0
  54. package/lib/threats/engine.d.mts +94 -0
  55. package/lib/threats/engine.d.mts.map +1 -0
  56. package/lib/threats/engine.mjs +167 -0
  57. package/lib/threats/engine.mjs.map +1 -0
  58. package/lib/threats/index.d.mts +11 -0
  59. package/lib/threats/index.mjs +11 -0
  60. package/lib/threats/rules/datastore.d.mts +13 -0
  61. package/lib/threats/rules/datastore.d.mts.map +1 -0
  62. package/lib/threats/rules/datastore.mjs +366 -0
  63. package/lib/threats/rules/datastore.mjs.map +1 -0
  64. package/lib/threats/rules/index.d.mts +54 -0
  65. package/lib/threats/rules/index.d.mts.map +1 -0
  66. package/lib/threats/rules/index.mjs +121 -0
  67. package/lib/threats/rules/index.mjs.map +1 -0
  68. package/lib/threats/rules/markup.d.mts +28 -0
  69. package/lib/threats/rules/markup.d.mts.map +1 -0
  70. package/lib/threats/rules/markup.mjs +373 -0
  71. package/lib/threats/rules/markup.mjs.map +1 -0
  72. package/lib/threats/rules/protocol.d.mts +49 -0
  73. package/lib/threats/rules/protocol.d.mts.map +1 -0
  74. package/lib/threats/rules/protocol.mjs +175 -0
  75. package/lib/threats/rules/protocol.mjs.map +1 -0
  76. package/lib/threats/rules/system.d.mts +19 -0
  77. package/lib/threats/rules/system.d.mts.map +1 -0
  78. package/lib/threats/rules/system.mjs +455 -0
  79. package/lib/threats/rules/system.mjs.map +1 -0
  80. package/lib/threats/rules/web.d.mts +26 -0
  81. package/lib/threats/rules/web.d.mts.map +1 -0
  82. package/lib/threats/rules/web.mjs +412 -0
  83. package/lib/threats/rules/web.mjs.map +1 -0
  84. package/lib/threats/scoring.d.mts +59 -0
  85. package/lib/threats/scoring.d.mts.map +1 -0
  86. package/lib/threats/scoring.mjs +111 -0
  87. package/lib/threats/scoring.mjs.map +1 -0
  88. package/lib/threats/types.d.mts +245 -0
  89. package/lib/threats/types.d.mts.map +1 -0
  90. package/lib/threats/types.mjs +52 -0
  91. package/lib/threats/types.mjs.map +1 -0
  92. package/lib/threats/variants.d.mts +57 -0
  93. package/lib/threats/variants.d.mts.map +1 -0
  94. package/lib/threats/variants.mjs +144 -0
  95. package/lib/threats/variants.mjs.map +1 -0
  96. package/lib/unicode/confusables.d.mts +82 -0
  97. package/lib/unicode/confusables.d.mts.map +1 -0
  98. package/lib/unicode/confusables.mjs +954 -0
  99. package/lib/unicode/confusables.mjs.map +1 -0
  100. package/lib/unicode/index.d.mts +126 -0
  101. package/lib/unicode/index.d.mts.map +1 -0
  102. package/lib/unicode/index.mjs +288 -0
  103. package/lib/unicode/index.mjs.map +1 -0
  104. package/lib/validators.d.mts +341 -164
  105. package/lib/validators.d.mts.map +1 -1
  106. package/lib/validators.mjs +519 -338
  107. package/lib/validators.mjs.map +1 -1
  108. package/package.json +35 -8
@@ -0,0 +1,245 @@
1
+ //#region src/threats/types.d.ts
2
+ /**
3
+ * Copyright 2026 ResQ Systems, Inc.
4
+ *
5
+ * Licensed under the Apache License, Version 2.0 (the "License");
6
+ * you may not use this file except in compliance with the License.
7
+ * You may obtain a copy of the License at
8
+ *
9
+ * http://www.apache.org/licenses/LICENSE-2.0
10
+ *
11
+ * Unless required by applicable law or agreed to in writing, software
12
+ * distributed under the License is distributed on an "AS IS" BASIS,
13
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
14
+ * See the License for the specific language governing permissions and
15
+ * limitations under the License.
16
+ */
17
+ /**
18
+ * @fileoverview Vocabulary for the threat rule engine — weakness categories, sink
19
+ * contexts, severity/confidence grades, and the {@link ThreatRule} /
20
+ * {@link ThreatFinding} shapes that carry them. Everything here is data-only so the
21
+ * rule catalog, the scanner, and the scoring policy can be reasoned about (and
22
+ * unit-tested) independently.
23
+ *
24
+ * Design note: rules are *telemetry and defense-in-depth*, not the primary control.
25
+ * The primary control for each category is the sink-appropriate one — parameterized
26
+ * queries for SQL, a parser-based sanitizer for HTML, canonicalize-and-contain for
27
+ * filesystem paths, argv-array process spawning for shells. Each rule names its own
28
+ * in {@link ThreatRule.primaryControl}.
29
+ *
30
+ * @module @resq-systems/security/threats/types
31
+ */
32
+ /**
33
+ * The closed set of weakness categories the engine recognizes.
34
+ *
35
+ * Serves as the discriminant of {@link ThreatFinding} and drives the exhaustive
36
+ * `switch` in `getThreatErrorMessage` — adding a variant here without a matching
37
+ * `case` there is a compile error via `assertNever`.
38
+ *
39
+ * Categories follow the OWASP Web Security Testing Guide's input-validation chapter
40
+ * so findings map onto an established taxonomy rather than package-local names.
41
+ */
42
+ type ThreatType = "xss" | "sql_injection" | "nosql_injection" | "command_injection" | "path_traversal" | "prototype_pollution" | "homoglyph" | "header_injection" | "ldap_injection" | "xpath_injection" | "xml_injection" | "template_injection" | "file_inclusion" | "ssrf" | "formula_injection" | "log_injection" | "prompt_injection" | "parameter_pollution" | "credential_exposure" | "jwt_tampering" | "double_encoding" | "resource_abuse";
43
+ /**
44
+ * The sink a value is destined for. Rules declare which contexts they apply to, and
45
+ * the scanner only evaluates rules matching the caller's declared contexts.
46
+ *
47
+ * This is the single most important false-positive control in the package: a
48
+ * biography containing `C:\Windows`, a support ticket containing `1=1`, and a code
49
+ * snippet containing `eval(` are all perfectly legitimate `general_text`. They are
50
+ * only suspicious when the value actually reaches a filesystem, SQL, or HTML sink.
51
+ *
52
+ * - `general_text` — free-form prose. Only near-universally hostile signals apply
53
+ * (invisible/bidirectional control characters, null bytes).
54
+ * - `html` — rendered as HTML or interpolated into markup.
55
+ * - `sql` / `nosql` — reaches a database query builder or driver.
56
+ * - `shell` — reaches a child process, especially with `shell: true`.
57
+ * - `filesystem` — becomes part of a path passed to `fs`.
58
+ * - `url` — fetched server-side, or used to build such a request.
59
+ * - `url_parameter` — a *single value* about to be concatenated into a query string.
60
+ * Distinct from `url`, where `&name=` is the normal grammar rather than evidence of
61
+ * an injected parameter — the same distinction `http_header` draws against a whole
62
+ * request.
63
+ * - `http_header` — written into an HTTP or email header value.
64
+ * - `jwt` — a JSON Web Token, or an already-decoded JWT header. Scoped narrowly on
65
+ * purpose: re-scanning a whole opaque token with every detector is the
66
+ * run-everything-against-everything anti-pattern this package exists to avoid. To
67
+ * check a `kid` or `jku` claim, extract it and declare *its* real sink
68
+ * (`filesystem`, `sql`, `url`).
69
+ * - `identifier` — a username, domain, org name, package name, or brand-adjacent
70
+ * label where confusable-glyph collisions matter.
71
+ * - `object_merge` — parsed into an object graph that is then merged, cloned, or
72
+ * assigned property-by-property (`Object.assign`, `lodash.merge`, query-string
73
+ * expansion). This is prototype pollution's actual sink.
74
+ * - `template` — concatenated into template *source* (as opposed to passed as data).
75
+ * - `xml` — parsed by an XML/SOAP/SVG parser.
76
+ * - `ldap` / `xpath` — becomes part of an LDAP filter/DN or an XPath expression.
77
+ * - `spreadsheet` — exported to CSV/XLSX and opened by a spreadsheet application.
78
+ * - `log` — written to a line-based or structured log sink.
79
+ * - `llm_prompt` — concatenated into an LLM prompt or reachable by a tool-using agent.
80
+ */
81
+ type ThreatContext = "general_text" | "html" | "sql" | "nosql" | "shell" | "filesystem" | "url" | "url_parameter" | "http_header" | "jwt" | "identifier" | "object_merge" | "template" | "xml" | "ldap" | "xpath" | "spreadsheet" | "log" | "llm_prompt";
82
+ /** Every {@link ThreatContext} value, for iteration and "scan every sink" callers. */
83
+ declare const ALL_THREAT_CONTEXTS: readonly ["general_text", "html", "sql", "nosql", "shell", "filesystem", "url", "url_parameter", "http_header", "jwt", "identifier", "object_merge", "template", "xml", "ldap", "xpath", "spreadsheet", "log", "llm_prompt"];
84
+ /**
85
+ * How bad the weakness is if the match is a true positive. Independent of how
86
+ * likely the match is to *be* a true positive — that is {@link ThreatConfidence}.
87
+ */
88
+ type ThreatSeverity = "low" | "medium" | "high" | "critical";
89
+ /**
90
+ * How likely a match is to be a real attack rather than benign content that
91
+ * happens to look like one.
92
+ *
93
+ * - `low` — fires on ordinary content regularly (e.g. the substring `1=1`).
94
+ * - `medium` — unusual in ordinary content but not impossible.
95
+ * - `high` — has essentially no benign explanation in the declared context.
96
+ */
97
+ type ThreatConfidence = "low" | "medium" | "high";
98
+ /** Base anomaly points contributed by a finding, before the confidence multiplier. */
99
+ declare const SEVERITY_WEIGHTS: Readonly<Record<ThreatSeverity, number>>;
100
+ /** Scales {@link SEVERITY_WEIGHTS} by how trustworthy the signature is. */
101
+ declare const CONFIDENCE_MULTIPLIERS: Readonly<Record<ThreatConfidence, number>>;
102
+ /** Ordering helper for `minSeverity` filtering. Higher is worse. */
103
+ declare const SEVERITY_ORDER: Readonly<Record<ThreatSeverity, number>>;
104
+ /**
105
+ * Which representation of the input a rule matched against.
106
+ *
107
+ * Attack signatures are routinely bypassed by encoding, so the scanner evaluates a
108
+ * small, bounded set of representations rather than the raw bytes alone. Findings
109
+ * record which one matched, so operators can tell "the request literally contained
110
+ * `../`" apart from "the request contained `%2e%2e%2f`, which only becomes `../` if
111
+ * a downstream component decodes it".
112
+ */
113
+ type InputVariantKind = "raw" | "nfc" | "nfkc" | "percent_decoded" | "html_decoded";
114
+ /** One representation of the scanned input. */
115
+ interface InputVariant {
116
+ /** Which transformation produced {@link InputVariant.value}. */
117
+ readonly kind: InputVariantKind;
118
+ /** The transformed string. */
119
+ readonly value: string;
120
+ }
121
+ /**
122
+ * A single signature in the catalog.
123
+ *
124
+ * @remarks
125
+ * `pattern` **must not** carry the global (`g`) or sticky (`y`) flag. Those flags
126
+ * make `RegExp` objects stateful via `lastIndex`, and the catalog is module-scoped
127
+ * and shared across every call — a stateful rule would silently skip matches on
128
+ * alternating invocations. `assertRuleCatalogIsValid` enforces this at load time.
129
+ */
130
+ interface ThreatRule {
131
+ /** Stable identifier (e.g. `PATH-TRAVERSAL-ENCODED-001`). Never reused once retired. */
132
+ readonly id: string;
133
+ /** Weakness category this rule evidences. */
134
+ readonly type: ThreatType;
135
+ /** Sinks for which this rule is meaningful. Never empty. */
136
+ readonly contexts: readonly ThreatContext[];
137
+ /** Impact if the match is real. */
138
+ readonly severity: ThreatSeverity;
139
+ /** Likelihood the match is real rather than benign lookalike content. */
140
+ readonly confidence: ThreatConfidence;
141
+ /** One-line description for operators and log lines. Not user-facing. */
142
+ readonly description: string;
143
+ /** MITRE CWE identifier for the weakness, when one applies cleanly. */
144
+ readonly cwe?: number;
145
+ /**
146
+ * The control that actually prevents this weakness. Signatures detect; they do
147
+ * not prevent. Surfaced on findings so a reviewer reading an alert is pointed at
148
+ * the fix rather than at a tighter regex.
149
+ */
150
+ readonly primaryControl: string;
151
+ /** The signature. Must be non-global and non-sticky — see the remark above. */
152
+ readonly pattern: RegExp;
153
+ /**
154
+ * Restrict this rule to specific input representations. Defaults to every
155
+ * variant. Used by rules whose whole purpose is detecting an encoding (e.g. the
156
+ * percent-encoded traversal rule, which is only meaningful against `raw`).
157
+ */
158
+ readonly variants?: readonly InputVariantKind[];
159
+ }
160
+ /** A rule that matched, plus where and in which representation. */
161
+ interface ThreatFinding {
162
+ /** {@link ThreatRule.id} of the rule that fired. */
163
+ readonly ruleId: string;
164
+ /** Copied from the rule — the weakness category. */
165
+ readonly type: ThreatType;
166
+ /** Copied from the rule. */
167
+ readonly severity: ThreatSeverity;
168
+ /** Copied from the rule. */
169
+ readonly confidence: ThreatConfidence;
170
+ /** Copied from the rule. Operator-facing; never render to end users. */
171
+ readonly description: string;
172
+ /** Copied from the rule, when present. */
173
+ readonly cwe?: number;
174
+ /** Copied from the rule — the control that actually fixes this. */
175
+ readonly primaryControl: string;
176
+ /** Which representation matched. */
177
+ readonly variant: InputVariantKind;
178
+ /** Matched substring, truncated to 50 characters so logs cannot be flooded. */
179
+ readonly matchedPattern?: string;
180
+ /** Match start offset *within the matched variant*, not within the raw input. */
181
+ readonly start?: number;
182
+ /** Match end offset (exclusive) within the matched variant. */
183
+ readonly end?: number;
184
+ }
185
+ /**
186
+ * Where an untrusted value entered the application.
187
+ *
188
+ * The counterpart to {@link ThreatContext}, which names the *sink*. A context says where
189
+ * a value is going and decides which rules run; a source says where it came from and
190
+ * decides nothing — it is carried through so a finding can be attributed later.
191
+ *
192
+ * Both halves together are what makes a finding actionable: "SQL keywords in a value
193
+ * bound to a SQL query" is a bug report, while "SQL keywords arriving in a query
194
+ * parameter from one account, forty times in a minute" is an incident.
195
+ */
196
+ type InputSource = "http.query" | "http.body" | "http.header" | "http.path" | "http.cookie" | "websocket.message" | "cli.argument" | "file.upload" | "database" | "external.api" | "internal";
197
+ /**
198
+ * Who and what a scan belongs to, carried through so findings can be correlated.
199
+ *
200
+ * Modelled on OWASP AppSensor, whose argument is that application-layer intrusion
201
+ * detection is about *sequences* attributed to an actor, not about individual strings. A
202
+ * single `review` verdict is ordinarily noise; thirty of them from one account inside a
203
+ * minute is an attack in progress, and nothing in a per-string API can express the
204
+ * difference.
205
+ *
206
+ * Every field is optional and none affects detection. The scanner does not resolve,
207
+ * validate or store any of them — it echoes them onto the result so a logging or SIEM
208
+ * pipeline can join findings to a request and an actor without threading a correlation id
209
+ * through its own call stack.
210
+ *
211
+ * Treat `actorId` as personal data: it reaches whatever sink the findings reach, so pass
212
+ * an opaque identifier rather than an email address, or redact downstream with
213
+ * `sanitizeForLogging`.
214
+ */
215
+ interface EventContext {
216
+ /** Correlates every finding raised while handling one request. */
217
+ readonly requestId?: string;
218
+ /** The authenticated principal, if any. Opaque — never an email address. */
219
+ readonly actorId?: string;
220
+ /** The session the request belongs to, for sequence detection across requests. */
221
+ readonly sessionId?: string;
222
+ /** Where the value entered the application. */
223
+ readonly source?: InputSource;
224
+ }
225
+ /**
226
+ * Policy outcome derived from the anomaly score.
227
+ *
228
+ * Scoring rather than first-match rejection is deliberate: any individual signature
229
+ * produces false positives, so a single low-confidence hit should raise a signal,
230
+ * not reject a form submission. This mirrors how OWASP CRS uses anomaly scoring with
231
+ * tunable thresholds instead of one-rule-one-block.
232
+ */
233
+ type ThreatVerdict = "allow" | "review" | "block";
234
+ /** Score thresholds separating {@link ThreatVerdict} bands. */
235
+ interface ThreatPolicy {
236
+ /** Score at or above which the verdict becomes `review`. Default `4`. */
237
+ readonly reviewAt: number;
238
+ /** Score at or above which the verdict becomes `block`. Default `8`. */
239
+ readonly blockAt: number;
240
+ }
241
+ /** Default thresholds: one high/high finding reviews, one critical finding blocks. */
242
+ declare const DEFAULT_THREAT_POLICY: ThreatPolicy;
243
+ //#endregion
244
+ export { ALL_THREAT_CONTEXTS, CONFIDENCE_MULTIPLIERS, DEFAULT_THREAT_POLICY, EventContext, InputSource, InputVariant, InputVariantKind, SEVERITY_ORDER, SEVERITY_WEIGHTS, ThreatConfidence, ThreatContext, ThreatFinding, ThreatPolicy, ThreatRule, ThreatSeverity, ThreatType, ThreatVerdict };
245
+ //# sourceMappingURL=types.d.mts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"types.d.mts","names":[],"sources":["../../src/threats/types.ts"],"mappings":";;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;KA4CY;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;KA8DA;;cAsBC;;;;;KA8BD;;;;;;;;;KAUA;;cAGC,kBAAkB,SAAS,OAAO;;cAQlC,wBAAwB,SAAS,OAAO;;cAOxC,gBAAgB,SAAS,OAAO;;;;;;;;;;KAoBjC;;UAGK;;WAEP,MAAM;;WAEN;;;;;;;;;;;UAgBO;;WAEP;;WAEA,MAAM;;WAEN,mBAAmB;;WAEnB,UAAU;;WAEV,YAAY;;WAEZ;;WAEA;;;;;;WAMA;;WAEA,SAAS;;;;;;WAMT,oBAAoB;;;UAIb;;WAEP;;WAEA,MAAM;;WAEN,UAAU;;WAEV,YAAY;;WAEZ;;WAEA;;WAEA;;WAEA,SAAS;;WAET;;WAEA;;WAEA;;;;;;;;;;;;;KAkBE;;;;;;;;;;;;;;;;;;;UA+BK;;WAEP;;WAEA;;WAEA;;WAEA,SAAS;;;;;;;;;;KAeP;;UAGK;;WAEP;;WAEA;;;cAIG,uBAAuB"}
@@ -0,0 +1,52 @@
1
+ //#region src/threats/types.ts
2
+ /** Every {@link ThreatContext} value, for iteration and "scan every sink" callers. */
3
+ const ALL_THREAT_CONTEXTS = [
4
+ "general_text",
5
+ "html",
6
+ "sql",
7
+ "nosql",
8
+ "shell",
9
+ "filesystem",
10
+ "url",
11
+ "url_parameter",
12
+ "http_header",
13
+ "jwt",
14
+ "identifier",
15
+ "object_merge",
16
+ "template",
17
+ "xml",
18
+ "ldap",
19
+ "xpath",
20
+ "spreadsheet",
21
+ "log",
22
+ "llm_prompt"
23
+ ];
24
+ /** Base anomaly points contributed by a finding, before the confidence multiplier. */
25
+ const SEVERITY_WEIGHTS = {
26
+ low: 1,
27
+ medium: 2,
28
+ high: 4,
29
+ critical: 8
30
+ };
31
+ /** Scales {@link SEVERITY_WEIGHTS} by how trustworthy the signature is. */
32
+ const CONFIDENCE_MULTIPLIERS = {
33
+ low: .5,
34
+ medium: 1,
35
+ high: 1.5
36
+ };
37
+ /** Ordering helper for `minSeverity` filtering. Higher is worse. */
38
+ const SEVERITY_ORDER = {
39
+ low: 0,
40
+ medium: 1,
41
+ high: 2,
42
+ critical: 3
43
+ };
44
+ /** Default thresholds: one high/high finding reviews, one critical finding blocks. */
45
+ const DEFAULT_THREAT_POLICY = {
46
+ reviewAt: 4,
47
+ blockAt: 8
48
+ };
49
+ //#endregion
50
+ export { ALL_THREAT_CONTEXTS, CONFIDENCE_MULTIPLIERS, DEFAULT_THREAT_POLICY, SEVERITY_ORDER, SEVERITY_WEIGHTS };
51
+
52
+ //# sourceMappingURL=types.mjs.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"types.mjs","names":[],"sources":["../../src/threats/types.ts"],"sourcesContent":["/**\n * Copyright 2026 ResQ Systems, Inc.\n *\n * Licensed under the Apache License, Version 2.0 (the \"License\");\n * you may not use this file except in compliance with the License.\n * You may obtain a copy of the License at\n *\n * http://www.apache.org/licenses/LICENSE-2.0\n *\n * Unless required by applicable law or agreed to in writing, software\n * distributed under the License is distributed on an \"AS IS\" BASIS,\n * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n * See the License for the specific language governing permissions and\n * limitations under the License.\n */\n\n/**\n * @fileoverview Vocabulary for the threat rule engine — weakness categories, sink\n * contexts, severity/confidence grades, and the {@link ThreatRule} /\n * {@link ThreatFinding} shapes that carry them. Everything here is data-only so the\n * rule catalog, the scanner, and the scoring policy can be reasoned about (and\n * unit-tested) independently.\n *\n * Design note: rules are *telemetry and defense-in-depth*, not the primary control.\n * The primary control for each category is the sink-appropriate one — parameterized\n * queries for SQL, a parser-based sanitizer for HTML, canonicalize-and-contain for\n * filesystem paths, argv-array process spawning for shells. Each rule names its own\n * in {@link ThreatRule.primaryControl}.\n *\n * @module @resq-systems/security/threats/types\n */\n\n//#region Categories\n\n/**\n * The closed set of weakness categories the engine recognizes.\n *\n * Serves as the discriminant of {@link ThreatFinding} and drives the exhaustive\n * `switch` in `getThreatErrorMessage` — adding a variant here without a matching\n * `case` there is a compile error via `assertNever`.\n *\n * Categories follow the OWASP Web Security Testing Guide's input-validation chapter\n * so findings map onto an established taxonomy rather than package-local names.\n */\nexport type ThreatType =\n\t| \"xss\"\n\t| \"sql_injection\"\n\t| \"nosql_injection\"\n\t| \"command_injection\"\n\t| \"path_traversal\"\n\t| \"prototype_pollution\"\n\t| \"homoglyph\"\n\t| \"header_injection\"\n\t| \"ldap_injection\"\n\t| \"xpath_injection\"\n\t| \"xml_injection\"\n\t| \"template_injection\"\n\t| \"file_inclusion\"\n\t| \"ssrf\"\n\t| \"formula_injection\"\n\t| \"log_injection\"\n\t| \"prompt_injection\"\n\t| \"parameter_pollution\"\n\t| \"credential_exposure\"\n\t| \"jwt_tampering\"\n\t| \"double_encoding\"\n\t| \"resource_abuse\";\n\n/**\n * The sink a value is destined for. Rules declare which contexts they apply to, and\n * the scanner only evaluates rules matching the caller's declared contexts.\n *\n * This is the single most important false-positive control in the package: a\n * biography containing `C:\\Windows`, a support ticket containing `1=1`, and a code\n * snippet containing `eval(` are all perfectly legitimate `general_text`. They are\n * only suspicious when the value actually reaches a filesystem, SQL, or HTML sink.\n *\n * - `general_text` — free-form prose. Only near-universally hostile signals apply\n * (invisible/bidirectional control characters, null bytes).\n * - `html` — rendered as HTML or interpolated into markup.\n * - `sql` / `nosql` — reaches a database query builder or driver.\n * - `shell` — reaches a child process, especially with `shell: true`.\n * - `filesystem` — becomes part of a path passed to `fs`.\n * - `url` — fetched server-side, or used to build such a request.\n * - `url_parameter` — a *single value* about to be concatenated into a query string.\n * Distinct from `url`, where `&name=` is the normal grammar rather than evidence of\n * an injected parameter — the same distinction `http_header` draws against a whole\n * request.\n * - `http_header` — written into an HTTP or email header value.\n * - `jwt` — a JSON Web Token, or an already-decoded JWT header. Scoped narrowly on\n * purpose: re-scanning a whole opaque token with every detector is the\n * run-everything-against-everything anti-pattern this package exists to avoid. To\n * check a `kid` or `jku` claim, extract it and declare *its* real sink\n * (`filesystem`, `sql`, `url`).\n * - `identifier` — a username, domain, org name, package name, or brand-adjacent\n * label where confusable-glyph collisions matter.\n * - `object_merge` — parsed into an object graph that is then merged, cloned, or\n * assigned property-by-property (`Object.assign`, `lodash.merge`, query-string\n * expansion). This is prototype pollution's actual sink.\n * - `template` — concatenated into template *source* (as opposed to passed as data).\n * - `xml` — parsed by an XML/SOAP/SVG parser.\n * - `ldap` / `xpath` — becomes part of an LDAP filter/DN or an XPath expression.\n * - `spreadsheet` — exported to CSV/XLSX and opened by a spreadsheet application.\n * - `log` — written to a line-based or structured log sink.\n * - `llm_prompt` — concatenated into an LLM prompt or reachable by a tool-using agent.\n */\nexport type ThreatContext =\n\t| \"general_text\"\n\t| \"html\"\n\t| \"sql\"\n\t| \"nosql\"\n\t| \"shell\"\n\t| \"filesystem\"\n\t| \"url\"\n\t| \"url_parameter\"\n\t| \"http_header\"\n\t| \"jwt\"\n\t| \"identifier\"\n\t| \"object_merge\"\n\t| \"template\"\n\t| \"xml\"\n\t| \"ldap\"\n\t| \"xpath\"\n\t| \"spreadsheet\"\n\t| \"log\"\n\t| \"llm_prompt\";\n\n/** Every {@link ThreatContext} value, for iteration and \"scan every sink\" callers. */\nexport const ALL_THREAT_CONTEXTS = [\n\t\"general_text\",\n\t\"html\",\n\t\"sql\",\n\t\"nosql\",\n\t\"shell\",\n\t\"filesystem\",\n\t\"url\",\n\t\"url_parameter\",\n\t\"http_header\",\n\t\"jwt\",\n\t\"identifier\",\n\t\"object_merge\",\n\t\"template\",\n\t\"xml\",\n\t\"ldap\",\n\t\"xpath\",\n\t\"spreadsheet\",\n\t\"log\",\n\t\"llm_prompt\",\n] as const satisfies readonly ThreatContext[];\n\n//#endregion\n\n//#region Grading\n\n/**\n * How bad the weakness is if the match is a true positive. Independent of how\n * likely the match is to *be* a true positive — that is {@link ThreatConfidence}.\n */\nexport type ThreatSeverity = \"low\" | \"medium\" | \"high\" | \"critical\";\n\n/**\n * How likely a match is to be a real attack rather than benign content that\n * happens to look like one.\n *\n * - `low` — fires on ordinary content regularly (e.g. the substring `1=1`).\n * - `medium` — unusual in ordinary content but not impossible.\n * - `high` — has essentially no benign explanation in the declared context.\n */\nexport type ThreatConfidence = \"low\" | \"medium\" | \"high\";\n\n/** Base anomaly points contributed by a finding, before the confidence multiplier. */\nexport const SEVERITY_WEIGHTS: Readonly<Record<ThreatSeverity, number>> = {\n\tlow: 1,\n\tmedium: 2,\n\thigh: 4,\n\tcritical: 8,\n};\n\n/** Scales {@link SEVERITY_WEIGHTS} by how trustworthy the signature is. */\nexport const CONFIDENCE_MULTIPLIERS: Readonly<Record<ThreatConfidence, number>> = {\n\tlow: 0.5,\n\tmedium: 1,\n\thigh: 1.5,\n};\n\n/** Ordering helper for `minSeverity` filtering. Higher is worse. */\nexport const SEVERITY_ORDER: Readonly<Record<ThreatSeverity, number>> = {\n\tlow: 0,\n\tmedium: 1,\n\thigh: 2,\n\tcritical: 3,\n};\n\n//#endregion\n\n//#region Variants\n\n/**\n * Which representation of the input a rule matched against.\n *\n * Attack signatures are routinely bypassed by encoding, so the scanner evaluates a\n * small, bounded set of representations rather than the raw bytes alone. Findings\n * record which one matched, so operators can tell \"the request literally contained\n * `../`\" apart from \"the request contained `%2e%2e%2f`, which only becomes `../` if\n * a downstream component decodes it\".\n */\nexport type InputVariantKind = \"raw\" | \"nfc\" | \"nfkc\" | \"percent_decoded\" | \"html_decoded\";\n\n/** One representation of the scanned input. */\nexport interface InputVariant {\n\t/** Which transformation produced {@link InputVariant.value}. */\n\treadonly kind: InputVariantKind;\n\t/** The transformed string. */\n\treadonly value: string;\n}\n\n//#endregion\n\n//#region Rules and findings\n\n/**\n * A single signature in the catalog.\n *\n * @remarks\n * `pattern` **must not** carry the global (`g`) or sticky (`y`) flag. Those flags\n * make `RegExp` objects stateful via `lastIndex`, and the catalog is module-scoped\n * and shared across every call — a stateful rule would silently skip matches on\n * alternating invocations. `assertRuleCatalogIsValid` enforces this at load time.\n */\nexport interface ThreatRule {\n\t/** Stable identifier (e.g. `PATH-TRAVERSAL-ENCODED-001`). Never reused once retired. */\n\treadonly id: string;\n\t/** Weakness category this rule evidences. */\n\treadonly type: ThreatType;\n\t/** Sinks for which this rule is meaningful. Never empty. */\n\treadonly contexts: readonly ThreatContext[];\n\t/** Impact if the match is real. */\n\treadonly severity: ThreatSeverity;\n\t/** Likelihood the match is real rather than benign lookalike content. */\n\treadonly confidence: ThreatConfidence;\n\t/** One-line description for operators and log lines. Not user-facing. */\n\treadonly description: string;\n\t/** MITRE CWE identifier for the weakness, when one applies cleanly. */\n\treadonly cwe?: number;\n\t/**\n\t * The control that actually prevents this weakness. Signatures detect; they do\n\t * not prevent. Surfaced on findings so a reviewer reading an alert is pointed at\n\t * the fix rather than at a tighter regex.\n\t */\n\treadonly primaryControl: string;\n\t/** The signature. Must be non-global and non-sticky — see the remark above. */\n\treadonly pattern: RegExp;\n\t/**\n\t * Restrict this rule to specific input representations. Defaults to every\n\t * variant. Used by rules whose whole purpose is detecting an encoding (e.g. the\n\t * percent-encoded traversal rule, which is only meaningful against `raw`).\n\t */\n\treadonly variants?: readonly InputVariantKind[];\n}\n\n/** A rule that matched, plus where and in which representation. */\nexport interface ThreatFinding {\n\t/** {@link ThreatRule.id} of the rule that fired. */\n\treadonly ruleId: string;\n\t/** Copied from the rule — the weakness category. */\n\treadonly type: ThreatType;\n\t/** Copied from the rule. */\n\treadonly severity: ThreatSeverity;\n\t/** Copied from the rule. */\n\treadonly confidence: ThreatConfidence;\n\t/** Copied from the rule. Operator-facing; never render to end users. */\n\treadonly description: string;\n\t/** Copied from the rule, when present. */\n\treadonly cwe?: number;\n\t/** Copied from the rule — the control that actually fixes this. */\n\treadonly primaryControl: string;\n\t/** Which representation matched. */\n\treadonly variant: InputVariantKind;\n\t/** Matched substring, truncated to 50 characters so logs cannot be flooded. */\n\treadonly matchedPattern?: string;\n\t/** Match start offset *within the matched variant*, not within the raw input. */\n\treadonly start?: number;\n\t/** Match end offset (exclusive) within the matched variant. */\n\treadonly end?: number;\n}\n\n//#endregion\n\n//#region Event correlation\n\n/**\n * Where an untrusted value entered the application.\n *\n * The counterpart to {@link ThreatContext}, which names the *sink*. A context says where\n * a value is going and decides which rules run; a source says where it came from and\n * decides nothing — it is carried through so a finding can be attributed later.\n *\n * Both halves together are what makes a finding actionable: \"SQL keywords in a value\n * bound to a SQL query\" is a bug report, while \"SQL keywords arriving in a query\n * parameter from one account, forty times in a minute\" is an incident.\n */\nexport type InputSource =\n\t| \"http.query\"\n\t| \"http.body\"\n\t| \"http.header\"\n\t| \"http.path\"\n\t| \"http.cookie\"\n\t| \"websocket.message\"\n\t| \"cli.argument\"\n\t| \"file.upload\"\n\t| \"database\"\n\t| \"external.api\"\n\t| \"internal\";\n\n/**\n * Who and what a scan belongs to, carried through so findings can be correlated.\n *\n * Modelled on OWASP AppSensor, whose argument is that application-layer intrusion\n * detection is about *sequences* attributed to an actor, not about individual strings. A\n * single `review` verdict is ordinarily noise; thirty of them from one account inside a\n * minute is an attack in progress, and nothing in a per-string API can express the\n * difference.\n *\n * Every field is optional and none affects detection. The scanner does not resolve,\n * validate or store any of them — it echoes them onto the result so a logging or SIEM\n * pipeline can join findings to a request and an actor without threading a correlation id\n * through its own call stack.\n *\n * Treat `actorId` as personal data: it reaches whatever sink the findings reach, so pass\n * an opaque identifier rather than an email address, or redact downstream with\n * `sanitizeForLogging`.\n */\nexport interface EventContext {\n\t/** Correlates every finding raised while handling one request. */\n\treadonly requestId?: string;\n\t/** The authenticated principal, if any. Opaque — never an email address. */\n\treadonly actorId?: string;\n\t/** The session the request belongs to, for sequence detection across requests. */\n\treadonly sessionId?: string;\n\t/** Where the value entered the application. */\n\treadonly source?: InputSource;\n}\n\n//#endregion\n\n//#region Verdicts\n\n/**\n * Policy outcome derived from the anomaly score.\n *\n * Scoring rather than first-match rejection is deliberate: any individual signature\n * produces false positives, so a single low-confidence hit should raise a signal,\n * not reject a form submission. This mirrors how OWASP CRS uses anomaly scoring with\n * tunable thresholds instead of one-rule-one-block.\n */\nexport type ThreatVerdict = \"allow\" | \"review\" | \"block\";\n\n/** Score thresholds separating {@link ThreatVerdict} bands. */\nexport interface ThreatPolicy {\n\t/** Score at or above which the verdict becomes `review`. Default `4`. */\n\treadonly reviewAt: number;\n\t/** Score at or above which the verdict becomes `block`. Default `8`. */\n\treadonly blockAt: number;\n}\n\n/** Default thresholds: one high/high finding reviews, one critical finding blocks. */\nexport const DEFAULT_THREAT_POLICY: ThreatPolicy = {\n\treviewAt: 4,\n\tblockAt: 8,\n};\n\n//#endregion\n"],"mappings":";;AAgIA,MAAa,sBAAsB;CAClC;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;AACD;;AAuBA,MAAa,mBAA6D;CACzE,KAAK;CACL,QAAQ;CACR,MAAM;CACN,UAAU;AACX;;AAGA,MAAa,yBAAqE;CACjF,KAAK;CACL,QAAQ;CACR,MAAM;AACP;;AAGA,MAAa,iBAA2D;CACvE,KAAK;CACL,QAAQ;CACR,MAAM;CACN,UAAU;AACX;;AA+KA,MAAa,wBAAsC;CAClD,UAAU;CACV,SAAS;AACV"}
@@ -0,0 +1,57 @@
1
+ import { InputVariant } from "./types.mjs";
2
+ //#region src/threats/variants.d.ts
3
+ /**
4
+ * Decode HTML character references in a single pass.
5
+ *
6
+ * Handles `&#60;`, `&#x3c;`, and the named subset in {@link NAMED_ENTITIES}, with or
7
+ * without the trailing semicolon — browsers accept several named references
8
+ * unterminated and attackers rely on it. Unrecognized references are left verbatim;
9
+ * this is best-effort detection support, not a rendering path.
10
+ *
11
+ * @param input - Raw string.
12
+ * @returns The string with recognized references replaced.
13
+ *
14
+ * @example
15
+ * ```ts
16
+ * decodeHtmlEntities("&#60;script&#62;"); // "<script>"
17
+ * decodeHtmlEntities("&unknownref;"); // "&unknownref;"
18
+ * ```
19
+ */
20
+ declare function decodeHtmlEntities(input: string): string;
21
+ /**
22
+ * Percent-decode a string, tolerating malformed sequences.
23
+ *
24
+ * `decodeURIComponent` throws `URIError` on an invalid escape (`%zz`, a lone `%`, a
25
+ * truncated surrogate pair). That failure is itself telemetry — well-formed clients
26
+ * do not emit it — so the caller gets `null` and the `raw` variant stays scannable
27
+ * instead of the whole scan aborting.
28
+ *
29
+ * @param input - Possibly percent-encoded string.
30
+ * @returns The decoded string, or `null` when the input contains no `%` or is not
31
+ * validly encoded.
32
+ */
33
+ declare function tryPercentDecode(input: string): string | null;
34
+ /**
35
+ * Build the representations to scan.
36
+ *
37
+ * Always includes `raw`. Adds `nfc`, `percent_decoded`, and `html_decoded` only when
38
+ * that transformation actually changes the string, so a plain ASCII value costs one
39
+ * pass rather than four.
40
+ *
41
+ * The original input is never mutated or replaced — callers keep the raw value for
42
+ * storage and display, and the variants exist solely so a rule can observe what the
43
+ * value would become at a decoding sink.
44
+ *
45
+ * @param input - Raw untrusted string.
46
+ * @returns One to four distinct representations, `raw` first.
47
+ *
48
+ * @example
49
+ * ```ts
50
+ * buildInputVariants("%2e%2e%2f");
51
+ * // [{ kind: "raw", value: "%2e%2e%2f" }, { kind: "percent_decoded", value: "../" }]
52
+ * ```
53
+ */
54
+ declare function buildInputVariants(input: string): readonly InputVariant[];
55
+ //#endregion
56
+ export { buildInputVariants, decodeHtmlEntities, tryPercentDecode };
57
+ //# sourceMappingURL=variants.d.mts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"variants.d.mts","names":[],"sources":["../../src/threats/variants.ts"],"mappings":";;;;;;;;;;;;;;;;;;;iBAqGgB,mBAAmB;;;;;;;;;;;;;iBA4CnB,iBAAiB;;;;;;;;;;;;;;;;;;;;;iBAiCjB,mBAAmB,yBAAyB"}
@@ -0,0 +1,144 @@
1
+ //#region src/threats/variants.ts
2
+ /**
3
+ * Named HTML entities worth decoding for detection purposes.
4
+ *
5
+ * Not a complete HTML5 entity table — there are roughly 2 200 of those — and not
6
+ * trying to be. It covers the characters that carry syntactic meaning in the sinks
7
+ * this package guards. Every other reference decodes to a character no rule looks
8
+ * for, so leaving it encoded cannot hide an attack.
9
+ */
10
+ const NAMED_ENTITIES = {
11
+ amp: "&",
12
+ lt: "<",
13
+ gt: ">",
14
+ quot: "\"",
15
+ apos: "'",
16
+ nbsp: " ",
17
+ sol: "/",
18
+ bsol: "\\",
19
+ colon: ":",
20
+ semi: ";",
21
+ lpar: "(",
22
+ rpar: ")",
23
+ lbrack: "[",
24
+ rbrack: "]",
25
+ lcub: "{",
26
+ rcub: "}",
27
+ excl: "!",
28
+ equals: "=",
29
+ dollar: "$",
30
+ commat: "@",
31
+ num: "#",
32
+ percnt: "%",
33
+ period: ".",
34
+ comma: ",",
35
+ ast: "*",
36
+ verbar: "|",
37
+ grave: "`",
38
+ Tab: " ",
39
+ NewLine: "\n"
40
+ };
41
+ /** Matches a numeric (decimal or hex) or named reference, with an optional semicolon. */
42
+ const ENTITY_PATTERN = /&(?:#[xX]([0-9a-fA-F]{1,6})|#(\d{1,7})|([a-zA-Z][a-zA-Z0-9]{1,31}));?/g;
43
+ /**
44
+ * Highest code point `String.fromCodePoint` accepts. Numeric references above this
45
+ * are left as written rather than throwing.
46
+ */
47
+ const MAX_CODE_POINT = 1114111;
48
+ /**
49
+ * Decode HTML character references in a single pass.
50
+ *
51
+ * Handles `&#60;`, `&#x3c;`, and the named subset in {@link NAMED_ENTITIES}, with or
52
+ * without the trailing semicolon — browsers accept several named references
53
+ * unterminated and attackers rely on it. Unrecognized references are left verbatim;
54
+ * this is best-effort detection support, not a rendering path.
55
+ *
56
+ * @param input - Raw string.
57
+ * @returns The string with recognized references replaced.
58
+ *
59
+ * @example
60
+ * ```ts
61
+ * decodeHtmlEntities("&#60;script&#62;"); // "<script>"
62
+ * decodeHtmlEntities("&unknownref;"); // "&unknownref;"
63
+ * ```
64
+ */
65
+ function decodeHtmlEntities(input) {
66
+ if (!input.includes("&")) return input;
67
+ return input.replace(ENTITY_PATTERN, (match, hex, dec, name) => {
68
+ if (hex !== void 0) {
69
+ const code = Number.parseInt(hex, 16);
70
+ return code <= MAX_CODE_POINT ? String.fromCodePoint(code) : match;
71
+ }
72
+ if (dec !== void 0) {
73
+ const code = Number.parseInt(dec, 10);
74
+ return code <= MAX_CODE_POINT ? String.fromCodePoint(code) : match;
75
+ }
76
+ if (name !== void 0) return Object.hasOwn(NAMED_ENTITIES, name) ? NAMED_ENTITIES[name] : match;
77
+ return match;
78
+ });
79
+ }
80
+ /**
81
+ * Percent-decode a string, tolerating malformed sequences.
82
+ *
83
+ * `decodeURIComponent` throws `URIError` on an invalid escape (`%zz`, a lone `%`, a
84
+ * truncated surrogate pair). That failure is itself telemetry — well-formed clients
85
+ * do not emit it — so the caller gets `null` and the `raw` variant stays scannable
86
+ * instead of the whole scan aborting.
87
+ *
88
+ * @param input - Possibly percent-encoded string.
89
+ * @returns The decoded string, or `null` when the input contains no `%` or is not
90
+ * validly encoded.
91
+ */
92
+ function tryPercentDecode(input) {
93
+ if (!input.includes("%")) return null;
94
+ try {
95
+ return decodeURIComponent(input);
96
+ } catch {
97
+ return null;
98
+ }
99
+ }
100
+ /**
101
+ * Build the representations to scan.
102
+ *
103
+ * Always includes `raw`. Adds `nfc`, `percent_decoded`, and `html_decoded` only when
104
+ * that transformation actually changes the string, so a plain ASCII value costs one
105
+ * pass rather than four.
106
+ *
107
+ * The original input is never mutated or replaced — callers keep the raw value for
108
+ * storage and display, and the variants exist solely so a rule can observe what the
109
+ * value would become at a decoding sink.
110
+ *
111
+ * @param input - Raw untrusted string.
112
+ * @returns One to four distinct representations, `raw` first.
113
+ *
114
+ * @example
115
+ * ```ts
116
+ * buildInputVariants("%2e%2e%2f");
117
+ * // [{ kind: "raw", value: "%2e%2e%2f" }, { kind: "percent_decoded", value: "../" }]
118
+ * ```
119
+ */
120
+ function buildInputVariants(input) {
121
+ const variants = [{
122
+ kind: "raw",
123
+ value: input
124
+ }];
125
+ const seen = /* @__PURE__ */ new Set([input]);
126
+ const add = (kind, value) => {
127
+ if (seen.has(value)) return;
128
+ seen.add(value);
129
+ variants.push({
130
+ kind,
131
+ value
132
+ });
133
+ };
134
+ add("nfc", input.normalize("NFC"));
135
+ add("nfkc", input.normalize("NFKC"));
136
+ const percentDecoded = tryPercentDecode(input);
137
+ if (percentDecoded !== null) add("percent_decoded", percentDecoded);
138
+ add("html_decoded", decodeHtmlEntities(input));
139
+ return variants;
140
+ }
141
+ //#endregion
142
+ export { buildInputVariants, decodeHtmlEntities, tryPercentDecode };
143
+
144
+ //# sourceMappingURL=variants.mjs.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"variants.mjs","names":[],"sources":["../../src/threats/variants.ts"],"sourcesContent":["/**\n * Copyright 2026 ResQ Systems, Inc.\n *\n * Licensed under the Apache License, Version 2.0 (the \"License\");\n * you may not use this file except in compliance with the License.\n * You may obtain a copy of the License at\n *\n * http://www.apache.org/licenses/LICENSE-2.0\n *\n * Unless required by applicable law or agreed to in writing, software\n * distributed under the License is distributed on an \"AS IS\" BASIS,\n * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n * See the License for the specific language governing permissions and\n * limitations under the License.\n */\n\n/**\n * @fileoverview Canonicalization variants — the small, bounded set of representations\n * the scanner evaluates so signatures are not trivially bypassed by encoding.\n *\n * The set is deliberately small and non-recursive. Decoding everything repeatedly\n * until it stops changing is its own bug class: it makes the scanner disagree with\n * the downstream component about what the value actually is, and it lets an attacker\n * pick whichever of several disagreeing interpretations suits them. Instead, each\n * variant is one transformation applied once, findings record which variant matched,\n * and payloads that only surface after multiple decodes get explicit rules such as\n * `PATH-TRAVERSAL-DOUBLE-ENCODED-001`.\n *\n * @module @resq-systems/security/threats/variants\n */\n\nimport type { InputVariant, InputVariantKind } from \"./types.js\";\n\n//#region HTML entity decoding\n\n/**\n * Named HTML entities worth decoding for detection purposes.\n *\n * Not a complete HTML5 entity table — there are roughly 2 200 of those — and not\n * trying to be. It covers the characters that carry syntactic meaning in the sinks\n * this package guards. Every other reference decodes to a character no rule looks\n * for, so leaving it encoded cannot hide an attack.\n */\nconst NAMED_ENTITIES: Readonly<Record<string, string>> = {\n\tamp: \"&\",\n\tlt: \"<\",\n\tgt: \">\",\n\tquot: '\"',\n\tapos: \"'\",\n\tnbsp: \" \",\n\tsol: \"/\",\n\tbsol: \"\\\\\",\n\tcolon: \":\",\n\tsemi: \";\",\n\tlpar: \"(\",\n\trpar: \")\",\n\tlbrack: \"[\",\n\trbrack: \"]\",\n\tlcub: \"{\",\n\trcub: \"}\",\n\texcl: \"!\",\n\tequals: \"=\",\n\tdollar: \"$\",\n\tcommat: \"@\",\n\tnum: \"#\",\n\tpercnt: \"%\",\n\tperiod: \".\",\n\tcomma: \",\",\n\tast: \"*\",\n\tverbar: \"|\",\n\tgrave: \"`\",\n\tTab: \"\\t\",\n\tNewLine: \"\\n\",\n};\n\n/** Matches a numeric (decimal or hex) or named reference, with an optional semicolon. */\nconst ENTITY_PATTERN = /&(?:#[xX]([0-9a-fA-F]{1,6})|#(\\d{1,7})|([a-zA-Z][a-zA-Z0-9]{1,31}));?/g;\n\n/**\n * Highest code point `String.fromCodePoint` accepts. Numeric references above this\n * are left as written rather than throwing.\n */\nconst MAX_CODE_POINT = 0x10ffff;\n\n/**\n * Decode HTML character references in a single pass.\n *\n * Handles `&#60;`, `&#x3c;`, and the named subset in {@link NAMED_ENTITIES}, with or\n * without the trailing semicolon — browsers accept several named references\n * unterminated and attackers rely on it. Unrecognized references are left verbatim;\n * this is best-effort detection support, not a rendering path.\n *\n * @param input - Raw string.\n * @returns The string with recognized references replaced.\n *\n * @example\n * ```ts\n * decodeHtmlEntities(\"&#60;script&#62;\"); // \"<script>\"\n * decodeHtmlEntities(\"&unknownref;\"); // \"&unknownref;\"\n * ```\n */\nexport function decodeHtmlEntities(input: string): string {\n\tif (!input.includes(\"&\")) return input;\n\n\t// `String.prototype.replace` resets a global pattern's lastIndex on entry, so the\n\t// module-scoped ENTITY_PATTERN carries no state between calls.\n\treturn input.replace(ENTITY_PATTERN, (match, hex?: string, dec?: string, name?: string) => {\n\t\tif (hex !== undefined) {\n\t\t\tconst code = Number.parseInt(hex, 16);\n\t\t\treturn code <= MAX_CODE_POINT ? String.fromCodePoint(code) : match;\n\t\t}\n\t\tif (dec !== undefined) {\n\t\t\tconst code = Number.parseInt(dec, 10);\n\t\t\treturn code <= MAX_CODE_POINT ? String.fromCodePoint(code) : match;\n\t\t}\n\t\tif (name !== undefined) {\n\t\t\t// Own-property check, not `?? match`: the name group matches `constructor`,\n\t\t\t// `toString`, `valueOf` and friends, which resolve up the prototype chain to a\n\t\t\t// function. `??` never fires for those, and `replace` then coerces the function\n\t\t\t// to its source text — `&constructor;` decoded to\n\t\t\t// `function Object() { [native code] }`, injecting braces and parens that trip\n\t\t\t// unrelated rules and corrupting the `html_decoded` variant for any input\n\t\t\t// carrying such a reference.\n\t\t\treturn Object.hasOwn(NAMED_ENTITIES, name) ? NAMED_ENTITIES[name] : match;\n\t\t}\n\t\treturn match;\n\t});\n}\n\n//#endregion\n\n//#region Percent decoding\n\n/**\n * Percent-decode a string, tolerating malformed sequences.\n *\n * `decodeURIComponent` throws `URIError` on an invalid escape (`%zz`, a lone `%`, a\n * truncated surrogate pair). That failure is itself telemetry — well-formed clients\n * do not emit it — so the caller gets `null` and the `raw` variant stays scannable\n * instead of the whole scan aborting.\n *\n * @param input - Possibly percent-encoded string.\n * @returns The decoded string, or `null` when the input contains no `%` or is not\n * validly encoded.\n */\nexport function tryPercentDecode(input: string): string | null {\n\tif (!input.includes(\"%\")) return null;\n\ttry {\n\t\treturn decodeURIComponent(input);\n\t} catch {\n\t\treturn null;\n\t}\n}\n\n//#endregion\n\n//#region Variant construction\n\n/**\n * Build the representations to scan.\n *\n * Always includes `raw`. Adds `nfc`, `percent_decoded`, and `html_decoded` only when\n * that transformation actually changes the string, so a plain ASCII value costs one\n * pass rather than four.\n *\n * The original input is never mutated or replaced — callers keep the raw value for\n * storage and display, and the variants exist solely so a rule can observe what the\n * value would become at a decoding sink.\n *\n * @param input - Raw untrusted string.\n * @returns One to four distinct representations, `raw` first.\n *\n * @example\n * ```ts\n * buildInputVariants(\"%2e%2e%2f\");\n * // [{ kind: \"raw\", value: \"%2e%2e%2f\" }, { kind: \"percent_decoded\", value: \"../\" }]\n * ```\n */\nexport function buildInputVariants(input: string): readonly InputVariant[] {\n\tconst variants: InputVariant[] = [{ kind: \"raw\", value: input }];\n\tconst seen = new Set<string>([input]);\n\n\tconst add = (kind: InputVariantKind, value: string): void => {\n\t\tif (seen.has(value)) return;\n\t\tseen.add(value);\n\t\tvariants.push({ kind, value });\n\t};\n\n\t// `normalize` throws only on an invalid form argument, never on content.\n\tadd(\"nfc\", input.normalize(\"NFC\"));\n\n\t// NFC is a documented no-op on compatibility characters, so fullwidth forms slipped\n\t// past every signature: `../../etc/passwd` and `<script>` both scanned clean.\n\t// NFKC folds U+FF01–FF5E onto ASCII, which is exactly the mapping a downstream\n\t// component performs when it normalizes before parsing. Kept as its own variant\n\t// rather than replacing `nfc`, because NFKC is lossy and must never be what the\n\t// caller stores.\n\tadd(\"nfkc\", input.normalize(\"NFKC\"));\n\n\tconst percentDecoded = tryPercentDecode(input);\n\tif (percentDecoded !== null) {\n\t\tadd(\"percent_decoded\", percentDecoded);\n\t}\n\n\tadd(\"html_decoded\", decodeHtmlEntities(input));\n\n\treturn variants;\n}\n\n//#endregion\n"],"mappings":";;;;;;;;;AA2CA,MAAM,iBAAmD;CACxD,KAAK;CACL,IAAI;CACJ,IAAI;CACJ,MAAM;CACN,MAAM;CACN,MAAM;CACN,KAAK;CACL,MAAM;CACN,OAAO;CACP,MAAM;CACN,MAAM;CACN,MAAM;CACN,QAAQ;CACR,QAAQ;CACR,MAAM;CACN,MAAM;CACN,MAAM;CACN,QAAQ;CACR,QAAQ;CACR,QAAQ;CACR,KAAK;CACL,QAAQ;CACR,QAAQ;CACR,OAAO;CACP,KAAK;CACL,QAAQ;CACR,OAAO;CACP,KAAK;CACL,SAAS;AACV;;AAGA,MAAM,iBAAiB;;;;;AAMvB,MAAM,iBAAiB;;;;;;;;;;;;;;;;;;AAmBvB,SAAgB,mBAAmB,OAAuB;CACzD,IAAI,CAAC,MAAM,SAAS,GAAG,GAAG,OAAO;CAIjC,OAAO,MAAM,QAAQ,iBAAiB,OAAO,KAAc,KAAc,SAAkB;EAC1F,IAAI,QAAQ,KAAA,GAAW;GACtB,MAAM,OAAO,OAAO,SAAS,KAAK,EAAE;GACpC,OAAO,QAAQ,iBAAiB,OAAO,cAAc,IAAI,IAAI;EAC9D;EACA,IAAI,QAAQ,KAAA,GAAW;GACtB,MAAM,OAAO,OAAO,SAAS,KAAK,EAAE;GACpC,OAAO,QAAQ,iBAAiB,OAAO,cAAc,IAAI,IAAI;EAC9D;EACA,IAAI,SAAS,KAAA,GAQZ,OAAO,OAAO,OAAO,gBAAgB,IAAI,IAAI,eAAe,QAAQ;EAErE,OAAO;CACR,CAAC;AACF;;;;;;;;;;;;;AAkBA,SAAgB,iBAAiB,OAA8B;CAC9D,IAAI,CAAC,MAAM,SAAS,GAAG,GAAG,OAAO;CACjC,IAAI;EACH,OAAO,mBAAmB,KAAK;CAChC,QAAQ;EACP,OAAO;CACR;AACD;;;;;;;;;;;;;;;;;;;;;AA0BA,SAAgB,mBAAmB,OAAwC;CAC1E,MAAM,WAA2B,CAAC;EAAE,MAAM;EAAO,OAAO;CAAM,CAAC;CAC/D,MAAM,uBAAO,IAAI,IAAY,CAAC,KAAK,CAAC;CAEpC,MAAM,OAAO,MAAwB,UAAwB;EAC5D,IAAI,KAAK,IAAI,KAAK,GAAG;EACrB,KAAK,IAAI,KAAK;EACd,SAAS,KAAK;GAAE;GAAM;EAAM,CAAC;CAC9B;CAGA,IAAI,OAAO,MAAM,UAAU,KAAK,CAAC;CAQjC,IAAI,QAAQ,MAAM,UAAU,MAAM,CAAC;CAEnC,MAAM,iBAAiB,iBAAiB,KAAK;CAC7C,IAAI,mBAAmB,MACtB,IAAI,mBAAmB,cAAc;CAGtC,IAAI,gBAAgB,mBAAmB,KAAK,CAAC;CAE7C,OAAO;AACR"}
@@ -0,0 +1,82 @@
1
+ //#region src/unicode/confusables.d.ts
2
+ /**
3
+ * Copyright 2026 ResQ Systems, Inc.
4
+ *
5
+ * Licensed under the Apache License, Version 2.0 (the "License");
6
+ * you may not use this file except in compliance with the License.
7
+ * You may obtain a copy of the License at
8
+ *
9
+ * http://www.apache.org/licenses/LICENSE-2.0
10
+ *
11
+ * Unless required by applicable law or agreed to in writing, software
12
+ * distributed under the License is distributed on an "AS IS" BASIS,
13
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
14
+ * See the License for the specific language governing permissions and
15
+ * limitations under the License.
16
+ */
17
+ /**
18
+ * Flattened lookup: confusable code point to its ASCII prototype.
19
+ *
20
+ * Later rows win on collision. That is intentional and harmless — a code point listed
21
+ * under two prototypes is confusable with both, so either answer yields the same
22
+ * collision behavior provided the choice is deterministic, which it is.
23
+ */
24
+ declare const CONFUSABLE_MAP: ReadonlyMap<number, string>;
25
+ /**
26
+ * Generate a UTS #39-style confusable skeleton.
27
+ *
28
+ * Pipeline: decompose to NFD, drop invisible and combining code points, fold each
29
+ * remaining code point through the confusable tables, recompose to NFC.
30
+ *
31
+ * Two strings with equal skeletons are visually confusable. That is *all* an equal
32
+ * skeleton means — in particular it does not mean the strings are equivalent, and the
33
+ * skeleton is neither a display nor a storage form. Accents are dropped and the
34
+ * `I/l/1` and `O/o/0` families collapse across case, so `José`, `Jose`, and `J0sé`
35
+ * share a skeleton by design.
36
+ *
37
+ * @param input - Raw string. Non-string or empty input yields `""`.
38
+ * @returns The comparison key.
39
+ *
40
+ * @example
41
+ * ```ts
42
+ * // Cyrillic а in an otherwise Latin string.
43
+ * getSkeleton("pаypal") === getSkeleton("paypal"); // true
44
+ * getSkeleton("paypaI") === getSkeleton("paypal"); // true — I folds to l
45
+ * ```
46
+ */
47
+ declare function getSkeleton(input: string): string;
48
+ /**
49
+ * Map non-ASCII lookalike characters onto their ASCII prototypes, preserving
50
+ * everything else.
51
+ *
52
+ * Unlike {@link getSkeleton} this is a *conservative* transform: accents and other
53
+ * combining marks survive, ASCII characters are never rewritten, and the result stays
54
+ * readable. `Ολγα` keeps its Greek letters only insofar as they are not Latin
55
+ * lookalikes; `café` stays `café`.
56
+ *
57
+ * Even so, prefer {@link getSkeleton} for collision checks and keep the original for
58
+ * display. Rewriting a user's identifier into a different string is a lossy operation
59
+ * that this function can only make *look* safe.
60
+ *
61
+ * @param input - Raw string. Non-string or empty input yields `""`.
62
+ * @returns NFC-composed string with non-ASCII confusables folded to ASCII.
63
+ *
64
+ * @example
65
+ * ```ts
66
+ * foldConfusables("pаypal"); // "paypal" — Cyrillic а folded
67
+ * foldConfusables("café"); // "café" — accent preserved
68
+ * foldConfusables("HELLO"); // "HELLO" — ASCII untouched
69
+ * ```
70
+ */
71
+ declare function foldConfusables(input: string): string;
72
+ /**
73
+ * Test whether two distinct strings are visually confusable.
74
+ *
75
+ * @param left - First string.
76
+ * @param right - Second string.
77
+ * @returns `true` when the strings differ but their skeletons match.
78
+ */
79
+ declare function areConfusable(left: string, right: string): boolean;
80
+ //#endregion
81
+ export { CONFUSABLE_MAP, areConfusable, foldConfusables, getSkeleton };
82
+ //# sourceMappingURL=confusables.d.mts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"confusables.d.mts","names":[],"sources":["../../src/unicode/confusables.ts"],"mappings":";;;;;;;;;;;;;;;;;;;;;;;cA6Ka,gBAAgB;;;;;;;;;;;;;;;;;;;;;;;iBAsKb,YAAY;;;;;;;;;;;;;;;;;;;;;;;;iBAyCZ,gBAAgB;;;;;;;;iBAsBhB,cAAc,cAAc"}