@endevops/effect-codec-xml 0.0.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/LICENSE-is-entities +21 -0
- package/LICENSE-is-xml-naming +21 -0
- package/README.md +415 -0
- package/dist/codec.d.ts +48 -0
- package/dist/codec.d.ts.map +1 -0
- package/dist/codec.js +63 -0
- package/dist/codec.js.map +1 -0
- package/dist/conventions.d.ts +88 -0
- package/dist/conventions.d.ts.map +1 -0
- package/dist/conventions.js +113 -0
- package/dist/conventions.js.map +1 -0
- package/dist/entities/entity-decoder.d.ts +333 -0
- package/dist/entities/entity-decoder.d.ts.map +1 -0
- package/dist/entities/entity-decoder.js +841 -0
- package/dist/entities/entity-decoder.js.map +1 -0
- package/dist/entities/entity-tables.js +16 -0
- package/dist/entities/entity-tables.js.map +1 -0
- package/dist/errors.d.ts +49 -0
- package/dist/errors.d.ts.map +1 -0
- package/dist/errors.js +48 -0
- package/dist/errors.js.map +1 -0
- package/dist/index.d.ts +11 -0
- package/dist/index.js +11 -0
- package/dist/namespaces.d.ts +101 -0
- package/dist/namespaces.d.ts.map +1 -0
- package/dist/namespaces.js +663 -0
- package/dist/namespaces.js.map +1 -0
- package/dist/naming.d.ts +149 -0
- package/dist/naming.d.ts.map +1 -0
- package/dist/naming.js +296 -0
- package/dist/naming.js.map +1 -0
- package/dist/parse.d.ts +75 -0
- package/dist/parse.d.ts.map +1 -0
- package/dist/parse.js +437 -0
- package/dist/parse.js.map +1 -0
- package/dist/render.d.ts +99 -0
- package/dist/render.d.ts.map +1 -0
- package/dist/render.js +509 -0
- package/dist/render.js.map +1 -0
- package/dist/xml-error.d.ts +172 -0
- package/dist/xml-error.d.ts.map +1 -0
- package/dist/xml-error.js +157 -0
- package/dist/xml-error.js.map +1 -0
- package/dist/xml-value.d.ts +42 -0
- package/dist/xml-value.d.ts.map +1 -0
- package/dist/xml-value.js +79 -0
- package/dist/xml-value.js.map +1 -0
- package/package.json +69 -0
- package/src/codec.ts +136 -0
- package/src/conventions.ts +145 -0
- package/src/entities/entity-decoder.ts +1248 -0
- package/src/entities/entity-tables.ts +18 -0
- package/src/errors.ts +55 -0
- package/src/index.ts +79 -0
- package/src/namespaces.ts +968 -0
- package/src/naming.ts +519 -0
- package/src/parse.ts +597 -0
- package/src/render.ts +708 -0
- package/src/xml-error.ts +168 -0
- package/src/xml-value.ts +108 -0
|
@@ -0,0 +1,841 @@
|
|
|
1
|
+
import { XmlError } from "../xml-error.js";
|
|
2
|
+
import { XML } from "./entity-tables.js";
|
|
3
|
+
import { Effect, Match, Predicate } from "effect";
|
|
4
|
+
//#region src/entities/entity-decoder.ts
|
|
5
|
+
const CODE_AMPERSAND = 38;
|
|
6
|
+
const CODE_SEMICOLON = 59;
|
|
7
|
+
const CODE_HASH = 35;
|
|
8
|
+
const CODE_LOWER_X = 120;
|
|
9
|
+
const CODE_UPPER_X = 88;
|
|
10
|
+
/**
|
|
11
|
+
* @description The widest entity name {@link EntityDecoder.decode} will look for, in characters. The forward scan for `;` stops once more than this many characters
|
|
12
|
+
* have passed since the `&`, so a longer run is not treated as one entity — it is copied through as literal text. The bound is what keeps a document
|
|
13
|
+
* with a megabyte of non-entity text between two ampersands from being sliced.
|
|
14
|
+
*/
|
|
15
|
+
const MAX_TOKEN_LENGTH = 32;
|
|
16
|
+
/**
|
|
17
|
+
* @description The largest codepoint `String.fromCodePoint` accepts, and the bound a numeric reference is checked against before it gets that far. Out of range is
|
|
18
|
+
* `leave`, not `remove`: the reference is preserved verbatim rather than deleted.
|
|
19
|
+
*/
|
|
20
|
+
const MAX_CODE_POINT = 1114111;
|
|
21
|
+
/**
|
|
22
|
+
* @description Returned by `#classifyNCR` for a codepoint that carries no minimum action level, which is what distinguishes "no restriction" from
|
|
23
|
+
* `NCR_LEVEL.allow` — both end up expanding, but only the first lets `numericAllowed: false` short-circuit the whole pipeline.
|
|
24
|
+
*/
|
|
25
|
+
const NO_MINIMUM_LEVEL = -1;
|
|
26
|
+
/**
|
|
27
|
+
* @description Characters that may not appear in an entity name registered through {@link EntityDecoder.setExternalEntities} or
|
|
28
|
+
* {@link EntityDecoder.addExternalEntity}. A name carrying one of these cannot be written as `&name;` at all, so registration refuses it rather than
|
|
29
|
+
* storing a name no document could ever reference. The set is the upstream string verbatim, including its duplicated backslash — a `Set` discards the
|
|
30
|
+
* duplicate, so the effective set is the eighteen characters below.
|
|
31
|
+
*/
|
|
32
|
+
const SPECIAL_CHARS = /* @__PURE__ */ new Set("!?\\/[]$%{}^&*()<>|+");
|
|
33
|
+
/**
|
|
34
|
+
* @description Injected at runtime: DOCTYPE entities for the current document, and persistent external entities. The untrusted tier, and the only one the
|
|
35
|
+
* expansion limits count by default.
|
|
36
|
+
*/
|
|
37
|
+
const LIMIT_TIER_EXTERNAL = "external";
|
|
38
|
+
/**
|
|
39
|
+
* @description Trusted: the five XML predefined entities, the caller's `namedEntities`, and every numeric reference.
|
|
40
|
+
*/
|
|
41
|
+
const LIMIT_TIER_BASE = "base";
|
|
42
|
+
/**
|
|
43
|
+
* @description Not a tier an entity belongs to but a switch on the tier filter. Selecting it makes every entity count against the limits regardless of where it
|
|
44
|
+
* came from.
|
|
45
|
+
*/
|
|
46
|
+
const LIMIT_TIER_ALL = "all";
|
|
47
|
+
/**
|
|
48
|
+
* @description The NCR action levels, in severity order. A higher number is a stricter action, and the resolver takes the maximum of the configured level and the
|
|
49
|
+
* minimum a codepoint range imposes, so a range can only ever make an entity stricter than the caller asked for — never more lenient.
|
|
50
|
+
*/
|
|
51
|
+
const NCR_LEVEL = Object.freeze({
|
|
52
|
+
allow: 0,
|
|
53
|
+
leave: 1,
|
|
54
|
+
remove: 2,
|
|
55
|
+
throw: 3
|
|
56
|
+
});
|
|
57
|
+
/**
|
|
58
|
+
* @description The C0 control codes XML 1.0 §2.2 permits as literal characters. Every other code in U+0001–U+001F is prohibited.
|
|
59
|
+
*/
|
|
60
|
+
const XML10_ALLOWED_C0 = /* @__PURE__ */ new Set([
|
|
61
|
+
9,
|
|
62
|
+
10,
|
|
63
|
+
13
|
|
64
|
+
]);
|
|
65
|
+
/**
|
|
66
|
+
* @description The three actions a registration hook can return, as a frozen object. Prefer it over the bare strings: the literals stay narrow string-literal
|
|
67
|
+
* types, so `ENTITY_ACTION.BLOK` fails to compile instead of silently registering the entity.
|
|
68
|
+
*
|
|
69
|
+
* @example
|
|
70
|
+
* ```typescript
|
|
71
|
+
* const decoder = new EntityDecoder({
|
|
72
|
+
* onInputEntity: () => ENTITY_ACTION.BLOCK,
|
|
73
|
+
* });
|
|
74
|
+
* ```;
|
|
75
|
+
*/
|
|
76
|
+
const ENTITY_ACTION = Object.freeze({
|
|
77
|
+
ALLOW: "allow",
|
|
78
|
+
BLOCK: "block",
|
|
79
|
+
THROW: "throw"
|
|
80
|
+
});
|
|
81
|
+
/**
|
|
82
|
+
* @description Reject an entity name that could never be written as a reference. `#` is refused positionally rather than by the character sweep, because a name
|
|
83
|
+
* starting with `#` is a numeric reference's token and would collide with `#resolveNCR`. Everything else is refused per character.
|
|
84
|
+
*
|
|
85
|
+
* @param name - The name to check.
|
|
86
|
+
*
|
|
87
|
+
* @returns An effect producing the name, unchanged, so the call can be inlined. Fails with {@link XmlError} and the `InvalidEntityName` reason,
|
|
88
|
+
* carrying the offending character. The `[EntityReplacer]` prefix in the message is preserved verbatim from the original throw, despite naming a
|
|
89
|
+
* class this decoder does not have — it is load-bearing for anything matching on it.
|
|
90
|
+
*/
|
|
91
|
+
const checkEntityName = (name) => {
|
|
92
|
+
if (name.charCodeAt(0) === CODE_HASH) return Effect.fail(new XmlError({
|
|
93
|
+
reason: {
|
|
94
|
+
_tag: "InvalidEntityName",
|
|
95
|
+
name,
|
|
96
|
+
character: "#"
|
|
97
|
+
},
|
|
98
|
+
message: `[EntityReplacer] Invalid character '#' in entity name: "${name}"`
|
|
99
|
+
}));
|
|
100
|
+
for (const ch of name) if (SPECIAL_CHARS.has(ch)) return Effect.fail(new XmlError({
|
|
101
|
+
reason: {
|
|
102
|
+
_tag: "InvalidEntityName",
|
|
103
|
+
name,
|
|
104
|
+
character: ch
|
|
105
|
+
},
|
|
106
|
+
message: `[EntityReplacer] Invalid character '${ch}' in entity name: "${name}"`
|
|
107
|
+
}));
|
|
108
|
+
return Effect.succeed(name);
|
|
109
|
+
};
|
|
110
|
+
/**
|
|
111
|
+
* @description Flatten registration maps into one name to string map, later maps winning over earlier ones for the same name. The result is a null-prototype
|
|
112
|
+
* object, not a `Map`. That is not incidental: a `Map` iterates in pure insertion order, while `Object.keys` lifts array-index-like names to the
|
|
113
|
+
* front in numeric order, and the registration hooks observe that order. A name of `"2"` registered after `"brand"` reaches the hook first here and
|
|
114
|
+
* second in a `Map`.
|
|
115
|
+
*
|
|
116
|
+
* @param maps - The maps to merge. A falsy entry — `null`, `undefined`, `''`, `0` — contributes nothing rather than throwing.
|
|
117
|
+
*
|
|
118
|
+
* @returns A null-prototype object of own string-valued entries. Nothing from `Object.prototype` can be read out of it, so a document naming
|
|
119
|
+
* `constructor` or `toString` finds nothing. Each entry is read through {@link flattenEntityValue}, so an entry that cannot be reduced to a string
|
|
120
|
+
* is absent rather than present-and-unusable.
|
|
121
|
+
*/
|
|
122
|
+
function mergeEntityMaps(...maps) {
|
|
123
|
+
const out = Object.create(null);
|
|
124
|
+
for (const map of maps) {
|
|
125
|
+
if (!map) continue;
|
|
126
|
+
for (const key of Object.keys(map)) {
|
|
127
|
+
const value = flattenEntityValue(map[key]);
|
|
128
|
+
if (value !== void 0) out[key] = value;
|
|
129
|
+
}
|
|
130
|
+
}
|
|
131
|
+
return out;
|
|
132
|
+
}
|
|
133
|
+
/**
|
|
134
|
+
* @description Reduce one registration entry to the string a reference to it expands to, or to nothing when the entry is a form the scanner has no use for. Three
|
|
135
|
+
* shapes survive: the string itself, and a `{ regex | regx, val }` envelope whose `val` is a string. Everything else — a number, `null`, `undefined`,
|
|
136
|
+
* a bare function, an envelope whose `val` is a function — has no string to substitute, so the name is dropped and a reference to it comes back out
|
|
137
|
+
* as the text it was written as. Dropping is silent on purpose: the runtime inspects whatever it is handed, and failing the construction over one
|
|
138
|
+
* unreadable entry would take every other entity in the table down with it.
|
|
139
|
+
*
|
|
140
|
+
* @param raw - The entry as it arrived, in whatever shape the caller supplied it — including no entry at all, which a table with a hole in it
|
|
141
|
+
* produces.
|
|
142
|
+
*
|
|
143
|
+
* @returns The replacement string, or `undefined` when the entry cannot be read. A name registered to the empty string yields `''`, which is why
|
|
144
|
+
* callers compare against `undefined` rather than testing for emptiness.
|
|
145
|
+
*/
|
|
146
|
+
function flattenEntityValue(raw) {
|
|
147
|
+
if (Predicate.isString(raw)) return raw;
|
|
148
|
+
if (Predicate.isNullish(raw) || !Predicate.isObject(raw) || raw.val === void 0) return void 0;
|
|
149
|
+
const val = raw.val;
|
|
150
|
+
return Predicate.isString(val) ? val : void 0;
|
|
151
|
+
}
|
|
152
|
+
/**
|
|
153
|
+
* @description Read one own entry out of a null-prototype entity map.
|
|
154
|
+
*
|
|
155
|
+
* @param map - The map to read.
|
|
156
|
+
* @param key - The entity name.
|
|
157
|
+
*
|
|
158
|
+
* @returns The registered string, or `undefined` when the name is not an own key. A name registered to the empty string returns `''`, which is why
|
|
159
|
+
* callers must compare against `undefined` rather than test for emptiness.
|
|
160
|
+
*/
|
|
161
|
+
function ownEntity(map, key) {
|
|
162
|
+
if (!Object.hasOwn(map, key)) return void 0;
|
|
163
|
+
return map[key];
|
|
164
|
+
}
|
|
165
|
+
/**
|
|
166
|
+
* @description Normalise the `applyLimitsTo` option into the set of tiers that count against the limits.
|
|
167
|
+
*
|
|
168
|
+
* @param raw - The configured value.
|
|
169
|
+
*
|
|
170
|
+
* @returns The tier set. An unrecognised string falls back to `external` rather than to no filtering at all, so a typo cannot silently disable the
|
|
171
|
+
* limits. An array is taken as given, which is why an empty array disables limit accounting entirely while an empty string falls back to
|
|
172
|
+
* `external`.
|
|
173
|
+
*/
|
|
174
|
+
function parseLimitTiers(raw) {
|
|
175
|
+
if (!raw || raw === LIMIT_TIER_EXTERNAL) return /* @__PURE__ */ new Set([LIMIT_TIER_EXTERNAL]);
|
|
176
|
+
if (raw === LIMIT_TIER_ALL) return /* @__PURE__ */ new Set([LIMIT_TIER_ALL]);
|
|
177
|
+
if (raw === LIMIT_TIER_BASE) return /* @__PURE__ */ new Set([LIMIT_TIER_BASE]);
|
|
178
|
+
if (Array.isArray(raw)) return new Set(raw);
|
|
179
|
+
return /* @__PURE__ */ new Set([LIMIT_TIER_EXTERNAL]);
|
|
180
|
+
}
|
|
181
|
+
/**
|
|
182
|
+
* @description Read one level out of {@link NCR_LEVEL} by name.
|
|
183
|
+
*
|
|
184
|
+
* @param name - The configured action name, or nothing.
|
|
185
|
+
* @param fallback - The level to use when the name is absent.
|
|
186
|
+
*
|
|
187
|
+
* @returns The level. A name the table does not carry also yields `fallback`, so a value outside the union degrades to the default action rather than
|
|
188
|
+
* to `NaN`.
|
|
189
|
+
*/
|
|
190
|
+
function ncrLevelOf(name, fallback) {
|
|
191
|
+
if (name === void 0) return fallback;
|
|
192
|
+
return NCR_LEVEL[name] ?? fallback;
|
|
193
|
+
}
|
|
194
|
+
/**
|
|
195
|
+
* @description Flatten the `ncr` option into the three numeric fields the decode loop reads, so nothing has to be re-derived per reference.
|
|
196
|
+
*
|
|
197
|
+
* @param ncr - The configured policy, or nothing.
|
|
198
|
+
*
|
|
199
|
+
* @returns The XML version, the base action level, and the null action level already clamped up to `remove`.
|
|
200
|
+
*/
|
|
201
|
+
function parseNCRConfig(ncr) {
|
|
202
|
+
if (!ncr) return {
|
|
203
|
+
xmlVersion: 1,
|
|
204
|
+
onLevel: NCR_LEVEL.allow,
|
|
205
|
+
nullLevel: NCR_LEVEL.remove
|
|
206
|
+
};
|
|
207
|
+
return {
|
|
208
|
+
xmlVersion: ncr.xmlVersion === 1.1 ? 1.1 : 1,
|
|
209
|
+
onLevel: ncrLevelOf(ncr.onNCR, NCR_LEVEL.allow),
|
|
210
|
+
nullLevel: Math.max(ncrLevelOf(ncr.nullNCR, NCR_LEVEL.remove), NCR_LEVEL.remove)
|
|
211
|
+
};
|
|
212
|
+
}
|
|
213
|
+
/**
|
|
214
|
+
* @description Resolve {@link EntityDecoderOptions.postCheck} to something the decode loop can call unconditionally, or to the identity function. The fallback
|
|
215
|
+
* exists so the path that actually scanned does not have to test for the option, while the two fast paths that return before the scan still skip the
|
|
216
|
+
* call entirely. A non-function value is treated as an absent option rather than rejected, matching the hook rules: a mistyped option disables its
|
|
217
|
+
* feature instead of failing the construction.
|
|
218
|
+
*
|
|
219
|
+
* @param raw - The configured hook, or nothing.
|
|
220
|
+
*
|
|
221
|
+
* @returns The hook itself, or a function returning its first argument.
|
|
222
|
+
*/
|
|
223
|
+
function readPostCheck(raw) {
|
|
224
|
+
if (Predicate.isFunction(raw)) return raw;
|
|
225
|
+
return (r) => r;
|
|
226
|
+
}
|
|
227
|
+
/**
|
|
228
|
+
* @description Resolve a registration hook option to something safe to call, under the same non-function rule as {@link readPostCheck}.
|
|
229
|
+
*
|
|
230
|
+
* @param raw - The configured hook, or nothing.
|
|
231
|
+
*
|
|
232
|
+
* @returns The hook itself, or `null` for an absent option and for a value that is not a function. `null` is what lets every registration path ask
|
|
233
|
+
* unconditionally: a hook that is not there accepts.
|
|
234
|
+
*/
|
|
235
|
+
function readHook(raw) {
|
|
236
|
+
if (Predicate.isFunction(raw)) return raw;
|
|
237
|
+
return null;
|
|
238
|
+
}
|
|
239
|
+
/**
|
|
240
|
+
* @description Read one of the two entity-name lists as a set, under the same missing-value rule as the other options: absent is empty, not an error. The
|
|
241
|
+
* `Array.isArray` test rather than a truthiness one is what keeps a mistyped list from reaching `new Set` and throwing there, so a caller's typo
|
|
242
|
+
* disables the list instead of taking the decoder down. Matching against a set is also why a name in both lists is decided by the order
|
|
243
|
+
* {@link EntityDecoder.decode} consults them in, not by the order the caller wrote them in.
|
|
244
|
+
*
|
|
245
|
+
* @param raw - The configured list, or nothing.
|
|
246
|
+
*
|
|
247
|
+
* @returns The names as a set, empty when the option is absent or is not an array.
|
|
248
|
+
*/
|
|
249
|
+
function readNameList(raw) {
|
|
250
|
+
if (Array.isArray(raw)) return new Set(raw);
|
|
251
|
+
return /* @__PURE__ */ new Set();
|
|
252
|
+
}
|
|
253
|
+
/**
|
|
254
|
+
* @description Scan forward from a `&` for the `;` that would close the reference it opens, giving up once more than {@link MAX_TOKEN_LENGTH} characters have
|
|
255
|
+
* passed since the `&`.
|
|
256
|
+
*
|
|
257
|
+
* @param str - The string being scanned.
|
|
258
|
+
* @param ampersand - The index of the `&`.
|
|
259
|
+
*
|
|
260
|
+
* @returns The index of the closing `;`, or `-1` when the run holds none inside the window. A `;` one character past the `&` is returned rather than
|
|
261
|
+
* refused: that is the empty token `&;`, and the caller decides it is not a reference.
|
|
262
|
+
*/
|
|
263
|
+
function scanTokenEnd(str, ampersand) {
|
|
264
|
+
const len = str.length;
|
|
265
|
+
let j = ampersand + 1;
|
|
266
|
+
while (j < len && str.charCodeAt(j) !== CODE_SEMICOLON && j - ampersand <= MAX_TOKEN_LENGTH) j++;
|
|
267
|
+
if (j >= len || str.charCodeAt(j) !== CODE_SEMICOLON) return -1;
|
|
268
|
+
return j;
|
|
269
|
+
}
|
|
270
|
+
/**
|
|
271
|
+
* @description Single-pass, zero-regex entity decoder for XML and HTML content.
|
|
272
|
+
*
|
|
273
|
+
* ### Entity lookup priority
|
|
274
|
+
*
|
|
275
|
+
* 1. **input / runtime** — injected per document through {@link EntityDecoder.addInputEntities}
|
|
276
|
+
* 2. **persistent external** — set through {@link EntityDecoder.setExternalEntities} and {@link EntityDecoder.addExternalEntity}, surviving
|
|
277
|
+
* {@link EntityDecoder.reset}
|
|
278
|
+
* 3. **base** — the five XML predefined entities plus the constructor's `namedEntities` Both input and external resolve as the `external` tier for limit
|
|
279
|
+
* purposes, because both are injected at runtime. Numeric references (`&#NNN;`, `&#xHH;`) resolve directly through `String.fromCodePoint` and are
|
|
280
|
+
* always `base` tier: they cannot recurse, so a limit that counted them would only punish a document that spells its characters out.
|
|
281
|
+
*
|
|
282
|
+
* ### Upstream behaviour preserved
|
|
283
|
+
*
|
|
284
|
+
* Several quirks of the original are kept deliberately, because a consumer's output already depends on them:
|
|
285
|
+
*
|
|
286
|
+
* - A value containing `&` is **not** filtered from `namedEntities` or `setExternalEntities`, contrary to the documentation. Only
|
|
287
|
+
* {@link EntityDecoder.addExternalEntity} checks, and it drops the entry rather than storing it, so the same name registered either way can resolve
|
|
288
|
+
* to nothing.
|
|
289
|
+
* - {@link EntityDecoderOptions.postCheck} is skipped entirely for input that never reaches the scan — an empty string, a non-string, or a string with
|
|
290
|
+
* no `&`.
|
|
291
|
+
* - {@link EntityDecoder.decode} returns a non-string argument unchanged, despite being typed `string`.
|
|
292
|
+
* - The expansion-limit errors are prefixed `EntityReplacer`, not `EntityDecoder`.
|
|
293
|
+
* - Nothing in XML 1.0 §2.2 is enforced for U+007F–U+009F or for the U+FFFE/U+FFFF noncharacters, and the sweep for `&` leaves a name of
|
|
294
|
+
* {@link MAX_TOKEN_LENGTH} + 1 characters unresolvable.
|
|
295
|
+
* - Numeric references are parsed with `parseInt`, so a leading space, sign, or trailing garbage is accepted: `&# 41;`, `&#x+41;` and `)zz;` all
|
|
296
|
+
* decode, and `�x41;` parses as a null reference rather than `A`.
|
|
297
|
+
* - {@link EntityDecoder.addInputEntities} validates no names, so a `#`-prefixed or `&`-bearing name registers without complaint, where the two
|
|
298
|
+
* external setters would throw.
|
|
299
|
+
*
|
|
300
|
+
* @example
|
|
301
|
+
* ```typescript
|
|
302
|
+
* const decoder = new EntityDecoder({ namedEntities: { copy: '©' } });
|
|
303
|
+
* decoder.setExternalEntities({ brand: 'Acme' });
|
|
304
|
+
* decoder.addInputEntities({ version: '1.0' });
|
|
305
|
+
*
|
|
306
|
+
* decoder.decode('&brand; v&version; ©'); // 'Acme v1.0 ©'
|
|
307
|
+
* decoder.decode('&#38;'); // '&&' — one pass, the output is never re-scanned
|
|
308
|
+
*
|
|
309
|
+
* decoder.reset(); // drops the input entities and the counters, keeps the external ones
|
|
310
|
+
* ```;
|
|
311
|
+
*/
|
|
312
|
+
var EntityDecoder = class EntityDecoder {
|
|
313
|
+
/**
|
|
314
|
+
* @description {@link EntityDecoderLimitOptions.maxTotalExpansions}, or `0` for unlimited. A negative number or `NaN` is also unlimited, since the decode loop
|
|
315
|
+
* tests `> 0`.
|
|
316
|
+
*/
|
|
317
|
+
#maxTotalExpansions;
|
|
318
|
+
/**
|
|
319
|
+
* @description {@link EntityDecoderLimitOptions.maxExpandedLength}, or `0` for unlimited, read the same way as `#maxTotalExpansions`.
|
|
320
|
+
*/
|
|
321
|
+
#maxExpandedLength;
|
|
322
|
+
/**
|
|
323
|
+
* @description {@link EntityDecoderOptions.postCheck}, or the identity function — so the decode loop can call it unconditionally on the path that actually
|
|
324
|
+
* scanned, and never on the two fast paths that return early.
|
|
325
|
+
*/
|
|
326
|
+
#postCheck;
|
|
327
|
+
/**
|
|
328
|
+
* @description The resolved tier filter. See {@link parseLimitTiers}.
|
|
329
|
+
*/
|
|
330
|
+
#limitTiers;
|
|
331
|
+
/**
|
|
332
|
+
* @description {@link EntityDecoderOptions.numericAllowed}. Only an explicit `false` turns it off, so an absent option cannot disable it.
|
|
333
|
+
*/
|
|
334
|
+
#numericAllowed;
|
|
335
|
+
/**
|
|
336
|
+
* @description The five XML predefined entities plus `namedEntities`, merged once at construction and never written again. The built-ins lose to a
|
|
337
|
+
* `namedEntities` entry of the same name, since it is merged second.
|
|
338
|
+
*/
|
|
339
|
+
#baseMap;
|
|
340
|
+
/**
|
|
341
|
+
* @description Persistent external entities, as a null-prototype object. Replaced wholesale by {@link EntityDecoder.setExternalEntities} and added to by
|
|
342
|
+
* {@link EntityDecoder.addExternalEntity}, and never touched by {@link EntityDecoder.reset} — that is the whole distinction from the input map.
|
|
343
|
+
*/
|
|
344
|
+
#externalMap;
|
|
345
|
+
/**
|
|
346
|
+
* @description DOCTYPE entities for the document being processed, as a null-prototype object. Wiped by both {@link EntityDecoder.reset} and
|
|
347
|
+
* {@link EntityDecoder.addInputEntities}.
|
|
348
|
+
*/
|
|
349
|
+
#inputMap;
|
|
350
|
+
/**
|
|
351
|
+
* @description Tracked expansions since the last reset. Cumulative across {@link EntityDecoder.decode} calls, which is what makes a limit a per-document budget
|
|
352
|
+
* rather than a per-call one. Deliberately not reset by a thrown limit error, so the over-limit count is what the error message reports.
|
|
353
|
+
*/
|
|
354
|
+
#totalExpansions;
|
|
355
|
+
/**
|
|
356
|
+
* @description Characters _added_ by expansion since the last reset, accumulated the same way as `#totalExpansions`. Only positive contributions are counted.
|
|
357
|
+
*/
|
|
358
|
+
#expandedLength;
|
|
359
|
+
/**
|
|
360
|
+
* @description {@link EntityDecoderOptions.remove} as a set, or empty. Checked before every other classification, so a name in here is deleted without the name
|
|
361
|
+
* ever being resolved.
|
|
362
|
+
*/
|
|
363
|
+
#removeSet;
|
|
364
|
+
/**
|
|
365
|
+
* @description {@link EntityDecoderOptions.leave} as a set, or empty. Checked after `remove` and before the numeric test, so a name in here is emitted as the
|
|
366
|
+
* original `&name;` text.
|
|
367
|
+
*/
|
|
368
|
+
#leaveSet;
|
|
369
|
+
/**
|
|
370
|
+
* @description The XML version governing numeric classification. Mutable, because a `<?xml version?>` declaration is normally only known after the decoder
|
|
371
|
+
* exists; see {@link EntityDecoder.setXmlVersion}.
|
|
372
|
+
*/
|
|
373
|
+
#ncrXmlVersion;
|
|
374
|
+
/**
|
|
375
|
+
* @description {@link EntityDecoderNCROptions.onNCR} as a level from {@link NCR_LEVEL}. A floor, not an override: the resolver takes the maximum of this and
|
|
376
|
+
* whatever minimum a codepoint range imposes.
|
|
377
|
+
*/
|
|
378
|
+
#ncrOnLevel;
|
|
379
|
+
/**
|
|
380
|
+
* @description {@link EntityDecoderNCROptions.nullNCR} as a level from {@link NCR_LEVEL}, already clamped to `remove` or stricter.
|
|
381
|
+
*/
|
|
382
|
+
#ncrNullLevel;
|
|
383
|
+
/**
|
|
384
|
+
* @description {@link EntityDecoderOptions.onExternalEntity}, or `null` when absent or not a function. A non-function is dropped rather than rejected, so a
|
|
385
|
+
* mistyped option disables the hook instead of failing the construction.
|
|
386
|
+
*/
|
|
387
|
+
#onExternalEntity;
|
|
388
|
+
/**
|
|
389
|
+
* @description {@link EntityDecoderOptions.onInputEntity}, or `null`, under the same non-function rule as `#onExternalEntity`.
|
|
390
|
+
*/
|
|
391
|
+
#onInputEntity;
|
|
392
|
+
/**
|
|
393
|
+
* @description Create a decoder. A factory rather than a constructor, because it refuses a `null` options object. Every field is optional, so `null` is not "a
|
|
394
|
+
* decoder with the defaults" — a caller who wrote it meant something the signature does not allow, and a decoder built from it would be
|
|
395
|
+
* indistinguishable from one built from `{}` while hiding the mistake. Saying so is worth a factory; `EntityDecoderOptions` is a plain object and
|
|
396
|
+
* nothing else about construction can fail.
|
|
397
|
+
*
|
|
398
|
+
* @example
|
|
399
|
+
* ```typescript
|
|
400
|
+
* const decoder = yield* EntityDecoder.make({ numericAllowed: false });
|
|
401
|
+
* yield* decoder.decode('café'); // 'café'
|
|
402
|
+
* ```;
|
|
403
|
+
*
|
|
404
|
+
* @param options - Configuration. See {@link EntityDecoderOptions}. Defaults to every field's own default.
|
|
405
|
+
*
|
|
406
|
+
* @returns An effect producing the decoder. Fails with {@link XmlError} and the `MissingOptions` reason for a `null`.
|
|
407
|
+
*/
|
|
408
|
+
static make = (options = {}) => Predicate.isNullish(options) ? Effect.fail(new XmlError({
|
|
409
|
+
reason: {
|
|
410
|
+
_tag: "MissingOptions",
|
|
411
|
+
parameter: "options"
|
|
412
|
+
},
|
|
413
|
+
message: "EntityDecoder.make: options is required. Use make({}) for a decoder with every default."
|
|
414
|
+
})) : Effect.succeed(new EntityDecoder(options));
|
|
415
|
+
/**
|
|
416
|
+
* @description Create a decoder. Every option is resolved here into the flat fields the decode loop reads, so nothing per-reference has to re-derive it. The
|
|
417
|
+
* options whose wrong type disables them rather than failing the construction — the two hooks, the two name lists — are read through
|
|
418
|
+
* {@link readHook} and {@link readNameList}, so that rule is written once instead of four times.
|
|
419
|
+
*
|
|
420
|
+
* @param resolved - Configuration, already checked. See {@link EntityDecoderOptions}.
|
|
421
|
+
*/
|
|
422
|
+
constructor(resolved) {
|
|
423
|
+
const limit = resolved.limit ?? {};
|
|
424
|
+
this.#maxTotalExpansions = limit.maxTotalExpansions || 0;
|
|
425
|
+
this.#maxExpandedLength = limit.maxExpandedLength || 0;
|
|
426
|
+
this.#postCheck = readPostCheck(resolved.postCheck);
|
|
427
|
+
this.#limitTiers = parseLimitTiers(limit.applyLimitsTo ?? LIMIT_TIER_EXTERNAL);
|
|
428
|
+
this.#numericAllowed = resolved.numericAllowed ?? true;
|
|
429
|
+
this.#baseMap = mergeEntityMaps(XML, resolved.namedEntities || null);
|
|
430
|
+
this.#externalMap = Object.create(null);
|
|
431
|
+
this.#inputMap = Object.create(null);
|
|
432
|
+
this.#totalExpansions = 0;
|
|
433
|
+
this.#expandedLength = 0;
|
|
434
|
+
this.#removeSet = readNameList(resolved.remove);
|
|
435
|
+
this.#leaveSet = readNameList(resolved.leave);
|
|
436
|
+
const ncrConfig = parseNCRConfig(resolved.ncr);
|
|
437
|
+
this.#ncrXmlVersion = ncrConfig.xmlVersion;
|
|
438
|
+
this.#ncrOnLevel = ncrConfig.onLevel;
|
|
439
|
+
this.#ncrNullLevel = ncrConfig.nullLevel;
|
|
440
|
+
this.#onExternalEntity = readHook(resolved.onExternalEntity);
|
|
441
|
+
this.#onInputEntity = readHook(resolved.onInputEntity);
|
|
442
|
+
}
|
|
443
|
+
/**
|
|
444
|
+
* @description Ask a registration hook about one name and value.
|
|
445
|
+
*
|
|
446
|
+
* @param hook - The hook, or `null`. A `null` hook accepts, which is what lets {@link EntityDecoder.addExternalEntity} call this unconditionally.
|
|
447
|
+
* @param name - The entity name, without `&` or `;`.
|
|
448
|
+
* @param value - The resolved value, after any `{ regex, val }` envelope was unwrapped.
|
|
449
|
+
* @param context - Which registration is in progress, for the error message.
|
|
450
|
+
*
|
|
451
|
+
* @returns An effect producing `true` to register, `false` to skip silently. Fails with {@link XmlError} and the `EntityRejected` reason when the
|
|
452
|
+
* hook returns `throw`. The message quotes the entity, so it is the only record left that a document was rejected.
|
|
453
|
+
*/
|
|
454
|
+
#applyRegistrationHook(hook, name, value, context) {
|
|
455
|
+
if (!hook) return Effect.succeed(true);
|
|
456
|
+
const action = hook(name, value);
|
|
457
|
+
if (action === ENTITY_ACTION.BLOCK) return Effect.succeed(false);
|
|
458
|
+
if (action === ENTITY_ACTION.THROW) return Effect.fail(new XmlError({
|
|
459
|
+
reason: {
|
|
460
|
+
_tag: "EntityRejected",
|
|
461
|
+
context,
|
|
462
|
+
name
|
|
463
|
+
},
|
|
464
|
+
message: `[EntityDecoder] Registration of ${context} entity "&${name};" was rejected by hook`
|
|
465
|
+
}));
|
|
466
|
+
return Effect.succeed(true);
|
|
467
|
+
}
|
|
468
|
+
/**
|
|
469
|
+
* @description Replace the whole set of persistent external entities. Every key is validated _before_ any value is read, so an invalid name fails even when its
|
|
470
|
+
* value is a form the merge would have dropped. A non-object or `null` map clears the set without validating anything.
|
|
471
|
+
*
|
|
472
|
+
* @param map - The entities to register, or nothing to clear.
|
|
473
|
+
*
|
|
474
|
+
* @returns An effect that registers the map. Fails with {@link XmlError} when a key contains a character from {@link SPECIAL_CHARS} or begins with
|
|
475
|
+
* `#` (`InvalidEntityName`), or when {@link EntityDecoderOptions.onExternalEntity} returns `throw` (`EntityRejected`). A rejection from the hook
|
|
476
|
+
* aborts before the assignment, so the previous map survives.
|
|
477
|
+
*/
|
|
478
|
+
setExternalEntities = Effect.fnUntraced(function* (map) {
|
|
479
|
+
if (map) for (const key of Object.keys(map)) yield* checkEntityName(key);
|
|
480
|
+
if (!this.#onExternalEntity) {
|
|
481
|
+
this.#externalMap = mergeEntityMaps(map);
|
|
482
|
+
return;
|
|
483
|
+
}
|
|
484
|
+
const flat = mergeEntityMaps(map);
|
|
485
|
+
const filtered = Object.create(null);
|
|
486
|
+
for (const [name, value] of Object.entries(flat)) if (yield* this.#applyRegistrationHook(this.#onExternalEntity, name, value, "external")) filtered[name] = value;
|
|
487
|
+
this.#externalMap = filtered;
|
|
488
|
+
});
|
|
489
|
+
/**
|
|
490
|
+
* @description Add one persistent external entity, keeping whatever is already registered. This is the only registration path that refuses a value containing
|
|
491
|
+
* `&`; the two map setters store one unchanged. The omission is upstream's, and it is kept: the same name registered through either route can end
|
|
492
|
+
* up resolving, or not resolving at all.
|
|
493
|
+
*
|
|
494
|
+
* @param key - The entity name, without `&` or `;`.
|
|
495
|
+
* @param value - The replacement text.
|
|
496
|
+
*
|
|
497
|
+
* @returns An effect that adds the entity. Fails with {@link XmlError} and the `InvalidEntityName` reason when `key` contains a character from
|
|
498
|
+
* {@link SPECIAL_CHARS} or begins with `#`, or with the `EntityRejected` reason when {@link EntityDecoderOptions.onExternalEntity} returns
|
|
499
|
+
* `throw`.
|
|
500
|
+
*/
|
|
501
|
+
addExternalEntity = Effect.fnUntraced(function* (key, value) {
|
|
502
|
+
yield* checkEntityName(key);
|
|
503
|
+
if (Predicate.isString(value) && value.indexOf("&") === -1) {
|
|
504
|
+
if (yield* this.#applyRegistrationHook(this.#onExternalEntity, key, value, "external")) this.#externalMap[key] = value;
|
|
505
|
+
}
|
|
506
|
+
});
|
|
507
|
+
/**
|
|
508
|
+
* @description Register the DOCTYPE entities for the document about to be decoded, replacing any previous set and clearing both counters. Unlike the external
|
|
509
|
+
* setters, no name is validated: a `#`-prefixed name, or one containing `&` or `<`, registers without complaint. A `#`-prefixed name is then
|
|
510
|
+
* unreachable, since `decode` routes `#`-prefixed tokens to the numeric pipeline first.
|
|
511
|
+
*
|
|
512
|
+
* @param map - The entities to register, or nothing to clear.
|
|
513
|
+
*
|
|
514
|
+
* @returns An effect that registers the map. Fails with {@link XmlError} and the `EntityRejected` reason when
|
|
515
|
+
* {@link EntityDecoderOptions.onInputEntity} returns `throw`. The counters have already been cleared by then.
|
|
516
|
+
*/
|
|
517
|
+
addInputEntities = Effect.fnUntraced(function* (map) {
|
|
518
|
+
this.#totalExpansions = 0;
|
|
519
|
+
this.#expandedLength = 0;
|
|
520
|
+
if (!this.#onInputEntity) {
|
|
521
|
+
this.#inputMap = mergeEntityMaps(map);
|
|
522
|
+
return;
|
|
523
|
+
}
|
|
524
|
+
const flat = mergeEntityMaps(map);
|
|
525
|
+
const filtered = Object.create(null);
|
|
526
|
+
for (const [name, value] of Object.entries(flat)) if (yield* this.#applyRegistrationHook(this.#onInputEntity, name, value, "input")) filtered[name] = value;
|
|
527
|
+
this.#inputMap = filtered;
|
|
528
|
+
});
|
|
529
|
+
/**
|
|
530
|
+
* @description Start a new document: drop the input entities and both counters. The persistent external entities, the base map, the limits, the NCR policy and
|
|
531
|
+
* the XML version all survive, which is the difference between this and constructing a fresh decoder.
|
|
532
|
+
*
|
|
533
|
+
* @returns This decoder, so a call can be chained onto the document it ends.
|
|
534
|
+
*/
|
|
535
|
+
reset() {
|
|
536
|
+
this.#inputMap = Object.create(null);
|
|
537
|
+
this.#totalExpansions = 0;
|
|
538
|
+
this.#expandedLength = 0;
|
|
539
|
+
return this;
|
|
540
|
+
}
|
|
541
|
+
/**
|
|
542
|
+
* @description Set the XML version used to classify numeric references, once a `<?xml version="…"?>` declaration has been read. Only the exact number `1.1`
|
|
543
|
+
* selects XML 1.1; `1.0`, `1.15`, `'1.1'` and `NaN` all become `1.0`, so the stricter classification is the default rather than the looser one.
|
|
544
|
+
*
|
|
545
|
+
* @param version - The declared version.
|
|
546
|
+
*
|
|
547
|
+
* @returns Nothing.
|
|
548
|
+
*/
|
|
549
|
+
setXmlVersion(version) {
|
|
550
|
+
this.#ncrXmlVersion = version === 1.1 ? 1.1 : 1;
|
|
551
|
+
}
|
|
552
|
+
/**
|
|
553
|
+
* @description Expand every entity reference in a string, in one pass. The output is never re-scanned, so no expansion can produce a _second_ one: a registered
|
|
554
|
+
* value that itself contains reference text reaches the caller as that literal text, unexpanded. What the limits bound is the growth of this single
|
|
555
|
+
* pass — how much one round of expansion can add. Three inputs return before the scan and therefore never reach
|
|
556
|
+
* {@link EntityDecoderOptions.postCheck}: a non-string, the empty string, and any string with no `&` in it. The scan itself is `#expandAll`; what
|
|
557
|
+
* this method adds is the three inputs that skip it and the single join of what it collected.
|
|
558
|
+
*
|
|
559
|
+
* @example
|
|
560
|
+
* ```typescript
|
|
561
|
+
* import { Effect } from 'effect';
|
|
562
|
+
* import { EntityDecoder } from '@endevops/effect-xml-codec';
|
|
563
|
+
*
|
|
564
|
+
* const decoder = new EntityDecoder({ namedEntities: { copy: '©' } });
|
|
565
|
+
* Effect.runSync(Effect.orElseSucceed(decoder.addExternalEntity('brand', 'Acme'), () => undefined));
|
|
566
|
+
* Effect.runSync(decoder.decode('&brand; ©')); // 'Acme ©'
|
|
567
|
+
* ```;
|
|
568
|
+
*
|
|
569
|
+
* @param str - The string to decode.
|
|
570
|
+
*
|
|
571
|
+
* @returns An effect producing the decoded string. A non-string argument comes back as the same non-string, which the `string` return type does not
|
|
572
|
+
* describe but callers passing untyped values depend on. Fails with {@link XmlError} when a numeric reference is prohibited under the configured
|
|
573
|
+
* policy (`ProhibitedCharacterReference`), or when a tracked tier would exceed {@link EntityDecoderLimitOptions.maxTotalExpansions}
|
|
574
|
+
* (`ExpansionLimitExceeded`) or {@link EntityDecoderLimitOptions.maxExpandedLength} (`ExpandedLengthLimitExceeded`). The two limit messages keep
|
|
575
|
+
* the `EntityReplacer` prefix from the original throw, which named a class this decoder does not have.
|
|
576
|
+
*/
|
|
577
|
+
decode = Effect.fnUntraced(function* (str) {
|
|
578
|
+
if (!Predicate.isString(str) || str.length === 0) return str;
|
|
579
|
+
if (str.indexOf("&") === -1) return str;
|
|
580
|
+
const chunks = yield* this.#expandAll(str);
|
|
581
|
+
const result = chunks.length === 0 ? str : chunks.join("");
|
|
582
|
+
return this.#postCheck(result, str);
|
|
583
|
+
});
|
|
584
|
+
/**
|
|
585
|
+
* @description Walk the string once and collect the pieces of every reference that resolved. Two advance rules make the walk terminate and keep it correct: an
|
|
586
|
+
* `&` that turns out to open nothing moves the cursor by one character rather than to the end of its run, so a second `&` in the same text is still
|
|
587
|
+
* found; and a reference that did resolve moves it to just past the `;`, so the text that was substituted for it is never looked at again — that is
|
|
588
|
+
* what makes the pass single, and a registered value containing `&` cannot expand a second level. What a reference becomes is `#resolveToken`'s to
|
|
589
|
+
* decide and what it costs is `#chargeExpansion`'s to apply, which leaves the scanning here as the only thing with a rule of its own.
|
|
590
|
+
*
|
|
591
|
+
* @param str - The string to expand. It always holds at least one `&` and is never empty, or the caller would have returned before reaching the
|
|
592
|
+
* walk.
|
|
593
|
+
*
|
|
594
|
+
* @returns An effect producing the pieces in order. The array is empty exactly when nothing was replaced, which the caller reads as "the input is
|
|
595
|
+
* its own result". Fails with {@link XmlError} and the reason the offending reference carries — `ProhibitedCharacterReference`,
|
|
596
|
+
* `ExpansionLimitExceeded` or `ExpandedLengthLimitExceeded`.
|
|
597
|
+
*/
|
|
598
|
+
#expandAll = Effect.fnUntraced(function* (str) {
|
|
599
|
+
const chunks = [];
|
|
600
|
+
const len = str.length;
|
|
601
|
+
let last = 0;
|
|
602
|
+
let i = 0;
|
|
603
|
+
while (i < len) {
|
|
604
|
+
if (str.charCodeAt(i) !== CODE_AMPERSAND) {
|
|
605
|
+
i++;
|
|
606
|
+
continue;
|
|
607
|
+
}
|
|
608
|
+
const end = scanTokenEnd(str, i);
|
|
609
|
+
if (end <= i + 1) {
|
|
610
|
+
i++;
|
|
611
|
+
continue;
|
|
612
|
+
}
|
|
613
|
+
const token = str.slice(i + 1, end);
|
|
614
|
+
const resolved = yield* this.#resolveToken(token);
|
|
615
|
+
if (resolved === void 0) {
|
|
616
|
+
i++;
|
|
617
|
+
continue;
|
|
618
|
+
}
|
|
619
|
+
if (i > last) chunks.push(str.slice(last, i));
|
|
620
|
+
chunks.push(resolved.value);
|
|
621
|
+
last = end + 1;
|
|
622
|
+
i = last;
|
|
623
|
+
yield* this.#chargeExpansion(token, resolved.value, resolved.tier);
|
|
624
|
+
}
|
|
625
|
+
if (last < len) chunks.push(str.slice(last));
|
|
626
|
+
return chunks;
|
|
627
|
+
});
|
|
628
|
+
/**
|
|
629
|
+
* @description Decide what one reference expands to. The lists and maps are consulted in the one order the runtime uses, and the first that matches wins:
|
|
630
|
+
*
|
|
631
|
+
* 1. `remove` — deleted outright, without the name ever being resolved, so the name need not exist.
|
|
632
|
+
* 2. `leave` — emitted as the original `&token;`, and charged to nothing.
|
|
633
|
+
* 3. A `#`-prefixed token — the numeric pipeline, which is the only one of the four that can fail. Classification runs before any decision about
|
|
634
|
+
* `numericAllowed`, because the ranges that carry a minimum have to be caught whichever way that option is set.
|
|
635
|
+
* 4. Anything else — resolved against the input map, then the external map, then the base map.
|
|
636
|
+
*
|
|
637
|
+
* @param token - The reference's token, e.g. `brand` or `#38`, with the `&` and the `;` already stripped. Never empty: the scanner drops `&;`
|
|
638
|
+
* before calling.
|
|
639
|
+
*
|
|
640
|
+
* @returns An effect producing what the reference expands to and the tier to charge it to, or `undefined` to leave it as written and charge it
|
|
641
|
+
* nothing. `undefined` covers all three ways of leaving a reference alone — a listed `leave` name, a numeric reference that is out of range, and
|
|
642
|
+
* a name registered nowhere — and none of them is distinguishable from outside. Fails with {@link XmlError} and the
|
|
643
|
+
* `ProhibitedCharacterReference` reason when the numeric policy throws on the codepoint.
|
|
644
|
+
*/
|
|
645
|
+
#resolveToken = Effect.fnUntraced(function* (token) {
|
|
646
|
+
if (this.#removeSet.has(token)) return {
|
|
647
|
+
value: "",
|
|
648
|
+
tier: LIMIT_TIER_EXTERNAL
|
|
649
|
+
};
|
|
650
|
+
if (this.#leaveSet.has(token)) return void 0;
|
|
651
|
+
if (token.charCodeAt(0) === CODE_HASH) {
|
|
652
|
+
const character = yield* this.#resolveNCR(token);
|
|
653
|
+
if (character === void 0) return void 0;
|
|
654
|
+
return {
|
|
655
|
+
value: character,
|
|
656
|
+
tier: LIMIT_TIER_BASE
|
|
657
|
+
};
|
|
658
|
+
}
|
|
659
|
+
return this.#resolveName(token);
|
|
660
|
+
});
|
|
661
|
+
/**
|
|
662
|
+
* @description Charge one expansion against the ceilings, or against neither. An expansion counts only when its tier passes `#tierCounts` and at least one
|
|
663
|
+
* ceiling is configured, so a decoder with no limits set does no accounting at all, and an entity in a tier the filter excludes is free. Each
|
|
664
|
+
* ceiling is guarded separately rather than left to its own check, because an unconfigured ceiling is not a ceiling of zero: `maxExpandedLength: 0`
|
|
665
|
+
* means unlimited, so a decoder with only a count limit must not accumulate length it will then be compared against.
|
|
666
|
+
*
|
|
667
|
+
* @param token - The reference's token, with the `&` and `;` stripped. Its width is the baseline the expansion is measured against.
|
|
668
|
+
* @param replacement - What the reference expanded to, including `''` for a removal.
|
|
669
|
+
* @param tier - The tier the expansion is charged to.
|
|
670
|
+
*
|
|
671
|
+
* @returns An effect that fails with {@link XmlError} once a ceiling is exceeded, and succeeds otherwise. The count is checked before the length,
|
|
672
|
+
* so a document that breaches both is reported against the count.
|
|
673
|
+
*/
|
|
674
|
+
#chargeExpansion = Effect.fnUntraced(function* (token, replacement, tier) {
|
|
675
|
+
const counts = this.#maxTotalExpansions > 0;
|
|
676
|
+
const grows = this.#maxExpandedLength > 0;
|
|
677
|
+
if (!counts && !grows) return;
|
|
678
|
+
if (!this.#tierCounts(tier)) return;
|
|
679
|
+
if (counts) yield* this.#countExpansion();
|
|
680
|
+
if (grows) yield* this.#countExpandedLength(token, replacement);
|
|
681
|
+
});
|
|
682
|
+
/**
|
|
683
|
+
* @description Add one expansion to the running total and compare it against {@link EntityDecoderLimitOptions.maxTotalExpansions}. The comparison is `>` rather
|
|
684
|
+
* than `>=`, so a limit of `n` allows exactly `n` expansions and throws on the `n + 1`th. That is a contract — the option's own documentation
|
|
685
|
+
* states it — and the kind of off-by-one a tidy-up changes by accident. The counter is deliberately not reset before failing: the over-limit total
|
|
686
|
+
* is what the error message reports, and {@link EntityDecoder.reset} is the caller's way to start a new document.
|
|
687
|
+
*
|
|
688
|
+
* @returns An effect that fails with {@link XmlError} and the `ExpansionLimitExceeded` reason once the count is past the ceiling, and succeeds
|
|
689
|
+
* otherwise. The `EntityReplacer` prefix in the message is preserved verbatim from the original throw, despite naming a class this decoder does
|
|
690
|
+
* not have.
|
|
691
|
+
*/
|
|
692
|
+
#countExpansion() {
|
|
693
|
+
this.#totalExpansions++;
|
|
694
|
+
if (this.#totalExpansions > this.#maxTotalExpansions) return Effect.fail(new XmlError({
|
|
695
|
+
reason: {
|
|
696
|
+
_tag: "ExpansionLimitExceeded",
|
|
697
|
+
actual: this.#totalExpansions,
|
|
698
|
+
limit: this.#maxTotalExpansions
|
|
699
|
+
},
|
|
700
|
+
message: `[EntityReplacer] Entity expansion count limit exceeded: ${this.#totalExpansions} > ${this.#maxTotalExpansions}`
|
|
701
|
+
}));
|
|
702
|
+
return Effect.void;
|
|
703
|
+
}
|
|
704
|
+
/**
|
|
705
|
+
* @description Add one expansion's surplus to the running total and compare it against {@link EntityDecoderLimitOptions.maxExpandedLength}. Only the surplus
|
|
706
|
+
* counts, and only upward: a reference whose replacement is no longer than the `&token;` it replaces contributes zero, and a shrinking one
|
|
707
|
+
* contributes nothing and cannot trip the limit at all. That is what makes the ceiling a bound on growth rather than on document size.
|
|
708
|
+
*
|
|
709
|
+
* @param token - The reference's token, with the `&` and `;` stripped. The two delimiters count towards what the expansion displaced.
|
|
710
|
+
* @param replacement - What the reference expanded to, including `''` for a removal.
|
|
711
|
+
*
|
|
712
|
+
* @returns An effect that fails with {@link XmlError} and the `ExpandedLengthLimitExceeded` reason once the total is past the ceiling, and succeeds
|
|
713
|
+
* otherwise. The `EntityReplacer` prefix in the message is preserved verbatim from the original throw, for the same reason as in
|
|
714
|
+
* `#countExpansion`.
|
|
715
|
+
*/
|
|
716
|
+
#countExpandedLength(token, replacement) {
|
|
717
|
+
const delta = replacement.length - (token.length + 2);
|
|
718
|
+
if (delta <= 0) return Effect.void;
|
|
719
|
+
this.#expandedLength += delta;
|
|
720
|
+
if (this.#expandedLength > this.#maxExpandedLength) return Effect.fail(new XmlError({
|
|
721
|
+
reason: {
|
|
722
|
+
_tag: "ExpandedLengthLimitExceeded",
|
|
723
|
+
actual: this.#expandedLength,
|
|
724
|
+
limit: this.#maxExpandedLength
|
|
725
|
+
},
|
|
726
|
+
message: `[EntityReplacer] Expanded content length limit exceeded: ${this.#expandedLength} > ${this.#maxExpandedLength}`
|
|
727
|
+
}));
|
|
728
|
+
return Effect.void;
|
|
729
|
+
}
|
|
730
|
+
/**
|
|
731
|
+
* @description Decide whether an entity of a given tier is charged against the limits.
|
|
732
|
+
*
|
|
733
|
+
* @param tier - The tier the replacement is charged to. Every expansion that reaches here carries one — a name deleted before it was ever resolved
|
|
734
|
+
* still carries the `external` tier — so there is no absent case to answer.
|
|
735
|
+
*
|
|
736
|
+
* @returns `true` when it counts. `'all'` short-circuits, so a filter naming every tier charges everything regardless of which map it came from.
|
|
737
|
+
*/
|
|
738
|
+
#tierCounts(tier) {
|
|
739
|
+
if (this.#limitTiers.has(LIMIT_TIER_ALL)) return true;
|
|
740
|
+
return this.#limitTiers.has(tier);
|
|
741
|
+
}
|
|
742
|
+
/**
|
|
743
|
+
* @description Resolve a named entity token, with the `&` and `;` already stripped.
|
|
744
|
+
*
|
|
745
|
+
* @param name - The token, e.g. `brand`.
|
|
746
|
+
*
|
|
747
|
+
* @returns The value and the tier to charge it to, or `undefined` when the name is registered nowhere. A name registered to the empty string
|
|
748
|
+
* resolves to `''` rather than to `undefined`, so it deletes the reference instead of leaving it alone.
|
|
749
|
+
*/
|
|
750
|
+
#resolveName(name) {
|
|
751
|
+
const fromInput = ownEntity(this.#inputMap, name);
|
|
752
|
+
if (fromInput !== void 0) return {
|
|
753
|
+
value: fromInput,
|
|
754
|
+
tier: LIMIT_TIER_EXTERNAL
|
|
755
|
+
};
|
|
756
|
+
const fromExternal = ownEntity(this.#externalMap, name);
|
|
757
|
+
if (fromExternal !== void 0) return {
|
|
758
|
+
value: fromExternal,
|
|
759
|
+
tier: LIMIT_TIER_EXTERNAL
|
|
760
|
+
};
|
|
761
|
+
const fromBase = ownEntity(this.#baseMap, name);
|
|
762
|
+
if (fromBase !== void 0) return {
|
|
763
|
+
value: fromBase,
|
|
764
|
+
tier: LIMIT_TIER_BASE
|
|
765
|
+
};
|
|
766
|
+
}
|
|
767
|
+
/**
|
|
768
|
+
* @description Find the strictest action a codepoint's range requires. Checked in this order:
|
|
769
|
+
*
|
|
770
|
+
* 1. U+0000 — governed by `nullNCR`, already clamped to `remove` or stricter
|
|
771
|
+
* 2. U+D800–U+DFFF — surrogates, always `remove`, under every policy and both XML versions
|
|
772
|
+
* 3. U+0001–U+001F other than tab, newline, carriage return — XML 1.0 only, `remove` Nothing else is classified. U+007F–U+009F (C1) and the
|
|
773
|
+
* U+FFFE/U+FFFF noncharacters are not checked, even though XML 1.0 §2.2 prohibits them and the `xmlVersion` option's own documentation claims C1
|
|
774
|
+
* is only permitted under 1.1. Both gaps are upstream's and are kept.
|
|
775
|
+
*
|
|
776
|
+
* @param cp - The codepoint.
|
|
777
|
+
*
|
|
778
|
+
* @returns The minimum level from {@link NCR_LEVEL}, or {@link NO_MINIMUM_LEVEL} when the codepoint carries none.
|
|
779
|
+
*/
|
|
780
|
+
#classifyNCR(cp) {
|
|
781
|
+
if (cp === 0) return this.#ncrNullLevel;
|
|
782
|
+
if (cp >= 55296 && cp <= 57343) return NCR_LEVEL.remove;
|
|
783
|
+
if (this.#ncrXmlVersion === 1 && cp >= 1 && cp <= 31 && !XML10_ALLOWED_C0.has(cp)) return NCR_LEVEL.remove;
|
|
784
|
+
return NO_MINIMUM_LEVEL;
|
|
785
|
+
}
|
|
786
|
+
/**
|
|
787
|
+
* @description Turn a resolved action level into a replacement.
|
|
788
|
+
*
|
|
789
|
+
* @param action - A level from {@link NCR_LEVEL}. A level outside the four known ones falls through to the allow behaviour, so a bad level cannot
|
|
790
|
+
* produce a wrong string — it can only fail open.
|
|
791
|
+
* @param token - The raw token, e.g. `#38`, for the error message.
|
|
792
|
+
* @param cp - The codepoint, for the error message.
|
|
793
|
+
*
|
|
794
|
+
* @returns An effect producing the character for `allow`, `''` for `remove`, and `undefined` for `leave` — which the caller reads as "emit the
|
|
795
|
+
* original `&token;`". Fails with {@link XmlError} and the `ProhibitedCharacterReference` reason for `throw`, naming both the token and the
|
|
796
|
+
* codepoint.
|
|
797
|
+
*/
|
|
798
|
+
#applyNCRAction(action, token, cp) {
|
|
799
|
+
return Match.value(action).pipe(Match.when(NCR_LEVEL.allow, () => Effect.succeed(String.fromCodePoint(cp))), Match.when(NCR_LEVEL.remove, () => Effect.succeed("")), Match.when(NCR_LEVEL.leave, () => Effect.succeed(void 0)), Match.when(NCR_LEVEL.throw, () => Effect.fail(new XmlError({
|
|
800
|
+
reason: {
|
|
801
|
+
_tag: "ProhibitedCharacterReference",
|
|
802
|
+
token,
|
|
803
|
+
codepoint: cp
|
|
804
|
+
},
|
|
805
|
+
message: `[EntityDecoder] Prohibited numeric character reference &${token}; (U+${cp.toString(16).toUpperCase().padStart(4, "0")})`
|
|
806
|
+
}))), Match.orElse(() => Effect.succeed(String.fromCodePoint(cp))));
|
|
807
|
+
}
|
|
808
|
+
/**
|
|
809
|
+
* @description The full numeric-reference pipeline for one `#`-prefixed token.
|
|
810
|
+
*
|
|
811
|
+
* 1. Parse the codepoint, decimal or hex.
|
|
812
|
+
* 2. Reject NaN, negatives, and anything above {@link MAX_CODE_POINT}, leaving the reference as written.
|
|
813
|
+
* 3. Classify the codepoint for a minimum level.
|
|
814
|
+
* 4. If `numericAllowed` is off and no minimum reaches `remove`, leave the reference as written.
|
|
815
|
+
* 5. Take the stricter of the configured level and the minimum.
|
|
816
|
+
* 6. Apply it. Step 4 is why `numericAllowed: false` does not neutralise `onNCR: 'throw'` for surrogates, the XML 1.0 C0 controls or null: their
|
|
817
|
+
* minimum already reaches `remove`, so they are handled no matter what the option says. It does neutralise the throw for every other codepoint.
|
|
818
|
+
* The parse is `parseInt`, which stops at the first character it cannot use. That is upstream's choice and it is permissive: a leading space or
|
|
819
|
+
* `+`, and trailing garbage, are all accepted, and a decimal token beginning `0x` parses as `0` rather than as hex.
|
|
820
|
+
*
|
|
821
|
+
* @param token - The raw token without `&` and `;`, e.g. `#38`, `#x26`, `#X26`.
|
|
822
|
+
*
|
|
823
|
+
* @returns An effect producing the replacement — the empty string meaning "delete" — or `undefined` to leave the reference as written. Fails with
|
|
824
|
+
* {@link XmlError} and the `ProhibitedCharacterReference` reason when the effective action is `throw`.
|
|
825
|
+
*/
|
|
826
|
+
#resolveNCR(token) {
|
|
827
|
+
const second = token.charCodeAt(1);
|
|
828
|
+
let cp;
|
|
829
|
+
if (second === CODE_LOWER_X || second === CODE_UPPER_X) cp = parseInt(token.slice(2), 16);
|
|
830
|
+
else cp = parseInt(token.slice(1), 10);
|
|
831
|
+
if (Number.isNaN(cp) || cp < 0 || cp > MAX_CODE_POINT) return Effect.succeed(void 0);
|
|
832
|
+
const minimum = this.#classifyNCR(cp);
|
|
833
|
+
if (!this.#numericAllowed && minimum < NCR_LEVEL.remove) return Effect.succeed(void 0);
|
|
834
|
+
const effective = minimum === NO_MINIMUM_LEVEL ? this.#ncrOnLevel : Math.max(this.#ncrOnLevel, minimum);
|
|
835
|
+
return this.#applyNCRAction(effective, token, cp);
|
|
836
|
+
}
|
|
837
|
+
};
|
|
838
|
+
//#endregion
|
|
839
|
+
export { ENTITY_ACTION, EntityDecoder };
|
|
840
|
+
|
|
841
|
+
//# sourceMappingURL=entity-decoder.js.map
|