@endevops/effect-codec-xml 0.0.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (61) hide show
  1. package/LICENSE +21 -0
  2. package/LICENSE-is-entities +21 -0
  3. package/LICENSE-is-xml-naming +21 -0
  4. package/README.md +415 -0
  5. package/dist/codec.d.ts +48 -0
  6. package/dist/codec.d.ts.map +1 -0
  7. package/dist/codec.js +63 -0
  8. package/dist/codec.js.map +1 -0
  9. package/dist/conventions.d.ts +88 -0
  10. package/dist/conventions.d.ts.map +1 -0
  11. package/dist/conventions.js +113 -0
  12. package/dist/conventions.js.map +1 -0
  13. package/dist/entities/entity-decoder.d.ts +333 -0
  14. package/dist/entities/entity-decoder.d.ts.map +1 -0
  15. package/dist/entities/entity-decoder.js +841 -0
  16. package/dist/entities/entity-decoder.js.map +1 -0
  17. package/dist/entities/entity-tables.js +16 -0
  18. package/dist/entities/entity-tables.js.map +1 -0
  19. package/dist/errors.d.ts +49 -0
  20. package/dist/errors.d.ts.map +1 -0
  21. package/dist/errors.js +48 -0
  22. package/dist/errors.js.map +1 -0
  23. package/dist/index.d.ts +11 -0
  24. package/dist/index.js +11 -0
  25. package/dist/namespaces.d.ts +101 -0
  26. package/dist/namespaces.d.ts.map +1 -0
  27. package/dist/namespaces.js +663 -0
  28. package/dist/namespaces.js.map +1 -0
  29. package/dist/naming.d.ts +149 -0
  30. package/dist/naming.d.ts.map +1 -0
  31. package/dist/naming.js +296 -0
  32. package/dist/naming.js.map +1 -0
  33. package/dist/parse.d.ts +75 -0
  34. package/dist/parse.d.ts.map +1 -0
  35. package/dist/parse.js +437 -0
  36. package/dist/parse.js.map +1 -0
  37. package/dist/render.d.ts +99 -0
  38. package/dist/render.d.ts.map +1 -0
  39. package/dist/render.js +509 -0
  40. package/dist/render.js.map +1 -0
  41. package/dist/xml-error.d.ts +172 -0
  42. package/dist/xml-error.d.ts.map +1 -0
  43. package/dist/xml-error.js +157 -0
  44. package/dist/xml-error.js.map +1 -0
  45. package/dist/xml-value.d.ts +42 -0
  46. package/dist/xml-value.d.ts.map +1 -0
  47. package/dist/xml-value.js +79 -0
  48. package/dist/xml-value.js.map +1 -0
  49. package/package.json +69 -0
  50. package/src/codec.ts +136 -0
  51. package/src/conventions.ts +145 -0
  52. package/src/entities/entity-decoder.ts +1248 -0
  53. package/src/entities/entity-tables.ts +18 -0
  54. package/src/errors.ts +55 -0
  55. package/src/index.ts +79 -0
  56. package/src/namespaces.ts +968 -0
  57. package/src/naming.ts +519 -0
  58. package/src/parse.ts +597 -0
  59. package/src/render.ts +708 -0
  60. package/src/xml-error.ts +168 -0
  61. package/src/xml-value.ts +108 -0
@@ -0,0 +1,1248 @@
1
+ // oxlint-disable effecttsgo/effect-succeed-with-void
2
+ // entity-decoder.ts
3
+ //
4
+ // Single-pass, zero-regex decoder. Scan for `&`, read to `;`, resolve, push chunks, join once. Ported from
5
+ // `@nodable/entities@2.2.0` (`src/EntityDecoder.js`) as native TypeScript.
6
+ //
7
+ // Three entity tiers exist and the distinction is the security model, not a performance detail. `input` and
8
+ // `external` entities are injected at runtime — DOCTYPE declarations, and whatever a caller hands the
9
+ // decoder — so they are the untrusted surface and are what the expansion limits count by default. `base` is
10
+ // the five XML predefined entities plus the caller's own `namedEntities`. Numeric references are always
11
+ // `base`: they cannot recurse.
12
+ //
13
+ // Behaviour is transcribed, not corrected. Several upstream quirks are load-bearing for the output a caller
14
+ // already sees — a `&` inside a registered value is not filtered the way the docs claim, `postCheck` is
15
+ // skipped on the two fast paths, C1 codepoints and the FFFE/FFFF noncharacters are not classified at all.
16
+ // Each is called out where it appears, and none of them is repaired, because this class sits in front of XXE
17
+ // and entity-expansion handling where a silent fix is a change to every consumer's output.
18
+
19
+ import { Effect, Match, Predicate } from 'effect';
20
+
21
+ import { XmlError } from '#/xml-error.ts';
22
+
23
+ import { XML as DEFAULT_XML_ENTITIES } from './entity-tables.ts';
24
+
25
+ // ---------------------------------------------------------------------------
26
+ // Character codes
27
+ //
28
+ // The scan is a hand-rolled `charCodeAt` loop, so the three codes it tests against are named here rather than
29
+ // written as literals. `&` opens a reference, `;` closes one, `#` is what makes a reference numeric.
30
+ // ---------------------------------------------------------------------------
31
+
32
+ const CODE_AMPERSAND = 38;
33
+ const CODE_SEMICOLON = 59;
34
+ const CODE_HASH = 35;
35
+ const CODE_LOWER_X = 120;
36
+ const CODE_UPPER_X = 88;
37
+
38
+ /**
39
+ * @description The widest entity name {@link EntityDecoder.decode} will look for, in characters. The forward scan for `;` stops once more than this many characters
40
+ * have passed since the `&`, so a longer run is not treated as one entity — it is copied through as literal text. The bound is what keeps a document
41
+ * with a megabyte of non-entity text between two ampersands from being sliced.
42
+ */
43
+ const MAX_TOKEN_LENGTH = 32;
44
+
45
+ /**
46
+ * @description The largest codepoint `String.fromCodePoint` accepts, and the bound a numeric reference is checked against before it gets that far. Out of range is
47
+ * `leave`, not `remove`: the reference is preserved verbatim rather than deleted.
48
+ */
49
+ const MAX_CODE_POINT = 0x10ffff;
50
+
51
+ /**
52
+ * @description Returned by `#classifyNCR` for a codepoint that carries no minimum action level, which is what distinguishes "no restriction" from
53
+ * `NCR_LEVEL.allow` — both end up expanding, but only the first lets `numericAllowed: false` short-circuit the whole pipeline.
54
+ */
55
+ const NO_MINIMUM_LEVEL = -1;
56
+
57
+ // ---------------------------------------------------------------------------
58
+ // Entity name validation
59
+ // ---------------------------------------------------------------------------
60
+
61
+ /**
62
+ * @description Characters that may not appear in an entity name registered through {@link EntityDecoder.setExternalEntities} or
63
+ * {@link EntityDecoder.addExternalEntity}. A name carrying one of these cannot be written as `&name;` at all, so registration refuses it rather than
64
+ * storing a name no document could ever reference. The set is the upstream string verbatim, including its duplicated backslash — a `Set` discards the
65
+ * duplicate, so the effective set is the eighteen characters below.
66
+ */
67
+ const SPECIAL_CHARS: ReadonlySet<string> = new Set('!?\\/[]$%{}^&*()<>|+');
68
+
69
+ // ---------------------------------------------------------------------------
70
+ // Limit tiers
71
+ // ---------------------------------------------------------------------------
72
+
73
+ /**
74
+ * @description Injected at runtime: DOCTYPE entities for the current document, and persistent external entities. The untrusted tier, and the only one the
75
+ * expansion limits count by default.
76
+ */
77
+ const LIMIT_TIER_EXTERNAL = 'external';
78
+
79
+ /**
80
+ * @description Trusted: the five XML predefined entities, the caller's `namedEntities`, and every numeric reference.
81
+ */
82
+ const LIMIT_TIER_BASE = 'base';
83
+
84
+ /**
85
+ * @description Not a tier an entity belongs to but a switch on the tier filter. Selecting it makes every entity count against the limits regardless of where it
86
+ * came from.
87
+ */
88
+ const LIMIT_TIER_ALL = 'all';
89
+
90
+ /**
91
+ * @description Which side of the trust boundary an entity came from, as `#tierCounts` and the limit errors name it.
92
+ */
93
+ type LimitTier = typeof LIMIT_TIER_ALL | typeof LIMIT_TIER_BASE | typeof LIMIT_TIER_EXTERNAL;
94
+
95
+ /**
96
+ * @description The NCR action levels, in severity order. A higher number is a stricter action, and the resolver takes the maximum of the configured level and the
97
+ * minimum a codepoint range imposes, so a range can only ever make an entity stricter than the caller asked for — never more lenient.
98
+ */
99
+ const NCR_LEVEL = Object.freeze({ allow: 0, leave: 1, remove: 2, throw: 3 });
100
+
101
+ /**
102
+ * @description The action names {@link EntityDecoderNCROptions.onNCR} accepts, matching the keys of {@link NCR_LEVEL}.
103
+ */
104
+ type NcrLevelName = keyof typeof NCR_LEVEL;
105
+
106
+ /**
107
+ * @description The XML version that governs which codepoint ranges a numeric reference is checked against. Narrowed to the two values the constructor and
108
+ * {@link EntityDecoder.setXmlVersion} can actually store, because `#classifyNCR` compares it with `=== 1.0` and a third value would silently disable
109
+ * the XML 1.0 C0 check.
110
+ */
111
+ type XmlVersion = 1 | 1.1;
112
+
113
+ /**
114
+ * @description The C0 control codes XML 1.0 §2.2 permits as literal characters. Every other code in U+0001–U+001F is prohibited.
115
+ */
116
+ const XML10_ALLOWED_C0: ReadonlySet<number> = new Set([0x09, 0x0a, 0x0d]);
117
+
118
+ // ---------------------------------------------------------------------------
119
+ // Hook actions
120
+ // ---------------------------------------------------------------------------
121
+
122
+ /**
123
+ * @description What an {@link EntityRegistrationHook} returns. Use {@link ENTITY_ACTION} rather than the bare strings, so a typo is a type error instead of an
124
+ * entity that is accepted by default.
125
+ */
126
+ export type EntityHookAction = 'allow' | 'block' | 'throw';
127
+
128
+ /**
129
+ * @description A function-valued entity replacement: the `val` of the legacy `{ regex, val }` form when it is not a string. This decoder cannot use one — a
130
+ * function has no meaning without the regex it was meant to be matched against — so such an entry is dropped at registration rather than expanded.
131
+ */
132
+ export type EntityValFn = (match: string, captured: string, ...rest: Array<unknown>) => string;
133
+
134
+ /**
135
+ * @description Called once per entity _at registration time_, never during {@link EntityDecoder.decode}. Receives the name without `&` and `;` and the resolved
136
+ * string value, after any `{ regex, val }` envelope has been unwrapped.
137
+ *
138
+ * @param name - The entity name, e.g. `brand`.
139
+ * @param value - The string the entity expands to.
140
+ *
141
+ * @returns The action to take. Anything other than `block` and `throw` is treated as `allow`, so an unrecognised return value never rejects an entity
142
+ * by accident.
143
+ */
144
+ export type EntityRegistrationHook = (name: string, value: string) => EntityHookAction;
145
+
146
+ /**
147
+ * @description The three actions a registration hook can return, as a frozen object. Prefer it over the bare strings: the literals stay narrow string-literal
148
+ * types, so `ENTITY_ACTION.BLOK` fails to compile instead of silently registering the entity.
149
+ *
150
+ * @example
151
+ * ```typescript
152
+ * const decoder = new EntityDecoder({
153
+ * onInputEntity: () => ENTITY_ACTION.BLOCK,
154
+ * });
155
+ * ```;
156
+ */
157
+ export const ENTITY_ACTION: Readonly<{ ALLOW: 'allow'; BLOCK: 'block'; THROW: 'throw' }> = Object.freeze({
158
+ ALLOW: 'allow',
159
+ BLOCK: 'block',
160
+ THROW: 'throw',
161
+ } as const);
162
+
163
+ // ---------------------------------------------------------------------------
164
+ // Option types
165
+ // ---------------------------------------------------------------------------
166
+
167
+ /**
168
+ * @description Which entity categories count toward the expansion limits.
169
+ *
170
+ * - `'external'` — only input/runtime + persistent external entities. The default, and the only one that ignores the built-in XML entities.
171
+ * - `'base'` — only the built-in XML entities, the caller's `namedEntities`, and numeric references.
172
+ * - `'all'` — every entity regardless of tier.
173
+ * - `Array<'external' | 'base'>` — an explicit combination. An empty array is honoured literally: nothing counts, so the limits can never trip.
174
+ */
175
+ export type ApplyLimitsTo = 'external' | 'base' | 'all' | Array<'external' | 'base'>;
176
+
177
+ /**
178
+ * @description Ceilings on what a single document's entity references may cost. Both are cumulative across {@link EntityDecoder.decode} calls until
179
+ * {@link EntityDecoder.reset}, and both default to `0`, meaning unlimited. `0` — and any negative or non-numeric value — is unlimited, because the
180
+ * runtime tests `> 0` rather than truthiness of the configured number.
181
+ */
182
+ export interface EntityDecoderLimitOptions {
183
+ /**
184
+ * @description Maximum number of tracked entity references expanded per document. The check is `> maxTotalExpansions`, so a limit of `2` allows two expansions
185
+ * and throws on the third.
186
+ *
187
+ * @default 0
188
+ */
189
+ maxTotalExpansions?: number;
190
+
191
+ /**
192
+ * @description Maximum number of characters _added_ by expansion per document. Only the surplus counts: a reference whose replacement is no longer than the
193
+ * `&token;` it replaces contributes zero, and a shrinking one contributes nothing and cannot trip the limit.
194
+ *
195
+ * @default 0
196
+ */
197
+ maxExpandedLength?: number;
198
+
199
+ /**
200
+ * @description Which tiers count against both limits. Defaults to `'external'`, which is what keeps the built-in entities — including every numeric reference —
201
+ * from being able to trip a limit on a document the caller already trusts.
202
+ *
203
+ * @default 'external'
204
+ */
205
+ applyLimitsTo?: ApplyLimitsTo;
206
+ }
207
+
208
+ /**
209
+ * @description Policy for numeric character references. The three fields are flattened into numeric levels at construction so the decode loop never re-reads the
210
+ * object.
211
+ */
212
+ export interface EntityDecoderNCROptions {
213
+ /**
214
+ * @description The XML version whose codepoint restrictions apply. `1.0` prohibits the C0 controls U+0001–U+001F other than tab, newline and carriage return;
215
+ * `1.1` does not, since it permits them when written as references. Any value other than `1.1` is read as `1.0`.
216
+ *
217
+ * @default 1.0
218
+ */
219
+ xmlVersion?: 1.0 | 1.1;
220
+
221
+ /**
222
+ * @description The base action for every numeric reference. Codepoint ranges that carry a minimum — surrogates always, the XML 1.0 C0 controls under `1.0`, and
223
+ * null under `nullNCR` — take the stricter of the two, so this is a floor and not an override.
224
+ *
225
+ * @default 'allow'
226
+ */
227
+ onNCR?: 'allow' | 'leave' | 'remove' | 'throw';
228
+
229
+ /**
230
+ * @description The action for U+0000. `'allow'` and `'leave'` are clamped up to `'remove'`, so a null reference is always at least deleted.
231
+ *
232
+ * @default 'remove'
233
+ */
234
+ nullNCR?: 'remove' | 'throw';
235
+ }
236
+
237
+ /**
238
+ * @description Construction options for {@link EntityDecoder}. Every field is optional, and the defaults are the permissive ones.
239
+ */
240
+ export interface EntityDecoderOptions {
241
+ /**
242
+ * @description Extra named entities merged into the `base` map alongside the five XML predefined ones. A string value is used directly; a `{ regex, val }` or `{
243
+ * regx, val }` envelope is unwrapped to its `val`. Anything else — a number, `null`, a function, an envelope whose `val` is a function — is
244
+ * dropped, leaving the name unresolvable rather than failing the construction. Upstream's documentation says a value containing `&` is skipped
245
+ * here, to prevent recursive expansion. It is not: the code stores the value unchanged, and only {@link EntityDecoder.addExternalEntity} checks for
246
+ * `&`. Preserved as-is; see the note on the class.
247
+ *
248
+ * @default null
249
+ */
250
+ namedEntities?: Record<string, string | { regex: RegExp; val: string | EntityValFn }> | null;
251
+
252
+ /**
253
+ * @description Called once on the finished string. Receives `(resolved, original)` and must return a string; return `original` to reject the expansion outright,
254
+ * or a sanitised form of `resolved` to clean it. It is _not_ called for a string that never reaches the scanning loop — an empty string, a
255
+ * non-string, or any string with no `&` in it. A caller relying on `postCheck` to sanitise therefore has to know that a string with no ampersand is
256
+ * never inspected.
257
+ *
258
+ * @default null
259
+ */
260
+ postCheck?: ((resolved: string, original: string) => string) | null;
261
+
262
+ /**
263
+ * @description Whether numeric references expand at all. Turning it off leaves every one of them in the output verbatim — _except_ the codepoints that carry a
264
+ * minimum action of `remove` or stricter, which are still handled, because that classification runs first and is what makes the option safe to rely
265
+ * on.
266
+ *
267
+ * @default true
268
+ */
269
+ numericAllowed?: boolean;
270
+
271
+ /**
272
+ * @description Names to keep as literal `&name;` text, matched against the token with no `&` or `;`. Numeric references are matched as `#38` or `#x26`.
273
+ *
274
+ * @default [ ]
275
+ */
276
+ leave?: Array<string>;
277
+
278
+ /**
279
+ * @description Names to delete outright, matched the same way as {@link EntityDecoderOptions.leave}. A removed reference is charged to the `external` tier even
280
+ * when the name is a built-in one, so a document full of removed built-ins can trip an `applyLimitsTo: 'external'` limit it would not otherwise be
281
+ * subject to. Preserved as-is; the only in-code comment claims the charge is for unknown references, which is not what distinguishes them.
282
+ *
283
+ * @default [ ]
284
+ */
285
+ remove?: Array<string>;
286
+
287
+ /**
288
+ * @description Ceilings on expansion count and expanded length. See {@link EntityDecoderLimitOptions}.
289
+ */
290
+ limit?: EntityDecoderLimitOptions;
291
+
292
+ /**
293
+ * @description Policy for numeric references. See {@link EntityDecoderNCROptions}.
294
+ */
295
+ ncr?: EntityDecoderNCROptions;
296
+
297
+ /**
298
+ * @description Called once per entity as it is registered through {@link EntityDecoder.setExternalEntities} or {@link EntityDecoder.addExternalEntity}. `block`
299
+ * skips the entity, `throw` aborts the whole registration, anything else registers it. With {@link EntityDecoder.setExternalEntities} a `throw`
300
+ * leaves the previous external map in place, because the replacement is only assigned once every entry has passed.
301
+ *
302
+ * @default null
303
+ */
304
+ onExternalEntity?: EntityRegistrationHook | null;
305
+
306
+ /**
307
+ * @description Called once per entity as it is registered through {@link EntityDecoder.addInputEntities}. Same contract as
308
+ * {@link EntityDecoderOptions.onExternalEntity}, and unlike it the hook is not the only filter — see the class note on name validation.
309
+ *
310
+ * @default null
311
+ */
312
+ onInputEntity?: EntityRegistrationHook | null;
313
+ }
314
+
315
+ // ---------------------------------------------------------------------------
316
+ // Internal types
317
+ //
318
+ // Looser than the exported option types on purpose. The runtime inspects whatever it is handed, and an entry it
319
+ // cannot read is dropped rather than rejected, so the helpers have to be able to describe an entry the public
320
+ // types claim cannot exist.
321
+ // ---------------------------------------------------------------------------
322
+
323
+ /**
324
+ * @description The value side of a registration map as the merge helper reads it: a ready string, or a `{ regex | regx, val }` envelope whose `val` may itself be
325
+ * absent, a string, or a function.
326
+ */
327
+ type EntityInputValue =
328
+ | string
329
+ | { readonly regex?: RegExp | undefined; readonly regx?: RegExp | undefined; readonly val?: string | EntityValFn | undefined };
330
+
331
+ /**
332
+ * @description One registration map, or nothing. `null` and `undefined` both mean "no entities here", which is how `setExternalEntities(null)` clears the map and
333
+ * how the constructor declines to pass `namedEntities`.
334
+ */
335
+ type EntityInputMap = Readonly<Record<string, EntityInputValue>> | null | undefined;
336
+
337
+ /**
338
+ * @description What one reference expanded to, tagged with the tier its limit accounting charges. The field is the replacement text itself for a named entity, the
339
+ * character for a numeric reference, and `''` for a removed one — the three shapes the walk pushes into its output.
340
+ */
341
+ type ResolvedEntity = { value: string; tier: LimitTier };
342
+
343
+ /**
344
+ * @description A registration context, as it appears in the error a rejecting hook produces.
345
+ */
346
+ type HookContext = 'external' | 'input';
347
+
348
+ // ---------------------------------------------------------------------------
349
+ // Helpers
350
+ // ---------------------------------------------------------------------------
351
+
352
+ /**
353
+ * @description Reject an entity name that could never be written as a reference. `#` is refused positionally rather than by the character sweep, because a name
354
+ * starting with `#` is a numeric reference's token and would collide with `#resolveNCR`. Everything else is refused per character.
355
+ *
356
+ * @param name - The name to check.
357
+ *
358
+ * @returns An effect producing the name, unchanged, so the call can be inlined. Fails with {@link XmlError} and the `InvalidEntityName` reason,
359
+ * carrying the offending character. The `[EntityReplacer]` prefix in the message is preserved verbatim from the original throw, despite naming a
360
+ * class this decoder does not have — it is load-bearing for anything matching on it.
361
+ */
362
+ const checkEntityName = (name: string): Effect.Effect<string, XmlError> => {
363
+ if (name.charCodeAt(0) === CODE_HASH) {
364
+ return Effect.fail(
365
+ new XmlError({
366
+ reason: { _tag: 'InvalidEntityName', name, character: '#' },
367
+ message: `[EntityReplacer] Invalid character '#' in entity name: "${name}"`,
368
+ })
369
+ );
370
+ }
371
+ for (const ch of name) {
372
+ if (SPECIAL_CHARS.has(ch)) {
373
+ return Effect.fail(
374
+ new XmlError({
375
+ reason: { _tag: 'InvalidEntityName', name, character: ch },
376
+ message: `[EntityReplacer] Invalid character '${ch}' in entity name: "${name}"`,
377
+ })
378
+ );
379
+ }
380
+ }
381
+ return Effect.succeed(name);
382
+ };
383
+
384
+ /**
385
+ * @description Flatten registration maps into one name to string map, later maps winning over earlier ones for the same name. The result is a null-prototype
386
+ * object, not a `Map`. That is not incidental: a `Map` iterates in pure insertion order, while `Object.keys` lifts array-index-like names to the
387
+ * front in numeric order, and the registration hooks observe that order. A name of `"2"` registered after `"brand"` reaches the hook first here and
388
+ * second in a `Map`.
389
+ *
390
+ * @param maps - The maps to merge. A falsy entry — `null`, `undefined`, `''`, `0` — contributes nothing rather than throwing.
391
+ *
392
+ * @returns A null-prototype object of own string-valued entries. Nothing from `Object.prototype` can be read out of it, so a document naming
393
+ * `constructor` or `toString` finds nothing. Each entry is read through {@link flattenEntityValue}, so an entry that cannot be reduced to a string
394
+ * is absent rather than present-and-unusable.
395
+ */
396
+ function mergeEntityMaps(...maps: ReadonlyArray<EntityInputMap>): Record<string, string> {
397
+ const out: Record<string, string> = Object.create(null);
398
+ for (const map of maps) {
399
+ if (!map) continue;
400
+ for (const key of Object.keys(map)) {
401
+ const value = flattenEntityValue(map[key]);
402
+ if (value !== undefined) out[key] = value;
403
+ }
404
+ }
405
+ return out;
406
+ }
407
+
408
+ /**
409
+ * @description Reduce one registration entry to the string a reference to it expands to, or to nothing when the entry is a form the scanner has no use for. Three
410
+ * shapes survive: the string itself, and a `{ regex | regx, val }` envelope whose `val` is a string. Everything else — a number, `null`, `undefined`,
411
+ * a bare function, an envelope whose `val` is a function — has no string to substitute, so the name is dropped and a reference to it comes back out
412
+ * as the text it was written as. Dropping is silent on purpose: the runtime inspects whatever it is handed, and failing the construction over one
413
+ * unreadable entry would take every other entity in the table down with it.
414
+ *
415
+ * @param raw - The entry as it arrived, in whatever shape the caller supplied it — including no entry at all, which a table with a hole in it
416
+ * produces.
417
+ *
418
+ * @returns The replacement string, or `undefined` when the entry cannot be read. A name registered to the empty string yields `''`, which is why
419
+ * callers compare against `undefined` rather than testing for emptiness.
420
+ */
421
+ function flattenEntityValue(raw: EntityInputValue | undefined): string | undefined {
422
+ if (Predicate.isString(raw)) return raw;
423
+
424
+ // The `raw &&` in upstream is a null check: every object is truthy, so it only ever rejects
425
+ // `null` and `undefined` here, and the object check then rejects a bare function value.
426
+ if (Predicate.isNullish(raw) || !Predicate.isObject(raw) || raw.val === undefined) return undefined;
427
+
428
+ const val = raw.val;
429
+ // A function `val` has no scanner equivalent and is dropped, upstream included.
430
+ return Predicate.isString(val) ? val : undefined;
431
+ }
432
+
433
+ /**
434
+ * @description Read one own entry out of a null-prototype entity map.
435
+ *
436
+ * @param map - The map to read.
437
+ * @param key - The entity name.
438
+ *
439
+ * @returns The registered string, or `undefined` when the name is not an own key. A name registered to the empty string returns `''`, which is why
440
+ * callers must compare against `undefined` rather than test for emptiness.
441
+ */
442
+ function ownEntity(map: Readonly<Record<string, string>>, key: string): string | undefined {
443
+ // Upstream tests `name in map`, which reads as "is this name present at all". `Object.hasOwn` is
444
+ // the same question asked explicitly, and it is the honest shape for the answer: an absent key
445
+ // yields `undefined` and a present one yields the stored string, so the `string | undefined` this
446
+ // returns is the real type rather than something an assertion has to paper over. The two maps are
447
+ // null-prototype objects, so `in` and `hasOwn` cannot disagree here.
448
+ if (!Object.hasOwn(map, key)) return undefined;
449
+ return map[key];
450
+ }
451
+
452
+ /**
453
+ * @description Normalise the `applyLimitsTo` option into the set of tiers that count against the limits.
454
+ *
455
+ * @param raw - The configured value.
456
+ *
457
+ * @returns The tier set. An unrecognised string falls back to `external` rather than to no filtering at all, so a typo cannot silently disable the
458
+ * limits. An array is taken as given, which is why an empty array disables limit accounting entirely while an empty string falls back to
459
+ * `external`.
460
+ */
461
+ function parseLimitTiers(raw: ApplyLimitsTo | undefined): ReadonlySet<LimitTier> {
462
+ if (!raw || raw === LIMIT_TIER_EXTERNAL) return new Set([LIMIT_TIER_EXTERNAL]);
463
+ if (raw === LIMIT_TIER_ALL) return new Set([LIMIT_TIER_ALL]);
464
+ if (raw === LIMIT_TIER_BASE) return new Set([LIMIT_TIER_BASE]);
465
+ if (Array.isArray(raw)) return new Set(raw);
466
+ return new Set([LIMIT_TIER_EXTERNAL]);
467
+ }
468
+
469
+ /**
470
+ * @description Read one level out of {@link NCR_LEVEL} by name.
471
+ *
472
+ * @param name - The configured action name, or nothing.
473
+ * @param fallback - The level to use when the name is absent.
474
+ *
475
+ * @returns The level. A name the table does not carry also yields `fallback`, so a value outside the union degrades to the default action rather than
476
+ * to `NaN`.
477
+ */
478
+ function ncrLevelOf(name: NcrLevelName | undefined, fallback: number): number {
479
+ if (name === undefined) return fallback;
480
+ return NCR_LEVEL[name] ?? fallback;
481
+ }
482
+
483
+ /**
484
+ * @description Flatten the `ncr` option into the three numeric fields the decode loop reads, so nothing has to be re-derived per reference.
485
+ *
486
+ * @param ncr - The configured policy, or nothing.
487
+ *
488
+ * @returns The XML version, the base action level, and the null action level already clamped up to `remove`.
489
+ */
490
+ function parseNCRConfig(ncr: EntityDecoderNCROptions | undefined): { xmlVersion: XmlVersion; onLevel: number; nullLevel: number } {
491
+ if (!ncr) {
492
+ return { xmlVersion: 1.0, onLevel: NCR_LEVEL.allow, nullLevel: NCR_LEVEL.remove };
493
+ }
494
+ const xmlVersion: XmlVersion = ncr.xmlVersion === 1.1 ? 1.1 : 1.0;
495
+ const onLevel = ncrLevelOf(ncr.onNCR, NCR_LEVEL.allow);
496
+ // Null is never safe to emit, so anything weaker than `remove` is raised to it before it is stored.
497
+ const nullLevel = Math.max(ncrLevelOf(ncr.nullNCR, NCR_LEVEL.remove), NCR_LEVEL.remove);
498
+ return { xmlVersion, onLevel, nullLevel };
499
+ }
500
+
501
+ /**
502
+ * @description Resolve {@link EntityDecoderOptions.postCheck} to something the decode loop can call unconditionally, or to the identity function. The fallback
503
+ * exists so the path that actually scanned does not have to test for the option, while the two fast paths that return before the scan still skip the
504
+ * call entirely. A non-function value is treated as an absent option rather than rejected, matching the hook rules: a mistyped option disables its
505
+ * feature instead of failing the construction.
506
+ *
507
+ * @param raw - The configured hook, or nothing.
508
+ *
509
+ * @returns The hook itself, or a function returning its first argument.
510
+ */
511
+ function readPostCheck(raw: EntityDecoderOptions['postCheck']): (resolved: string, original: string) => string {
512
+ if (Predicate.isFunction(raw)) return raw;
513
+ return r => r;
514
+ }
515
+
516
+ /**
517
+ * @description Resolve a registration hook option to something safe to call, under the same non-function rule as {@link readPostCheck}.
518
+ *
519
+ * @param raw - The configured hook, or nothing.
520
+ *
521
+ * @returns The hook itself, or `null` for an absent option and for a value that is not a function. `null` is what lets every registration path ask
522
+ * unconditionally: a hook that is not there accepts.
523
+ */
524
+ function readHook(raw: EntityRegistrationHook | null | undefined): EntityRegistrationHook | null {
525
+ if (Predicate.isFunction(raw)) return raw;
526
+ return null;
527
+ }
528
+
529
+ /**
530
+ * @description Read one of the two entity-name lists as a set, under the same missing-value rule as the other options: absent is empty, not an error. The
531
+ * `Array.isArray` test rather than a truthiness one is what keeps a mistyped list from reaching `new Set` and throwing there, so a caller's typo
532
+ * disables the list instead of taking the decoder down. Matching against a set is also why a name in both lists is decided by the order
533
+ * {@link EntityDecoder.decode} consults them in, not by the order the caller wrote them in.
534
+ *
535
+ * @param raw - The configured list, or nothing.
536
+ *
537
+ * @returns The names as a set, empty when the option is absent or is not an array.
538
+ */
539
+ function readNameList(raw: Array<string> | undefined): ReadonlySet<string> {
540
+ if (Array.isArray(raw)) return new Set(raw);
541
+ return new Set();
542
+ }
543
+
544
+ /**
545
+ * @description Scan forward from a `&` for the `;` that would close the reference it opens, giving up once more than {@link MAX_TOKEN_LENGTH} characters have
546
+ * passed since the `&`.
547
+ *
548
+ * @param str - The string being scanned.
549
+ * @param ampersand - The index of the `&`.
550
+ *
551
+ * @returns The index of the closing `;`, or `-1` when the run holds none inside the window. A `;` one character past the `&` is returned rather than
552
+ * refused: that is the empty token `&;`, and the caller decides it is not a reference.
553
+ */
554
+ function scanTokenEnd(str: string, ampersand: number): number {
555
+ const len = str.length;
556
+ let j = ampersand + 1;
557
+ while (j < len && str.charCodeAt(j) !== CODE_SEMICOLON && j - ampersand <= MAX_TOKEN_LENGTH) j++;
558
+ if (j >= len || str.charCodeAt(j) !== CODE_SEMICOLON) return -1;
559
+ return j;
560
+ }
561
+
562
+ /**
563
+ * @description Single-pass, zero-regex entity decoder for XML and HTML content.
564
+ *
565
+ * ### Entity lookup priority
566
+ *
567
+ * 1. **input / runtime** — injected per document through {@link EntityDecoder.addInputEntities}
568
+ * 2. **persistent external** — set through {@link EntityDecoder.setExternalEntities} and {@link EntityDecoder.addExternalEntity}, surviving
569
+ * {@link EntityDecoder.reset}
570
+ * 3. **base** — the five XML predefined entities plus the constructor's `namedEntities` Both input and external resolve as the `external` tier for limit
571
+ * purposes, because both are injected at runtime. Numeric references (`&#NNN;`, `&#xHH;`) resolve directly through `String.fromCodePoint` and are
572
+ * always `base` tier: they cannot recurse, so a limit that counted them would only punish a document that spells its characters out.
573
+ *
574
+ * ### Upstream behaviour preserved
575
+ *
576
+ * Several quirks of the original are kept deliberately, because a consumer's output already depends on them:
577
+ *
578
+ * - A value containing `&` is **not** filtered from `namedEntities` or `setExternalEntities`, contrary to the documentation. Only
579
+ * {@link EntityDecoder.addExternalEntity} checks, and it drops the entry rather than storing it, so the same name registered either way can resolve
580
+ * to nothing.
581
+ * - {@link EntityDecoderOptions.postCheck} is skipped entirely for input that never reaches the scan — an empty string, a non-string, or a string with
582
+ * no `&`.
583
+ * - {@link EntityDecoder.decode} returns a non-string argument unchanged, despite being typed `string`.
584
+ * - The expansion-limit errors are prefixed `EntityReplacer`, not `EntityDecoder`.
585
+ * - Nothing in XML 1.0 §2.2 is enforced for U+007F–U+009F or for the U+FFFE/U+FFFF noncharacters, and the sweep for `&` leaves a name of
586
+ * {@link MAX_TOKEN_LENGTH} + 1 characters unresolvable.
587
+ * - Numeric references are parsed with `parseInt`, so a leading space, sign, or trailing garbage is accepted: `&# 41;`, `&#x+41;` and `&#41zz;` all
588
+ * decode, and `&#0x41;` parses as a null reference rather than `A`.
589
+ * - {@link EntityDecoder.addInputEntities} validates no names, so a `#`-prefixed or `&`-bearing name registers without complaint, where the two
590
+ * external setters would throw.
591
+ *
592
+ * @example
593
+ * ```typescript
594
+ * const decoder = new EntityDecoder({ namedEntities: { copy: '©' } });
595
+ * decoder.setExternalEntities({ brand: 'Acme' });
596
+ * decoder.addInputEntities({ version: '1.0' });
597
+ *
598
+ * decoder.decode('&brand; v&version; &copy;'); // 'Acme v1.0 ©'
599
+ * decoder.decode('&#x26;#38;'); // '&&' — one pass, the output is never re-scanned
600
+ *
601
+ * decoder.reset(); // drops the input entities and the counters, keeps the external ones
602
+ * ```;
603
+ */
604
+ export class EntityDecoder {
605
+ /**
606
+ * @description {@link EntityDecoderLimitOptions.maxTotalExpansions}, or `0` for unlimited. A negative number or `NaN` is also unlimited, since the decode loop
607
+ * tests `> 0`.
608
+ */
609
+ readonly #maxTotalExpansions: number;
610
+
611
+ /**
612
+ * @description {@link EntityDecoderLimitOptions.maxExpandedLength}, or `0` for unlimited, read the same way as `#maxTotalExpansions`.
613
+ */
614
+ readonly #maxExpandedLength: number;
615
+
616
+ /**
617
+ * @description {@link EntityDecoderOptions.postCheck}, or the identity function — so the decode loop can call it unconditionally on the path that actually
618
+ * scanned, and never on the two fast paths that return early.
619
+ */
620
+ readonly #postCheck: (resolved: string, original: string) => string;
621
+
622
+ /**
623
+ * @description The resolved tier filter. See {@link parseLimitTiers}.
624
+ */
625
+ readonly #limitTiers: ReadonlySet<LimitTier>;
626
+
627
+ /**
628
+ * @description {@link EntityDecoderOptions.numericAllowed}. Only an explicit `false` turns it off, so an absent option cannot disable it.
629
+ */
630
+ readonly #numericAllowed: boolean;
631
+
632
+ /**
633
+ * @description The five XML predefined entities plus `namedEntities`, merged once at construction and never written again. The built-ins lose to a
634
+ * `namedEntities` entry of the same name, since it is merged second.
635
+ */
636
+ readonly #baseMap: Record<string, string>;
637
+
638
+ /**
639
+ * @description Persistent external entities, as a null-prototype object. Replaced wholesale by {@link EntityDecoder.setExternalEntities} and added to by
640
+ * {@link EntityDecoder.addExternalEntity}, and never touched by {@link EntityDecoder.reset} — that is the whole distinction from the input map.
641
+ */
642
+ #externalMap: Record<string, string>;
643
+
644
+ /**
645
+ * @description DOCTYPE entities for the document being processed, as a null-prototype object. Wiped by both {@link EntityDecoder.reset} and
646
+ * {@link EntityDecoder.addInputEntities}.
647
+ */
648
+ #inputMap: Record<string, string>;
649
+
650
+ /**
651
+ * @description Tracked expansions since the last reset. Cumulative across {@link EntityDecoder.decode} calls, which is what makes a limit a per-document budget
652
+ * rather than a per-call one. Deliberately not reset by a thrown limit error, so the over-limit count is what the error message reports.
653
+ */
654
+ #totalExpansions: number;
655
+
656
+ /**
657
+ * @description Characters _added_ by expansion since the last reset, accumulated the same way as `#totalExpansions`. Only positive contributions are counted.
658
+ */
659
+ #expandedLength: number;
660
+
661
+ /**
662
+ * @description {@link EntityDecoderOptions.remove} as a set, or empty. Checked before every other classification, so a name in here is deleted without the name
663
+ * ever being resolved.
664
+ */
665
+ readonly #removeSet: ReadonlySet<string>;
666
+
667
+ /**
668
+ * @description {@link EntityDecoderOptions.leave} as a set, or empty. Checked after `remove` and before the numeric test, so a name in here is emitted as the
669
+ * original `&name;` text.
670
+ */
671
+ readonly #leaveSet: ReadonlySet<string>;
672
+
673
+ /**
674
+ * @description The XML version governing numeric classification. Mutable, because a `<?xml version?>` declaration is normally only known after the decoder
675
+ * exists; see {@link EntityDecoder.setXmlVersion}.
676
+ */
677
+ #ncrXmlVersion: XmlVersion;
678
+
679
+ /**
680
+ * @description {@link EntityDecoderNCROptions.onNCR} as a level from {@link NCR_LEVEL}. A floor, not an override: the resolver takes the maximum of this and
681
+ * whatever minimum a codepoint range imposes.
682
+ */
683
+ readonly #ncrOnLevel: number;
684
+
685
+ /**
686
+ * @description {@link EntityDecoderNCROptions.nullNCR} as a level from {@link NCR_LEVEL}, already clamped to `remove` or stricter.
687
+ */
688
+ readonly #ncrNullLevel: number;
689
+
690
+ /**
691
+ * @description {@link EntityDecoderOptions.onExternalEntity}, or `null` when absent or not a function. A non-function is dropped rather than rejected, so a
692
+ * mistyped option disables the hook instead of failing the construction.
693
+ */
694
+ readonly #onExternalEntity: EntityRegistrationHook | null;
695
+
696
+ /**
697
+ * @description {@link EntityDecoderOptions.onInputEntity}, or `null`, under the same non-function rule as `#onExternalEntity`.
698
+ */
699
+ readonly #onInputEntity: EntityRegistrationHook | null;
700
+
701
+ /**
702
+ * @description Create a decoder. A factory rather than a constructor, because it refuses a `null` options object. Every field is optional, so `null` is not "a
703
+ * decoder with the defaults" — a caller who wrote it meant something the signature does not allow, and a decoder built from it would be
704
+ * indistinguishable from one built from `{}` while hiding the mistake. Saying so is worth a factory; `EntityDecoderOptions` is a plain object and
705
+ * nothing else about construction can fail.
706
+ *
707
+ * @example
708
+ * ```typescript
709
+ * const decoder = yield* EntityDecoder.make({ numericAllowed: false });
710
+ * yield* decoder.decode('caf&eacute;'); // 'café'
711
+ * ```;
712
+ *
713
+ * @param options - Configuration. See {@link EntityDecoderOptions}. Defaults to every field's own default.
714
+ *
715
+ * @returns An effect producing the decoder. Fails with {@link XmlError} and the `MissingOptions` reason for a `null`.
716
+ */
717
+ static make = (options: EntityDecoderOptions = {}): Effect.Effect<EntityDecoder, XmlError> =>
718
+ Predicate.isNullish(options)
719
+ ? Effect.fail(
720
+ new XmlError({
721
+ reason: { _tag: 'MissingOptions', parameter: 'options' },
722
+ message: 'EntityDecoder.make: options is required. Use make({}) for a decoder with every default.',
723
+ })
724
+ )
725
+ : Effect.succeed(new EntityDecoder(options));
726
+
727
+ /**
728
+ * @description Create a decoder. Every option is resolved here into the flat fields the decode loop reads, so nothing per-reference has to re-derive it. The
729
+ * options whose wrong type disables them rather than failing the construction — the two hooks, the two name lists — are read through
730
+ * {@link readHook} and {@link readNameList}, so that rule is written once instead of four times.
731
+ *
732
+ * @param resolved - Configuration, already checked. See {@link EntityDecoderOptions}.
733
+ */
734
+ private constructor(resolved: EntityDecoderOptions) {
735
+ // `options.limit` is read first, deliberately: it is the first property the original touched, so
736
+ // the property a `null` would have faulted on, and keeping that order means the reason still
737
+ // names it. The option stays a local — every value the decode loop needs is flattened out of it
738
+ // below, so retaining it on the instance would only be a way to observe the option back.
739
+ const limit = resolved.limit ?? {};
740
+ this.#maxTotalExpansions = limit.maxTotalExpansions || 0;
741
+ this.#maxExpandedLength = limit.maxExpandedLength || 0;
742
+ this.#postCheck = readPostCheck(resolved.postCheck);
743
+ this.#limitTiers = parseLimitTiers(limit.applyLimitsTo ?? LIMIT_TIER_EXTERNAL);
744
+ this.#numericAllowed = resolved.numericAllowed ?? true;
745
+ this.#baseMap = mergeEntityMaps(DEFAULT_XML_ENTITIES, resolved.namedEntities || null);
746
+
747
+ this.#externalMap = Object.create(null);
748
+ this.#inputMap = Object.create(null);
749
+ this.#totalExpansions = 0;
750
+ this.#expandedLength = 0;
751
+
752
+ this.#removeSet = readNameList(resolved.remove);
753
+ this.#leaveSet = readNameList(resolved.leave);
754
+
755
+ const ncrConfig = parseNCRConfig(resolved.ncr);
756
+ this.#ncrXmlVersion = ncrConfig.xmlVersion;
757
+ this.#ncrOnLevel = ncrConfig.onLevel;
758
+ this.#ncrNullLevel = ncrConfig.nullLevel;
759
+
760
+ this.#onExternalEntity = readHook(resolved.onExternalEntity);
761
+ this.#onInputEntity = readHook(resolved.onInputEntity);
762
+ }
763
+
764
+ /**
765
+ * @description Ask a registration hook about one name and value.
766
+ *
767
+ * @param hook - The hook, or `null`. A `null` hook accepts, which is what lets {@link EntityDecoder.addExternalEntity} call this unconditionally.
768
+ * @param name - The entity name, without `&` or `;`.
769
+ * @param value - The resolved value, after any `{ regex, val }` envelope was unwrapped.
770
+ * @param context - Which registration is in progress, for the error message.
771
+ *
772
+ * @returns An effect producing `true` to register, `false` to skip silently. Fails with {@link XmlError} and the `EntityRejected` reason when the
773
+ * hook returns `throw`. The message quotes the entity, so it is the only record left that a document was rejected.
774
+ */
775
+ #applyRegistrationHook(hook: EntityRegistrationHook | null, name: string, value: string, context: HookContext): Effect.Effect<boolean, XmlError> {
776
+ if (!hook) return Effect.succeed(true); // no hook to ask
777
+ const action = hook(name, value);
778
+ if (action === ENTITY_ACTION.BLOCK) return Effect.succeed(false);
779
+ if (action === ENTITY_ACTION.THROW) {
780
+ return Effect.fail(
781
+ new XmlError({
782
+ reason: { _tag: 'EntityRejected', context, name },
783
+ message: `[EntityDecoder] Registration of ${context} entity "&${name};" was rejected by hook`,
784
+ })
785
+ );
786
+ }
787
+ return Effect.succeed(true); // ALLOW, and anything unrecognised, accepts
788
+ }
789
+
790
+ /**
791
+ * @description Replace the whole set of persistent external entities. Every key is validated _before_ any value is read, so an invalid name fails even when its
792
+ * value is a form the merge would have dropped. A non-object or `null` map clears the set without validating anything.
793
+ *
794
+ * @param map - The entities to register, or nothing to clear.
795
+ *
796
+ * @returns An effect that registers the map. Fails with {@link XmlError} when a key contains a character from {@link SPECIAL_CHARS} or begins with
797
+ * `#` (`InvalidEntityName`), or when {@link EntityDecoderOptions.onExternalEntity} returns `throw` (`EntityRejected`). A rejection from the hook
798
+ * aborts before the assignment, so the previous map survives.
799
+ */
800
+ setExternalEntities = Effect.fnUntraced(function* (
801
+ this: EntityDecoder,
802
+ map: Record<string, string | { regex: RegExp; val: string | EntityValFn }>
803
+ ): Effect.fn.Return<void, XmlError> {
804
+ if (map) {
805
+ for (const key of Object.keys(map)) {
806
+ yield* checkEntityName(key);
807
+ }
808
+ }
809
+ if (!this.#onExternalEntity) {
810
+ this.#externalMap = mergeEntityMaps(map);
811
+ return;
812
+ }
813
+ // With a hook, values are flattened first and the hook sees what will actually be stored.
814
+ const flat = mergeEntityMaps(map);
815
+ const filtered: Record<string, string> = Object.create(null);
816
+ for (const [name, value] of Object.entries(flat)) {
817
+ if (yield* this.#applyRegistrationHook(this.#onExternalEntity, name, value, 'external')) {
818
+ filtered[name] = value;
819
+ }
820
+ }
821
+ this.#externalMap = filtered;
822
+ });
823
+
824
+ /**
825
+ * @description Add one persistent external entity, keeping whatever is already registered. This is the only registration path that refuses a value containing
826
+ * `&`; the two map setters store one unchanged. The omission is upstream's, and it is kept: the same name registered through either route can end
827
+ * up resolving, or not resolving at all.
828
+ *
829
+ * @param key - The entity name, without `&` or `;`.
830
+ * @param value - The replacement text.
831
+ *
832
+ * @returns An effect that adds the entity. Fails with {@link XmlError} and the `InvalidEntityName` reason when `key` contains a character from
833
+ * {@link SPECIAL_CHARS} or begins with `#`, or with the `EntityRejected` reason when {@link EntityDecoderOptions.onExternalEntity} returns
834
+ * `throw`.
835
+ */
836
+ addExternalEntity = Effect.fnUntraced(function* (this: EntityDecoder, key: string, value: string): Effect.fn.Return<void, XmlError> {
837
+ yield* checkEntityName(key);
838
+ // The two guards are unreachable from typed code — `value` is a `string` — and are kept for
839
+ // untyped callers, which is the only way to reach them.
840
+ if (Predicate.isString(value) && value.indexOf('&') === -1) {
841
+ if (yield* this.#applyRegistrationHook(this.#onExternalEntity, key, value, 'external')) {
842
+ this.#externalMap[key] = value;
843
+ }
844
+ }
845
+ });
846
+
847
+ /**
848
+ * @description Register the DOCTYPE entities for the document about to be decoded, replacing any previous set and clearing both counters. Unlike the external
849
+ * setters, no name is validated: a `#`-prefixed name, or one containing `&` or `<`, registers without complaint. A `#`-prefixed name is then
850
+ * unreachable, since `decode` routes `#`-prefixed tokens to the numeric pipeline first.
851
+ *
852
+ * @param map - The entities to register, or nothing to clear.
853
+ *
854
+ * @returns An effect that registers the map. Fails with {@link XmlError} and the `EntityRejected` reason when
855
+ * {@link EntityDecoderOptions.onInputEntity} returns `throw`. The counters have already been cleared by then.
856
+ */
857
+ addInputEntities = Effect.fnUntraced(function* (
858
+ this: EntityDecoder,
859
+ map: Record<string, string | { regx: RegExp; val: string | EntityValFn } | { regex: RegExp; val: string | EntityValFn }>
860
+ ): Effect.fn.Return<void, XmlError> {
861
+ // Cleared first and unconditionally, so registering entities is itself the start of a new
862
+ // document's budget — including when the call goes on to fail.
863
+ this.#totalExpansions = 0;
864
+ this.#expandedLength = 0;
865
+ if (!this.#onInputEntity) {
866
+ this.#inputMap = mergeEntityMaps(map);
867
+ return;
868
+ }
869
+ const flat = mergeEntityMaps(map);
870
+ const filtered: Record<string, string> = Object.create(null);
871
+ for (const [name, value] of Object.entries(flat)) {
872
+ if (yield* this.#applyRegistrationHook(this.#onInputEntity, name, value, 'input')) {
873
+ filtered[name] = value;
874
+ }
875
+ }
876
+ this.#inputMap = filtered;
877
+ });
878
+
879
+ /**
880
+ * @description Start a new document: drop the input entities and both counters. The persistent external entities, the base map, the limits, the NCR policy and
881
+ * the XML version all survive, which is the difference between this and constructing a fresh decoder.
882
+ *
883
+ * @returns This decoder, so a call can be chained onto the document it ends.
884
+ */
885
+ reset(): this {
886
+ this.#inputMap = Object.create(null);
887
+ this.#totalExpansions = 0;
888
+ this.#expandedLength = 0;
889
+ return this;
890
+ }
891
+
892
+ /**
893
+ * @description Set the XML version used to classify numeric references, once a `<?xml version="…"?>` declaration has been read. Only the exact number `1.1`
894
+ * selects XML 1.1; `1.0`, `1.15`, `'1.1'` and `NaN` all become `1.0`, so the stricter classification is the default rather than the looser one.
895
+ *
896
+ * @param version - The declared version.
897
+ *
898
+ * @returns Nothing.
899
+ */
900
+ setXmlVersion(version: number): void {
901
+ this.#ncrXmlVersion = version === 1.1 ? 1.1 : 1.0;
902
+ }
903
+
904
+ /**
905
+ * @description Expand every entity reference in a string, in one pass. The output is never re-scanned, so no expansion can produce a _second_ one: a registered
906
+ * value that itself contains reference text reaches the caller as that literal text, unexpanded. What the limits bound is the growth of this single
907
+ * pass — how much one round of expansion can add. Three inputs return before the scan and therefore never reach
908
+ * {@link EntityDecoderOptions.postCheck}: a non-string, the empty string, and any string with no `&` in it. The scan itself is `#expandAll`; what
909
+ * this method adds is the three inputs that skip it and the single join of what it collected.
910
+ *
911
+ * @example
912
+ * ```typescript
913
+ * import { Effect } from 'effect';
914
+ * import { EntityDecoder } from '@endevops/effect-xml-codec';
915
+ *
916
+ * const decoder = new EntityDecoder({ namedEntities: { copy: '©' } });
917
+ * Effect.runSync(Effect.orElseSucceed(decoder.addExternalEntity('brand', 'Acme'), () => undefined));
918
+ * Effect.runSync(decoder.decode('&brand; &copy;')); // 'Acme ©'
919
+ * ```;
920
+ *
921
+ * @param str - The string to decode.
922
+ *
923
+ * @returns An effect producing the decoded string. A non-string argument comes back as the same non-string, which the `string` return type does not
924
+ * describe but callers passing untyped values depend on. Fails with {@link XmlError} when a numeric reference is prohibited under the configured
925
+ * policy (`ProhibitedCharacterReference`), or when a tracked tier would exceed {@link EntityDecoderLimitOptions.maxTotalExpansions}
926
+ * (`ExpansionLimitExceeded`) or {@link EntityDecoderLimitOptions.maxExpandedLength} (`ExpandedLengthLimitExceeded`). The two limit messages keep
927
+ * the `EntityReplacer` prefix from the original throw, which named a class this decoder does not have.
928
+ */
929
+ decode = Effect.fnUntraced(function* (this: EntityDecoder, str: string): Effect.fn.Return<string, XmlError> {
930
+ if (!Predicate.isString(str) || str.length === 0) return str;
931
+ if (str.indexOf('&') === -1) return str; // nothing here can be a reference
932
+
933
+ const chunks = yield* this.#expandAll(str);
934
+
935
+ // `chunks` is empty exactly when nothing was replaced, in which case the input is its own result.
936
+ const result = chunks.length === 0 ? str : chunks.join('');
937
+
938
+ return this.#postCheck(result, str);
939
+ });
940
+
941
+ /**
942
+ * @description Walk the string once and collect the pieces of every reference that resolved. Two advance rules make the walk terminate and keep it correct: an
943
+ * `&` that turns out to open nothing moves the cursor by one character rather than to the end of its run, so a second `&` in the same text is still
944
+ * found; and a reference that did resolve moves it to just past the `;`, so the text that was substituted for it is never looked at again — that is
945
+ * what makes the pass single, and a registered value containing `&` cannot expand a second level. What a reference becomes is `#resolveToken`'s to
946
+ * decide and what it costs is `#chargeExpansion`'s to apply, which leaves the scanning here as the only thing with a rule of its own.
947
+ *
948
+ * @param str - The string to expand. It always holds at least one `&` and is never empty, or the caller would have returned before reaching the
949
+ * walk.
950
+ *
951
+ * @returns An effect producing the pieces in order. The array is empty exactly when nothing was replaced, which the caller reads as "the input is
952
+ * its own result". Fails with {@link XmlError} and the reason the offending reference carries — `ProhibitedCharacterReference`,
953
+ * `ExpansionLimitExceeded` or `ExpandedLengthLimitExceeded`.
954
+ */
955
+ #expandAll = Effect.fnUntraced(function* (this: EntityDecoder, str: string): Effect.fn.Return<Array<string>, XmlError> {
956
+ const chunks: Array<string> = [];
957
+ const len = str.length;
958
+ let last = 0; // start of the next unprocessed literal run
959
+ let i = 0;
960
+
961
+ while (i < len) {
962
+ if (str.charCodeAt(i) !== CODE_AMPERSAND) {
963
+ i++;
964
+ continue;
965
+ }
966
+
967
+ const end = scanTokenEnd(str, i);
968
+ if (end <= i + 1) {
969
+ // Nothing to resolve: no `;` inside the scan window, or an empty token (`&;`). A bare ampersand
970
+ // rather than a reference, so advance past the `&` only and let the rest of the run be copied.
971
+ i++;
972
+ continue;
973
+ }
974
+
975
+ const token = str.slice(i + 1, end);
976
+ const resolved = yield* this.#resolveToken(token);
977
+ if (resolved === undefined) {
978
+ // Left, unparseable or unknown: leave the text alone and resume scanning just after the `&`.
979
+ i++;
980
+ continue;
981
+ }
982
+
983
+ if (i > last) chunks.push(str.slice(last, i));
984
+ chunks.push(resolved.value);
985
+ last = end + 1;
986
+ i = last;
987
+
988
+ yield* this.#chargeExpansion(token, resolved.value, resolved.tier);
989
+ }
990
+
991
+ if (last < len) chunks.push(str.slice(last));
992
+
993
+ return chunks;
994
+ });
995
+
996
+ /**
997
+ * @description Decide what one reference expands to. The lists and maps are consulted in the one order the runtime uses, and the first that matches wins:
998
+ *
999
+ * 1. `remove` — deleted outright, without the name ever being resolved, so the name need not exist.
1000
+ * 2. `leave` — emitted as the original `&token;`, and charged to nothing.
1001
+ * 3. A `#`-prefixed token — the numeric pipeline, which is the only one of the four that can fail. Classification runs before any decision about
1002
+ * `numericAllowed`, because the ranges that carry a minimum have to be caught whichever way that option is set.
1003
+ * 4. Anything else — resolved against the input map, then the external map, then the base map.
1004
+ *
1005
+ * @param token - The reference's token, e.g. `brand` or `#38`, with the `&` and the `;` already stripped. Never empty: the scanner drops `&;`
1006
+ * before calling.
1007
+ *
1008
+ * @returns An effect producing what the reference expands to and the tier to charge it to, or `undefined` to leave it as written and charge it
1009
+ * nothing. `undefined` covers all three ways of leaving a reference alone — a listed `leave` name, a numeric reference that is out of range, and
1010
+ * a name registered nowhere — and none of them is distinguishable from outside. Fails with {@link XmlError} and the
1011
+ * `ProhibitedCharacterReference` reason when the numeric policy throws on the codepoint.
1012
+ */
1013
+ #resolveToken = Effect.fnUntraced(function* (this: EntityDecoder, token: string): Effect.fn.Return<ResolvedEntity | undefined, XmlError> {
1014
+ if (this.#removeSet.has(token)) {
1015
+ // Deleted without being resolved, so the name need not exist. Upstream guards this charge with
1016
+ // `if (tier === undefined)`, and its `tier` is declared without an initialiser, so the branch is
1017
+ // unconditionally taken and the charge always lands on `external` — whatever tier the name would
1018
+ // have resolved in. That is why a document full of removed built-ins can trip an `external` limit
1019
+ // nothing it wrote could otherwise reach. Kept as written, since that is a behaviour a caller may
1020
+ // already be relying on.
1021
+ return { value: '', tier: LIMIT_TIER_EXTERNAL };
1022
+ }
1023
+
1024
+ // Emitted as the original `&token;`. The walk advances only past the `&` and leaves the `;` to be
1025
+ // copied by the next literal run, which is what makes the text come back unchanged.
1026
+ if (this.#leaveSet.has(token)) return undefined;
1027
+
1028
+ if (token.charCodeAt(0) === CODE_HASH) {
1029
+ const character = yield* this.#resolveNCR(token);
1030
+ // `''` for a removal and the character for an allow are both real replacements; `undefined` is the
1031
+ // numeric pipeline's own way of saying "leave it as written".
1032
+ if (character === undefined) return undefined;
1033
+ return { value: character, tier: LIMIT_TIER_BASE };
1034
+ }
1035
+
1036
+ return this.#resolveName(token);
1037
+ });
1038
+
1039
+ /**
1040
+ * @description Charge one expansion against the ceilings, or against neither. An expansion counts only when its tier passes `#tierCounts` and at least one
1041
+ * ceiling is configured, so a decoder with no limits set does no accounting at all, and an entity in a tier the filter excludes is free. Each
1042
+ * ceiling is guarded separately rather than left to its own check, because an unconfigured ceiling is not a ceiling of zero: `maxExpandedLength: 0`
1043
+ * means unlimited, so a decoder with only a count limit must not accumulate length it will then be compared against.
1044
+ *
1045
+ * @param token - The reference's token, with the `&` and `;` stripped. Its width is the baseline the expansion is measured against.
1046
+ * @param replacement - What the reference expanded to, including `''` for a removal.
1047
+ * @param tier - The tier the expansion is charged to.
1048
+ *
1049
+ * @returns An effect that fails with {@link XmlError} once a ceiling is exceeded, and succeeds otherwise. The count is checked before the length,
1050
+ * so a document that breaches both is reported against the count.
1051
+ */
1052
+ #chargeExpansion = Effect.fnUntraced(function* (
1053
+ this: EntityDecoder,
1054
+ token: string,
1055
+ replacement: string,
1056
+ tier: LimitTier
1057
+ ): Effect.fn.Return<void, XmlError> {
1058
+ const counts = this.#maxTotalExpansions > 0;
1059
+ const grows = this.#maxExpandedLength > 0;
1060
+ if (!counts && !grows) return;
1061
+ if (!this.#tierCounts(tier)) return;
1062
+
1063
+ if (counts) yield* this.#countExpansion();
1064
+ if (grows) yield* this.#countExpandedLength(token, replacement);
1065
+ });
1066
+
1067
+ /**
1068
+ * @description Add one expansion to the running total and compare it against {@link EntityDecoderLimitOptions.maxTotalExpansions}. The comparison is `>` rather
1069
+ * than `>=`, so a limit of `n` allows exactly `n` expansions and throws on the `n + 1`th. That is a contract — the option's own documentation
1070
+ * states it — and the kind of off-by-one a tidy-up changes by accident. The counter is deliberately not reset before failing: the over-limit total
1071
+ * is what the error message reports, and {@link EntityDecoder.reset} is the caller's way to start a new document.
1072
+ *
1073
+ * @returns An effect that fails with {@link XmlError} and the `ExpansionLimitExceeded` reason once the count is past the ceiling, and succeeds
1074
+ * otherwise. The `EntityReplacer` prefix in the message is preserved verbatim from the original throw, despite naming a class this decoder does
1075
+ * not have.
1076
+ */
1077
+ #countExpansion(): Effect.Effect<void, XmlError> {
1078
+ this.#totalExpansions++;
1079
+ if (this.#totalExpansions > this.#maxTotalExpansions) {
1080
+ return Effect.fail(
1081
+ new XmlError({
1082
+ reason: { _tag: 'ExpansionLimitExceeded', actual: this.#totalExpansions, limit: this.#maxTotalExpansions },
1083
+ message: `[EntityReplacer] Entity expansion count limit exceeded: ${this.#totalExpansions} > ${this.#maxTotalExpansions}`,
1084
+ })
1085
+ );
1086
+ }
1087
+ return Effect.void;
1088
+ }
1089
+
1090
+ /**
1091
+ * @description Add one expansion's surplus to the running total and compare it against {@link EntityDecoderLimitOptions.maxExpandedLength}. Only the surplus
1092
+ * counts, and only upward: a reference whose replacement is no longer than the `&token;` it replaces contributes zero, and a shrinking one
1093
+ * contributes nothing and cannot trip the limit at all. That is what makes the ceiling a bound on growth rather than on document size.
1094
+ *
1095
+ * @param token - The reference's token, with the `&` and `;` stripped. The two delimiters count towards what the expansion displaced.
1096
+ * @param replacement - What the reference expanded to, including `''` for a removal.
1097
+ *
1098
+ * @returns An effect that fails with {@link XmlError} and the `ExpandedLengthLimitExceeded` reason once the total is past the ceiling, and succeeds
1099
+ * otherwise. The `EntityReplacer` prefix in the message is preserved verbatim from the original throw, for the same reason as in
1100
+ * `#countExpansion`.
1101
+ */
1102
+ #countExpandedLength(token: string, replacement: string): Effect.Effect<void, XmlError> {
1103
+ const delta = replacement.length - (token.length + 2);
1104
+ if (delta <= 0) return Effect.void;
1105
+
1106
+ this.#expandedLength += delta;
1107
+ if (this.#expandedLength > this.#maxExpandedLength) {
1108
+ return Effect.fail(
1109
+ new XmlError({
1110
+ reason: { _tag: 'ExpandedLengthLimitExceeded', actual: this.#expandedLength, limit: this.#maxExpandedLength },
1111
+ message: `[EntityReplacer] Expanded content length limit exceeded: ${this.#expandedLength} > ${this.#maxExpandedLength}`,
1112
+ })
1113
+ );
1114
+ }
1115
+ return Effect.void;
1116
+ }
1117
+
1118
+ /**
1119
+ * @description Decide whether an entity of a given tier is charged against the limits.
1120
+ *
1121
+ * @param tier - The tier the replacement is charged to. Every expansion that reaches here carries one — a name deleted before it was ever resolved
1122
+ * still carries the `external` tier — so there is no absent case to answer.
1123
+ *
1124
+ * @returns `true` when it counts. `'all'` short-circuits, so a filter naming every tier charges everything regardless of which map it came from.
1125
+ */
1126
+ #tierCounts(tier: LimitTier): boolean {
1127
+ if (this.#limitTiers.has(LIMIT_TIER_ALL)) return true;
1128
+ return this.#limitTiers.has(tier);
1129
+ }
1130
+
1131
+ /**
1132
+ * @description Resolve a named entity token, with the `&` and `;` already stripped.
1133
+ *
1134
+ * @param name - The token, e.g. `brand`.
1135
+ *
1136
+ * @returns The value and the tier to charge it to, or `undefined` when the name is registered nowhere. A name registered to the empty string
1137
+ * resolves to `''` rather than to `undefined`, so it deletes the reference instead of leaving it alone.
1138
+ */
1139
+ #resolveName(name: string): ResolvedEntity | undefined {
1140
+ // Input and external share the `external` tier: both are injected at runtime, and that is the
1141
+ // surface the limits exist to bound.
1142
+ const fromInput = ownEntity(this.#inputMap, name);
1143
+ if (fromInput !== undefined) return { value: fromInput, tier: LIMIT_TIER_EXTERNAL };
1144
+
1145
+ const fromExternal = ownEntity(this.#externalMap, name);
1146
+ if (fromExternal !== undefined) return { value: fromExternal, tier: LIMIT_TIER_EXTERNAL };
1147
+
1148
+ const fromBase = ownEntity(this.#baseMap, name);
1149
+ if (fromBase !== undefined) return { value: fromBase, tier: LIMIT_TIER_BASE };
1150
+
1151
+ return undefined;
1152
+ }
1153
+
1154
+ /**
1155
+ * @description Find the strictest action a codepoint's range requires. Checked in this order:
1156
+ *
1157
+ * 1. U+0000 — governed by `nullNCR`, already clamped to `remove` or stricter
1158
+ * 2. U+D800–U+DFFF — surrogates, always `remove`, under every policy and both XML versions
1159
+ * 3. U+0001–U+001F other than tab, newline, carriage return — XML 1.0 only, `remove` Nothing else is classified. U+007F–U+009F (C1) and the
1160
+ * U+FFFE/U+FFFF noncharacters are not checked, even though XML 1.0 §2.2 prohibits them and the `xmlVersion` option's own documentation claims C1
1161
+ * is only permitted under 1.1. Both gaps are upstream's and are kept.
1162
+ *
1163
+ * @param cp - The codepoint.
1164
+ *
1165
+ * @returns The minimum level from {@link NCR_LEVEL}, or {@link NO_MINIMUM_LEVEL} when the codepoint carries none.
1166
+ */
1167
+ #classifyNCR(cp: number): number {
1168
+ if (cp === 0) return this.#ncrNullLevel;
1169
+
1170
+ if (cp >= 0xd800 && cp <= 0xdfff) return NCR_LEVEL.remove;
1171
+
1172
+ if (this.#ncrXmlVersion === 1.0 && cp >= 0x01 && cp <= 0x1f && !XML10_ALLOWED_C0.has(cp)) {
1173
+ return NCR_LEVEL.remove;
1174
+ }
1175
+
1176
+ return NO_MINIMUM_LEVEL;
1177
+ }
1178
+
1179
+ /**
1180
+ * @description Turn a resolved action level into a replacement.
1181
+ *
1182
+ * @param action - A level from {@link NCR_LEVEL}. A level outside the four known ones falls through to the allow behaviour, so a bad level cannot
1183
+ * produce a wrong string — it can only fail open.
1184
+ * @param token - The raw token, e.g. `#38`, for the error message.
1185
+ * @param cp - The codepoint, for the error message.
1186
+ *
1187
+ * @returns An effect producing the character for `allow`, `''` for `remove`, and `undefined` for `leave` — which the caller reads as "emit the
1188
+ * original `&token;`". Fails with {@link XmlError} and the `ProhibitedCharacterReference` reason for `throw`, naming both the token and the
1189
+ * codepoint.
1190
+ */
1191
+ #applyNCRAction(action: number, token: string, cp: number): Effect.Effect<string | undefined, XmlError> {
1192
+ return Match.value(action).pipe(
1193
+ Match.when(NCR_LEVEL.allow, () => Effect.succeed(String.fromCodePoint(cp))),
1194
+ Match.when(NCR_LEVEL.remove, () => Effect.succeed('')),
1195
+ // oxlint-disable-next-line effecttsgo/effect-succeed-with-void
1196
+ Match.when(NCR_LEVEL.leave, () => Effect.succeed(undefined)),
1197
+ Match.when(NCR_LEVEL.throw, () =>
1198
+ Effect.fail(
1199
+ new XmlError({
1200
+ reason: { _tag: 'ProhibitedCharacterReference', token, codepoint: cp },
1201
+ message: `[EntityDecoder] Prohibited numeric character reference &${token}; ` + `(U+${cp.toString(16).toUpperCase().padStart(4, '0')})`,
1202
+ })
1203
+ )
1204
+ ),
1205
+ Match.orElse(() => Effect.succeed(String.fromCodePoint(cp)))
1206
+ );
1207
+ }
1208
+
1209
+ /**
1210
+ * @description The full numeric-reference pipeline for one `#`-prefixed token.
1211
+ *
1212
+ * 1. Parse the codepoint, decimal or hex.
1213
+ * 2. Reject NaN, negatives, and anything above {@link MAX_CODE_POINT}, leaving the reference as written.
1214
+ * 3. Classify the codepoint for a minimum level.
1215
+ * 4. If `numericAllowed` is off and no minimum reaches `remove`, leave the reference as written.
1216
+ * 5. Take the stricter of the configured level and the minimum.
1217
+ * 6. Apply it. Step 4 is why `numericAllowed: false` does not neutralise `onNCR: 'throw'` for surrogates, the XML 1.0 C0 controls or null: their
1218
+ * minimum already reaches `remove`, so they are handled no matter what the option says. It does neutralise the throw for every other codepoint.
1219
+ * The parse is `parseInt`, which stops at the first character it cannot use. That is upstream's choice and it is permissive: a leading space or
1220
+ * `+`, and trailing garbage, are all accepted, and a decimal token beginning `0x` parses as `0` rather than as hex.
1221
+ *
1222
+ * @param token - The raw token without `&` and `;`, e.g. `#38`, `#x26`, `#X26`.
1223
+ *
1224
+ * @returns An effect producing the replacement — the empty string meaning "delete" — or `undefined` to leave the reference as written. Fails with
1225
+ * {@link XmlError} and the `ProhibitedCharacterReference` reason when the effective action is `throw`.
1226
+ */
1227
+ #resolveNCR(token: string): Effect.Effect<string | undefined, XmlError> {
1228
+ const second = token.charCodeAt(1);
1229
+ let cp: number;
1230
+ if (second === CODE_LOWER_X || second === CODE_UPPER_X) {
1231
+ cp = parseInt(token.slice(2), 16);
1232
+ } else {
1233
+ cp = parseInt(token.slice(1), 10);
1234
+ }
1235
+
1236
+ // Out of range is `leave` rather than `remove`: an unparseable reference is text, and
1237
+ // deleting a document's characters because one of them was malformed is not a safe default.
1238
+ if (Number.isNaN(cp) || cp < 0 || cp > MAX_CODE_POINT) return Effect.succeed(undefined);
1239
+
1240
+ const minimum = this.#classifyNCR(cp);
1241
+
1242
+ if (!this.#numericAllowed && minimum < NCR_LEVEL.remove) return Effect.succeed(undefined);
1243
+
1244
+ const effective = minimum === NO_MINIMUM_LEVEL ? this.#ncrOnLevel : Math.max(this.#ncrOnLevel, minimum);
1245
+
1246
+ return this.#applyNCRAction(effective, token, cp);
1247
+ }
1248
+ }