@endevops/effect-codec-xml 0.0.1 → 0.1.0-beta.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +63 -62
- package/dist/codec.d.ts +17 -9
- package/dist/codec.d.ts.map +1 -1
- package/dist/codec.js +25 -16
- package/dist/codec.js.map +1 -1
- package/dist/conventions.d.ts +4 -4
- package/dist/conventions.js +7 -7
- package/dist/conventions.js.map +1 -1
- package/dist/entities/entity-decoder.d.ts +33 -33
- package/dist/entities/entity-decoder.d.ts.map +1 -1
- package/dist/entities/entity-decoder.js +63 -64
- package/dist/entities/entity-decoder.js.map +1 -1
- package/dist/errors.d.ts +2 -2
- package/dist/errors.js +2 -2
- package/dist/errors.js.map +1 -1
- package/dist/namespaces.js +2 -2
- package/dist/namespaces.js.map +1 -1
- package/dist/naming.d.ts +6 -6
- package/dist/naming.d.ts.map +1 -1
- package/dist/naming.js +3 -3
- package/dist/naming.js.map +1 -1
- package/dist/parse.d.ts +5 -5
- package/dist/parse.js +14 -14
- package/dist/parse.js.map +1 -1
- package/dist/render.d.ts +1 -1
- package/dist/render.d.ts.map +1 -1
- package/dist/render.js +28 -28
- package/dist/render.js.map +1 -1
- package/dist/xml-error.d.ts +9 -9
- package/dist/xml-error.js +18 -18
- package/dist/xml-error.js.map +1 -1
- package/dist/xml-value.d.ts +7 -7
- package/dist/xml-value.d.ts.map +1 -1
- package/dist/xml-value.js +6 -7
- package/dist/xml-value.js.map +1 -1
- package/package.json +1 -1
- package/src/codec.ts +91 -71
- package/src/conventions.ts +7 -7
- package/src/entities/entity-decoder.ts +98 -99
- package/src/errors.ts +3 -3
- package/src/index.ts +3 -3
- package/src/namespaces.ts +16 -16
- package/src/naming.ts +34 -35
- package/src/parse.ts +26 -26
- package/src/render.ts +44 -44
- package/src/xml-error.ts +18 -18
- package/src/xml-value.ts +10 -11
|
@@ -5,13 +5,13 @@
|
|
|
5
5
|
// `@nodable/entities@2.2.0` (`src/EntityDecoder.js`) as native TypeScript.
|
|
6
6
|
//
|
|
7
7
|
// Three entity tiers exist and the distinction is the security model, not a performance detail. `input` and
|
|
8
|
-
// `external` entities are injected at runtime
|
|
9
|
-
// decoder
|
|
8
|
+
// `external` entities are injected at runtime (DOCTYPE declarations, and whatever a caller hands the
|
|
9
|
+
// decoder), so they are the untrusted surface and are what the expansion limits count by default. `base` is
|
|
10
10
|
// the five XML predefined entities plus the caller's own `namedEntities`. Numeric references are always
|
|
11
11
|
// `base`: they cannot recurse.
|
|
12
12
|
//
|
|
13
|
-
// Behaviour is transcribed, not corrected. Several upstream quirks are
|
|
14
|
-
// already sees
|
|
13
|
+
// Behaviour is transcribed, not corrected. Several upstream quirks are relied on by the output a caller
|
|
14
|
+
// already sees. A `&` inside a registered value is not filtered the way the docs claim, `postCheck` is
|
|
15
15
|
// skipped on the two fast paths, C1 codepoints and the FFFE/FFFF noncharacters are not classified at all.
|
|
16
16
|
// Each is called out where it appears, and none of them is repaired, because this class sits in front of XXE
|
|
17
17
|
// and entity-expansion handling where a silent fix is a change to every consumer's output.
|
|
@@ -26,7 +26,7 @@ import { XML as DEFAULT_XML_ENTITIES } from './entity-tables.ts';
|
|
|
26
26
|
// Character codes
|
|
27
27
|
//
|
|
28
28
|
// The scan is a hand-rolled `charCodeAt` loop, so the three codes it tests against are named here rather than
|
|
29
|
-
// written as literals. `&` opens a reference, `;` closes one, `#`
|
|
29
|
+
// written as literals. `&` opens a reference, `;` closes one, `#` marks a reference as numeric.
|
|
30
30
|
// ---------------------------------------------------------------------------
|
|
31
31
|
|
|
32
32
|
const CODE_AMPERSAND = 38;
|
|
@@ -37,8 +37,8 @@ const CODE_UPPER_X = 88;
|
|
|
37
37
|
|
|
38
38
|
/**
|
|
39
39
|
* @description The widest entity name {@link EntityDecoder.decode} will look for, in characters. The forward scan for `;` stops once more than this many characters
|
|
40
|
-
* have passed since the `&`, so a longer run is not treated as one entity
|
|
41
|
-
*
|
|
40
|
+
* have passed since the `&`, so a longer run is not treated as one entity. It is copied through as literal text. The bound keeps a document with a
|
|
41
|
+
* megabyte of non-entity text between two ampersands from being sliced.
|
|
42
42
|
*/
|
|
43
43
|
const MAX_TOKEN_LENGTH = 32;
|
|
44
44
|
|
|
@@ -49,8 +49,8 @@ const MAX_TOKEN_LENGTH = 32;
|
|
|
49
49
|
const MAX_CODE_POINT = 0x10ffff;
|
|
50
50
|
|
|
51
51
|
/**
|
|
52
|
-
* @description Returned by `#classifyNCR` for a codepoint that carries no minimum action level
|
|
53
|
-
*
|
|
52
|
+
* @description Returned by `#classifyNCR` for a codepoint that carries no minimum action level. This value distinguishes "no restriction" from `NCR_LEVEL.allow`:
|
|
53
|
+
* both end up expanding, but only the first lets `numericAllowed: false` short-circuit the whole pipeline.
|
|
54
54
|
*/
|
|
55
55
|
const NO_MINIMUM_LEVEL = -1;
|
|
56
56
|
|
|
@@ -61,7 +61,7 @@ const NO_MINIMUM_LEVEL = -1;
|
|
|
61
61
|
/**
|
|
62
62
|
* @description Characters that may not appear in an entity name registered through {@link EntityDecoder.setExternalEntities} or
|
|
63
63
|
* {@link EntityDecoder.addExternalEntity}. A name carrying one of these cannot be written as `&name;` at all, so registration refuses it rather than
|
|
64
|
-
* storing a name no document could ever reference. The set is the upstream string verbatim, including its duplicated backslash
|
|
64
|
+
* storing a name no document could ever reference. The set is the upstream string verbatim, including its duplicated backslash. A `Set` discards the
|
|
65
65
|
* duplicate, so the effective set is the eighteen characters below.
|
|
66
66
|
*/
|
|
67
67
|
const SPECIAL_CHARS: ReadonlySet<string> = new Set('!?\\/[]$%{}^&*()<>|+');
|
|
@@ -94,7 +94,7 @@ type LimitTier = typeof LIMIT_TIER_ALL | typeof LIMIT_TIER_BASE | typeof LIMIT_T
|
|
|
94
94
|
|
|
95
95
|
/**
|
|
96
96
|
* @description The NCR action levels, in severity order. A higher number is a stricter action, and the resolver takes the maximum of the configured level and the
|
|
97
|
-
* minimum a codepoint range imposes
|
|
97
|
+
* minimum a codepoint range imposes. A range can therefore only make an entity stricter than the caller asked for, never more lenient.
|
|
98
98
|
*/
|
|
99
99
|
const NCR_LEVEL = Object.freeze({ allow: 0, leave: 1, remove: 2, throw: 3 });
|
|
100
100
|
|
|
@@ -111,7 +111,7 @@ type NcrLevelName = keyof typeof NCR_LEVEL;
|
|
|
111
111
|
type XmlVersion = 1 | 1.1;
|
|
112
112
|
|
|
113
113
|
/**
|
|
114
|
-
* @description The C0 control codes XML 1.0 §2.2 permits as literal characters. Every other code in U+0001
|
|
114
|
+
* @description The C0 control codes XML 1.0 §2.2 permits as literal characters. Every other code in U+0001 to U+001F is prohibited.
|
|
115
115
|
*/
|
|
116
116
|
const XML10_ALLOWED_C0: ReadonlySet<number> = new Set([0x09, 0x0a, 0x0d]);
|
|
117
117
|
|
|
@@ -126,8 +126,8 @@ const XML10_ALLOWED_C0: ReadonlySet<number> = new Set([0x09, 0x0a, 0x0d]);
|
|
|
126
126
|
export type EntityHookAction = 'allow' | 'block' | 'throw';
|
|
127
127
|
|
|
128
128
|
/**
|
|
129
|
-
* @description A function-valued entity replacement: the `val` of the legacy `{ regex, val }` form when it is not a string. This decoder cannot use one
|
|
130
|
-
* function has no meaning without the regex it was meant to be matched against
|
|
129
|
+
* @description A function-valued entity replacement: the `val` of the legacy `{ regex, val }` form when it is not a string. This decoder cannot use one. A
|
|
130
|
+
* function has no meaning without the regex it was meant to be matched against, so such an entry is dropped at registration rather than expanded.
|
|
131
131
|
*/
|
|
132
132
|
export type EntityValFn = (match: string, captured: string, ...rest: Array<unknown>) => string;
|
|
133
133
|
|
|
@@ -167,16 +167,16 @@ export const ENTITY_ACTION: Readonly<{ ALLOW: 'allow'; BLOCK: 'block'; THROW: 't
|
|
|
167
167
|
/**
|
|
168
168
|
* @description Which entity categories count toward the expansion limits.
|
|
169
169
|
*
|
|
170
|
-
* - `'external'
|
|
171
|
-
* - `'base'
|
|
172
|
-
* - `'all'
|
|
173
|
-
* - `Array<'external' | 'base'
|
|
170
|
+
* - `'external'`: only input/runtime + persistent external entities. The default, and the only one that ignores the built-in XML entities.
|
|
171
|
+
* - `'base'`: only the built-in XML entities, the caller's `namedEntities`, and numeric references.
|
|
172
|
+
* - `'all'`: every entity regardless of tier.
|
|
173
|
+
* - `Array<'external' | 'base'>`: an explicit combination. An empty array is honoured literally: nothing counts, so the limits can never trip.
|
|
174
174
|
*/
|
|
175
175
|
export type ApplyLimitsTo = 'external' | 'base' | 'all' | Array<'external' | 'base'>;
|
|
176
176
|
|
|
177
177
|
/**
|
|
178
178
|
* @description Ceilings on what a single document's entity references may cost. Both are cumulative across {@link EntityDecoder.decode} calls until
|
|
179
|
-
* {@link EntityDecoder.reset}, and both default to `0`, meaning unlimited. `0
|
|
179
|
+
* {@link EntityDecoder.reset}, and both default to `0`, meaning unlimited. `0`, and any negative or non-numeric value, is unlimited, because the
|
|
180
180
|
* runtime tests `> 0` rather than truthiness of the configured number.
|
|
181
181
|
*/
|
|
182
182
|
export interface EntityDecoderLimitOptions {
|
|
@@ -197,8 +197,8 @@ export interface EntityDecoderLimitOptions {
|
|
|
197
197
|
maxExpandedLength?: number;
|
|
198
198
|
|
|
199
199
|
/**
|
|
200
|
-
* @description Which tiers count against both limits. Defaults to `'external'`, which
|
|
201
|
-
*
|
|
200
|
+
* @description Which tiers count against both limits. Defaults to `'external'`, which keeps the built-in entities, including every numeric reference, from being
|
|
201
|
+
* able to trip a limit on a document the caller already trusts.
|
|
202
202
|
*
|
|
203
203
|
* @default 'external'
|
|
204
204
|
*/
|
|
@@ -211,7 +211,7 @@ export interface EntityDecoderLimitOptions {
|
|
|
211
211
|
*/
|
|
212
212
|
export interface EntityDecoderNCROptions {
|
|
213
213
|
/**
|
|
214
|
-
* @description The XML version whose codepoint restrictions apply. `1.0` prohibits the C0 controls U+0001
|
|
214
|
+
* @description The XML version whose codepoint restrictions apply. `1.0` prohibits the C0 controls U+0001 to U+001F other than tab, newline and carriage return;
|
|
215
215
|
* `1.1` does not, since it permits them when written as references. Any value other than `1.1` is read as `1.0`.
|
|
216
216
|
*
|
|
217
217
|
* @default 1.0
|
|
@@ -219,8 +219,8 @@ export interface EntityDecoderNCROptions {
|
|
|
219
219
|
xmlVersion?: 1.0 | 1.1;
|
|
220
220
|
|
|
221
221
|
/**
|
|
222
|
-
* @description The base action for every numeric reference. Codepoint ranges that carry a minimum
|
|
223
|
-
* null under `nullNCR`
|
|
222
|
+
* @description The base action for every numeric reference. Codepoint ranges that carry a minimum (surrogates always, the XML 1.0 C0 controls under `1.0`, and
|
|
223
|
+
* null under `nullNCR`) take the stricter of the two, so this is a floor and not an override.
|
|
224
224
|
*
|
|
225
225
|
* @default 'allow'
|
|
226
226
|
*/
|
|
@@ -240,10 +240,10 @@ export interface EntityDecoderNCROptions {
|
|
|
240
240
|
export interface EntityDecoderOptions {
|
|
241
241
|
/**
|
|
242
242
|
* @description Extra named entities merged into the `base` map alongside the five XML predefined ones. A string value is used directly; a `{ regex, val }` or `{
|
|
243
|
-
* regx, val }` envelope is unwrapped to its `val`. Anything else
|
|
244
|
-
*
|
|
245
|
-
*
|
|
246
|
-
*
|
|
243
|
+
* regx, val }` envelope is unwrapped to its `val`. Anything else (a number, `null`, a function, an envelope whose `val` is a function) is dropped,
|
|
244
|
+
* leaving the name unresolvable rather than failing the construction. Upstream's documentation says a value containing `&` is skipped here, to
|
|
245
|
+
* prevent recursive expansion. It is not: the code stores the value unchanged, and only {@link EntityDecoder.addExternalEntity} checks for `&`.
|
|
246
|
+
* Preserved as-is; see the note on the class.
|
|
247
247
|
*
|
|
248
248
|
* @default null
|
|
249
249
|
*/
|
|
@@ -251,7 +251,7 @@ export interface EntityDecoderOptions {
|
|
|
251
251
|
|
|
252
252
|
/**
|
|
253
253
|
* @description Called once on the finished string. Receives `(resolved, original)` and must return a string; return `original` to reject the expansion outright,
|
|
254
|
-
* or a sanitised form of `resolved` to clean it. It is _not_ called for a string that never reaches the scanning loop
|
|
254
|
+
* or a sanitised form of `resolved` to clean it. It is _not_ called for a string that never reaches the scanning loop: an empty string, a
|
|
255
255
|
* non-string, or any string with no `&` in it. A caller relying on `postCheck` to sanitise therefore has to know that a string with no ampersand is
|
|
256
256
|
* never inspected.
|
|
257
257
|
*
|
|
@@ -260,8 +260,8 @@ export interface EntityDecoderOptions {
|
|
|
260
260
|
postCheck?: ((resolved: string, original: string) => string) | null;
|
|
261
261
|
|
|
262
262
|
/**
|
|
263
|
-
* @description Whether numeric references expand at all. Turning it off leaves every one of them in the output verbatim
|
|
264
|
-
* minimum action of `remove` or stricter, which are still handled, because that classification runs first
|
|
263
|
+
* @description Whether numeric references expand at all. Turning it off leaves every one of them in the output verbatim, _except_ the codepoints that carry a
|
|
264
|
+
* minimum action of `remove` or stricter, which are still handled, because that classification runs first. This is why the option is safe to rely
|
|
265
265
|
* on.
|
|
266
266
|
*
|
|
267
267
|
* @default true
|
|
@@ -278,7 +278,7 @@ export interface EntityDecoderOptions {
|
|
|
278
278
|
/**
|
|
279
279
|
* @description Names to delete outright, matched the same way as {@link EntityDecoderOptions.leave}. A removed reference is charged to the `external` tier even
|
|
280
280
|
* when the name is a built-in one, so a document full of removed built-ins can trip an `applyLimitsTo: 'external'` limit it would not otherwise be
|
|
281
|
-
* subject to. Preserved as-is; the only in-code comment claims the charge is for unknown references, which
|
|
281
|
+
* subject to. Preserved as-is; the only in-code comment claims the charge is for unknown references, which does not distinguish them.
|
|
282
282
|
*
|
|
283
283
|
* @default [ ]
|
|
284
284
|
*/
|
|
@@ -305,7 +305,7 @@ export interface EntityDecoderOptions {
|
|
|
305
305
|
|
|
306
306
|
/**
|
|
307
307
|
* @description Called once per entity as it is registered through {@link EntityDecoder.addInputEntities}. Same contract as
|
|
308
|
-
* {@link EntityDecoderOptions.onExternalEntity}, and unlike it the hook is not the only filter
|
|
308
|
+
* {@link EntityDecoderOptions.onExternalEntity}, and unlike it the hook is not the only filter. See the class note on name validation.
|
|
309
309
|
*
|
|
310
310
|
* @default null
|
|
311
311
|
*/
|
|
@@ -336,7 +336,7 @@ type EntityInputMap = Readonly<Record<string, EntityInputValue>> | null | undefi
|
|
|
336
336
|
|
|
337
337
|
/**
|
|
338
338
|
* @description What one reference expanded to, tagged with the tier its limit accounting charges. The field is the replacement text itself for a named entity, the
|
|
339
|
-
* character for a numeric reference, and `''` for a removed one
|
|
339
|
+
* character for a numeric reference, and `''` for a removed one. Those are the three shapes the walk pushes into its output.
|
|
340
340
|
*/
|
|
341
341
|
type ResolvedEntity = { value: string; tier: LimitTier };
|
|
342
342
|
|
|
@@ -357,7 +357,7 @@ type HookContext = 'external' | 'input';
|
|
|
357
357
|
*
|
|
358
358
|
* @returns An effect producing the name, unchanged, so the call can be inlined. Fails with {@link XmlError} and the `InvalidEntityName` reason,
|
|
359
359
|
* carrying the offending character. The `[EntityReplacer]` prefix in the message is preserved verbatim from the original throw, despite naming a
|
|
360
|
-
* class this decoder does not have
|
|
360
|
+
* class this decoder does not have. It is relied on by anything matching the message text.
|
|
361
361
|
*/
|
|
362
362
|
const checkEntityName = (name: string): Effect.Effect<string, XmlError> => {
|
|
363
363
|
if (name.charCodeAt(0) === CODE_HASH) {
|
|
@@ -383,11 +383,11 @@ const checkEntityName = (name: string): Effect.Effect<string, XmlError> => {
|
|
|
383
383
|
|
|
384
384
|
/**
|
|
385
385
|
* @description Flatten registration maps into one name to string map, later maps winning over earlier ones for the same name. The result is a null-prototype
|
|
386
|
-
* object, not a `Map`. That is
|
|
387
|
-
*
|
|
388
|
-
*
|
|
386
|
+
* object, not a `Map`. That is intentional. A `Map` iterates in pure insertion order, while `Object.keys` lifts array-index-like names to the front
|
|
387
|
+
* in numeric order, and the registration hooks observe that order. A name of `"2"` registered after `"brand"` reaches the hook first here and second
|
|
388
|
+
* in a `Map`.
|
|
389
389
|
*
|
|
390
|
-
* @param maps - The maps to merge. A falsy entry
|
|
390
|
+
* @param maps - The maps to merge. A falsy entry (`null`, `undefined`, `''`, `0`) contributes nothing rather than throwing.
|
|
391
391
|
*
|
|
392
392
|
* @returns A null-prototype object of own string-valued entries. Nothing from `Object.prototype` can be read out of it, so a document naming
|
|
393
393
|
* `constructor` or `toString` finds nothing. Each entry is read through {@link flattenEntityValue}, so an entry that cannot be reduced to a string
|
|
@@ -407,15 +407,15 @@ function mergeEntityMaps(...maps: ReadonlyArray<EntityInputMap>): Record<string,
|
|
|
407
407
|
|
|
408
408
|
/**
|
|
409
409
|
* @description Reduce one registration entry to the string a reference to it expands to, or to nothing when the entry is a form the scanner has no use for. Three
|
|
410
|
-
* shapes survive: the string itself, and a `{ regex | regx, val }` envelope whose `val` is a string. Everything else
|
|
411
|
-
* a bare function, an envelope whose `val` is a function
|
|
412
|
-
*
|
|
410
|
+
* shapes survive: the string itself, and a `{ regex | regx, val }` envelope whose `val` is a string. Everything else (a number, `null`, `undefined`,
|
|
411
|
+
* a bare function, an envelope whose `val` is a function) has no string to substitute, so the name is dropped and a reference to it comes back out as
|
|
412
|
+
* the text it was written as. Dropping is silent on purpose: the runtime inspects whatever it is handed, and failing the construction over one
|
|
413
413
|
* unreadable entry would take every other entity in the table down with it.
|
|
414
414
|
*
|
|
415
|
-
* @param raw - The entry as it arrived, in whatever shape the caller supplied it
|
|
415
|
+
* @param raw - The entry as it arrived, in whatever shape the caller supplied it, including no entry at all, which a table with a hole in it
|
|
416
416
|
* produces.
|
|
417
417
|
*
|
|
418
|
-
* @returns The replacement string, or `undefined` when the entry cannot be read. A name registered to the empty string yields `''
|
|
418
|
+
* @returns The replacement string, or `undefined` when the entry cannot be read. A name registered to the empty string yields `''`. That is why
|
|
419
419
|
* callers compare against `undefined` rather than testing for emptiness.
|
|
420
420
|
*/
|
|
421
421
|
function flattenEntityValue(raw: EntityInputValue | undefined): string | undefined {
|
|
@@ -436,12 +436,12 @@ function flattenEntityValue(raw: EntityInputValue | undefined): string | undefin
|
|
|
436
436
|
* @param map - The map to read.
|
|
437
437
|
* @param key - The entity name.
|
|
438
438
|
*
|
|
439
|
-
* @returns The registered string, or `undefined` when the name is not an own key. A name registered to the empty string returns `''
|
|
439
|
+
* @returns The registered string, or `undefined` when the name is not an own key. A name registered to the empty string returns `''`. That is why
|
|
440
440
|
* callers must compare against `undefined` rather than test for emptiness.
|
|
441
441
|
*/
|
|
442
442
|
function ownEntity(map: Readonly<Record<string, string>>, key: string): string | undefined {
|
|
443
443
|
// Upstream tests `name in map`, which reads as "is this name present at all". `Object.hasOwn` is
|
|
444
|
-
// the same question asked explicitly, and it is the
|
|
444
|
+
// the same question asked explicitly, and it is the right shape for the answer: an absent key
|
|
445
445
|
// yields `undefined` and a present one yields the stored string, so the `string | undefined` this
|
|
446
446
|
// returns is the real type rather than something an assertion has to paper over. The two maps are
|
|
447
447
|
// null-prototype objects, so `in` and `hasOwn` cannot disagree here.
|
|
@@ -455,8 +455,7 @@ function ownEntity(map: Readonly<Record<string, string>>, key: string): string |
|
|
|
455
455
|
* @param raw - The configured value.
|
|
456
456
|
*
|
|
457
457
|
* @returns The tier set. An unrecognised string falls back to `external` rather than to no filtering at all, so a typo cannot silently disable the
|
|
458
|
-
* limits. An array is taken as given
|
|
459
|
-
* `external`.
|
|
458
|
+
* limits. An array is taken as given. An empty array therefore disables limit accounting entirely, while an empty string falls back to `external`.
|
|
460
459
|
*/
|
|
461
460
|
function parseLimitTiers(raw: ApplyLimitsTo | undefined): ReadonlySet<LimitTier> {
|
|
462
461
|
if (!raw || raw === LIMIT_TIER_EXTERNAL) return new Set([LIMIT_TIER_EXTERNAL]);
|
|
@@ -518,7 +517,7 @@ function readPostCheck(raw: EntityDecoderOptions['postCheck']): (resolved: strin
|
|
|
518
517
|
*
|
|
519
518
|
* @param raw - The configured hook, or nothing.
|
|
520
519
|
*
|
|
521
|
-
* @returns The hook itself, or `null` for an absent option and for a value that is not a function. `null`
|
|
520
|
+
* @returns The hook itself, or `null` for an absent option and for a value that is not a function. `null` lets every registration path ask
|
|
522
521
|
* unconditionally: a hook that is not there accepts.
|
|
523
522
|
*/
|
|
524
523
|
function readHook(raw: EntityRegistrationHook | null | undefined): EntityRegistrationHook | null {
|
|
@@ -528,9 +527,9 @@ function readHook(raw: EntityRegistrationHook | null | undefined): EntityRegistr
|
|
|
528
527
|
|
|
529
528
|
/**
|
|
530
529
|
* @description Read one of the two entity-name lists as a set, under the same missing-value rule as the other options: absent is empty, not an error. The
|
|
531
|
-
* `Array.isArray` test rather than a truthiness one
|
|
532
|
-
*
|
|
533
|
-
*
|
|
530
|
+
* `Array.isArray` test rather than a truthiness one keeps a mistyped list from reaching `new Set` and throwing there, so a caller's typo disables the
|
|
531
|
+
* list instead of taking the decoder down. Matching against a set also means a name in both lists is decided by the order {@link EntityDecoder.decode}
|
|
532
|
+
* consults them in, not by the order the caller wrote them in.
|
|
534
533
|
*
|
|
535
534
|
* @param raw - The configured list, or nothing.
|
|
536
535
|
*
|
|
@@ -564,25 +563,25 @@ function scanTokenEnd(str: string, ampersand: number): number {
|
|
|
564
563
|
*
|
|
565
564
|
* ### Entity lookup priority
|
|
566
565
|
*
|
|
567
|
-
* 1. **input / runtime
|
|
568
|
-
* 2. **persistent external
|
|
566
|
+
* 1. **input / runtime**: injected per document through {@link EntityDecoder.addInputEntities}
|
|
567
|
+
* 2. **persistent external**: set through {@link EntityDecoder.setExternalEntities} and {@link EntityDecoder.addExternalEntity}, surviving
|
|
569
568
|
* {@link EntityDecoder.reset}
|
|
570
|
-
* 3. **base
|
|
569
|
+
* 3. **base**: the five XML predefined entities plus the constructor's `namedEntities` Both input and external resolve as the `external` tier for limit
|
|
571
570
|
* purposes, because both are injected at runtime. Numeric references (`&#NNN;`, `&#xHH;`) resolve directly through `String.fromCodePoint` and are
|
|
572
571
|
* always `base` tier: they cannot recurse, so a limit that counted them would only punish a document that spells its characters out.
|
|
573
572
|
*
|
|
574
573
|
* ### Upstream behaviour preserved
|
|
575
574
|
*
|
|
576
|
-
* Several quirks of the original are kept
|
|
575
|
+
* Several quirks of the original are kept intentionally, because a consumer's output already depends on them:
|
|
577
576
|
*
|
|
578
577
|
* - A value containing `&` is **not** filtered from `namedEntities` or `setExternalEntities`, contrary to the documentation. Only
|
|
579
578
|
* {@link EntityDecoder.addExternalEntity} checks, and it drops the entry rather than storing it, so the same name registered either way can resolve
|
|
580
579
|
* to nothing.
|
|
581
|
-
* - {@link EntityDecoderOptions.postCheck} is skipped entirely for input that never reaches the scan
|
|
580
|
+
* - {@link EntityDecoderOptions.postCheck} is skipped entirely for input that never reaches the scan: an empty string, a non-string, or a string with
|
|
582
581
|
* no `&`.
|
|
583
582
|
* - {@link EntityDecoder.decode} returns a non-string argument unchanged, despite being typed `string`.
|
|
584
583
|
* - The expansion-limit errors are prefixed `EntityReplacer`, not `EntityDecoder`.
|
|
585
|
-
* - Nothing in XML 1.0 §2.2 is enforced for U+007F
|
|
584
|
+
* - Nothing in XML 1.0 §2.2 is enforced for U+007F to U+009F or for the U+FFFE/U+FFFF noncharacters, and the sweep for `&` leaves a name of
|
|
586
585
|
* {@link MAX_TOKEN_LENGTH} + 1 characters unresolvable.
|
|
587
586
|
* - Numeric references are parsed with `parseInt`, so a leading space, sign, or trailing garbage is accepted: `&# 41;`, `&#x+41;` and `)zz;` all
|
|
588
587
|
* decode, and `�x41;` parses as a null reference rather than `A`.
|
|
@@ -596,7 +595,7 @@ function scanTokenEnd(str: string, ampersand: number): number {
|
|
|
596
595
|
* decoder.addInputEntities({ version: '1.0' });
|
|
597
596
|
*
|
|
598
597
|
* decoder.decode('&brand; v&version; ©'); // 'Acme v1.0 ©'
|
|
599
|
-
* decoder.decode('&#38;'); // '&&'
|
|
598
|
+
* decoder.decode('&#38;'); // '&&', one pass, the output is never re-scanned
|
|
600
599
|
*
|
|
601
600
|
* decoder.reset(); // drops the input entities and the counters, keeps the external ones
|
|
602
601
|
* ```;
|
|
@@ -614,7 +613,7 @@ export class EntityDecoder {
|
|
|
614
613
|
readonly #maxExpandedLength: number;
|
|
615
614
|
|
|
616
615
|
/**
|
|
617
|
-
* @description {@link EntityDecoderOptions.postCheck}, or the identity function
|
|
616
|
+
* @description {@link EntityDecoderOptions.postCheck}, or the identity function. That lets the decode loop call it unconditionally on the path that actually
|
|
618
617
|
* scanned, and never on the two fast paths that return early.
|
|
619
618
|
*/
|
|
620
619
|
readonly #postCheck: (resolved: string, original: string) => string;
|
|
@@ -637,7 +636,7 @@ export class EntityDecoder {
|
|
|
637
636
|
|
|
638
637
|
/**
|
|
639
638
|
* @description Persistent external entities, as a null-prototype object. Replaced wholesale by {@link EntityDecoder.setExternalEntities} and added to by
|
|
640
|
-
* {@link EntityDecoder.addExternalEntity}, and never touched by {@link EntityDecoder.reset}
|
|
639
|
+
* {@link EntityDecoder.addExternalEntity}, and never touched by {@link EntityDecoder.reset}. That is the whole distinction from the input map.
|
|
641
640
|
*/
|
|
642
641
|
#externalMap: Record<string, string>;
|
|
643
642
|
|
|
@@ -648,8 +647,8 @@ export class EntityDecoder {
|
|
|
648
647
|
#inputMap: Record<string, string>;
|
|
649
648
|
|
|
650
649
|
/**
|
|
651
|
-
* @description Tracked expansions since the last reset. Cumulative across {@link EntityDecoder.decode} calls, which
|
|
652
|
-
*
|
|
650
|
+
* @description Tracked expansions since the last reset. Cumulative across {@link EntityDecoder.decode} calls, which makes a limit a per-document budget rather
|
|
651
|
+
* than a per-call one. Intentionally not reset by a thrown limit error, so the error message reports the over-limit count.
|
|
653
652
|
*/
|
|
654
653
|
#totalExpansions: number;
|
|
655
654
|
|
|
@@ -700,7 +699,7 @@ export class EntityDecoder {
|
|
|
700
699
|
|
|
701
700
|
/**
|
|
702
701
|
* @description Create a decoder. A factory rather than a constructor, because it refuses a `null` options object. Every field is optional, so `null` is not "a
|
|
703
|
-
* decoder with the defaults"
|
|
702
|
+
* decoder with the defaults". A caller who wrote it meant something the signature does not allow, and a decoder built from it would be
|
|
704
703
|
* indistinguishable from one built from `{}` while hiding the mistake. Saying so is worth a factory; `EntityDecoderOptions` is a plain object and
|
|
705
704
|
* nothing else about construction can fail.
|
|
706
705
|
*
|
|
@@ -726,15 +725,15 @@ export class EntityDecoder {
|
|
|
726
725
|
|
|
727
726
|
/**
|
|
728
727
|
* @description Create a decoder. Every option is resolved here into the flat fields the decode loop reads, so nothing per-reference has to re-derive it. The
|
|
729
|
-
* options whose wrong type disables them rather than failing the construction
|
|
730
|
-
*
|
|
728
|
+
* options whose wrong type disables them rather than failing the construction (the two hooks, the two name lists) are read through {@link readHook}
|
|
729
|
+
* and {@link readNameList}, so that rule is written once instead of four times.
|
|
731
730
|
*
|
|
732
731
|
* @param resolved - Configuration, already checked. See {@link EntityDecoderOptions}.
|
|
733
732
|
*/
|
|
734
733
|
private constructor(resolved: EntityDecoderOptions) {
|
|
735
|
-
// `options.limit` is read first,
|
|
734
|
+
// `options.limit` is read first, intentionally: it is the first property the original touched, so
|
|
736
735
|
// the property a `null` would have faulted on, and keeping that order means the reason still
|
|
737
|
-
// names it. The option stays a local
|
|
736
|
+
// names it. The option stays a local. Every value the decode loop needs is flattened out of it
|
|
738
737
|
// below, so retaining it on the instance would only be a way to observe the option back.
|
|
739
738
|
const limit = resolved.limit ?? {};
|
|
740
739
|
this.#maxTotalExpansions = limit.maxTotalExpansions || 0;
|
|
@@ -764,7 +763,7 @@ export class EntityDecoder {
|
|
|
764
763
|
/**
|
|
765
764
|
* @description Ask a registration hook about one name and value.
|
|
766
765
|
*
|
|
767
|
-
* @param hook - The hook, or `null`. A `null` hook accepts,
|
|
766
|
+
* @param hook - The hook, or `null`. A `null` hook accepts, so {@link EntityDecoder.addExternalEntity} can call this unconditionally.
|
|
768
767
|
* @param name - The entity name, without `&` or `;`.
|
|
769
768
|
* @param value - The resolved value, after any `{ regex, val }` envelope was unwrapped.
|
|
770
769
|
* @param context - Which registration is in progress, for the error message.
|
|
@@ -773,7 +772,7 @@ export class EntityDecoder {
|
|
|
773
772
|
* hook returns `throw`. The message quotes the entity, so it is the only record left that a document was rejected.
|
|
774
773
|
*/
|
|
775
774
|
#applyRegistrationHook(hook: EntityRegistrationHook | null, name: string, value: string, context: HookContext): Effect.Effect<boolean, XmlError> {
|
|
776
|
-
if (!hook) return Effect.succeed(true); //
|
|
775
|
+
if (!hook) return Effect.succeed(true); // nothing to ask
|
|
777
776
|
const action = hook(name, value);
|
|
778
777
|
if (action === ENTITY_ACTION.BLOCK) return Effect.succeed(false);
|
|
779
778
|
if (action === ENTITY_ACTION.THROW) {
|
|
@@ -835,7 +834,7 @@ export class EntityDecoder {
|
|
|
835
834
|
*/
|
|
836
835
|
addExternalEntity = Effect.fnUntraced(function* (this: EntityDecoder, key: string, value: string): Effect.fn.Return<void, XmlError> {
|
|
837
836
|
yield* checkEntityName(key);
|
|
838
|
-
// The two guards are unreachable from typed code
|
|
837
|
+
// The two guards are unreachable from typed code (`value` is a `string`) and are kept for
|
|
839
838
|
// untyped callers, which is the only way to reach them.
|
|
840
839
|
if (Predicate.isString(value) && value.indexOf('&') === -1) {
|
|
841
840
|
if (yield* this.#applyRegistrationHook(this.#onExternalEntity, key, value, 'external')) {
|
|
@@ -859,7 +858,7 @@ export class EntityDecoder {
|
|
|
859
858
|
map: Record<string, string | { regx: RegExp; val: string | EntityValFn } | { regex: RegExp; val: string | EntityValFn }>
|
|
860
859
|
): Effect.fn.Return<void, XmlError> {
|
|
861
860
|
// Cleared first and unconditionally, so registering entities is itself the start of a new
|
|
862
|
-
// document's budget
|
|
861
|
+
// document's budget, including when the call goes on to fail.
|
|
863
862
|
this.#totalExpansions = 0;
|
|
864
863
|
this.#expandedLength = 0;
|
|
865
864
|
if (!this.#onInputEntity) {
|
|
@@ -904,14 +903,14 @@ export class EntityDecoder {
|
|
|
904
903
|
/**
|
|
905
904
|
* @description Expand every entity reference in a string, in one pass. The output is never re-scanned, so no expansion can produce a _second_ one: a registered
|
|
906
905
|
* value that itself contains reference text reaches the caller as that literal text, unexpanded. What the limits bound is the growth of this single
|
|
907
|
-
* pass
|
|
906
|
+
* pass, meaning how much one round of expansion can add. Three inputs return before the scan and therefore never reach
|
|
908
907
|
* {@link EntityDecoderOptions.postCheck}: a non-string, the empty string, and any string with no `&` in it. The scan itself is `#expandAll`; what
|
|
909
908
|
* this method adds is the three inputs that skip it and the single join of what it collected.
|
|
910
909
|
*
|
|
911
910
|
* @example
|
|
912
911
|
* ```typescript
|
|
913
912
|
* import { Effect } from 'effect';
|
|
914
|
-
* import { EntityDecoder } from '@endevops/effect-xml
|
|
913
|
+
* import { EntityDecoder } from '@endevops/effect-codec-xml';
|
|
915
914
|
*
|
|
916
915
|
* const decoder = new EntityDecoder({ namedEntities: { copy: '©' } });
|
|
917
916
|
* Effect.runSync(Effect.orElseSucceed(decoder.addExternalEntity('brand', 'Acme'), () => undefined));
|
|
@@ -941,15 +940,15 @@ export class EntityDecoder {
|
|
|
941
940
|
/**
|
|
942
941
|
* @description Walk the string once and collect the pieces of every reference that resolved. Two advance rules make the walk terminate and keep it correct: an
|
|
943
942
|
* `&` that turns out to open nothing moves the cursor by one character rather than to the end of its run, so a second `&` in the same text is still
|
|
944
|
-
* found; and a reference that did resolve moves it to just past the `;`, so the text that was substituted for it is never looked at again
|
|
945
|
-
*
|
|
946
|
-
* decide and what it costs is `#chargeExpansion`'s to apply, which leaves the scanning here as the only thing with a rule of its own.
|
|
943
|
+
* found; and a reference that did resolve moves it to just past the `;`, so the text that was substituted for it is never looked at again. The pass
|
|
944
|
+
* stays single because of that, and a registered value containing `&` cannot expand a second level. What a reference becomes is `#resolveToken`'s
|
|
945
|
+
* to decide and what it costs is `#chargeExpansion`'s to apply, which leaves the scanning here as the only thing with a rule of its own.
|
|
947
946
|
*
|
|
948
947
|
* @param str - The string to expand. It always holds at least one `&` and is never empty, or the caller would have returned before reaching the
|
|
949
948
|
* walk.
|
|
950
949
|
*
|
|
951
950
|
* @returns An effect producing the pieces in order. The array is empty exactly when nothing was replaced, which the caller reads as "the input is
|
|
952
|
-
* its own result". Fails with {@link XmlError} and the reason the offending reference carries
|
|
951
|
+
* its own result". Fails with {@link XmlError} and the reason the offending reference carries: `ProhibitedCharacterReference`,
|
|
953
952
|
* `ExpansionLimitExceeded` or `ExpandedLengthLimitExceeded`.
|
|
954
953
|
*/
|
|
955
954
|
#expandAll = Effect.fnUntraced(function* (this: EntityDecoder, str: string): Effect.fn.Return<Array<string>, XmlError> {
|
|
@@ -996,25 +995,25 @@ export class EntityDecoder {
|
|
|
996
995
|
/**
|
|
997
996
|
* @description Decide what one reference expands to. The lists and maps are consulted in the one order the runtime uses, and the first that matches wins:
|
|
998
997
|
*
|
|
999
|
-
* 1. `remove
|
|
1000
|
-
* 2. `leave
|
|
1001
|
-
* 3. A `#`-prefixed token
|
|
998
|
+
* 1. `remove`: deleted outright, without the name ever being resolved, so the name need not exist.
|
|
999
|
+
* 2. `leave`: emitted as the original `&token;`, and charged to nothing.
|
|
1000
|
+
* 3. A `#`-prefixed token: the numeric pipeline, which is the only one of the four that can fail. Classification runs before any decision about
|
|
1002
1001
|
* `numericAllowed`, because the ranges that carry a minimum have to be caught whichever way that option is set.
|
|
1003
|
-
* 4. Anything else
|
|
1002
|
+
* 4. Anything else: resolved against the input map, then the external map, then the base map.
|
|
1004
1003
|
*
|
|
1005
1004
|
* @param token - The reference's token, e.g. `brand` or `#38`, with the `&` and the `;` already stripped. Never empty: the scanner drops `&;`
|
|
1006
1005
|
* before calling.
|
|
1007
1006
|
*
|
|
1008
1007
|
* @returns An effect producing what the reference expands to and the tier to charge it to, or `undefined` to leave it as written and charge it
|
|
1009
|
-
* nothing. `undefined` covers all three ways of leaving a reference alone
|
|
1010
|
-
*
|
|
1011
|
-
*
|
|
1008
|
+
* nothing. `undefined` covers all three ways of leaving a reference alone (a listed `leave` name, a numeric reference that is out of range, and a
|
|
1009
|
+
* name registered nowhere), and none of them is distinguishable from outside. Fails with {@link XmlError} and the `ProhibitedCharacterReference`
|
|
1010
|
+
* reason when the numeric policy throws on the codepoint.
|
|
1012
1011
|
*/
|
|
1013
1012
|
#resolveToken = Effect.fnUntraced(function* (this: EntityDecoder, token: string): Effect.fn.Return<ResolvedEntity | undefined, XmlError> {
|
|
1014
1013
|
if (this.#removeSet.has(token)) {
|
|
1015
1014
|
// Deleted without being resolved, so the name need not exist. Upstream guards this charge with
|
|
1016
1015
|
// `if (tier === undefined)`, and its `tier` is declared without an initialiser, so the branch is
|
|
1017
|
-
// unconditionally taken and the charge always lands on `external
|
|
1016
|
+
// unconditionally taken and the charge always lands on `external`, whatever tier the name would
|
|
1018
1017
|
// have resolved in. That is why a document full of removed built-ins can trip an `external` limit
|
|
1019
1018
|
// nothing it wrote could otherwise reach. Kept as written, since that is a behaviour a caller may
|
|
1020
1019
|
// already be relying on.
|
|
@@ -1022,7 +1021,7 @@ export class EntityDecoder {
|
|
|
1022
1021
|
}
|
|
1023
1022
|
|
|
1024
1023
|
// Emitted as the original `&token;`. The walk advances only past the `&` and leaves the `;` to be
|
|
1025
|
-
// copied by the next literal run, which
|
|
1024
|
+
// copied by the next literal run, which keeps the text coming back unchanged.
|
|
1026
1025
|
if (this.#leaveSet.has(token)) return undefined;
|
|
1027
1026
|
|
|
1028
1027
|
if (token.charCodeAt(0) === CODE_HASH) {
|
|
@@ -1066,9 +1065,9 @@ export class EntityDecoder {
|
|
|
1066
1065
|
|
|
1067
1066
|
/**
|
|
1068
1067
|
* @description Add one expansion to the running total and compare it against {@link EntityDecoderLimitOptions.maxTotalExpansions}. The comparison is `>` rather
|
|
1069
|
-
* than `>=`, so a limit of `n` allows exactly `n` expansions and throws on the `n + 1`th. That is a contract
|
|
1070
|
-
* states it
|
|
1071
|
-
*
|
|
1068
|
+
* than `>=`, so a limit of `n` allows exactly `n` expansions and throws on the `n + 1`th. That is a contract, and the option's own documentation
|
|
1069
|
+
* states it. It is also the kind of off-by-one a tidy-up changes by accident. The counter is intentionally not reset before failing: the over-limit
|
|
1070
|
+
* total reports the over-limit total, and {@link EntityDecoder.reset} is the caller's way to start a new document.
|
|
1072
1071
|
*
|
|
1073
1072
|
* @returns An effect that fails with {@link XmlError} and the `ExpansionLimitExceeded` reason once the count is past the ceiling, and succeeds
|
|
1074
1073
|
* otherwise. The `EntityReplacer` prefix in the message is preserved verbatim from the original throw, despite naming a class this decoder does
|
|
@@ -1090,7 +1089,7 @@ export class EntityDecoder {
|
|
|
1090
1089
|
/**
|
|
1091
1090
|
* @description Add one expansion's surplus to the running total and compare it against {@link EntityDecoderLimitOptions.maxExpandedLength}. Only the surplus
|
|
1092
1091
|
* counts, and only upward: a reference whose replacement is no longer than the `&token;` it replaces contributes zero, and a shrinking one
|
|
1093
|
-
* contributes nothing and cannot trip the limit at all. That
|
|
1092
|
+
* contributes nothing and cannot trip the limit at all. That keeps the ceiling a bound on growth rather than on document size.
|
|
1094
1093
|
*
|
|
1095
1094
|
* @param token - The reference's token, with the `&` and `;` stripped. The two delimiters count towards what the expansion displaced.
|
|
1096
1095
|
* @param replacement - What the reference expanded to, including `''` for a removal.
|
|
@@ -1118,8 +1117,8 @@ export class EntityDecoder {
|
|
|
1118
1117
|
/**
|
|
1119
1118
|
* @description Decide whether an entity of a given tier is charged against the limits.
|
|
1120
1119
|
*
|
|
1121
|
-
* @param tier - The tier the replacement is charged to. Every expansion that reaches here carries one
|
|
1122
|
-
* still carries the `external` tier
|
|
1120
|
+
* @param tier - The tier the replacement is charged to. Every expansion that reaches here carries one (a name deleted before it was ever resolved
|
|
1121
|
+
* still carries the `external` tier), so there is no absent case to answer.
|
|
1123
1122
|
*
|
|
1124
1123
|
* @returns `true` when it counts. `'all'` short-circuits, so a filter naming every tier charges everything regardless of which map it came from.
|
|
1125
1124
|
*/
|
|
@@ -1154,9 +1153,9 @@ export class EntityDecoder {
|
|
|
1154
1153
|
/**
|
|
1155
1154
|
* @description Find the strictest action a codepoint's range requires. Checked in this order:
|
|
1156
1155
|
*
|
|
1157
|
-
* 1. U+0000
|
|
1158
|
-
* 2. U+D800
|
|
1159
|
-
* 3. U+0001
|
|
1156
|
+
* 1. U+0000: governed by `nullNCR`, already clamped to `remove` or stricter
|
|
1157
|
+
* 2. U+D800 to U+DFFF: surrogates, always `remove`, under every policy and both XML versions
|
|
1158
|
+
* 3. U+0001 to U+001F other than tab, newline, carriage return: XML 1.0 only, `remove` Nothing else is classified. U+007F to U+009F (C1) and the
|
|
1160
1159
|
* U+FFFE/U+FFFF noncharacters are not checked, even though XML 1.0 §2.2 prohibits them and the `xmlVersion` option's own documentation claims C1
|
|
1161
1160
|
* is only permitted under 1.1. Both gaps are upstream's and are kept.
|
|
1162
1161
|
*
|
|
@@ -1180,11 +1179,11 @@ export class EntityDecoder {
|
|
|
1180
1179
|
* @description Turn a resolved action level into a replacement.
|
|
1181
1180
|
*
|
|
1182
1181
|
* @param action - A level from {@link NCR_LEVEL}. A level outside the four known ones falls through to the allow behaviour, so a bad level cannot
|
|
1183
|
-
* produce a wrong string
|
|
1182
|
+
* produce a wrong string. It can only fail open.
|
|
1184
1183
|
* @param token - The raw token, e.g. `#38`, for the error message.
|
|
1185
1184
|
* @param cp - The codepoint, for the error message.
|
|
1186
1185
|
*
|
|
1187
|
-
* @returns An effect producing the character for `allow`, `''` for `remove`, and `undefined` for `leave
|
|
1186
|
+
* @returns An effect producing the character for `allow`, `''` for `remove`, and `undefined` for `leave`, which the caller reads as "emit the
|
|
1188
1187
|
* original `&token;`". Fails with {@link XmlError} and the `ProhibitedCharacterReference` reason for `throw`, naming both the token and the
|
|
1189
1188
|
* codepoint.
|
|
1190
1189
|
*/
|
|
@@ -1221,7 +1220,7 @@ export class EntityDecoder {
|
|
|
1221
1220
|
*
|
|
1222
1221
|
* @param token - The raw token without `&` and `;`, e.g. `#38`, `#x26`, `#X26`.
|
|
1223
1222
|
*
|
|
1224
|
-
* @returns An effect producing the replacement
|
|
1223
|
+
* @returns An effect producing the replacement (the empty string meaning "delete") or `undefined` to leave the reference as written. Fails with
|
|
1225
1224
|
* {@link XmlError} and the `ProhibitedCharacterReference` reason when the effective action is `throw`.
|
|
1226
1225
|
*/
|
|
1227
1226
|
#resolveNCR(token: string): Effect.Effect<string | undefined, XmlError> {
|
package/src/errors.ts
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
// The one way XML serialization can fail that a `SchemaIssue.Issue` does not already describe.
|
|
2
2
|
//
|
|
3
|
-
// A schema mismatch
|
|
3
|
+
// A schema mismatch, such as a `number` where the document says `text`, is a
|
|
4
4
|
// `SchemaIssue.Issue` and comes from Effect's own parser. What is left is the
|
|
5
5
|
// part Effect knows nothing about: a document that is not well-formed XML, and a
|
|
6
6
|
// field name that cannot be written as one.
|
|
@@ -12,7 +12,7 @@ import { Schema } from 'effect';
|
|
|
12
12
|
*
|
|
13
13
|
* @example
|
|
14
14
|
* ```typescript
|
|
15
|
-
* import { XmlParseError } from '@endevops/effect-xml
|
|
15
|
+
* import { XmlParseError } from '@endevops/effect-codec-xml';
|
|
16
16
|
*
|
|
17
17
|
* const error = new XmlParseError({ message: 'Unclosed element', position: 12, input: '<a><b>' });
|
|
18
18
|
* ```;
|
|
@@ -42,7 +42,7 @@ export class XmlParseError extends Schema.TaggedError<XmlParseError>()('XmlParse
|
|
|
42
42
|
*
|
|
43
43
|
* @example
|
|
44
44
|
* ```typescript
|
|
45
|
-
* import { XmlRenderError } from '@endevops/effect-xml
|
|
45
|
+
* import { XmlRenderError } from '@endevops/effect-codec-xml';
|
|
46
46
|
*
|
|
47
47
|
* const error = new XmlRenderError({ message: 'Invalid XML name "not a name"' });
|
|
48
48
|
* ```;
|
package/src/index.ts
CHANGED
|
@@ -1,14 +1,14 @@
|
|
|
1
1
|
/**
|
|
2
2
|
* @description A round-trip Effect Schema codec for XML. `toCodecXml(schema)` returns a `Schema` whose `Encoded` is XML text, so `Schema.encodeSync` writes a
|
|
3
|
-
* document and `Schema.decodeSync` reads one back,
|
|
4
|
-
*
|
|
3
|
+
* document and `Schema.decodeSync` reads one back, as `Schema.toCodecJson` does for JSON. Attributes are the fields whose names start with `@`, so
|
|
4
|
+
* `@xmlns` is written as `xmlns="…"`, and `#text` holds an element's character data. A schema node annotated with `xmlNamespace` is placed in that
|
|
5
5
|
* namespace, and the codec resolves the document's own prefixes back to it. `renderXml` and `parseXml` are the text layer the codec runs underneath,
|
|
6
6
|
* and remain available on their own.
|
|
7
7
|
*
|
|
8
8
|
* @example
|
|
9
9
|
* ```typescript
|
|
10
10
|
* import { Schema } from 'effect';
|
|
11
|
-
* import { toCodecXml } from '@endevops/effect-xml
|
|
11
|
+
* import { toCodecXml } from '@endevops/effect-codec-xml';
|
|
12
12
|
*
|
|
13
13
|
* const Book = Schema.Struct({
|
|
14
14
|
* '@id': Schema.String,
|