@endevops/effect-codec-xml 0.0.1 → 0.1.0-beta.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +63 -62
- package/dist/codec.d.ts +17 -9
- package/dist/codec.d.ts.map +1 -1
- package/dist/codec.js +25 -16
- package/dist/codec.js.map +1 -1
- package/dist/conventions.d.ts +4 -4
- package/dist/conventions.js +7 -7
- package/dist/conventions.js.map +1 -1
- package/dist/entities/entity-decoder.d.ts +33 -33
- package/dist/entities/entity-decoder.d.ts.map +1 -1
- package/dist/entities/entity-decoder.js +63 -64
- package/dist/entities/entity-decoder.js.map +1 -1
- package/dist/errors.d.ts +2 -2
- package/dist/errors.js +2 -2
- package/dist/errors.js.map +1 -1
- package/dist/namespaces.js +2 -2
- package/dist/namespaces.js.map +1 -1
- package/dist/naming.d.ts +6 -6
- package/dist/naming.d.ts.map +1 -1
- package/dist/naming.js +3 -3
- package/dist/naming.js.map +1 -1
- package/dist/parse.d.ts +5 -5
- package/dist/parse.js +14 -14
- package/dist/parse.js.map +1 -1
- package/dist/render.d.ts +1 -1
- package/dist/render.d.ts.map +1 -1
- package/dist/render.js +28 -28
- package/dist/render.js.map +1 -1
- package/dist/xml-error.d.ts +9 -9
- package/dist/xml-error.js +18 -18
- package/dist/xml-error.js.map +1 -1
- package/dist/xml-value.d.ts +7 -7
- package/dist/xml-value.d.ts.map +1 -1
- package/dist/xml-value.js +6 -7
- package/dist/xml-value.js.map +1 -1
- package/package.json +1 -1
- package/src/codec.ts +91 -71
- package/src/conventions.ts +7 -7
- package/src/entities/entity-decoder.ts +98 -99
- package/src/errors.ts +3 -3
- package/src/index.ts +3 -3
- package/src/namespaces.ts +16 -16
- package/src/naming.ts +34 -35
- package/src/parse.ts +26 -26
- package/src/render.ts +44 -44
- package/src/xml-error.ts +18 -18
- package/src/xml-value.ts +10 -11
package/src/namespaces.ts
CHANGED
|
@@ -19,21 +19,21 @@
|
|
|
19
19
|
// - `xmlValue` marks one field as the element's character data, the `#text`
|
|
20
20
|
// value, for an element that also carries attributes or children.
|
|
21
21
|
//
|
|
22
|
-
//
|
|
23
|
-
//
|
|
24
|
-
//
|
|
25
|
-
//
|
|
22
|
+
// An element's namespace is inherited by its descendants, as an XML default
|
|
23
|
+
// namespace is. An attribute never inherits: it is in a namespace only when it
|
|
24
|
+
// is annotated with one explicitly, because a default namespace does not apply
|
|
25
|
+
// to attributes.
|
|
26
26
|
//
|
|
27
|
-
// On
|
|
28
|
-
//
|
|
29
|
-
//
|
|
30
|
-
//
|
|
31
|
-
// declaration attributes are dropped from the decoded value;
|
|
32
|
-
//
|
|
27
|
+
// On encode, each element writes its own declaration when the prefix or default
|
|
28
|
+
// is not already in scope. On decode, the parser's own declarations are read
|
|
29
|
+
// into scope and every name is resolved to its URI, so a document that binds the
|
|
30
|
+
// same URI to a different prefix still decodes to the same value. The
|
|
31
|
+
// declaration attributes are dropped from the decoded value; the codec manages
|
|
32
|
+
// them, not the schema.
|
|
33
33
|
//
|
|
34
|
-
// The plan is built per local name,
|
|
35
|
-
// name cannot belong to two namespaces in one codec;
|
|
36
|
-
// codec is built rather than
|
|
34
|
+
// The plan is built per local name, and a schema field is one local name. One
|
|
35
|
+
// local name cannot belong to two namespaces in one codec; the plan reports that
|
|
36
|
+
// when the codec is built rather than guessing.
|
|
37
37
|
|
|
38
38
|
import type { Schema } from 'effect';
|
|
39
39
|
|
|
@@ -396,7 +396,7 @@ interface Scan {
|
|
|
396
396
|
readonly problems: Array<string>;
|
|
397
397
|
/**
|
|
398
398
|
* @description The AST nodes on the current scan path. A recursive schema terminates because the `Suspend` node is still on the path when its thunk is reached,
|
|
399
|
-
* and a schema reused under two sibling paths is scanned once per path because each node is removed again on the way out.
|
|
399
|
+
* and a schema reused under two sibling paths is scanned once per path because each node is removed again on the way back out.
|
|
400
400
|
*/
|
|
401
401
|
readonly seen: Set<SchemaAST.AST>;
|
|
402
402
|
}
|
|
@@ -634,7 +634,7 @@ const scanNode = (scan: Scan, ast: SchemaAST.AST, inherited: XmlNamespace | unde
|
|
|
634
634
|
};
|
|
635
635
|
|
|
636
636
|
/**
|
|
637
|
-
* @description Collects the namespace and name of every field in a schema.
|
|
637
|
+
* @description Collects the namespace and name of every field in a schema. Descendant elements inherit a namespace, as they inherit a default namespace, and an
|
|
638
638
|
* element field records its own namespace, so encode and decode can find it by the local name alone.
|
|
639
639
|
*
|
|
640
640
|
* @param schema - The schema to walk.
|
|
@@ -900,7 +900,7 @@ const schemaKey = (plan: NamespacePlan, parent: string, key: string, scope: Reco
|
|
|
900
900
|
/**
|
|
901
901
|
* @description The scope a child element resolves its own name against: the declarations it carries on itself, layered over the parent scope. An element may
|
|
902
902
|
* declare the prefix it uses on the element itself, so its own name is read with those bindings in scope. A repeated element arrives as an array, so
|
|
903
|
-
* the first member stands in for the run
|
|
903
|
+
* the first member stands in for the run; every member describes the same element and carries the same declaration.
|
|
904
904
|
*
|
|
905
905
|
* @param child - The child value.
|
|
906
906
|
* @param scope - The bindings in scope above the child.
|
package/src/naming.ts
CHANGED
|
@@ -8,10 +8,10 @@
|
|
|
8
8
|
//
|
|
9
9
|
// The five predicates and `sanitize` are plain synchronous functions: a regex
|
|
10
10
|
// test cannot fail and a character substitution has nothing to fail about, so
|
|
11
|
-
// there is no effect to model. `validate` does have one failure to report
|
|
11
|
+
// there is no effect to model. `validate` does have one failure to report: an
|
|
12
12
|
// unknown production, unreachable from TypeScript where `Production` is a
|
|
13
13
|
// closed union but reachable for an untyped JavaScript caller, or a value that
|
|
14
|
-
// crossed a boundary as `unknown
|
|
14
|
+
// crossed a boundary as `unknown`. It answers with an `Effect` whose error
|
|
15
15
|
// channel is that {@link XmlError}. The value it produces is still a plain
|
|
16
16
|
// result.
|
|
17
17
|
|
|
@@ -20,7 +20,7 @@ import { Effect } from 'effect';
|
|
|
20
20
|
import { XmlError } from '#/xml-error.ts';
|
|
21
21
|
|
|
22
22
|
/**
|
|
23
|
-
* @description The XML specification version a production is validated against. The two differ only in their non-ASCII character ranges
|
|
23
|
+
* @description The XML specification version a production is validated against. The two differ only in their non-ASCII character ranges. See {@link getRegexes}.
|
|
24
24
|
*/
|
|
25
25
|
export type XmlVersion = '1.0' | '1.1';
|
|
26
26
|
|
|
@@ -42,7 +42,7 @@ export interface ValidationOptions {
|
|
|
42
42
|
/**
|
|
43
43
|
* @description Restrict matching to the ASCII subset of the NameStartChar/NameChar productions and skip unicode-aware regex matching entirely. Faster,
|
|
44
44
|
* especially for XML 1.1 (which otherwise requires the `/u` regex flag), but rejects legitimate non-ASCII XML names. Off by default for backward
|
|
45
|
-
* compatibility
|
|
45
|
+
* compatibility. Opt in only when inputs are known to be ASCII. Defaults to false.
|
|
46
46
|
*/
|
|
47
47
|
asciiOnly?: boolean;
|
|
48
48
|
}
|
|
@@ -60,7 +60,7 @@ export interface SanitizeOptions {
|
|
|
60
60
|
*/
|
|
61
61
|
asciiOnly?: boolean;
|
|
62
62
|
/**
|
|
63
|
-
* @description Accepted and ignored. Sanitizing is not version-dependent
|
|
63
|
+
* @description Accepted and ignored. Sanitizing is not version-dependent: the character set it considers illegal is the union of both versions. But the option
|
|
64
64
|
* is part of the published signature, so dropping it would break callers that pass it through a shared options object.
|
|
65
65
|
*/
|
|
66
66
|
xmlVersion?: XmlVersion;
|
|
@@ -94,7 +94,7 @@ const PRODUCTIONS = ['name', 'ncName', 'qName', 'nmToken', 'nmTokens'] as const
|
|
|
94
94
|
type ProductionRegexes = Record<Production, RegExp>;
|
|
95
95
|
|
|
96
96
|
// ---------------------------------------------------------------------------
|
|
97
|
-
// Character class strings
|
|
97
|
+
// Character class strings: XML 1.0
|
|
98
98
|
//
|
|
99
99
|
// NameStartChar ::= ":" | [A-Z] | "_" | [a-z]
|
|
100
100
|
// | [#xC0-#xD6] | [#xD8-#xF6] | [#xF8-#x2FF]
|
|
@@ -127,7 +127,7 @@ const nameStartChar10 =
|
|
|
127
127
|
const nameChar10 = nameStartChar10 + '\\-\\.\\d' + '\u00B7' + '\u0300-\u036F' + '\u203F-\u2040';
|
|
128
128
|
|
|
129
129
|
// ---------------------------------------------------------------------------
|
|
130
|
-
// Character class strings
|
|
130
|
+
// Character class strings: XML 1.1
|
|
131
131
|
//
|
|
132
132
|
// Differences from XML 1.0:
|
|
133
133
|
//
|
|
@@ -138,7 +138,7 @@ const nameChar10 = nameStartChar10 + '\\-\\.\\d' + '\u00B7' + '\u0300-\u036F' +
|
|
|
138
138
|
//
|
|
139
139
|
// 1.0 tops out at \uFFFD (BMP only)
|
|
140
140
|
// 1.1 adds \u{10000}-\u{EFFFF} (supplementary planes)
|
|
141
|
-
// These require the /u flag on the RegExp
|
|
141
|
+
// These require the /u flag on the RegExp. See buildRegexes below.
|
|
142
142
|
//
|
|
143
143
|
// NameChar:
|
|
144
144
|
// 1.1 adds \u0487 (Combining Cyrillic Millions Sign, added in Unicode 4.0)
|
|
@@ -146,7 +146,7 @@ const nameChar10 = nameStartChar10 + '\\-\\.\\d' + '\u00B7' + '\u0300-\u036F' +
|
|
|
146
146
|
|
|
147
147
|
const nameStartChar11 =
|
|
148
148
|
':A-Za-z_' +
|
|
149
|
-
'\u00C0-\u02FF' + // merged
|
|
149
|
+
'\u00C0-\u02FF' + // merged: 1.0 had three split ranges here
|
|
150
150
|
'\u0370-\u037D' +
|
|
151
151
|
'\u037F-\u0486\u0488-\u1FFF' + // split to exclude \u0487 (combining mark, never a NameStartChar)
|
|
152
152
|
'\u200C-\u200D' +
|
|
@@ -155,21 +155,21 @@ const nameStartChar11 =
|
|
|
155
155
|
'\u3001-\uD7FF' +
|
|
156
156
|
'\uF900-\uFDCF' +
|
|
157
157
|
'\uFDF0-\uFFFD' +
|
|
158
|
-
'\u{10000}-\u{EFFFF}'; // supplementary planes
|
|
158
|
+
'\u{10000}-\u{EFFFF}'; // supplementary planes: REQUIRES /u flag on RegExp
|
|
159
159
|
|
|
160
160
|
const nameChar11 =
|
|
161
161
|
nameStartChar11 +
|
|
162
162
|
'\\-\\.\\d' +
|
|
163
163
|
'\u00B7' +
|
|
164
164
|
'\u0300-\u036F' +
|
|
165
|
-
'\u0487' + // Combining Cyrillic Millions Sign
|
|
165
|
+
'\u0487' + // Combining Cyrillic Millions Sign: valid in 1.1, not 1.0
|
|
166
166
|
'\u203F-\u2040';
|
|
167
167
|
|
|
168
168
|
// ---------------------------------------------------------------------------
|
|
169
169
|
// Regex builders
|
|
170
170
|
//
|
|
171
|
-
// XML 1.0 regexes: no flags
|
|
172
|
-
// XML 1.1 regexes: /u flag
|
|
171
|
+
// XML 1.0 regexes: no flags, BMP only, standard JS regex behaviour.
|
|
172
|
+
// XML 1.1 regexes: /u flag, required for \u{10000}-\u{EFFFF} to match actual
|
|
173
173
|
// supplementary code points rather than lone surrogates (which are illegal XML).
|
|
174
174
|
// ---------------------------------------------------------------------------
|
|
175
175
|
|
|
@@ -196,34 +196,33 @@ const buildRegexes = (startChar: string, char: string, flags = ''): ProductionRe
|
|
|
196
196
|
};
|
|
197
197
|
};
|
|
198
198
|
|
|
199
|
-
const regexes10 = buildRegexes(nameStartChar10, nameChar10); // no /u
|
|
200
|
-
const regexes11 = buildRegexes(nameStartChar11, nameChar11, 'u'); // /u
|
|
199
|
+
const regexes10 = buildRegexes(nameStartChar10, nameChar10); // no /u, BMP only
|
|
200
|
+
const regexes11 = buildRegexes(nameStartChar11, nameChar11, 'u'); // /u enables \u{10000}-\u{EFFFF}
|
|
201
201
|
|
|
202
202
|
// ---------------------------------------------------------------------------
|
|
203
203
|
// ASCII-only fast path (opt-in, off by default)
|
|
204
204
|
//
|
|
205
|
-
// The XML 1.0
|
|
206
|
-
//
|
|
207
|
-
//
|
|
208
|
-
//
|
|
209
|
-
//
|
|
205
|
+
// The XML 1.0 and 1.1 NameStartChar/NameChar productions differ only in their
|
|
206
|
+
// non-ASCII ranges: merged vs split Latin-1 ranges, \u0487, and supplementary
|
|
207
|
+
// planes. Restricted to ASCII, both versions collapse to the same character
|
|
208
|
+
// classes, so one regex pair covers both xmlVersion values and no /u flag is
|
|
209
|
+
// needed.
|
|
210
210
|
//
|
|
211
|
-
//
|
|
212
|
-
//
|
|
213
|
-
//
|
|
214
|
-
//
|
|
215
|
-
//
|
|
216
|
-
//
|
|
217
|
-
//
|
|
218
|
-
//
|
|
219
|
-
//
|
|
220
|
-
// default.
|
|
211
|
+
// Unicode-aware regexes (the /u flag, required for XML 1.1's supplementary-
|
|
212
|
+
// plane range) are measurably slower in V8 than plain non-unicode regexes on
|
|
213
|
+
// the same input, even when the input is pure ASCII. For the common case
|
|
214
|
+
// (HTML/SVG ids, XML tags) names are ASCII, so callers who know this can opt
|
|
215
|
+
// in to skip the unicode-aware matching path entirely. The win is conditional:
|
|
216
|
+
// it applies mainly to XML 1.1 input (avoids /u), or at scale where the larger
|
|
217
|
+
// unicode character classes add engine overhead. The option also changes
|
|
218
|
+
// behaviour (it rejects legitimate non-ASCII XML 1.0/1.1 names), so it must
|
|
219
|
+
// never be enabled silently. That is why the default is off.
|
|
221
220
|
// ---------------------------------------------------------------------------
|
|
222
221
|
|
|
223
222
|
const nameStartCharAscii = ':A-Za-z_';
|
|
224
223
|
const nameCharAscii = nameStartCharAscii + '\\-\\.\\d';
|
|
225
224
|
|
|
226
|
-
const regexesAscii = buildRegexes(nameStartCharAscii, nameCharAscii); // no /u
|
|
225
|
+
const regexesAscii = buildRegexes(nameStartCharAscii, nameCharAscii); // no /u, ASCII only
|
|
227
226
|
|
|
228
227
|
/**
|
|
229
228
|
* @description The compiled regex set for a version/ASCII combination. Only three sets are ever built, at module load; this is a lookup, not a compile.
|
|
@@ -242,7 +241,7 @@ const getRegexes = (xmlVersion: XmlVersion = '1.0', asciiOnly = false): Producti
|
|
|
242
241
|
// Boolean validators
|
|
243
242
|
//
|
|
244
243
|
// One plain predicate per production. A regex test cannot fail, and every one of these is called per name
|
|
245
|
-
// inside a parser's or codec's hot loop, so a boolean is the
|
|
244
|
+
// inside a parser's or codec's hot loop, so a boolean is the direct answer and there is no effect to
|
|
246
245
|
// allocate or run.
|
|
247
246
|
// ---------------------------------------------------------------------------
|
|
248
247
|
|
|
@@ -296,7 +295,7 @@ export const isNmToken = (str: string, { xmlVersion = '1.0', asciiOnly = false }
|
|
|
296
295
|
getRegexes(xmlVersion, asciiOnly).nmToken.test(str);
|
|
297
296
|
|
|
298
297
|
/**
|
|
299
|
-
* @description Whether the string is a valid NMTokens value
|
|
298
|
+
* @description Whether the string is a valid NMTokens value: a whitespace-separated list of NMToken values. Used for: DTD NMTOKENS attribute values.
|
|
300
299
|
*
|
|
301
300
|
* @param str - The candidate list.
|
|
302
301
|
* @param opts - `asciiOnly` skips unicode-aware matching, ASCII names only (default false).
|
|
@@ -455,7 +454,7 @@ const diagnoseWith = (str: string, production: Production, isValid: boolean, asc
|
|
|
455
454
|
* @example
|
|
456
455
|
* ```typescript
|
|
457
456
|
* import { Effect } from 'effect';
|
|
458
|
-
* import { validate } from '@endevops/effect-xml
|
|
457
|
+
* import { validate } from '@endevops/effect-codec-xml';
|
|
459
458
|
*
|
|
460
459
|
* Effect.runSync(validate('not a name', 'ncName'));
|
|
461
460
|
* // { valid: false, production: 'ncName', input: 'not a name', reason: 'First character " " is not a valid NameStartChar', position: 0 }
|
|
@@ -466,7 +465,7 @@ const diagnoseWith = (str: string, production: Production, isValid: boolean, asc
|
|
|
466
465
|
* @param opts - Version and ASCII-only selection, as for the boolean validators.
|
|
467
466
|
*
|
|
468
467
|
* @returns An effect producing a discriminated result: the plain triple when valid, or the offending `reason` and `position` when not. A name that
|
|
469
|
-
* fails to validate is a `valid: false` result, not a failure
|
|
468
|
+
* fails to validate is a `valid: false` result, not a failure: an invalid name is the question being answered. The effect fails only with an
|
|
470
469
|
* {@link XmlError} and the `InvalidProduction` reason for an unknown production, which is unreachable from TypeScript and is the guard for untyped
|
|
471
470
|
* JavaScript callers.
|
|
472
471
|
*/
|
package/src/parse.ts
CHANGED
|
@@ -34,7 +34,7 @@ export interface XmlParseOptions {
|
|
|
34
34
|
* @description Keep the whitespace at the edges of every text run.
|
|
35
35
|
*
|
|
36
36
|
* @default false\
|
|
37
|
-
* which trims it
|
|
37
|
+
* which trims it. Trimming makes a pretty-printed document
|
|
38
38
|
* read as the same value as an unindented one, because the indentation around a child element and around a closing tag lands at the edges of its
|
|
39
39
|
* parent's text. Whitespace _inside_ a run is content and is never touched either way, so `'one two'` and a paragraph with a newline in the middle
|
|
40
40
|
* of it survive. Set it to `true` to keep leading and trailing spaces in text exactly as written, at the cost of a document that was laid out on
|
|
@@ -68,13 +68,13 @@ export interface XmlParseOptions {
|
|
|
68
68
|
/**
|
|
69
69
|
* @description Parses an XML document into its root element's content.\
|
|
70
70
|
* The walk itself is synchronous, but it reports a malformed document by failing with an {@link XmlParseError} rather than by throwing, so the failure lands in the effect's error channel where `catchTag`, `retry` and a fallback can all
|
|
71
|
-
* see it. A failed parse is an expected outcome of reading untrusted text
|
|
71
|
+
* see it. A failed parse is an expected outcome of reading untrusted text, and those combinators key off it, so only a defect would hide it.
|
|
72
72
|
* The span is the boundary a performance trace hangs off: it carries the document's length, which is the size that drives the parser's cost, so a
|
|
73
73
|
* slow parse in a profile can be attributed to the input that produced it. A caller that wants the value outside an `Effect` uses
|
|
74
74
|
* {@link parseXmlDocument}, which runs the same walk synchronously and throws instead. The walk is plain recursive descent rather than a chain of
|
|
75
|
-
* `yield*`es. Publicly `parseXml` is still an `Effect
|
|
76
|
-
* into the typed error channel
|
|
77
|
-
* those, and
|
|
75
|
+
* `yield*`es. Publicly `parseXml` is still an `Effect`: it suspends the walk so it runs lazily under the span, and folds the failure the walk throws
|
|
76
|
+
* into the typed error channel, but inside a document there is no effect boundary per tag, attribute or text run. A 500-row report is thousands of
|
|
77
|
+
* those, and one fiber step per construct dominated the `parse 500 rows` benchmark. The typed failure survives: the walk throws an
|
|
78
78
|
* {@link XmlParseError} and `parseXml` catches it into `Effect.fail`.
|
|
79
79
|
*
|
|
80
80
|
* @param text - The document to read.
|
|
@@ -140,8 +140,8 @@ const SLASH = 47;
|
|
|
140
140
|
const EQUALS = 61;
|
|
141
141
|
|
|
142
142
|
/**
|
|
143
|
-
* @description One element as the parser saw it: the name it was written under, and the value it holds. Carrying the name alongside the value
|
|
144
|
-
*
|
|
143
|
+
* @description One element as the parser saw it: the name it was written under, and the value it holds. Carrying the name alongside the value lets the parent file
|
|
144
|
+
* it correctly, because the value alone cannot say: a text-only element reduces to a bare string.
|
|
145
145
|
*/
|
|
146
146
|
interface Element {
|
|
147
147
|
readonly name: string;
|
|
@@ -179,9 +179,9 @@ interface Content {
|
|
|
179
179
|
}
|
|
180
180
|
|
|
181
181
|
/**
|
|
182
|
-
* @description What sits at the cursor inside an element's body. Naming what is there before deciding what to do with it
|
|
183
|
-
*
|
|
184
|
-
*
|
|
182
|
+
* @description What sits at the cursor inside an element's body. Naming what is there before deciding what to do with it lets the content loop stay a dispatch:
|
|
183
|
+
* each construct is recognised in one place, against the ones that cannot be confused with it, rather than by a chain of `startsWith` guesses where
|
|
184
|
+
* each had to remember what the last had already ruled out.
|
|
185
185
|
*/
|
|
186
186
|
type Construct = 'text' | 'close' | 'comment' | 'cdata' | 'instruction' | 'child';
|
|
187
187
|
|
|
@@ -213,7 +213,7 @@ const parseDocument = (text: string, options: XmlParseOptions): XmlDocument => {
|
|
|
213
213
|
const nameOptions = { mode: resolved.name, xmlVersion: resolved.xmlVersion };
|
|
214
214
|
|
|
215
215
|
/**
|
|
216
|
-
* @description Names already resolved by this parse. A document repeats names
|
|
216
|
+
* @description Names already resolved by this parse. A document repeats names (every one of five hundred rows has a `sku`), and a validator that ran per
|
|
217
217
|
* occurrence would pay for the same answer five hundred times.
|
|
218
218
|
*/
|
|
219
219
|
const nameCache = new Map<string, string>();
|
|
@@ -258,7 +258,7 @@ const parseDocument = (text: string, options: XmlParseOptions): XmlDocument => {
|
|
|
258
258
|
};
|
|
259
259
|
|
|
260
260
|
/**
|
|
261
|
-
* @description Consumes whitespace, comments, processing instructions and a DOCTYPE, leaving the cursor on the first character that is none of them
|
|
261
|
+
* @description Consumes whitespace, comments, processing instructions and a DOCTYPE, leaving the cursor on the first character that is none of them, or at the
|
|
262
262
|
* end of the document.
|
|
263
263
|
*/
|
|
264
264
|
const skipMisc = (): void => {
|
|
@@ -294,7 +294,7 @@ const parseDocument = (text: string, options: XmlParseOptions): XmlDocument => {
|
|
|
294
294
|
const start = at;
|
|
295
295
|
while (at < text.length) {
|
|
296
296
|
const char = text.charCodeAt(at);
|
|
297
|
-
// Whitespace, `/`, `=` and `>` all end a name. Stopping on `/` and `>`
|
|
297
|
+
// Whitespace, `/`, `=` and `>` all end a name. Stopping on `/` and `>` lets `<a/>` and `<a>` share one loop.
|
|
298
298
|
if (isWhitespace(char) || char === SLASH || char === EQUALS || char === GT) {
|
|
299
299
|
break;
|
|
300
300
|
}
|
|
@@ -322,7 +322,7 @@ const parseDocument = (text: string, options: XmlParseOptions): XmlDocument => {
|
|
|
322
322
|
at++;
|
|
323
323
|
|
|
324
324
|
const end = text.indexOf(quote ?? '', at);
|
|
325
|
-
// A raw quote cannot appear inside a quoted value
|
|
325
|
+
// A raw quote cannot appear inside a quoted value (it would have to be written `"`), so the next quote of the same kind always closes it.
|
|
326
326
|
if (end === -1) {
|
|
327
327
|
throw new XmlParseError({ message: `Unterminated value for attribute "${name}"`, position: at, input: text });
|
|
328
328
|
}
|
|
@@ -425,8 +425,8 @@ const parseDocument = (text: string, options: XmlParseOptions): XmlDocument => {
|
|
|
425
425
|
/**
|
|
426
426
|
* @description What the cursor is sitting on inside an element's body. The two things the loop cannot read are refused here rather than in it: running out of
|
|
427
427
|
* document and a declaration, which is markup the parser does not accept inside an element. Recognising the constructs that _are_ read is the rest,
|
|
428
|
-
* and the order
|
|
429
|
-
*
|
|
428
|
+
* and the order rules out the shorter prefixes first: `</` before `<?` before any other `<!`, and `<![CDATA[` before the `<!` that would otherwise
|
|
429
|
+
* match it.
|
|
430
430
|
*
|
|
431
431
|
* @param name - The name the enclosing element's start tag gave it, for the unterminated-body message.
|
|
432
432
|
*
|
|
@@ -458,7 +458,7 @@ const parseDocument = (text: string, options: XmlParseOptions): XmlDocument => {
|
|
|
458
458
|
};
|
|
459
459
|
|
|
460
460
|
/**
|
|
461
|
-
* @description Consumes a `</name>`, checking
|
|
461
|
+
* @description Consumes a `</name>`, checking as it goes that it is the tag that closes this element and that it is well-formed.
|
|
462
462
|
*
|
|
463
463
|
* @param name - The name the start tag gave the element, which the closing tag has to match.
|
|
464
464
|
*/
|
|
@@ -490,7 +490,7 @@ const parseDocument = (text: string, options: XmlParseOptions): XmlDocument => {
|
|
|
490
490
|
};
|
|
491
491
|
|
|
492
492
|
/**
|
|
493
|
-
* @description Reads a `<![CDATA[…]]>` section. CDATA is character data, and character data is what it holds, so it joins the element's text as it stands
|
|
493
|
+
* @description Reads a `<![CDATA[…]]>` section. CDATA is character data, and character data is what it holds, so it joins the element's text as it stands. The
|
|
494
494
|
* entities in it are literal text and must not be expanded.
|
|
495
495
|
*
|
|
496
496
|
* @returns The section's contents.
|
|
@@ -522,15 +522,15 @@ const parseDocument = (text: string, options: XmlParseOptions): XmlDocument => {
|
|
|
522
522
|
*/
|
|
523
523
|
const finishElement = (record: Record<string, XmlValue>, hasAttributes: boolean, text: string, hasChildren: boolean): XmlValue => {
|
|
524
524
|
// Whitespace at the edges of a text run is dropped unless the caller asked to
|
|
525
|
-
// keep it.
|
|
525
|
+
// keep it. Trimming here makes a pretty-printed document round trip: the
|
|
526
526
|
// indentation a renderer puts around a child element and around a closing tag
|
|
527
527
|
// lands at the edges of its parent's text, and trimming removes exactly that
|
|
528
|
-
// and nothing else. Whitespace *inside* the run
|
|
529
|
-
// newline in the middle of a paragraph
|
|
528
|
+
// and nothing else. Whitespace *inside* the run (between two words, or a
|
|
529
|
+
// newline in the middle of a paragraph) is content and stays.
|
|
530
530
|
const content = resolved.preserveWhitespace ? text : text.trim();
|
|
531
531
|
|
|
532
532
|
if (!hasAttributes && !hasChildren) {
|
|
533
|
-
// A leaf is character data on its own. Returning the string rather than a `{ '#text': … }` record
|
|
533
|
+
// A leaf is character data on its own. Returning the string rather than a `{ '#text': … }` record lets
|
|
534
534
|
// `Schema.Struct({ name: Schema.String })` round-trip.
|
|
535
535
|
return content;
|
|
536
536
|
}
|
|
@@ -577,10 +577,10 @@ const parseDocumentResult = (text: string, options: XmlParseOptions): Result.Res
|
|
|
577
577
|
};
|
|
578
578
|
|
|
579
579
|
/**
|
|
580
|
-
* @description Decodes character references, falling back to the raw text when the reference is not one the decoder recognises. The fallback
|
|
581
|
-
*
|
|
582
|
-
*
|
|
583
|
-
*
|
|
580
|
+
* @description Decodes character references, falling back to the raw text when the reference is not one the decoder recognises. The fallback keeps a bare `&`
|
|
581
|
+
* survivable: the decoder treats it as a malformed reference and fails, and a document containing one is far more likely to be worth reading than to
|
|
582
|
+
* be rejected. The `&` is escaped on the way out, so the value still round-trips. The decoder answers with an `Effect`, and this is the one place a
|
|
583
|
+
* parse still runs one. It is only reached when the raw text holds an `&`, since the common case returns before it, and the effect is synchronous, so
|
|
584
584
|
* the run is cheap next to the decoder's own work.
|
|
585
585
|
*
|
|
586
586
|
* @param raw - Text read straight from the source, with references unexpanded.
|
package/src/render.ts
CHANGED
|
@@ -7,23 +7,23 @@
|
|
|
7
7
|
// string.
|
|
8
8
|
//
|
|
9
9
|
// The walk itself is plain synchronous functions rather than a chain of
|
|
10
|
-
// `yield*`es. Publicly `renderXml` is still an `Effect
|
|
10
|
+
// `yield*`es. Publicly `renderXml` is still an `Effect`: it suspends the walk so
|
|
11
11
|
// it runs lazily, and folds the one failure the walk can report into the typed
|
|
12
|
-
// error channel
|
|
13
|
-
// or per attribute. A 500-row report is thousands of elements, and
|
|
14
|
-
//
|
|
15
|
-
//
|
|
16
|
-
//
|
|
12
|
+
// error channel, but inside a document there is no effect boundary per element
|
|
13
|
+
// or per attribute. A 500-row report is thousands of elements, and one fiber
|
|
14
|
+
// step per element dominated the `render 500 rows` benchmark. The typed failure
|
|
15
|
+
// survives: the walk throws an {@link XmlRenderError} and `renderXml` catches
|
|
16
|
+
// it into `Effect.fail`.
|
|
17
17
|
//
|
|
18
18
|
// Escaping is the part that scales with the size of the document rather than
|
|
19
19
|
// with its structure, and it is written out here rather than delegated, for a
|
|
20
20
|
// measured reason. The entity encoder that used to live beside this package
|
|
21
21
|
// escaped by applying five sequential global replacements, one per character,
|
|
22
22
|
// so a document with a single `&` in twenty thousand characters was scanned
|
|
23
|
-
// five times over to change one byte
|
|
24
|
-
// `bench/codec.bench.ts` measure. The table below covers
|
|
25
|
-
// characters that encoder escaped, and the explicit expectations
|
|
26
|
-
// `test/render.spec.ts` pin the fast path.
|
|
23
|
+
// five times over to change one byte. The `render 20k` rows in
|
|
24
|
+
// `bench/codec.bench.ts` measure that five-pass cost. The table below covers
|
|
25
|
+
// the same five characters that encoder escaped, and the explicit expectations
|
|
26
|
+
// in `test/render.spec.ts` pin the fast path.
|
|
27
27
|
|
|
28
28
|
import { Effect, Predicate, Result } from 'effect';
|
|
29
29
|
|
|
@@ -65,11 +65,12 @@ const TEXT_TABLE = buildTable({});
|
|
|
65
65
|
const ATTRIBUTE_TABLE = buildTable(ATTRIBUTE_WHITESPACE);
|
|
66
66
|
|
|
67
67
|
/**
|
|
68
|
-
* @description The characters each table escapes, as a pattern rather than as a set of replacement passes.
|
|
69
|
-
* clean text cheap: V8 compiles a single character class into a scan that is several times faster than a JavaScript loop
|
|
70
|
-
* code unit at a time, and clean text is most text. `render 20k of clean text` in `bench/codec.bench.ts`
|
|
71
|
-
* loop over the same twenty thousand characters is roughly two and a half times slower. Neither pattern is
|
|
72
|
-
* always starts at the beginning. One module-level instance of each is therefore safe to reuse, and nothing
|
|
68
|
+
* @description The characters each table escapes, as a pattern rather than as a set of replacement passes. A single pattern finds the first character that needs
|
|
69
|
+
* replacing, which keeps clean text cheap: V8 compiles a single character class into a scan that is several times faster than a JavaScript loop
|
|
70
|
+
* reading the same string a code unit at a time, and clean text is most text. The `render 20k of clean text` benchmark in `bench/codec.bench.ts`
|
|
71
|
+
* measures that difference: a hand-written loop over the same twenty thousand characters is roughly two and a half times slower. Neither pattern is
|
|
72
|
+
* global, so `exec` ignores `lastIndex` and always starts at the beginning. One module-level instance of each is therefore safe to reuse, and nothing
|
|
73
|
+
* has to be reset between calls.
|
|
73
74
|
*/
|
|
74
75
|
const TEXT_UNSAFE = /[<>&"']/;
|
|
75
76
|
const ATTRIBUTE_UNSAFE = /[<>&"'\n\r\t]/;
|
|
@@ -171,9 +172,9 @@ interface ResolvedOptions {
|
|
|
171
172
|
/**
|
|
172
173
|
* @description Builds the name resolver for one render. Every element and every attribute name goes through here, and a document repeats names: a thousand
|
|
173
174
|
* `<item>` elements, or the same `id` on every row. A validator that runs a regex per occurrence pays that cost a thousand times for one answer, so
|
|
174
|
-
* the first result is remembered and the rest are lookups.
|
|
175
|
-
*
|
|
176
|
-
*
|
|
175
|
+
* the first result is remembered and the rest are lookups. Keeping the mode and version in one place also stops a caller from resolving a name with
|
|
176
|
+
* different settings than the render it is part of. A cache miss calls {@link resolveNameSync}, which throws an {@link XmlParseError} in `'error'`
|
|
177
|
+
* mode; {@link renderXml} catches it and reports it as an {@link XmlRenderError}.
|
|
177
178
|
*
|
|
178
179
|
* @param options - Resolved render options.
|
|
179
180
|
*
|
|
@@ -193,8 +194,7 @@ const makeNamer = (options: Omit<ResolvedOptions, 'namer' | 'lineAt'>): ((name:
|
|
|
193
194
|
/**
|
|
194
195
|
* @description A boolean option's value, with an absent one read as the default. The three boolean options are spelled through here rather than through a `??` of
|
|
195
196
|
* their own, so the table below reads as a list of what each option _is_ instead of a list of nine separate decisions about what an omitted option
|
|
196
|
-
* means
|
|
197
|
-
* read.
|
|
197
|
+
* means, and a reader looking for "which options are on by default" finds three words rather than three mixes of `?? true` and `?? false` to read.
|
|
198
198
|
*
|
|
199
199
|
* @param value - The option as the caller wrote it, or `undefined` when the caller left it out.
|
|
200
200
|
* @param fallback - The value to use when the caller left it out.
|
|
@@ -265,10 +265,10 @@ export const escapeAttribute = (value: string): string => escape(value, ATTRIBUT
|
|
|
265
265
|
|
|
266
266
|
/**
|
|
267
267
|
* @description Replaces every character the table has an entry for, in one pass over the string. The pattern finds the first character that needs replacing, and a
|
|
268
|
-
* string with none is handed straight back
|
|
269
|
-
*
|
|
270
|
-
*
|
|
271
|
-
*
|
|
268
|
+
* string with none is handed straight back. That is the common case, and the pattern exists to keep it fast. From there the rest of the string is
|
|
269
|
+
* copied in runs between the replacements rather than a character at a time, so the cost is one pattern scan, one copy, and one concatenation per
|
|
270
|
+
* replacement, rather than a whole pass per character class. Only ASCII is looked up. XML carries every other character natively, and a code unit
|
|
271
|
+
* above 127 has no entity an XML parser is required to know.
|
|
272
272
|
*
|
|
273
273
|
* @param value - The text to escape.
|
|
274
274
|
* @param pattern - Matches the first character that needs replacing.
|
|
@@ -302,7 +302,7 @@ const escape = (value: string, pattern: RegExp, table: ReadonlyArray<string | un
|
|
|
302
302
|
* A record becomes an element:
|
|
303
303
|
*
|
|
304
304
|
* - `@`-prefixed keys become attributes, the reserved `#text` key becomes character data, and every other key becomes a child element.
|
|
305
|
-
* - An array repeats its name
|
|
305
|
+
* - An array repeats its name. A document whose root value is an array wraps it in the root element and names each member `itemName`.
|
|
306
306
|
* - A string is character data. The walk is synchronous, and what can go wrong is reported by throwing an {@link XmlRenderError}; {@link renderXml}
|
|
307
307
|
* folds that into the effect's typed error channel. A caller not already in an `Effect` runs it with `Effect.runSync`, which throws the failure it
|
|
308
308
|
* produced.
|
|
@@ -379,9 +379,9 @@ const render = (value: XmlValue, options: XmlRenderOptions): string => {
|
|
|
379
379
|
};
|
|
380
380
|
|
|
381
381
|
/**
|
|
382
|
-
* @description Renders one named element and its subtree. The value an {@link XmlValue} holds decides which of the four shapes below it takes
|
|
383
|
-
* children, character data, an absent field, or a record
|
|
384
|
-
*
|
|
382
|
+
* @description Renders one named element and its subtree. The value an {@link XmlValue} holds decides which of the four shapes below it takes: a repeated run of
|
|
383
|
+
* children, character data, an absent field, or a record. Each of those is written by a function of its own, so this one is the dispatch rather than
|
|
384
|
+
* the document.
|
|
385
385
|
*
|
|
386
386
|
* @param out - The chunk buffer to append to.
|
|
387
387
|
* @param name - The element name, not yet resolved.
|
|
@@ -397,8 +397,8 @@ const renderElement = (out: Array<string>, name: string, value: XmlValue, depth:
|
|
|
397
397
|
return;
|
|
398
398
|
}
|
|
399
399
|
|
|
400
|
-
// An element opens its own line rather than having its caller do it,
|
|
401
|
-
//
|
|
400
|
+
// An element opens its own line rather than having its caller do it, so a
|
|
401
|
+
// repeated run of children stays on separate lines. The root is the
|
|
402
402
|
// one element that has nothing in front of it.
|
|
403
403
|
if (options.format && depth > 0) openLine(out, depth, options);
|
|
404
404
|
|
|
@@ -417,7 +417,7 @@ const renderElement = (out: Array<string>, name: string, value: XmlValue, depth:
|
|
|
417
417
|
|
|
418
418
|
/**
|
|
419
419
|
* @description Refuses to walk deeper than the render allows. A value can nest without end, and every one of those levels costs a stack frame here, so the cap is
|
|
420
|
-
* checked
|
|
420
|
+
* checked as the walk descends rather than trusted to the caller.
|
|
421
421
|
*
|
|
422
422
|
* @param depth - The depth about to be written.
|
|
423
423
|
* @param options - Resolved render options.
|
|
@@ -459,8 +459,8 @@ const renderRepeated = (out: Array<string>, name: string, members: ReadonlyArray
|
|
|
459
459
|
|
|
460
460
|
/**
|
|
461
461
|
* @description Renders an element whose value is character data, or nothing. An empty string is character data that happens to be empty, and an element holding
|
|
462
|
-
* none of it is the same element as one holding nothing at all
|
|
463
|
-
* that never went through the schema
|
|
462
|
+
* none of it is the same element as one holding nothing at all, as is an `undefined` element, which is an absent one. The renderer is handed values
|
|
463
|
+
* that never went through the schema (a caller building a document by hand), so the absent case is reachable, and an empty element is the correct
|
|
464
464
|
* rendering of both.
|
|
465
465
|
*
|
|
466
466
|
* @param out - The chunk buffer to append to.
|
|
@@ -538,9 +538,9 @@ interface Fields {
|
|
|
538
538
|
* @description One pass over a record's keys, collecting all three roles at once: the attributes are rendered as they are found, the child names are set aside for
|
|
539
539
|
* the pass that writes them, and the text key is left to {@link textOf}. A pass for the attributes, a pass for the children and an index for the text
|
|
540
540
|
* instead walks the keys three times and allocates the key array twice, which on a document of a few thousand elements is thousands of allocations
|
|
541
|
-
* for nothing. Sorting is off by default, and the default path is the one
|
|
541
|
+
* for nothing. Sorting is off by default, and the default path is the important one, so the attributes are built as they are found and there is
|
|
542
542
|
* nothing to sort. When it is on, the attribute keys are collected instead and rendered afterwards in sorted order, which costs an array per element
|
|
543
|
-
* and
|
|
543
|
+
* and gives output independent of the order the fields happened to be declared in.
|
|
544
544
|
*
|
|
545
545
|
* @param record - The element's value.
|
|
546
546
|
* @param options - Resolved render options.
|
|
@@ -557,10 +557,10 @@ const collectFields = (record: XmlRecord, options: ResolvedOptions): Fields => {
|
|
|
557
557
|
const key = keys[i] as string;
|
|
558
558
|
const child = record[key];
|
|
559
559
|
|
|
560
|
-
// An absent field is not written at all,
|
|
561
|
-
//
|
|
562
|
-
//
|
|
563
|
-
//
|
|
560
|
+
// An absent field is not written at all, so an unset optional attribute
|
|
561
|
+
// stays out of the document rather than appearing as `a=""`, and an absent
|
|
562
|
+
// child stays out rather than appearing as `<a/>`. The text key is read by
|
|
563
|
+
// `textOf` either way, so skipping it here costs nothing.
|
|
564
564
|
if (child === undefined) continue;
|
|
565
565
|
|
|
566
566
|
if (isAttributeKey(key)) {
|
|
@@ -647,9 +647,9 @@ const openLine = (out: Array<string>, depth: number, options: ResolvedOptions):
|
|
|
647
647
|
|
|
648
648
|
/**
|
|
649
649
|
* @description Writes an element with no content, in whichever of the two forms the options ask for. Every path that produces an element with nothing in it goes
|
|
650
|
-
* through here, so the self-closing decision is made in exactly one place. That
|
|
651
|
-
* empty string, an absent value, an empty array, and a record whose fields are all absent
|
|
652
|
-
*
|
|
650
|
+
* through here, so the self-closing decision is made in exactly one place. That is important because "nothing in it" arrives four different ways: an
|
|
651
|
+
* empty string, an absent value, an empty array, and a record whose fields are all absent. Four separate decisions are four chances for one of them
|
|
652
|
+
* to write the long form by accident.
|
|
653
653
|
*
|
|
654
654
|
* @param out - The chunk buffer to append to.
|
|
655
655
|
* @param tag - The element's name, already resolved.
|
|
@@ -690,8 +690,8 @@ const attributeText = (value: XmlValue): string => {
|
|
|
690
690
|
};
|
|
691
691
|
|
|
692
692
|
/**
|
|
693
|
-
* @description Renders a leaf that is not a string as the character data an XML document can hold. A schema-derived value never reaches here
|
|
694
|
-
* `Schema.toCodecStringTree` has already turned every scalar into a string
|
|
693
|
+
* @description Renders a leaf that is not a string as the character data an XML document can hold. A schema-derived value never reaches here, because
|
|
694
|
+
* `Schema.toCodecStringTree` has already turned every scalar into a string, so this is for values a caller built by hand. A value with no sensible
|
|
695
695
|
* text form is rendered as nothing rather than as `[object Object]`, which would silently write a document that parses back to something else.
|
|
696
696
|
*
|
|
697
697
|
* @param value - The leaf to render.
|