@endevops/effect-codec-xml 0.0.1 → 0.1.0-beta.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (50) hide show
  1. package/README.md +81 -62
  2. package/dist/codec.d.ts +17 -9
  3. package/dist/codec.d.ts.map +1 -1
  4. package/dist/codec.js +29 -18
  5. package/dist/codec.js.map +1 -1
  6. package/dist/conventions.d.ts +4 -4
  7. package/dist/conventions.js +7 -7
  8. package/dist/conventions.js.map +1 -1
  9. package/dist/entities/entity-decoder.d.ts +33 -33
  10. package/dist/entities/entity-decoder.d.ts.map +1 -1
  11. package/dist/entities/entity-decoder.js +63 -64
  12. package/dist/entities/entity-decoder.js.map +1 -1
  13. package/dist/errors.d.ts +2 -2
  14. package/dist/errors.js +2 -2
  15. package/dist/errors.js.map +1 -1
  16. package/dist/namespaces.js +40 -13
  17. package/dist/namespaces.js.map +1 -1
  18. package/dist/naming.d.ts +6 -6
  19. package/dist/naming.d.ts.map +1 -1
  20. package/dist/naming.js +3 -3
  21. package/dist/naming.js.map +1 -1
  22. package/dist/parse.d.ts +5 -5
  23. package/dist/parse.js +14 -14
  24. package/dist/parse.js.map +1 -1
  25. package/dist/plain-value.js +242 -0
  26. package/dist/plain-value.js.map +1 -0
  27. package/dist/render.d.ts +1 -1
  28. package/dist/render.d.ts.map +1 -1
  29. package/dist/render.js +28 -28
  30. package/dist/render.js.map +1 -1
  31. package/dist/xml-error.d.ts +9 -9
  32. package/dist/xml-error.js +18 -18
  33. package/dist/xml-error.js.map +1 -1
  34. package/dist/xml-value.d.ts +7 -7
  35. package/dist/xml-value.d.ts.map +1 -1
  36. package/dist/xml-value.js +6 -7
  37. package/dist/xml-value.js.map +1 -1
  38. package/package.json +1 -1
  39. package/src/codec.ts +98 -71
  40. package/src/conventions.ts +7 -7
  41. package/src/entities/entity-decoder.ts +98 -99
  42. package/src/errors.ts +3 -3
  43. package/src/index.ts +3 -3
  44. package/src/namespaces.ts +62 -36
  45. package/src/naming.ts +34 -35
  46. package/src/parse.ts +26 -26
  47. package/src/plain-value.ts +312 -0
  48. package/src/render.ts +44 -44
  49. package/src/xml-error.ts +18 -18
  50. package/src/xml-value.ts +10 -11
package/src/namespaces.ts CHANGED
@@ -19,21 +19,21 @@
19
19
  // - `xmlValue` marks one field as the element's character data, the `#text`
20
20
  // value, for an element that also carries attributes or children.
21
21
  //
22
- // The namespace of an element is inherited by its descendants, the way an XML
23
- // default namespace is. An attribute never inherits: it is in a namespace only
24
- // when it is annotated with one explicitly, because a default namespace does
25
- // not apply to attributes.
22
+ // An element's namespace is inherited by its descendants, as an XML default
23
+ // namespace is. An attribute never inherits: it is in a namespace only when it
24
+ // is annotated with one explicitly, because a default namespace does not apply
25
+ // to attributes.
26
26
  //
27
- // On the way out, each element writes its own declaration when the prefix or
28
- // default is not already in scope. On the way in, the parser's own declarations
29
- // are read into scope and every name is resolved to its URI, so a document that
30
- // binds the same URI to a different prefix still decodes to the same value. The
31
- // declaration attributes are dropped from the decoded value; they are the
32
- // codec's to manage, not the schema's.
27
+ // On encode, each element writes its own declaration when the prefix or default
28
+ // is not already in scope. On decode, the parser's own declarations are read
29
+ // into scope and every name is resolved to its URI, so a document that binds the
30
+ // same URI to a different prefix still decodes to the same value. The
31
+ // declaration attributes are dropped from the decoded value; the codec manages
32
+ // them, not the schema.
33
33
  //
34
- // The plan is built per local name, which is what a schema field is. One local
35
- // name cannot belong to two namespaces in one codec; that is reported when the
36
- // codec is built rather than guessed at.
34
+ // The plan is built per local name, and a schema field is one local name. One
35
+ // local name cannot belong to two namespaces in one codec; the plan reports that
36
+ // when the codec is built rather than guessing.
37
37
 
38
38
  import type { Schema } from 'effect';
39
39
 
@@ -396,7 +396,7 @@ interface Scan {
396
396
  readonly problems: Array<string>;
397
397
  /**
398
398
  * @description The AST nodes on the current scan path. A recursive schema terminates because the `Suspend` node is still on the path when its thunk is reached,
399
- * and a schema reused under two sibling paths is scanned once per path because each node is removed again on the way out.
399
+ * and a schema reused under two sibling paths is scanned once per path because each node is removed again on the way back out.
400
400
  */
401
401
  readonly seen: Set<SchemaAST.AST>;
402
402
  }
@@ -634,7 +634,7 @@ const scanNode = (scan: Scan, ast: SchemaAST.AST, inherited: XmlNamespace | unde
634
634
  };
635
635
 
636
636
  /**
637
- * @description Collects the namespace and name of every field in a schema. A namespace is inherited by descendant elements, the way a default namespace is, and an
637
+ * @description Collects the namespace and name of every field in a schema. Descendant elements inherit a namespace, as they inherit a default namespace, and an
638
638
  * element field records its own namespace, so encode and decode can find it by the local name alone.
639
639
  *
640
640
  * @param schema - The schema to walk.
@@ -900,7 +900,7 @@ const schemaKey = (plan: NamespacePlan, parent: string, key: string, scope: Reco
900
900
  /**
901
901
  * @description The scope a child element resolves its own name against: the declarations it carries on itself, layered over the parent scope. An element may
902
902
  * declare the prefix it uses on the element itself, so its own name is read with those bindings in scope. A repeated element arrives as an array, so
903
- * the first member stands in for the run — every member describes the same element and carries the same declaration.
903
+ * the first member stands in for the run; every member describes the same element and carries the same declaration.
904
904
  *
905
905
  * @param child - The child value.
906
906
  * @param scope - The bindings in scope above the child.
@@ -915,31 +915,32 @@ const childScopeOf = (child: XmlValue, scope: Record<string, string | undefined>
915
915
  };
916
916
 
917
917
  /**
918
- * @description Rewrites a wire value tree back to the schema's local names, resolving every name against the declarations the document carries and dropping those
919
- * declarations. Character data maps to the value field of the element it belongs to. A record left holding only character data collapses back to that
920
- * string, which is how a namespaced leaf stays a `Schema.String`.
918
+ * @description Character data read back as a value tree. An element the schema reads as a struct with a value field derives that field from the element's
919
+ * character data, and the parser reduces an element with no attributes and no children to a bare string, so the string is put back under the value's
920
+ * key for that struct to read.
921
921
  *
922
- * @param value - The parsed wire tree.
922
+ * @param value - The character data the parser produced.
923
+ * @param plan - The namespace plan.
924
+ * @param elementPath - The path of the element the character data belongs to.
925
+ *
926
+ * @returns The character data, under the element's value key when it has one.
927
+ */
928
+ const decodeText = (value: string, plan: NamespacePlan, elementPath: string): XmlValue => {
929
+ const valueKey = plan.valueByElement.get(elementPath);
930
+ return valueKey !== undefined ? { [valueKey]: value } : value;
931
+ };
932
+
933
+ /**
934
+ * @description One record read back: its declarations dropped, its keys resolved to the schema's names, and its character data placed under the value field.
935
+ *
936
+ * @param record - The record to read.
923
937
  * @param plan - The namespace plan.
924
938
  * @param scope - The prefix bindings in scope above this element.
925
939
  * @param elementPath - The path of this element, or the root sentinel.
926
940
  *
927
- * @returns The value tree, keyed by the schema's names.
941
+ * @returns The record, keyed by the schema's names, or the string it collapses to.
928
942
  */
929
- export const decodeNames = (value: XmlValue, plan: NamespacePlan, scope: Record<string, string | undefined>, elementPath: string): XmlValue => {
930
- if (Array.isArray(value)) {
931
- return value.map(member => decodeNames(member, plan, scope, elementPath));
932
- }
933
- if (Predicate.isString(value) || Predicate.isUndefined(value)) {
934
- return value;
935
- }
936
- if (!Predicate.isObject(value)) {
937
- return value;
938
- }
939
-
940
- // `Predicate.isObject` narrows to a generic index signature, so the value
941
- // tree's own record type is named here.
942
- const record = value as XmlRecord;
943
+ const decodeRecord = (record: XmlRecord, plan: NamespacePlan, scope: Record<string, string | undefined>, elementPath: string): XmlValue => {
943
944
  const inner = scopeOf(record, scope);
944
945
  const valueKey = plan.valueByElement.get(elementPath);
945
946
  const out: Record<string, XmlValue> = {};
@@ -950,7 +951,7 @@ export const decodeNames = (value: XmlValue, plan: NamespacePlan, scope: Record<
950
951
  }
951
952
 
952
953
  if (key === TEXT_KEY) {
953
- out[valueKey ?? TEXT_KEY] = decodeNames(child, plan, inner, elementPath);
954
+ out[valueKey ?? TEXT_KEY] = child;
954
955
  continue;
955
956
  }
956
957
 
@@ -966,3 +967,28 @@ export const decodeNames = (value: XmlValue, plan: NamespacePlan, scope: Record<
966
967
  }
967
968
  return out;
968
969
  };
970
+
971
+ /**
972
+ * @description Rewrites a wire value tree back to the schema's local names, resolving every name against the declarations the document carries and dropping those
973
+ * declarations. Character data maps to the value field of the element it belongs to. A record left holding only character data collapses back to that
974
+ * string, which is how a namespaced leaf stays a `Schema.String`.
975
+ *
976
+ * @param value - The parsed wire tree.
977
+ * @param plan - The namespace plan.
978
+ * @param scope - The prefix bindings in scope above this element.
979
+ * @param elementPath - The path of this element, or the root sentinel.
980
+ *
981
+ * @returns The value tree, keyed by the schema's names.
982
+ */
983
+ export const decodeNames = (value: XmlValue, plan: NamespacePlan, scope: Record<string, string | undefined>, elementPath: string): XmlValue => {
984
+ if (Array.isArray(value)) {
985
+ return value.map(member => decodeNames(member, plan, scope, elementPath));
986
+ }
987
+ if (Predicate.isString(value)) {
988
+ return decodeText(value, plan, elementPath);
989
+ }
990
+ if (!Predicate.isObject(value)) {
991
+ return value;
992
+ }
993
+ return decodeRecord(value as XmlRecord, plan, scope, elementPath);
994
+ };
package/src/naming.ts CHANGED
@@ -8,10 +8,10 @@
8
8
  //
9
9
  // The five predicates and `sanitize` are plain synchronous functions: a regex
10
10
  // test cannot fail and a character substitution has nothing to fail about, so
11
- // there is no effect to model. `validate` does have one failure to report — an
11
+ // there is no effect to model. `validate` does have one failure to report: an
12
12
  // unknown production, unreachable from TypeScript where `Production` is a
13
13
  // closed union but reachable for an untyped JavaScript caller, or a value that
14
- // crossed a boundary as `unknown` — so it answers with an `Effect` whose error
14
+ // crossed a boundary as `unknown`. It answers with an `Effect` whose error
15
15
  // channel is that {@link XmlError}. The value it produces is still a plain
16
16
  // result.
17
17
 
@@ -20,7 +20,7 @@ import { Effect } from 'effect';
20
20
  import { XmlError } from '#/xml-error.ts';
21
21
 
22
22
  /**
23
- * @description The XML specification version a production is validated against. The two differ only in their non-ASCII character ranges — see {@link getRegexes}.
23
+ * @description The XML specification version a production is validated against. The two differ only in their non-ASCII character ranges. See {@link getRegexes}.
24
24
  */
25
25
  export type XmlVersion = '1.0' | '1.1';
26
26
 
@@ -42,7 +42,7 @@ export interface ValidationOptions {
42
42
  /**
43
43
  * @description Restrict matching to the ASCII subset of the NameStartChar/NameChar productions and skip unicode-aware regex matching entirely. Faster,
44
44
  * especially for XML 1.1 (which otherwise requires the `/u` regex flag), but rejects legitimate non-ASCII XML names. Off by default for backward
45
- * compatibility — opt in only when inputs are known to be ASCII. Defaults to false.
45
+ * compatibility. Opt in only when inputs are known to be ASCII. Defaults to false.
46
46
  */
47
47
  asciiOnly?: boolean;
48
48
  }
@@ -60,7 +60,7 @@ export interface SanitizeOptions {
60
60
  */
61
61
  asciiOnly?: boolean;
62
62
  /**
63
- * @description Accepted and ignored. Sanitizing is not version-dependent — the character set it considers illegal is the union of both versions — but the option
63
+ * @description Accepted and ignored. Sanitizing is not version-dependent: the character set it considers illegal is the union of both versions. But the option
64
64
  * is part of the published signature, so dropping it would break callers that pass it through a shared options object.
65
65
  */
66
66
  xmlVersion?: XmlVersion;
@@ -94,7 +94,7 @@ const PRODUCTIONS = ['name', 'ncName', 'qName', 'nmToken', 'nmTokens'] as const
94
94
  type ProductionRegexes = Record<Production, RegExp>;
95
95
 
96
96
  // ---------------------------------------------------------------------------
97
- // Character class strings — XML 1.0
97
+ // Character class strings: XML 1.0
98
98
  //
99
99
  // NameStartChar ::= ":" | [A-Z] | "_" | [a-z]
100
100
  // | [#xC0-#xD6] | [#xD8-#xF6] | [#xF8-#x2FF]
@@ -127,7 +127,7 @@ const nameStartChar10 =
127
127
  const nameChar10 = nameStartChar10 + '\\-\\.\\d' + '\u00B7' + '\u0300-\u036F' + '\u203F-\u2040';
128
128
 
129
129
  // ---------------------------------------------------------------------------
130
- // Character class strings — XML 1.1
130
+ // Character class strings: XML 1.1
131
131
  //
132
132
  // Differences from XML 1.0:
133
133
  //
@@ -138,7 +138,7 @@ const nameChar10 = nameStartChar10 + '\\-\\.\\d' + '\u00B7' + '\u0300-\u036F' +
138
138
  //
139
139
  // 1.0 tops out at \uFFFD (BMP only)
140
140
  // 1.1 adds \u{10000}-\u{EFFFF} (supplementary planes)
141
- // These require the /u flag on the RegExp — see buildRegexes below.
141
+ // These require the /u flag on the RegExp. See buildRegexes below.
142
142
  //
143
143
  // NameChar:
144
144
  // 1.1 adds \u0487 (Combining Cyrillic Millions Sign, added in Unicode 4.0)
@@ -146,7 +146,7 @@ const nameChar10 = nameStartChar10 + '\\-\\.\\d' + '\u00B7' + '\u0300-\u036F' +
146
146
 
147
147
  const nameStartChar11 =
148
148
  ':A-Za-z_' +
149
- '\u00C0-\u02FF' + // merged — 1.0 had three split ranges here
149
+ '\u00C0-\u02FF' + // merged: 1.0 had three split ranges here
150
150
  '\u0370-\u037D' +
151
151
  '\u037F-\u0486\u0488-\u1FFF' + // split to exclude \u0487 (combining mark, never a NameStartChar)
152
152
  '\u200C-\u200D' +
@@ -155,21 +155,21 @@ const nameStartChar11 =
155
155
  '\u3001-\uD7FF' +
156
156
  '\uF900-\uFDCF' +
157
157
  '\uFDF0-\uFFFD' +
158
- '\u{10000}-\u{EFFFF}'; // supplementary planes — REQUIRES /u flag on RegExp
158
+ '\u{10000}-\u{EFFFF}'; // supplementary planes: REQUIRES /u flag on RegExp
159
159
 
160
160
  const nameChar11 =
161
161
  nameStartChar11 +
162
162
  '\\-\\.\\d' +
163
163
  '\u00B7' +
164
164
  '\u0300-\u036F' +
165
- '\u0487' + // Combining Cyrillic Millions Sign — valid in 1.1, not 1.0
165
+ '\u0487' + // Combining Cyrillic Millions Sign: valid in 1.1, not 1.0
166
166
  '\u203F-\u2040';
167
167
 
168
168
  // ---------------------------------------------------------------------------
169
169
  // Regex builders
170
170
  //
171
- // XML 1.0 regexes: no flags — BMP only, standard JS regex behaviour.
172
- // XML 1.1 regexes: /u flag — required for \u{10000}-\u{EFFFF} to match actual
171
+ // XML 1.0 regexes: no flags, BMP only, standard JS regex behaviour.
172
+ // XML 1.1 regexes: /u flag, required for \u{10000}-\u{EFFFF} to match actual
173
173
  // supplementary code points rather than lone surrogates (which are illegal XML).
174
174
  // ---------------------------------------------------------------------------
175
175
 
@@ -196,34 +196,33 @@ const buildRegexes = (startChar: string, char: string, flags = ''): ProductionRe
196
196
  };
197
197
  };
198
198
 
199
- const regexes10 = buildRegexes(nameStartChar10, nameChar10); // no /u — BMP only
200
- const regexes11 = buildRegexes(nameStartChar11, nameChar11, 'u'); // /u — enables \u{10000}-\u{EFFFF}
199
+ const regexes10 = buildRegexes(nameStartChar10, nameChar10); // no /u, BMP only
200
+ const regexes11 = buildRegexes(nameStartChar11, nameChar11, 'u'); // /u enables \u{10000}-\u{EFFFF}
201
201
 
202
202
  // ---------------------------------------------------------------------------
203
203
  // ASCII-only fast path (opt-in, off by default)
204
204
  //
205
- // The XML 1.0 vs 1.1 NameStartChar/NameChar productions differ *only* in
206
- // their non-ASCII ranges (merged vs split Latin-1 ranges, \u0487, and
207
- // supplementary planes). Restricted to ASCII, both versions collapse to the
208
- // same character classes, so a single regex pair covers both xmlVersion
209
- // values — no /u flag needed.
205
+ // The XML 1.0 and 1.1 NameStartChar/NameChar productions differ only in their
206
+ // non-ASCII ranges: merged vs split Latin-1 ranges, \u0487, and supplementary
207
+ // planes. Restricted to ASCII, both versions collapse to the same character
208
+ // classes, so one regex pair covers both xmlVersion values and no /u flag is
209
+ // needed.
210
210
  //
211
- // Rationale: unicode-aware regexes (the /u flag, required for XML 1.1's
212
- // supplementary-plane range) are measurably slower in V8 than plain
213
- // non-unicode regexes on the same input, even when the input is pure ASCII.
214
- // For the common case — HTML/SVG ids, XML tags — names are ASCII, so callers
215
- // who know this can opt in to skip the unicode-aware matching path entirely.
216
- // This is a real but *conditional* win: mainly for XML 1.1 input (avoids /u),
217
- // or at scale where the larger unicode character classes add engine
218
- // overhead. It also changes behaviour (rejects legitimate non-ASCII XML
219
- // 1.0/1.1 names), so it must never be silently enabled — hence off by
220
- // default.
211
+ // Unicode-aware regexes (the /u flag, required for XML 1.1's supplementary-
212
+ // plane range) are measurably slower in V8 than plain non-unicode regexes on
213
+ // the same input, even when the input is pure ASCII. For the common case
214
+ // (HTML/SVG ids, XML tags) names are ASCII, so callers who know this can opt
215
+ // in to skip the unicode-aware matching path entirely. The win is conditional:
216
+ // it applies mainly to XML 1.1 input (avoids /u), or at scale where the larger
217
+ // unicode character classes add engine overhead. The option also changes
218
+ // behaviour (it rejects legitimate non-ASCII XML 1.0/1.1 names), so it must
219
+ // never be enabled silently. That is why the default is off.
221
220
  // ---------------------------------------------------------------------------
222
221
 
223
222
  const nameStartCharAscii = ':A-Za-z_';
224
223
  const nameCharAscii = nameStartCharAscii + '\\-\\.\\d';
225
224
 
226
- const regexesAscii = buildRegexes(nameStartCharAscii, nameCharAscii); // no /u — ASCII only
225
+ const regexesAscii = buildRegexes(nameStartCharAscii, nameCharAscii); // no /u, ASCII only
227
226
 
228
227
  /**
229
228
  * @description The compiled regex set for a version/ASCII combination. Only three sets are ever built, at module load; this is a lookup, not a compile.
@@ -242,7 +241,7 @@ const getRegexes = (xmlVersion: XmlVersion = '1.0', asciiOnly = false): Producti
242
241
  // Boolean validators
243
242
  //
244
243
  // One plain predicate per production. A regex test cannot fail, and every one of these is called per name
245
- // inside a parser's or codec's hot loop, so a boolean is the honest answer and there is no effect to
244
+ // inside a parser's or codec's hot loop, so a boolean is the direct answer and there is no effect to
246
245
  // allocate or run.
247
246
  // ---------------------------------------------------------------------------
248
247
 
@@ -296,7 +295,7 @@ export const isNmToken = (str: string, { xmlVersion = '1.0', asciiOnly = false }
296
295
  getRegexes(xmlVersion, asciiOnly).nmToken.test(str);
297
296
 
298
297
  /**
299
- * @description Whether the string is a valid NMTokens value — a whitespace-separated list of NMToken values. Used for: DTD NMTOKENS attribute values.
298
+ * @description Whether the string is a valid NMTokens value: a whitespace-separated list of NMToken values. Used for: DTD NMTOKENS attribute values.
300
299
  *
301
300
  * @param str - The candidate list.
302
301
  * @param opts - `asciiOnly` skips unicode-aware matching, ASCII names only (default false).
@@ -455,7 +454,7 @@ const diagnoseWith = (str: string, production: Production, isValid: boolean, asc
455
454
  * @example
456
455
  * ```typescript
457
456
  * import { Effect } from 'effect';
458
- * import { validate } from '@endevops/effect-xml-codec';
457
+ * import { validate } from '@endevops/effect-codec-xml';
459
458
  *
460
459
  * Effect.runSync(validate('not a name', 'ncName'));
461
460
  * // { valid: false, production: 'ncName', input: 'not a name', reason: 'First character " " is not a valid NameStartChar', position: 0 }
@@ -466,7 +465,7 @@ const diagnoseWith = (str: string, production: Production, isValid: boolean, asc
466
465
  * @param opts - Version and ASCII-only selection, as for the boolean validators.
467
466
  *
468
467
  * @returns An effect producing a discriminated result: the plain triple when valid, or the offending `reason` and `position` when not. A name that
469
- * fails to validate is a `valid: false` result, not a failure — an invalid name is the question being answered. The effect fails only with an
468
+ * fails to validate is a `valid: false` result, not a failure: an invalid name is the question being answered. The effect fails only with an
470
469
  * {@link XmlError} and the `InvalidProduction` reason for an unknown production, which is unreachable from TypeScript and is the guard for untyped
471
470
  * JavaScript callers.
472
471
  */
package/src/parse.ts CHANGED
@@ -34,7 +34,7 @@ export interface XmlParseOptions {
34
34
  * @description Keep the whitespace at the edges of every text run.
35
35
  *
36
36
  * @default false\
37
- * which trims it — and trimming is what makes a pretty-printed document
37
+ * which trims it. Trimming makes a pretty-printed document
38
38
  * read as the same value as an unindented one, because the indentation around a child element and around a closing tag lands at the edges of its
39
39
  * parent's text. Whitespace _inside_ a run is content and is never touched either way, so `'one two'` and a paragraph with a newline in the middle
40
40
  * of it survive. Set it to `true` to keep leading and trailing spaces in text exactly as written, at the cost of a document that was laid out on
@@ -68,13 +68,13 @@ export interface XmlParseOptions {
68
68
  /**
69
69
  * @description Parses an XML document into its root element's content.\
70
70
  * The walk itself is synchronous, but it reports a malformed document by failing with an {@link XmlParseError} rather than by throwing, so the failure lands in the effect's error channel where `catchTag`, `retry` and a fallback can all
71
- * see it. A failed parse is an expected outcome of reading untrusted text — it is what those combinators key off — and only a defect would hide it.
71
+ * see it. A failed parse is an expected outcome of reading untrusted text, and those combinators key off it, so only a defect would hide it.
72
72
  * The span is the boundary a performance trace hangs off: it carries the document's length, which is the size that drives the parser's cost, so a
73
73
  * slow parse in a profile can be attributed to the input that produced it. A caller that wants the value outside an `Effect` uses
74
74
  * {@link parseXmlDocument}, which runs the same walk synchronously and throws instead. The walk is plain recursive descent rather than a chain of
75
- * `yield*`es. Publicly `parseXml` is still an `Effect` — it suspends the walk so it runs lazily under the span, and folds the failure the walk throws
76
- * into the typed error channel — but inside a document there is no effect boundary per tag, attribute or text run. A 500-row report is thousands of
77
- * those, and a fiber step for each of them was most of what the `parse 500 rows` row measured. The typed failure survives: the walk throws an
75
+ * `yield*`es. Publicly `parseXml` is still an `Effect`: it suspends the walk so it runs lazily under the span, and folds the failure the walk throws
76
+ * into the typed error channel, but inside a document there is no effect boundary per tag, attribute or text run. A 500-row report is thousands of
77
+ * those, and one fiber step per construct dominated the `parse 500 rows` benchmark. The typed failure survives: the walk throws an
78
78
  * {@link XmlParseError} and `parseXml` catches it into `Effect.fail`.
79
79
  *
80
80
  * @param text - The document to read.
@@ -140,8 +140,8 @@ const SLASH = 47;
140
140
  const EQUALS = 61;
141
141
 
142
142
  /**
143
- * @description One element as the parser saw it: the name it was written under, and the value it holds. Carrying the name alongside the value is what lets the
144
- * parent file it correctly — the value alone cannot say, because a text-only element reduces to a bare string.
143
+ * @description One element as the parser saw it: the name it was written under, and the value it holds. Carrying the name alongside the value lets the parent file
144
+ * it correctly, because the value alone cannot say: a text-only element reduces to a bare string.
145
145
  */
146
146
  interface Element {
147
147
  readonly name: string;
@@ -179,9 +179,9 @@ interface Content {
179
179
  }
180
180
 
181
181
  /**
182
- * @description What sits at the cursor inside an element's body. Naming what is there before deciding what to do with it is what lets the content loop stay a
183
- * dispatch: each construct is recognised in one place, against the ones that cannot be confused with it, rather than by a chain of `startsWith`
184
- * guesses where each had to remember what the last had already ruled out.
182
+ * @description What sits at the cursor inside an element's body. Naming what is there before deciding what to do with it lets the content loop stay a dispatch:
183
+ * each construct is recognised in one place, against the ones that cannot be confused with it, rather than by a chain of `startsWith` guesses where
184
+ * each had to remember what the last had already ruled out.
185
185
  */
186
186
  type Construct = 'text' | 'close' | 'comment' | 'cdata' | 'instruction' | 'child';
187
187
 
@@ -213,7 +213,7 @@ const parseDocument = (text: string, options: XmlParseOptions): XmlDocument => {
213
213
  const nameOptions = { mode: resolved.name, xmlVersion: resolved.xmlVersion };
214
214
 
215
215
  /**
216
- * @description Names already resolved by this parse. A document repeats names — every one of five hundred rows has a `sku` — and a validator that ran per
216
+ * @description Names already resolved by this parse. A document repeats names (every one of five hundred rows has a `sku`), and a validator that ran per
217
217
  * occurrence would pay for the same answer five hundred times.
218
218
  */
219
219
  const nameCache = new Map<string, string>();
@@ -258,7 +258,7 @@ const parseDocument = (text: string, options: XmlParseOptions): XmlDocument => {
258
258
  };
259
259
 
260
260
  /**
261
- * @description Consumes whitespace, comments, processing instructions and a DOCTYPE, leaving the cursor on the first character that is none of them — or at the
261
+ * @description Consumes whitespace, comments, processing instructions and a DOCTYPE, leaving the cursor on the first character that is none of them, or at the
262
262
  * end of the document.
263
263
  */
264
264
  const skipMisc = (): void => {
@@ -294,7 +294,7 @@ const parseDocument = (text: string, options: XmlParseOptions): XmlDocument => {
294
294
  const start = at;
295
295
  while (at < text.length) {
296
296
  const char = text.charCodeAt(at);
297
- // Whitespace, `/`, `=` and `>` all end a name. Stopping on `/` and `>` is what lets `<a/>` and `<a>` share one loop.
297
+ // Whitespace, `/`, `=` and `>` all end a name. Stopping on `/` and `>` lets `<a/>` and `<a>` share one loop.
298
298
  if (isWhitespace(char) || char === SLASH || char === EQUALS || char === GT) {
299
299
  break;
300
300
  }
@@ -322,7 +322,7 @@ const parseDocument = (text: string, options: XmlParseOptions): XmlDocument => {
322
322
  at++;
323
323
 
324
324
  const end = text.indexOf(quote ?? '', at);
325
- // A raw quote cannot appear inside a quoted value — it would have to be written `&quot;` — so the next quote of the same kind always closes it.
325
+ // A raw quote cannot appear inside a quoted value (it would have to be written `&quot;`), so the next quote of the same kind always closes it.
326
326
  if (end === -1) {
327
327
  throw new XmlParseError({ message: `Unterminated value for attribute "${name}"`, position: at, input: text });
328
328
  }
@@ -425,8 +425,8 @@ const parseDocument = (text: string, options: XmlParseOptions): XmlDocument => {
425
425
  /**
426
426
  * @description What the cursor is sitting on inside an element's body. The two things the loop cannot read are refused here rather than in it: running out of
427
427
  * document and a declaration, which is markup the parser does not accept inside an element. Recognising the constructs that _are_ read is the rest,
428
- * and the order is the one that rules out the shorter prefixes first — `</` before `<?` before any other `<!`, and `<![CDATA[` before the `<!` that
429
- * would otherwise match it.
428
+ * and the order rules out the shorter prefixes first: `</` before `<?` before any other `<!`, and `<![CDATA[` before the `<!` that would otherwise
429
+ * match it.
430
430
  *
431
431
  * @param name - The name the enclosing element's start tag gave it, for the unterminated-body message.
432
432
  *
@@ -458,7 +458,7 @@ const parseDocument = (text: string, options: XmlParseOptions): XmlDocument => {
458
458
  };
459
459
 
460
460
  /**
461
- * @description Consumes a `</name>`, checking on the way that it is the tag that closes this element and that it is well-formed.
461
+ * @description Consumes a `</name>`, checking as it goes that it is the tag that closes this element and that it is well-formed.
462
462
  *
463
463
  * @param name - The name the start tag gave the element, which the closing tag has to match.
464
464
  */
@@ -490,7 +490,7 @@ const parseDocument = (text: string, options: XmlParseOptions): XmlDocument => {
490
490
  };
491
491
 
492
492
  /**
493
- * @description Reads a `<![CDATA[…]]>` section. CDATA is character data, and character data is what it holds, so it joins the element's text as it stands — the
493
+ * @description Reads a `<![CDATA[…]]>` section. CDATA is character data, and character data is what it holds, so it joins the element's text as it stands. The
494
494
  * entities in it are literal text and must not be expanded.
495
495
  *
496
496
  * @returns The section's contents.
@@ -522,15 +522,15 @@ const parseDocument = (text: string, options: XmlParseOptions): XmlDocument => {
522
522
  */
523
523
  const finishElement = (record: Record<string, XmlValue>, hasAttributes: boolean, text: string, hasChildren: boolean): XmlValue => {
524
524
  // Whitespace at the edges of a text run is dropped unless the caller asked to
525
- // keep it. This is what makes a pretty-printed document round trip: the
525
+ // keep it. Trimming here makes a pretty-printed document round trip: the
526
526
  // indentation a renderer puts around a child element and around a closing tag
527
527
  // lands at the edges of its parent's text, and trimming removes exactly that
528
- // and nothing else. Whitespace *inside* the run — between two words, or a
529
- // newline in the middle of a paragraph — is content and stays.
528
+ // and nothing else. Whitespace *inside* the run (between two words, or a
529
+ // newline in the middle of a paragraph) is content and stays.
530
530
  const content = resolved.preserveWhitespace ? text : text.trim();
531
531
 
532
532
  if (!hasAttributes && !hasChildren) {
533
- // A leaf is character data on its own. Returning the string rather than a `{ '#text': … }` record is what lets
533
+ // A leaf is character data on its own. Returning the string rather than a `{ '#text': … }` record lets
534
534
  // `Schema.Struct({ name: Schema.String })` round-trip.
535
535
  return content;
536
536
  }
@@ -577,10 +577,10 @@ const parseDocumentResult = (text: string, options: XmlParseOptions): Result.Res
577
577
  };
578
578
 
579
579
  /**
580
- * @description Decodes character references, falling back to the raw text when the reference is not one the decoder recognises. The fallback is what makes a bare
581
- * `&` survivable: the decoder treats it as a malformed reference and fails, and a document containing one is far more likely to be worth reading than
582
- * to be rejected. The `&` is escaped on the way out, so the value still round-trips. The decoder answers with an `Effect`, and this is the one place
583
- * a parse still runs one. It is only reached when the raw text holds an `&` — the common case returns before it — and the effect is synchronous, so
580
+ * @description Decodes character references, falling back to the raw text when the reference is not one the decoder recognises. The fallback keeps a bare `&`
581
+ * survivable: the decoder treats it as a malformed reference and fails, and a document containing one is far more likely to be worth reading than to
582
+ * be rejected. The `&` is escaped on the way out, so the value still round-trips. The decoder answers with an `Effect`, and this is the one place a
583
+ * parse still runs one. It is only reached when the raw text holds an `&`, since the common case returns before it, and the effect is synchronous, so
584
584
  * the run is cheap next to the decoder's own work.
585
585
  *
586
586
  * @param raw - Text read straight from the source, with references unexpanded.