@endevops/effect-codec-xml 0.0.1 → 0.1.0-beta.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (50) hide show
  1. package/README.md +81 -62
  2. package/dist/codec.d.ts +17 -9
  3. package/dist/codec.d.ts.map +1 -1
  4. package/dist/codec.js +29 -18
  5. package/dist/codec.js.map +1 -1
  6. package/dist/conventions.d.ts +4 -4
  7. package/dist/conventions.js +7 -7
  8. package/dist/conventions.js.map +1 -1
  9. package/dist/entities/entity-decoder.d.ts +33 -33
  10. package/dist/entities/entity-decoder.d.ts.map +1 -1
  11. package/dist/entities/entity-decoder.js +63 -64
  12. package/dist/entities/entity-decoder.js.map +1 -1
  13. package/dist/errors.d.ts +2 -2
  14. package/dist/errors.js +2 -2
  15. package/dist/errors.js.map +1 -1
  16. package/dist/namespaces.js +40 -13
  17. package/dist/namespaces.js.map +1 -1
  18. package/dist/naming.d.ts +6 -6
  19. package/dist/naming.d.ts.map +1 -1
  20. package/dist/naming.js +3 -3
  21. package/dist/naming.js.map +1 -1
  22. package/dist/parse.d.ts +5 -5
  23. package/dist/parse.js +14 -14
  24. package/dist/parse.js.map +1 -1
  25. package/dist/plain-value.js +242 -0
  26. package/dist/plain-value.js.map +1 -0
  27. package/dist/render.d.ts +1 -1
  28. package/dist/render.d.ts.map +1 -1
  29. package/dist/render.js +28 -28
  30. package/dist/render.js.map +1 -1
  31. package/dist/xml-error.d.ts +9 -9
  32. package/dist/xml-error.js +18 -18
  33. package/dist/xml-error.js.map +1 -1
  34. package/dist/xml-value.d.ts +7 -7
  35. package/dist/xml-value.d.ts.map +1 -1
  36. package/dist/xml-value.js +6 -7
  37. package/dist/xml-value.js.map +1 -1
  38. package/package.json +1 -1
  39. package/src/codec.ts +98 -71
  40. package/src/conventions.ts +7 -7
  41. package/src/entities/entity-decoder.ts +98 -99
  42. package/src/errors.ts +3 -3
  43. package/src/index.ts +3 -3
  44. package/src/namespaces.ts +62 -36
  45. package/src/naming.ts +34 -35
  46. package/src/parse.ts +26 -26
  47. package/src/plain-value.ts +312 -0
  48. package/src/render.ts +44 -44
  49. package/src/xml-error.ts +18 -18
  50. package/src/xml-value.ts +10 -11
package/dist/parse.js.map CHANGED
@@ -1 +1 @@
1
- {"version":3,"file":"parse.js","names":[],"sources":["../src/parse.ts"],"sourcesContent":["import { Effect, Predicate, Result } from 'effect';\n\nimport type { NameMode } from './conventions.ts';\nimport type { XmlVersion } from './naming.ts';\nimport type { XmlValue } from './xml-value.ts';\n\nimport { ATTRIBUTE_PREFIX, resolveNameSync, TEXT_KEY } from './conventions.ts';\nimport { EntityDecoder } from './entities/entity-decoder.ts';\nimport { XmlParseError } from './errors.ts';\n\nconst decoder = EntityDecoder.make().pipe(Effect.runSync);\n\n/**\n * @description A parsed document: the root element's name, and its content as an {@link XmlValue}.\n */\nexport interface XmlDocument {\n /**\n * @description The root element's name as it appeared in the source, after name resolution.\n */\n readonly name: string;\n\n /**\n * @description The root element's content. The root's own name is not part of it, the same way a schema's encoded form does not carry a name for the value it\n * describes.\n */\n readonly value: XmlValue;\n}\n\n/**\n * @description Options for {@link parseXml} and {@link parseXmlDocument}.\n */\nexport interface XmlParseOptions {\n /**\n * @description Keep the whitespace at the edges of every text run.\n *\n * @default false\\\n * which trims it — and trimming is what makes a pretty-printed document\n * read as the same value as an unindented one, because the indentation around a child element and around a closing tag lands at the edges of its\n * parent's text. Whitespace _inside_ a run is content and is never touched either way, so `'one two'` and a paragraph with a newline in the middle\n * of it survive. Set it to `true` to keep leading and trailing spaces in text exactly as written, at the cost of a document that was laid out on\n * several lines no longer reading the same as one that was not.\n */\n readonly preserveWhitespace?: boolean | undefined;\n\n /**\n * @description How deep to nest before giving up. Guards against a document crafted to exhaust the stack.\n *\n * @default 256\n */\n readonly maxDepth?: number | undefined;\n\n /**\n * @description What to do with an element or attribute name that is not a legal XML name.\n *\n * @default 'repair'\\\n * the same default {@link renderXml} uses, so a name that renders and a name that parses come out the same.\n */\n readonly name?: NameMode | undefined;\n\n /**\n * @description XML version to validate names against.\n *\n * @default '1.0'\n */\n readonly xmlVersion?: XmlVersion | undefined;\n}\n\n/**\n * @description Parses an XML document into its root element's content.\\\n * The walk itself is synchronous, but it reports a malformed document by failing with an {@link XmlParseError} rather than by throwing, so the failure lands in the effect's error channel where `catchTag`, `retry` and a fallback can all\n * see it. A failed parse is an expected outcome of reading untrusted text — it is what those combinators key off — and only a defect would hide it.\n * The span is the boundary a performance trace hangs off: it carries the document's length, which is the size that drives the parser's cost, so a\n * slow parse in a profile can be attributed to the input that produced it. A caller that wants the value outside an `Effect` uses\n * {@link parseXmlDocument}, which runs the same walk synchronously and throws instead. The walk is plain recursive descent rather than a chain of\n * `yield*`es. Publicly `parseXml` is still an `Effect` — it suspends the walk so it runs lazily under the span, and folds the failure the walk throws\n * into the typed error channel — but inside a document there is no effect boundary per tag, attribute or text run. A 500-row report is thousands of\n * those, and a fiber step for each of them was most of what the `parse 500 rows` row measured. The typed failure survives: the walk throws an\n * {@link XmlParseError} and `parseXml` catches it into `Effect.fail`.\n *\n * @param text - The document to read.\n * @param options - Whitespace, depth and name-handling settings.\n *\n * @returns An effect producing the root element's content as an {@link XmlValue}.\n */\nexport const parseXml = (text: string, options: XmlParseOptions = {}): Effect.Effect<XmlValue, XmlParseError> =>\n Effect.suspend(() => Effect.fromResult(parseDocumentResult(text, options))).pipe(\n Effect.map(document => document.value),\n Effect.withSpan('XmlCodec.parseXml', { attributes: { 'xml.length': text.length } })\n );\n\n/**\n * @description Parses an XML document, keeping the root element's name. This is the synchronous form of {@link parseXml}: it runs the same walk and throws the\n * {@link XmlParseError} the effect would have failed with, for a caller that is not already in an `Effect`.\n *\n * @deprecated\n *\n * @param text - The document to read.\n * @param options - Whitespace, depth and name-handling settings.\n *\n * @returns The root element's name and content.\n *\n * @throws {XmlParseError} When the document is not well-formed.\n */\nexport const parseXmlDocument = (text: string, options: XmlParseOptions = {}): XmlDocument => parseDocument(text, options);\n\n/**\n * @description Options every parse call needs, with the defaults already applied.\n */\ninterface ResolvedOptions {\n readonly preserveWhitespace: boolean;\n readonly maxDepth: number;\n readonly name: NameMode;\n readonly xmlVersion: XmlVersion;\n}\n\n/**\n * @description Whether a character is XML whitespace. XML defines exactly four, and they are the only ones a parser may treat as insignificant.\n *\n * @param code - A UTF-16 code unit.\n *\n * @returns Whether the character is XML whitespace.\n */\nconst isWhitespace = (code: number): boolean => code === 32 || code === 10 || code === 9 || code === 13;\n\n/**\n * @description `<`\n */\nconst LT = 60;\n/**\n * @description `>`\n */\nconst GT = 62;\n/**\n * @description `/`\n */\nconst SLASH = 47;\n/**\n * @description `=`\n */\nconst EQUALS = 61;\n\n/**\n * @description One element as the parser saw it: the name it was written under, and the value it holds. Carrying the name alongside the value is what lets the\n * parent file it correctly — the value alone cannot say, because a text-only element reduces to a bare string.\n */\ninterface Element {\n readonly name: string;\n readonly value: XmlValue;\n}\n\n/**\n * @description What a start tag yielded: the record its attributes went into, which becomes the element's value, and whether the tag closed itself.\n */\ninterface StartTag {\n readonly record: Record<string, XmlValue>;\n readonly selfClosing: boolean;\n\n /**\n * @description Whether the tag carried any attribute. Counted as they are read rather than asked of the record afterwards, which would mean a key array per\n * element.\n */\n readonly hasAttributes: boolean;\n}\n\n/**\n * @description An element's body as the parser read it: the character data it accumulated, and whether any child element went into the record.\n */\ninterface Content {\n /**\n * @description Every text run in the body, concatenated in the order they appeared.\n */\n readonly text: string;\n\n /**\n * @description Whether the body held at least one child element. Counted rather than asked of the record afterwards, because a child whose fields are all absent\n * leaves no trace of itself in the record and must still keep its element from being written self-closing.\n */\n readonly hasChildren: boolean;\n}\n\n/**\n * @description What sits at the cursor inside an element's body. Naming what is there before deciding what to do with it is what lets the content loop stay a\n * dispatch: each construct is recognised in one place, against the ones that cannot be confused with it, rather than by a chain of `startsWith`\n * guesses where each had to remember what the last had already ruled out.\n */\ntype Construct = 'text' | 'close' | 'comment' | 'cdata' | 'instruction' | 'child';\n\n/**\n * @description Parses a whole document: a prolog, exactly one root element, and nothing but whitespace after it. The walk is synchronous and reports a malformed\n * document by throwing an {@link XmlParseError}; {@link parseXml} folds that into the effect's typed error channel.\n *\n * @param text - The document to read.\n * @param options - Whitespace, depth and name-handling settings.\n *\n * @returns The root element's name and content.\n *\n * @throws {XmlParseError} When the document is not well-formed.\n */\nconst parseDocument = (text: string, options: XmlParseOptions): XmlDocument => {\n const resolved: ResolvedOptions = {\n preserveWhitespace: options.preserveWhitespace ?? false,\n maxDepth: options.maxDepth ?? 256,\n name: options.name ?? 'repair',\n xmlVersion: options.xmlVersion ?? '1.0',\n };\n\n let at = 0;\n\n /**\n * @description The options every name is resolved with, built once. They cannot change during a parse, and building them per name would allocate one object per\n * element and per attribute in the document.\n */\n const nameOptions = { mode: resolved.name, xmlVersion: resolved.xmlVersion };\n\n /**\n * @description Names already resolved by this parse. A document repeats names — every one of five hundred rows has a `sku` — and a validator that ran per\n * occurrence would pay for the same answer five hundred times.\n */\n const nameCache = new Map<string, string>();\n\n const resolve = (raw: string, what: string, position: number): string => {\n const cached = nameCache.get(raw);\n if (cached !== undefined) return cached;\n\n // `resolveNameSync` reports an illegal name by throwing an `XmlParseError`; the failure is reworded\n // here so it names the position in the document and whether the name belonged to an element or\n // an attribute, which a generic name resolver cannot know.\n let name: string;\n try {\n name = resolveNameSync(raw, nameOptions);\n } catch (failure) {\n const reason = Predicate.isError(failure) ? failure.message : String(failure);\n throw new XmlParseError({ message: `${what} ${JSON.stringify(raw)} is not a legal XML name: ${reason}`, position, input: text });\n }\n\n nameCache.set(raw, name);\n return name;\n };\n\n /**\n * @description Reads to the end of a `<!-- -->`, `<? ?>` or `<!DOCTYPE >` construct, and reports the one past its last character.\n */\n const skipUntil = (marker: string, start: number, what: string): number => {\n const end = text.indexOf(marker, start);\n if (end === -1) throw new XmlParseError({ message: `Unterminated ${what}`, position: start, input: text });\n return end + marker.length;\n };\n\n const skipDoctype = (start: number): number => {\n let depth = 0;\n for (let i = start + 9; i < text.length; i++) {\n const char = text[i];\n if (char === '[') depth++;\n else if (char === ']') depth--;\n else if (char === '>' && depth <= 0) return i + 1;\n }\n throw new XmlParseError({ message: 'Unterminated DOCTYPE declaration', position: start, input: text });\n };\n\n /**\n * @description Consumes whitespace, comments, processing instructions and a DOCTYPE, leaving the cursor on the first character that is none of them — or at the\n * end of the document.\n */\n const skipMisc = (): void => {\n for (;;) {\n while (at < text.length && isWhitespace(text.charCodeAt(at))) {\n at++;\n }\n\n if (at >= text.length) {\n return; // whitespace ran to the end of the document: consumed, and that is the end\n }\n\n if (text.charCodeAt(at) !== LT) {\n return; // real content: leave the cursor on it for the caller\n }\n\n if (text.startsWith('<!--', at)) {\n at = skipUntil('-->', at + 4, 'comment');\n } else if (text.startsWith('<?', at)) {\n at = skipUntil('?>', at + 2, 'processing instruction');\n } else if (text.startsWith('<!DOCTYPE', at)) {\n at = skipDoctype(at);\n } else {\n return; // the start of the root element, or of a closing tag\n }\n }\n };\n\n /**\n * @description Reads a name up to the character that ends it, advancing the cursor past it.\n */\n const readName = (what: string): string => {\n const start = at;\n while (at < text.length) {\n const char = text.charCodeAt(at);\n // Whitespace, `/`, `=` and `>` all end a name. Stopping on `/` and `>` is what lets `<a/>` and `<a>` share one loop.\n if (isWhitespace(char) || char === SLASH || char === EQUALS || char === GT) {\n break;\n }\n at++;\n }\n if (at === start) {\n throw new XmlParseError({ message: `Expected a ${what}`, position: start, input: text });\n }\n return text.slice(start, at);\n };\n\n const skipSpaces = (): void => {\n while (at < text.length && isWhitespace(text.charCodeAt(at))) at++;\n };\n\n const readAttributeValue = (name: string, nameStart: number): string => {\n const quote = text[at];\n // `indexOf` below is only reached once `quote` is known to be a real quote,\n // which the guard establishes; the `?? ''` is unreachable and exists only to\n // keep the type of the index lookup a `string`.\n if (quote !== '\"' && quote !== \"'\") {\n throw new XmlParseError({ message: `Attribute \"${name}\" has no quoted value`, position: nameStart, input: text });\n }\n\n at++;\n\n const end = text.indexOf(quote ?? '', at);\n // A raw quote cannot appear inside a quoted value — it would have to be written `&quot;` — so the next quote of the same kind always closes it.\n if (end === -1) {\n throw new XmlParseError({ message: `Unterminated value for attribute \"${name}\"`, position: at, input: text });\n }\n\n const raw = text.slice(at, end);\n at = end + 1;\n\n return decodeEntities(raw);\n };\n\n const readStartTag = (): StartTag => {\n // Built as the record the element will end up holding rather than as a\n // separate set of attributes, so that folding the text and the children into\n // it later costs no copy. One object per element instead of two.\n const record: Record<string, XmlValue> = {};\n let hasAttributes = false;\n for (;;) {\n skipSpaces();\n if (at >= text.length) throw new XmlParseError({ message: 'Unterminated start tag', position: at, input: text });\n if (text.charCodeAt(at) === GT) {\n at++;\n return { record, selfClosing: false, hasAttributes };\n }\n if (text.charCodeAt(at) === SLASH && text[at + 1] === '>') {\n at += 2;\n return { record, selfClosing: true, hasAttributes };\n }\n const nameStart = at;\n const name = resolve(readName('attribute name'), 'Attribute', nameStart);\n skipSpaces();\n if (text.charCodeAt(at) !== EQUALS) throw new XmlParseError({ message: `Attribute \"${name}\" has no \"=\"`, position: at, input: text });\n at++;\n skipSpaces();\n record[ATTRIBUTE_PREFIX + name] = readAttributeValue(name, nameStart);\n hasAttributes = true;\n }\n };\n\n const readElement = (depth: number): Element => {\n if (depth > resolved.maxDepth)\n throw new XmlParseError({ message: `Element nesting exceeded maxDepth (${resolved.maxDepth})`, position: at, input: text });\n if (text.charCodeAt(at) !== LT) throw new XmlParseError({ message: 'Expected an element', position: at, input: text });\n at++;\n\n const name = resolve(readName('element name'), 'Element', at);\n const { record, selfClosing, hasAttributes } = readStartTag();\n\n if (selfClosing) return { name, value: finishElement(record, hasAttributes, '', false) };\n\n // The parser folds character data and child elements into the record the\n // start tag produced, as it goes rather than in passes, because the order\n // they appear in is the only order available: attributes always come first on\n // the tag, but text and children interleave freely.\n const content = readContent(name, record, depth);\n\n return { name, value: finishElement(record, hasAttributes, content.text, content.hasChildren) };\n };\n\n /**\n * @description Reads an element's body up to and including its closing tag, folding what it finds into the record the start tag produced. Returns when the\n * closing tag has been consumed; failing on it is {@link readClosingTag}'s job, so that a mismatched or unclosed tag is reported the same way\n * wherever it was found.\n *\n * @param name - The name the start tag gave the element, which its closing tag has to match.\n * @param record - The record to fold the children into.\n * @param depth - The depth the element sits at; its children are one deeper.\n *\n * @returns The body as character data, and whether it held any child element.\n */\n const readContent = (name: string, record: Record<string, XmlValue>, depth: number): Content => {\n let childText = '';\n let hasChildren = false;\n\n for (;;) {\n switch (classifyContent(name)) {\n case 'text':\n childText += readTextRun();\n break;\n case 'close':\n readClosingTag(name);\n return { text: childText, hasChildren };\n case 'comment':\n at = skipUntil('-->', at + 4, 'comment');\n break;\n case 'cdata':\n childText += readCdata();\n break;\n case 'instruction':\n at = skipUntil('?>', at + 2, 'processing instruction');\n break;\n case 'child': {\n hasChildren = true;\n addChild(record, readElement(depth + 1));\n break;\n }\n }\n }\n };\n\n /**\n * @description What the cursor is sitting on inside an element's body. The two things the loop cannot read are refused here rather than in it: running out of\n * document and a declaration, which is markup the parser does not accept inside an element. Recognising the constructs that _are_ read is the rest,\n * and the order is the one that rules out the shorter prefixes first — `</` before `<?` before any other `<!`, and `<![CDATA[` before the `<!` that\n * would otherwise match it.\n *\n * @param name - The name the enclosing element's start tag gave it, for the unterminated-body message.\n *\n * @returns What the cursor is on.\n */\n const classifyContent = (name: string): Construct => {\n if (at >= text.length) {\n throw new XmlParseError({ message: `Unclosed element <${name}>`, position: at, input: text });\n }\n if (text.charCodeAt(at) !== LT) {\n return 'text';\n }\n if (text.startsWith('</', at)) {\n return 'close';\n }\n if (text.startsWith('<!--', at)) {\n return 'comment';\n }\n if (text.startsWith('<![CDATA[', at)) {\n return 'cdata';\n }\n if (text.startsWith('<?', at)) {\n return 'instruction';\n }\n if (text.startsWith('<!', at)) {\n throw new XmlParseError({ message: 'A declaration is not allowed inside an element', position: at, input: text });\n }\n return 'child';\n };\n\n /**\n * @description Consumes a `</name>`, checking on the way that it is the tag that closes this element and that it is well-formed.\n *\n * @param name - The name the start tag gave the element, which the closing tag has to match.\n */\n const readClosingTag = (name: string): void => {\n const closeStart = at;\n at += 2;\n const closing = readName('element name');\n if (closing !== name) {\n throw new XmlParseError({ message: `Closing tag </${closing}> does not match <${name}>`, position: closeStart, input: text });\n }\n skipSpaces();\n if (text.charCodeAt(at) !== GT) {\n throw new XmlParseError({ message: `Malformed closing tag </${closing}>`, position: at, input: text });\n }\n at++;\n };\n\n /**\n * @description Reads the run of character data up to the next `<`, or to the end of the document.\n *\n * @returns The run, with its character references expanded.\n */\n const readTextRun = (): string => {\n const next = text.indexOf('<', at);\n const end = next === -1 ? text.length : next;\n const run = decodeEntities(text.slice(at, end));\n at = end;\n return run;\n };\n\n /**\n * @description Reads a `<![CDATA[…]]>` section. CDATA is character data, and character data is what it holds, so it joins the element's text as it stands — the\n * entities in it are literal text and must not be expanded.\n *\n * @returns The section's contents.\n */\n const readCdata = (): string => {\n const end = text.indexOf(']]>', at + 9);\n if (end === -1) throw new XmlParseError({ message: 'Unterminated CDATA section', position: at, input: text });\n const data = text.slice(at + 9, end);\n at = end + 3;\n return data;\n };\n\n /**\n * @description Adds a child to its parent's record. Two children under one name make an array, and the first one does not: a schema can tell a repeated field\n * from a single one by the shape, and an array of one is not what a single value encodes to.\n *\n * @param record - The parent's record, added to in place.\n * @param child - The child element as it was read.\n */\n const addChild = (record: Record<string, XmlValue>, child: Element): void => {\n const existing = record[child.name];\n if (existing === undefined) record[child.name] = child.value;\n else if (Array.isArray(existing)) (existing as Array<XmlValue>).push(child.value);\n else record[child.name] = [existing, child.value];\n };\n\n /**\n * @description Decides what an element with the given attributes, text and children reduces to.\n */\n const finishElement = (record: Record<string, XmlValue>, hasAttributes: boolean, text: string, hasChildren: boolean): XmlValue => {\n // Whitespace at the edges of a text run is dropped unless the caller asked to\n // keep it. This is what makes a pretty-printed document round trip: the\n // indentation a renderer puts around a child element and around a closing tag\n // lands at the edges of its parent's text, and trimming removes exactly that\n // and nothing else. Whitespace *inside* the run — between two words, or a\n // newline in the middle of a paragraph — is content and stays.\n const content = resolved.preserveWhitespace ? text : text.trim();\n\n if (!hasAttributes && !hasChildren) {\n // A leaf is character data on its own. Returning the string rather than a `{ '#text': … }` record is what lets\n // `Schema.Struct({ name: Schema.String })` round-trip.\n return content;\n }\n\n // Folded in place: the record is the one the start tag built and that the children were added to, so there is\n // nothing left to copy.\n if (content !== '') record[TEXT_KEY] = content;\n return record;\n };\n\n skipMisc();\n if (at >= text.length || text.charCodeAt(at) !== LT) {\n throw new XmlParseError({ message: 'Document has no root element', position: at, input: text });\n }\n\n const root = readElement(0);\n\n skipMisc();\n if (at < text.length) {\n throw new XmlParseError({ message: 'Unexpected content after the root element', position: at, input: text });\n }\n\n return { name: root.name, value: root.value };\n};\n\n/**\n * @description Runs the synchronous walk and folds the one failure it reports into a {@link Result}, which {@link parseXml} turns back into an `Effect`. Kept\n * separate so the walk itself can throw without the public API ever throwing.\n *\n * @param text - The document to read.\n * @param options - The options as the caller wrote them.\n *\n * @returns The document, or the failure to report.\n */\nconst parseDocumentResult = (text: string, options: XmlParseOptions): Result.Result<XmlDocument, XmlParseError> => {\n try {\n return Result.succeed(parseDocument(text, options));\n } catch (cause) {\n if (cause instanceof XmlParseError) {\n return Result.fail(cause);\n }\n return Result.fail(new XmlParseError({ message: Predicate.isError(cause) ? cause.message : String(cause), position: -1, input: text }));\n }\n};\n\n/**\n * @description Decodes character references, falling back to the raw text when the reference is not one the decoder recognises. The fallback is what makes a bare\n * `&` survivable: the decoder treats it as a malformed reference and fails, and a document containing one is far more likely to be worth reading than\n * to be rejected. The `&` is escaped on the way out, so the value still round-trips. The decoder answers with an `Effect`, and this is the one place\n * a parse still runs one. It is only reached when the raw text holds an `&` — the common case returns before it — and the effect is synchronous, so\n * the run is cheap next to the decoder's own work.\n *\n * @param raw - Text read straight from the source, with references unexpanded.\n *\n * @returns The decoded text, which cannot fail.\n */\nconst decodeEntities = (raw: string): string => {\n if (raw.indexOf('&') === -1) return raw; // nothing to expand: the common case, and no work\n // `orElseSucceed` rather than `try`/`catch`: the decoder reports a malformed\n // reference by failing in its error channel, and a document containing a bare\n // `&` is far more likely to be worth reading than to be rejected. The `&` is\n // escaped on the way out, so the value still round-trips.\n return Effect.runSync(Effect.orElseSucceed(decoder.decode(raw), () => raw));\n};\n"],"mappings":";;;;;AAUA,MAAM,UAAU,cAAc,KAAK,CAAC,CAAC,KAAK,OAAO,OAAO;;;;;;;;;;;;;;;;;;AA0ExD,MAAa,YAAY,MAAc,UAA2B,CAAC,MACjE,OAAO,cAAc,OAAO,WAAW,oBAAoB,MAAM,OAAO,CAAC,CAAC,CAAC,CAAC,KAC1E,OAAO,KAAI,aAAY,SAAS,KAAK,GACrC,OAAO,SAAS,qBAAqB,EAAE,YAAY,EAAE,cAAc,KAAK,OAAO,EAAE,CAAC,CACpF;;;;;;;;AAkCF,MAAM,gBAAgB,SAA0B,SAAS,MAAM,SAAS,MAAM,SAAS,KAAK,SAAS;;;;AAKrG,MAAM,KAAK;;;;AAIX,MAAM,KAAK;;;;AAIX,MAAM,QAAQ;;;;AAId,MAAM,SAAS;;;;;;;;;;;;AA2Df,MAAM,iBAAiB,MAAc,YAA0C;CAC7E,MAAM,WAA4B;EAChC,oBAAoB,QAAQ,sBAAsB;EAClD,UAAU,QAAQ,YAAY;EAC9B,MAAM,QAAQ,QAAQ;EACtB,YAAY,QAAQ,cAAc;CACpC;CAEA,IAAI,KAAK;;;;;CAMT,MAAM,cAAc;EAAE,MAAM,SAAS;EAAM,YAAY,SAAS;CAAW;;;;;CAM3E,MAAM,4BAAY,IAAI,IAAoB;CAE1C,MAAM,WAAW,KAAa,MAAc,aAA6B;EACvE,MAAM,SAAS,UAAU,IAAI,GAAG;EAChC,IAAI,WAAW,KAAA,GAAW,OAAO;EAKjC,IAAI;EACJ,IAAI;GACF,OAAO,gBAAgB,KAAK,WAAW;EACzC,SAAS,SAAS;GAChB,MAAM,SAAS,UAAU,QAAQ,OAAO,IAAI,QAAQ,UAAU,OAAO,OAAO;GAC5E,MAAM,IAAI,cAAc;IAAE,SAAS,GAAG,KAAK,GAAG,KAAK,UAAU,GAAG,EAAE,4BAA4B;IAAU;IAAU,OAAO;GAAK,CAAC;EACjI;EAEA,UAAU,IAAI,KAAK,IAAI;EACvB,OAAO;CACT;;;;CAKA,MAAM,aAAa,QAAgB,OAAe,SAAyB;EACzE,MAAM,MAAM,KAAK,QAAQ,QAAQ,KAAK;EACtC,IAAI,QAAQ,IAAI,MAAM,IAAI,cAAc;GAAE,SAAS,gBAAgB;GAAQ,UAAU;GAAO,OAAO;EAAK,CAAC;EACzG,OAAO,MAAM,OAAO;CACtB;CAEA,MAAM,eAAe,UAA0B;EAC7C,IAAI,QAAQ;EACZ,KAAK,IAAI,IAAI,QAAQ,GAAG,IAAI,KAAK,QAAQ,KAAK;GAC5C,MAAM,OAAO,KAAK;GAClB,IAAI,SAAS,KAAK;QACb,IAAI,SAAS,KAAK;QAClB,IAAI,SAAS,OAAO,SAAS,GAAG,OAAO,IAAI;EAClD;EACA,MAAM,IAAI,cAAc;GAAE,SAAS;GAAoC,UAAU;GAAO,OAAO;EAAK,CAAC;CACvG;;;;;CAMA,MAAM,iBAAuB;EAC3B,SAAS;GACP,OAAO,KAAK,KAAK,UAAU,aAAa,KAAK,WAAW,EAAE,CAAC,GACzD;GAGF,IAAI,MAAM,KAAK,QACb;GAGF,IAAI,KAAK,WAAW,EAAE,MAAM,IAC1B;GAGF,IAAI,KAAK,WAAW,QAAQ,EAAE,GAC5B,KAAK,UAAU,OAAO,KAAK,GAAG,SAAS;QAClC,IAAI,KAAK,WAAW,MAAM,EAAE,GACjC,KAAK,UAAU,MAAM,KAAK,GAAG,wBAAwB;QAChD,IAAI,KAAK,WAAW,aAAa,EAAE,GACxC,KAAK,YAAY,EAAE;QAEnB;EAEJ;CACF;;;;CAKA,MAAM,YAAY,SAAyB;EACzC,MAAM,QAAQ;EACd,OAAO,KAAK,KAAK,QAAQ;GACvB,MAAM,OAAO,KAAK,WAAW,EAAE;GAE/B,IAAI,aAAa,IAAI,KAAK,SAAS,SAAS,SAAS,UAAU,SAAS,IACtE;GAEF;EACF;EACA,IAAI,OAAO,OACT,MAAM,IAAI,cAAc;GAAE,SAAS,cAAc;GAAQ,UAAU;GAAO,OAAO;EAAK,CAAC;EAEzF,OAAO,KAAK,MAAM,OAAO,EAAE;CAC7B;CAEA,MAAM,mBAAyB;EAC7B,OAAO,KAAK,KAAK,UAAU,aAAa,KAAK,WAAW,EAAE,CAAC,GAAG;CAChE;CAEA,MAAM,sBAAsB,MAAc,cAA8B;EACtE,MAAM,QAAQ,KAAK;EAInB,IAAI,UAAU,QAAO,UAAU,KAC7B,MAAM,IAAI,cAAc;GAAE,SAAS,cAAc,KAAK;GAAwB,UAAU;GAAW,OAAO;EAAK,CAAC;EAGlH;EAEA,MAAM,MAAM,KAAK,QAAQ,SAAS,IAAI,EAAE;EAExC,IAAI,QAAQ,IACV,MAAM,IAAI,cAAc;GAAE,SAAS,qCAAqC,KAAK;GAAI,UAAU;GAAI,OAAO;EAAK,CAAC;EAG9G,MAAM,MAAM,KAAK,MAAM,IAAI,GAAG;EAC9B,KAAK,MAAM;EAEX,OAAO,eAAe,GAAG;CAC3B;CAEA,MAAM,qBAA+B;EAInC,MAAM,SAAmC,CAAC;EAC1C,IAAI,gBAAgB;EACpB,SAAS;GACP,WAAW;GACX,IAAI,MAAM,KAAK,QAAQ,MAAM,IAAI,cAAc;IAAE,SAAS;IAA0B,UAAU;IAAI,OAAO;GAAK,CAAC;GAC/G,IAAI,KAAK,WAAW,EAAE,MAAM,IAAI;IAC9B;IACA,OAAO;KAAE;KAAQ,aAAa;KAAO;IAAc;GACrD;GACA,IAAI,KAAK,WAAW,EAAE,MAAM,SAAS,KAAK,KAAK,OAAO,KAAK;IACzD,MAAM;IACN,OAAO;KAAE;KAAQ,aAAa;KAAM;IAAc;GACpD;GACA,MAAM,YAAY;GAClB,MAAM,OAAO,QAAQ,SAAS,gBAAgB,GAAG,aAAa,SAAS;GACvE,WAAW;GACX,IAAI,KAAK,WAAW,EAAE,MAAM,QAAQ,MAAM,IAAI,cAAc;IAAE,SAAS,cAAc,KAAK;IAAe,UAAU;IAAI,OAAO;GAAK,CAAC;GACpI;GACA,WAAW;GACX,OAAA,MAA0B,QAAQ,mBAAmB,MAAM,SAAS;GACpE,gBAAgB;EAClB;CACF;CAEA,MAAM,eAAe,UAA2B;EAC9C,IAAI,QAAQ,SAAS,UACnB,MAAM,IAAI,cAAc;GAAE,SAAS,sCAAsC,SAAS,SAAS;GAAI,UAAU;GAAI,OAAO;EAAK,CAAC;EAC5H,IAAI,KAAK,WAAW,EAAE,MAAM,IAAI,MAAM,IAAI,cAAc;GAAE,SAAS;GAAuB,UAAU;GAAI,OAAO;EAAK,CAAC;EACrH;EAEA,MAAM,OAAO,QAAQ,SAAS,cAAc,GAAG,WAAW,EAAE;EAC5D,MAAM,EAAE,QAAQ,aAAa,kBAAkB,aAAa;EAE5D,IAAI,aAAa,OAAO;GAAE;GAAM,OAAO,cAAc,QAAQ,eAAe,IAAI,KAAK;EAAE;EAMvF,MAAM,UAAU,YAAY,MAAM,QAAQ,KAAK;EAE/C,OAAO;GAAE;GAAM,OAAO,cAAc,QAAQ,eAAe,QAAQ,MAAM,QAAQ,WAAW;EAAE;CAChG;;;;;;;;;;;;CAaA,MAAM,eAAe,MAAc,QAAkC,UAA2B;EAC9F,IAAI,YAAY;EAChB,IAAI,cAAc;EAElB,SACE,QAAQ,gBAAgB,IAAI,GAA5B;GACE,KAAK;IACH,aAAa,YAAY;IACzB;GACF,KAAK;IACH,eAAe,IAAI;IACnB,OAAO;KAAE,MAAM;KAAW;IAAY;GACxC,KAAK;IACH,KAAK,UAAU,OAAO,KAAK,GAAG,SAAS;IACvC;GACF,KAAK;IACH,aAAa,UAAU;IACvB;GACF,KAAK;IACH,KAAK,UAAU,MAAM,KAAK,GAAG,wBAAwB;IACrD;GACF,KAAK;IACH,cAAc;IACd,SAAS,QAAQ,YAAY,QAAQ,CAAC,CAAC;EAG3C;CAEJ;;;;;;;;;;;CAYA,MAAM,mBAAmB,SAA4B;EACnD,IAAI,MAAM,KAAK,QACb,MAAM,IAAI,cAAc;GAAE,SAAS,qBAAqB,KAAK;GAAI,UAAU;GAAI,OAAO;EAAK,CAAC;EAE9F,IAAI,KAAK,WAAW,EAAE,MAAM,IAC1B,OAAO;EAET,IAAI,KAAK,WAAW,MAAM,EAAE,GAC1B,OAAO;EAET,IAAI,KAAK,WAAW,QAAQ,EAAE,GAC5B,OAAO;EAET,IAAI,KAAK,WAAW,aAAa,EAAE,GACjC,OAAO;EAET,IAAI,KAAK,WAAW,MAAM,EAAE,GAC1B,OAAO;EAET,IAAI,KAAK,WAAW,MAAM,EAAE,GAC1B,MAAM,IAAI,cAAc;GAAE,SAAS;GAAkD,UAAU;GAAI,OAAO;EAAK,CAAC;EAElH,OAAO;CACT;;;;;;CAOA,MAAM,kBAAkB,SAAuB;EAC7C,MAAM,aAAa;EACnB,MAAM;EACN,MAAM,UAAU,SAAS,cAAc;EACvC,IAAI,YAAY,MACd,MAAM,IAAI,cAAc;GAAE,SAAS,iBAAiB,QAAQ,oBAAoB,KAAK;GAAI,UAAU;GAAY,OAAO;EAAK,CAAC;EAE9H,WAAW;EACX,IAAI,KAAK,WAAW,EAAE,MAAM,IAC1B,MAAM,IAAI,cAAc;GAAE,SAAS,2BAA2B,QAAQ;GAAI,UAAU;GAAI,OAAO;EAAK,CAAC;EAEvG;CACF;;;;;;CAOA,MAAM,oBAA4B;EAChC,MAAM,OAAO,KAAK,QAAQ,KAAK,EAAE;EACjC,MAAM,MAAM,SAAS,KAAK,KAAK,SAAS;EACxC,MAAM,MAAM,eAAe,KAAK,MAAM,IAAI,GAAG,CAAC;EAC9C,KAAK;EACL,OAAO;CACT;;;;;;;CAQA,MAAM,kBAA0B;EAC9B,MAAM,MAAM,KAAK,QAAQ,OAAO,KAAK,CAAC;EACtC,IAAI,QAAQ,IAAI,MAAM,IAAI,cAAc;GAAE,SAAS;GAA8B,UAAU;GAAI,OAAO;EAAK,CAAC;EAC5G,MAAM,OAAO,KAAK,MAAM,KAAK,GAAG,GAAG;EACnC,KAAK,MAAM;EACX,OAAO;CACT;;;;;;;;CASA,MAAM,YAAY,QAAkC,UAAyB;EAC3E,MAAM,WAAW,OAAO,MAAM;EAC9B,IAAI,aAAa,KAAA,GAAW,OAAO,MAAM,QAAQ,MAAM;OAClD,IAAI,MAAM,QAAQ,QAAQ,GAAG,SAA8B,KAAK,MAAM,KAAK;OAC3E,OAAO,MAAM,QAAQ,CAAC,UAAU,MAAM,KAAK;CAClD;;;;CAKA,MAAM,iBAAiB,QAAkC,eAAwB,MAAc,gBAAmC;EAOhI,MAAM,UAAU,SAAS,qBAAqB,OAAO,KAAK,KAAK;EAE/D,IAAI,CAAC,iBAAiB,CAAC,aAGrB,OAAO;EAKT,IAAI,YAAY,IAAI,OAAO,YAAY;EACvC,OAAO;CACT;CAEA,SAAS;CACT,IAAI,MAAM,KAAK,UAAU,KAAK,WAAW,EAAE,MAAM,IAC/C,MAAM,IAAI,cAAc;EAAE,SAAS;EAAgC,UAAU;EAAI,OAAO;CAAK,CAAC;CAGhG,MAAM,OAAO,YAAY,CAAC;CAE1B,SAAS;CACT,IAAI,KAAK,KAAK,QACZ,MAAM,IAAI,cAAc;EAAE,SAAS;EAA6C,UAAU;EAAI,OAAO;CAAK,CAAC;CAG7G,OAAO;EAAE,MAAM,KAAK;EAAM,OAAO,KAAK;CAAM;AAC9C;;;;;;;;;;AAWA,MAAM,uBAAuB,MAAc,YAAwE;CACjH,IAAI;EACF,OAAO,OAAO,QAAQ,cAAc,MAAM,OAAO,CAAC;CACpD,SAAS,OAAO;EACd,IAAI,iBAAiB,eACnB,OAAO,OAAO,KAAK,KAAK;EAE1B,OAAO,OAAO,KAAK,IAAI,cAAc;GAAE,SAAS,UAAU,QAAQ,KAAK,IAAI,MAAM,UAAU,OAAO,KAAK;GAAG,UAAU;GAAI,OAAO;EAAK,CAAC,CAAC;CACxI;AACF;;;;;;;;;;;;AAaA,MAAM,kBAAkB,QAAwB;CAC9C,IAAI,IAAI,QAAQ,GAAG,MAAM,IAAI,OAAO;CAKpC,OAAO,OAAO,QAAQ,OAAO,cAAc,QAAQ,OAAO,GAAG,SAAS,GAAG,CAAC;AAC5E"}
1
+ {"version":3,"file":"parse.js","names":[],"sources":["../src/parse.ts"],"sourcesContent":["import { Effect, Predicate, Result } from 'effect';\n\nimport type { NameMode } from './conventions.ts';\nimport type { XmlVersion } from './naming.ts';\nimport type { XmlValue } from './xml-value.ts';\n\nimport { ATTRIBUTE_PREFIX, resolveNameSync, TEXT_KEY } from './conventions.ts';\nimport { EntityDecoder } from './entities/entity-decoder.ts';\nimport { XmlParseError } from './errors.ts';\n\nconst decoder = EntityDecoder.make().pipe(Effect.runSync);\n\n/**\n * @description A parsed document: the root element's name, and its content as an {@link XmlValue}.\n */\nexport interface XmlDocument {\n /**\n * @description The root element's name as it appeared in the source, after name resolution.\n */\n readonly name: string;\n\n /**\n * @description The root element's content. The root's own name is not part of it, the same way a schema's encoded form does not carry a name for the value it\n * describes.\n */\n readonly value: XmlValue;\n}\n\n/**\n * @description Options for {@link parseXml} and {@link parseXmlDocument}.\n */\nexport interface XmlParseOptions {\n /**\n * @description Keep the whitespace at the edges of every text run.\n *\n * @default false\\\n * which trims it. Trimming makes a pretty-printed document\n * read as the same value as an unindented one, because the indentation around a child element and around a closing tag lands at the edges of its\n * parent's text. Whitespace _inside_ a run is content and is never touched either way, so `'one two'` and a paragraph with a newline in the middle\n * of it survive. Set it to `true` to keep leading and trailing spaces in text exactly as written, at the cost of a document that was laid out on\n * several lines no longer reading the same as one that was not.\n */\n readonly preserveWhitespace?: boolean | undefined;\n\n /**\n * @description How deep to nest before giving up. Guards against a document crafted to exhaust the stack.\n *\n * @default 256\n */\n readonly maxDepth?: number | undefined;\n\n /**\n * @description What to do with an element or attribute name that is not a legal XML name.\n *\n * @default 'repair'\\\n * the same default {@link renderXml} uses, so a name that renders and a name that parses come out the same.\n */\n readonly name?: NameMode | undefined;\n\n /**\n * @description XML version to validate names against.\n *\n * @default '1.0'\n */\n readonly xmlVersion?: XmlVersion | undefined;\n}\n\n/**\n * @description Parses an XML document into its root element's content.\\\n * The walk itself is synchronous, but it reports a malformed document by failing with an {@link XmlParseError} rather than by throwing, so the failure lands in the effect's error channel where `catchTag`, `retry` and a fallback can all\n * see it. A failed parse is an expected outcome of reading untrusted text, and those combinators key off it, so only a defect would hide it.\n * The span is the boundary a performance trace hangs off: it carries the document's length, which is the size that drives the parser's cost, so a\n * slow parse in a profile can be attributed to the input that produced it. A caller that wants the value outside an `Effect` uses\n * {@link parseXmlDocument}, which runs the same walk synchronously and throws instead. The walk is plain recursive descent rather than a chain of\n * `yield*`es. Publicly `parseXml` is still an `Effect`: it suspends the walk so it runs lazily under the span, and folds the failure the walk throws\n * into the typed error channel, but inside a document there is no effect boundary per tag, attribute or text run. A 500-row report is thousands of\n * those, and one fiber step per construct dominated the `parse 500 rows` benchmark. The typed failure survives: the walk throws an\n * {@link XmlParseError} and `parseXml` catches it into `Effect.fail`.\n *\n * @param text - The document to read.\n * @param options - Whitespace, depth and name-handling settings.\n *\n * @returns An effect producing the root element's content as an {@link XmlValue}.\n */\nexport const parseXml = (text: string, options: XmlParseOptions = {}): Effect.Effect<XmlValue, XmlParseError> =>\n Effect.suspend(() => Effect.fromResult(parseDocumentResult(text, options))).pipe(\n Effect.map(document => document.value),\n Effect.withSpan('XmlCodec.parseXml', { attributes: { 'xml.length': text.length } })\n );\n\n/**\n * @description Parses an XML document, keeping the root element's name. This is the synchronous form of {@link parseXml}: it runs the same walk and throws the\n * {@link XmlParseError} the effect would have failed with, for a caller that is not already in an `Effect`.\n *\n * @deprecated\n *\n * @param text - The document to read.\n * @param options - Whitespace, depth and name-handling settings.\n *\n * @returns The root element's name and content.\n *\n * @throws {XmlParseError} When the document is not well-formed.\n */\nexport const parseXmlDocument = (text: string, options: XmlParseOptions = {}): XmlDocument => parseDocument(text, options);\n\n/**\n * @description Options every parse call needs, with the defaults already applied.\n */\ninterface ResolvedOptions {\n readonly preserveWhitespace: boolean;\n readonly maxDepth: number;\n readonly name: NameMode;\n readonly xmlVersion: XmlVersion;\n}\n\n/**\n * @description Whether a character is XML whitespace. XML defines exactly four, and they are the only ones a parser may treat as insignificant.\n *\n * @param code - A UTF-16 code unit.\n *\n * @returns Whether the character is XML whitespace.\n */\nconst isWhitespace = (code: number): boolean => code === 32 || code === 10 || code === 9 || code === 13;\n\n/**\n * @description `<`\n */\nconst LT = 60;\n/**\n * @description `>`\n */\nconst GT = 62;\n/**\n * @description `/`\n */\nconst SLASH = 47;\n/**\n * @description `=`\n */\nconst EQUALS = 61;\n\n/**\n * @description One element as the parser saw it: the name it was written under, and the value it holds. Carrying the name alongside the value lets the parent file\n * it correctly, because the value alone cannot say: a text-only element reduces to a bare string.\n */\ninterface Element {\n readonly name: string;\n readonly value: XmlValue;\n}\n\n/**\n * @description What a start tag yielded: the record its attributes went into, which becomes the element's value, and whether the tag closed itself.\n */\ninterface StartTag {\n readonly record: Record<string, XmlValue>;\n readonly selfClosing: boolean;\n\n /**\n * @description Whether the tag carried any attribute. Counted as they are read rather than asked of the record afterwards, which would mean a key array per\n * element.\n */\n readonly hasAttributes: boolean;\n}\n\n/**\n * @description An element's body as the parser read it: the character data it accumulated, and whether any child element went into the record.\n */\ninterface Content {\n /**\n * @description Every text run in the body, concatenated in the order they appeared.\n */\n readonly text: string;\n\n /**\n * @description Whether the body held at least one child element. Counted rather than asked of the record afterwards, because a child whose fields are all absent\n * leaves no trace of itself in the record and must still keep its element from being written self-closing.\n */\n readonly hasChildren: boolean;\n}\n\n/**\n * @description What sits at the cursor inside an element's body. Naming what is there before deciding what to do with it lets the content loop stay a dispatch:\n * each construct is recognised in one place, against the ones that cannot be confused with it, rather than by a chain of `startsWith` guesses where\n * each had to remember what the last had already ruled out.\n */\ntype Construct = 'text' | 'close' | 'comment' | 'cdata' | 'instruction' | 'child';\n\n/**\n * @description Parses a whole document: a prolog, exactly one root element, and nothing but whitespace after it. The walk is synchronous and reports a malformed\n * document by throwing an {@link XmlParseError}; {@link parseXml} folds that into the effect's typed error channel.\n *\n * @param text - The document to read.\n * @param options - Whitespace, depth and name-handling settings.\n *\n * @returns The root element's name and content.\n *\n * @throws {XmlParseError} When the document is not well-formed.\n */\nconst parseDocument = (text: string, options: XmlParseOptions): XmlDocument => {\n const resolved: ResolvedOptions = {\n preserveWhitespace: options.preserveWhitespace ?? false,\n maxDepth: options.maxDepth ?? 256,\n name: options.name ?? 'repair',\n xmlVersion: options.xmlVersion ?? '1.0',\n };\n\n let at = 0;\n\n /**\n * @description The options every name is resolved with, built once. They cannot change during a parse, and building them per name would allocate one object per\n * element and per attribute in the document.\n */\n const nameOptions = { mode: resolved.name, xmlVersion: resolved.xmlVersion };\n\n /**\n * @description Names already resolved by this parse. A document repeats names (every one of five hundred rows has a `sku`), and a validator that ran per\n * occurrence would pay for the same answer five hundred times.\n */\n const nameCache = new Map<string, string>();\n\n const resolve = (raw: string, what: string, position: number): string => {\n const cached = nameCache.get(raw);\n if (cached !== undefined) return cached;\n\n // `resolveNameSync` reports an illegal name by throwing an `XmlParseError`; the failure is reworded\n // here so it names the position in the document and whether the name belonged to an element or\n // an attribute, which a generic name resolver cannot know.\n let name: string;\n try {\n name = resolveNameSync(raw, nameOptions);\n } catch (failure) {\n const reason = Predicate.isError(failure) ? failure.message : String(failure);\n throw new XmlParseError({ message: `${what} ${JSON.stringify(raw)} is not a legal XML name: ${reason}`, position, input: text });\n }\n\n nameCache.set(raw, name);\n return name;\n };\n\n /**\n * @description Reads to the end of a `<!-- -->`, `<? ?>` or `<!DOCTYPE >` construct, and reports the one past its last character.\n */\n const skipUntil = (marker: string, start: number, what: string): number => {\n const end = text.indexOf(marker, start);\n if (end === -1) throw new XmlParseError({ message: `Unterminated ${what}`, position: start, input: text });\n return end + marker.length;\n };\n\n const skipDoctype = (start: number): number => {\n let depth = 0;\n for (let i = start + 9; i < text.length; i++) {\n const char = text[i];\n if (char === '[') depth++;\n else if (char === ']') depth--;\n else if (char === '>' && depth <= 0) return i + 1;\n }\n throw new XmlParseError({ message: 'Unterminated DOCTYPE declaration', position: start, input: text });\n };\n\n /**\n * @description Consumes whitespace, comments, processing instructions and a DOCTYPE, leaving the cursor on the first character that is none of them, or at the\n * end of the document.\n */\n const skipMisc = (): void => {\n for (;;) {\n while (at < text.length && isWhitespace(text.charCodeAt(at))) {\n at++;\n }\n\n if (at >= text.length) {\n return; // whitespace ran to the end of the document: consumed, and that is the end\n }\n\n if (text.charCodeAt(at) !== LT) {\n return; // real content: leave the cursor on it for the caller\n }\n\n if (text.startsWith('<!--', at)) {\n at = skipUntil('-->', at + 4, 'comment');\n } else if (text.startsWith('<?', at)) {\n at = skipUntil('?>', at + 2, 'processing instruction');\n } else if (text.startsWith('<!DOCTYPE', at)) {\n at = skipDoctype(at);\n } else {\n return; // the start of the root element, or of a closing tag\n }\n }\n };\n\n /**\n * @description Reads a name up to the character that ends it, advancing the cursor past it.\n */\n const readName = (what: string): string => {\n const start = at;\n while (at < text.length) {\n const char = text.charCodeAt(at);\n // Whitespace, `/`, `=` and `>` all end a name. Stopping on `/` and `>` lets `<a/>` and `<a>` share one loop.\n if (isWhitespace(char) || char === SLASH || char === EQUALS || char === GT) {\n break;\n }\n at++;\n }\n if (at === start) {\n throw new XmlParseError({ message: `Expected a ${what}`, position: start, input: text });\n }\n return text.slice(start, at);\n };\n\n const skipSpaces = (): void => {\n while (at < text.length && isWhitespace(text.charCodeAt(at))) at++;\n };\n\n const readAttributeValue = (name: string, nameStart: number): string => {\n const quote = text[at];\n // `indexOf` below is only reached once `quote` is known to be a real quote,\n // which the guard establishes; the `?? ''` is unreachable and exists only to\n // keep the type of the index lookup a `string`.\n if (quote !== '\"' && quote !== \"'\") {\n throw new XmlParseError({ message: `Attribute \"${name}\" has no quoted value`, position: nameStart, input: text });\n }\n\n at++;\n\n const end = text.indexOf(quote ?? '', at);\n // A raw quote cannot appear inside a quoted value (it would have to be written `&quot;`), so the next quote of the same kind always closes it.\n if (end === -1) {\n throw new XmlParseError({ message: `Unterminated value for attribute \"${name}\"`, position: at, input: text });\n }\n\n const raw = text.slice(at, end);\n at = end + 1;\n\n return decodeEntities(raw);\n };\n\n const readStartTag = (): StartTag => {\n // Built as the record the element will end up holding rather than as a\n // separate set of attributes, so that folding the text and the children into\n // it later costs no copy. One object per element instead of two.\n const record: Record<string, XmlValue> = {};\n let hasAttributes = false;\n for (;;) {\n skipSpaces();\n if (at >= text.length) throw new XmlParseError({ message: 'Unterminated start tag', position: at, input: text });\n if (text.charCodeAt(at) === GT) {\n at++;\n return { record, selfClosing: false, hasAttributes };\n }\n if (text.charCodeAt(at) === SLASH && text[at + 1] === '>') {\n at += 2;\n return { record, selfClosing: true, hasAttributes };\n }\n const nameStart = at;\n const name = resolve(readName('attribute name'), 'Attribute', nameStart);\n skipSpaces();\n if (text.charCodeAt(at) !== EQUALS) throw new XmlParseError({ message: `Attribute \"${name}\" has no \"=\"`, position: at, input: text });\n at++;\n skipSpaces();\n record[ATTRIBUTE_PREFIX + name] = readAttributeValue(name, nameStart);\n hasAttributes = true;\n }\n };\n\n const readElement = (depth: number): Element => {\n if (depth > resolved.maxDepth)\n throw new XmlParseError({ message: `Element nesting exceeded maxDepth (${resolved.maxDepth})`, position: at, input: text });\n if (text.charCodeAt(at) !== LT) throw new XmlParseError({ message: 'Expected an element', position: at, input: text });\n at++;\n\n const name = resolve(readName('element name'), 'Element', at);\n const { record, selfClosing, hasAttributes } = readStartTag();\n\n if (selfClosing) return { name, value: finishElement(record, hasAttributes, '', false) };\n\n // The parser folds character data and child elements into the record the\n // start tag produced, as it goes rather than in passes, because the order\n // they appear in is the only order available: attributes always come first on\n // the tag, but text and children interleave freely.\n const content = readContent(name, record, depth);\n\n return { name, value: finishElement(record, hasAttributes, content.text, content.hasChildren) };\n };\n\n /**\n * @description Reads an element's body up to and including its closing tag, folding what it finds into the record the start tag produced. Returns when the\n * closing tag has been consumed; failing on it is {@link readClosingTag}'s job, so that a mismatched or unclosed tag is reported the same way\n * wherever it was found.\n *\n * @param name - The name the start tag gave the element, which its closing tag has to match.\n * @param record - The record to fold the children into.\n * @param depth - The depth the element sits at; its children are one deeper.\n *\n * @returns The body as character data, and whether it held any child element.\n */\n const readContent = (name: string, record: Record<string, XmlValue>, depth: number): Content => {\n let childText = '';\n let hasChildren = false;\n\n for (;;) {\n switch (classifyContent(name)) {\n case 'text':\n childText += readTextRun();\n break;\n case 'close':\n readClosingTag(name);\n return { text: childText, hasChildren };\n case 'comment':\n at = skipUntil('-->', at + 4, 'comment');\n break;\n case 'cdata':\n childText += readCdata();\n break;\n case 'instruction':\n at = skipUntil('?>', at + 2, 'processing instruction');\n break;\n case 'child': {\n hasChildren = true;\n addChild(record, readElement(depth + 1));\n break;\n }\n }\n }\n };\n\n /**\n * @description What the cursor is sitting on inside an element's body. The two things the loop cannot read are refused here rather than in it: running out of\n * document and a declaration, which is markup the parser does not accept inside an element. Recognising the constructs that _are_ read is the rest,\n * and the order rules out the shorter prefixes first: `</` before `<?` before any other `<!`, and `<![CDATA[` before the `<!` that would otherwise\n * match it.\n *\n * @param name - The name the enclosing element's start tag gave it, for the unterminated-body message.\n *\n * @returns What the cursor is on.\n */\n const classifyContent = (name: string): Construct => {\n if (at >= text.length) {\n throw new XmlParseError({ message: `Unclosed element <${name}>`, position: at, input: text });\n }\n if (text.charCodeAt(at) !== LT) {\n return 'text';\n }\n if (text.startsWith('</', at)) {\n return 'close';\n }\n if (text.startsWith('<!--', at)) {\n return 'comment';\n }\n if (text.startsWith('<![CDATA[', at)) {\n return 'cdata';\n }\n if (text.startsWith('<?', at)) {\n return 'instruction';\n }\n if (text.startsWith('<!', at)) {\n throw new XmlParseError({ message: 'A declaration is not allowed inside an element', position: at, input: text });\n }\n return 'child';\n };\n\n /**\n * @description Consumes a `</name>`, checking as it goes that it is the tag that closes this element and that it is well-formed.\n *\n * @param name - The name the start tag gave the element, which the closing tag has to match.\n */\n const readClosingTag = (name: string): void => {\n const closeStart = at;\n at += 2;\n const closing = readName('element name');\n if (closing !== name) {\n throw new XmlParseError({ message: `Closing tag </${closing}> does not match <${name}>`, position: closeStart, input: text });\n }\n skipSpaces();\n if (text.charCodeAt(at) !== GT) {\n throw new XmlParseError({ message: `Malformed closing tag </${closing}>`, position: at, input: text });\n }\n at++;\n };\n\n /**\n * @description Reads the run of character data up to the next `<`, or to the end of the document.\n *\n * @returns The run, with its character references expanded.\n */\n const readTextRun = (): string => {\n const next = text.indexOf('<', at);\n const end = next === -1 ? text.length : next;\n const run = decodeEntities(text.slice(at, end));\n at = end;\n return run;\n };\n\n /**\n * @description Reads a `<![CDATA[…]]>` section. CDATA is character data, and character data is what it holds, so it joins the element's text as it stands. The\n * entities in it are literal text and must not be expanded.\n *\n * @returns The section's contents.\n */\n const readCdata = (): string => {\n const end = text.indexOf(']]>', at + 9);\n if (end === -1) throw new XmlParseError({ message: 'Unterminated CDATA section', position: at, input: text });\n const data = text.slice(at + 9, end);\n at = end + 3;\n return data;\n };\n\n /**\n * @description Adds a child to its parent's record. Two children under one name make an array, and the first one does not: a schema can tell a repeated field\n * from a single one by the shape, and an array of one is not what a single value encodes to.\n *\n * @param record - The parent's record, added to in place.\n * @param child - The child element as it was read.\n */\n const addChild = (record: Record<string, XmlValue>, child: Element): void => {\n const existing = record[child.name];\n if (existing === undefined) record[child.name] = child.value;\n else if (Array.isArray(existing)) (existing as Array<XmlValue>).push(child.value);\n else record[child.name] = [existing, child.value];\n };\n\n /**\n * @description Decides what an element with the given attributes, text and children reduces to.\n */\n const finishElement = (record: Record<string, XmlValue>, hasAttributes: boolean, text: string, hasChildren: boolean): XmlValue => {\n // Whitespace at the edges of a text run is dropped unless the caller asked to\n // keep it. Trimming here makes a pretty-printed document round trip: the\n // indentation a renderer puts around a child element and around a closing tag\n // lands at the edges of its parent's text, and trimming removes exactly that\n // and nothing else. Whitespace *inside* the run (between two words, or a\n // newline in the middle of a paragraph) is content and stays.\n const content = resolved.preserveWhitespace ? text : text.trim();\n\n if (!hasAttributes && !hasChildren) {\n // A leaf is character data on its own. Returning the string rather than a `{ '#text': … }` record lets\n // `Schema.Struct({ name: Schema.String })` round-trip.\n return content;\n }\n\n // Folded in place: the record is the one the start tag built and that the children were added to, so there is\n // nothing left to copy.\n if (content !== '') record[TEXT_KEY] = content;\n return record;\n };\n\n skipMisc();\n if (at >= text.length || text.charCodeAt(at) !== LT) {\n throw new XmlParseError({ message: 'Document has no root element', position: at, input: text });\n }\n\n const root = readElement(0);\n\n skipMisc();\n if (at < text.length) {\n throw new XmlParseError({ message: 'Unexpected content after the root element', position: at, input: text });\n }\n\n return { name: root.name, value: root.value };\n};\n\n/**\n * @description Runs the synchronous walk and folds the one failure it reports into a {@link Result}, which {@link parseXml} turns back into an `Effect`. Kept\n * separate so the walk itself can throw without the public API ever throwing.\n *\n * @param text - The document to read.\n * @param options - The options as the caller wrote them.\n *\n * @returns The document, or the failure to report.\n */\nconst parseDocumentResult = (text: string, options: XmlParseOptions): Result.Result<XmlDocument, XmlParseError> => {\n try {\n return Result.succeed(parseDocument(text, options));\n } catch (cause) {\n if (cause instanceof XmlParseError) {\n return Result.fail(cause);\n }\n return Result.fail(new XmlParseError({ message: Predicate.isError(cause) ? cause.message : String(cause), position: -1, input: text }));\n }\n};\n\n/**\n * @description Decodes character references, falling back to the raw text when the reference is not one the decoder recognises. The fallback keeps a bare `&`\n * survivable: the decoder treats it as a malformed reference and fails, and a document containing one is far more likely to be worth reading than to\n * be rejected. The `&` is escaped on the way out, so the value still round-trips. The decoder answers with an `Effect`, and this is the one place a\n * parse still runs one. It is only reached when the raw text holds an `&`, since the common case returns before it, and the effect is synchronous, so\n * the run is cheap next to the decoder's own work.\n *\n * @param raw - Text read straight from the source, with references unexpanded.\n *\n * @returns The decoded text, which cannot fail.\n */\nconst decodeEntities = (raw: string): string => {\n if (raw.indexOf('&') === -1) return raw; // nothing to expand: the common case, and no work\n // `orElseSucceed` rather than `try`/`catch`: the decoder reports a malformed\n // reference by failing in its error channel, and a document containing a bare\n // `&` is far more likely to be worth reading than to be rejected. The `&` is\n // escaped on the way out, so the value still round-trips.\n return Effect.runSync(Effect.orElseSucceed(decoder.decode(raw), () => raw));\n};\n"],"mappings":";;;;;AAUA,MAAM,UAAU,cAAc,KAAK,CAAC,CAAC,KAAK,OAAO,OAAO;;;;;;;;;;;;;;;;;;AA0ExD,MAAa,YAAY,MAAc,UAA2B,CAAC,MACjE,OAAO,cAAc,OAAO,WAAW,oBAAoB,MAAM,OAAO,CAAC,CAAC,CAAC,CAAC,KAC1E,OAAO,KAAI,aAAY,SAAS,KAAK,GACrC,OAAO,SAAS,qBAAqB,EAAE,YAAY,EAAE,cAAc,KAAK,OAAO,EAAE,CAAC,CACpF;;;;;;;;AAkCF,MAAM,gBAAgB,SAA0B,SAAS,MAAM,SAAS,MAAM,SAAS,KAAK,SAAS;;;;AAKrG,MAAM,KAAK;;;;AAIX,MAAM,KAAK;;;;AAIX,MAAM,QAAQ;;;;AAId,MAAM,SAAS;;;;;;;;;;;;AA2Df,MAAM,iBAAiB,MAAc,YAA0C;CAC7E,MAAM,WAA4B;EAChC,oBAAoB,QAAQ,sBAAsB;EAClD,UAAU,QAAQ,YAAY;EAC9B,MAAM,QAAQ,QAAQ;EACtB,YAAY,QAAQ,cAAc;CACpC;CAEA,IAAI,KAAK;;;;;CAMT,MAAM,cAAc;EAAE,MAAM,SAAS;EAAM,YAAY,SAAS;CAAW;;;;;CAM3E,MAAM,4BAAY,IAAI,IAAoB;CAE1C,MAAM,WAAW,KAAa,MAAc,aAA6B;EACvE,MAAM,SAAS,UAAU,IAAI,GAAG;EAChC,IAAI,WAAW,KAAA,GAAW,OAAO;EAKjC,IAAI;EACJ,IAAI;GACF,OAAO,gBAAgB,KAAK,WAAW;EACzC,SAAS,SAAS;GAChB,MAAM,SAAS,UAAU,QAAQ,OAAO,IAAI,QAAQ,UAAU,OAAO,OAAO;GAC5E,MAAM,IAAI,cAAc;IAAE,SAAS,GAAG,KAAK,GAAG,KAAK,UAAU,GAAG,EAAE,4BAA4B;IAAU;IAAU,OAAO;GAAK,CAAC;EACjI;EAEA,UAAU,IAAI,KAAK,IAAI;EACvB,OAAO;CACT;;;;CAKA,MAAM,aAAa,QAAgB,OAAe,SAAyB;EACzE,MAAM,MAAM,KAAK,QAAQ,QAAQ,KAAK;EACtC,IAAI,QAAQ,IAAI,MAAM,IAAI,cAAc;GAAE,SAAS,gBAAgB;GAAQ,UAAU;GAAO,OAAO;EAAK,CAAC;EACzG,OAAO,MAAM,OAAO;CACtB;CAEA,MAAM,eAAe,UAA0B;EAC7C,IAAI,QAAQ;EACZ,KAAK,IAAI,IAAI,QAAQ,GAAG,IAAI,KAAK,QAAQ,KAAK;GAC5C,MAAM,OAAO,KAAK;GAClB,IAAI,SAAS,KAAK;QACb,IAAI,SAAS,KAAK;QAClB,IAAI,SAAS,OAAO,SAAS,GAAG,OAAO,IAAI;EAClD;EACA,MAAM,IAAI,cAAc;GAAE,SAAS;GAAoC,UAAU;GAAO,OAAO;EAAK,CAAC;CACvG;;;;;CAMA,MAAM,iBAAuB;EAC3B,SAAS;GACP,OAAO,KAAK,KAAK,UAAU,aAAa,KAAK,WAAW,EAAE,CAAC,GACzD;GAGF,IAAI,MAAM,KAAK,QACb;GAGF,IAAI,KAAK,WAAW,EAAE,MAAM,IAC1B;GAGF,IAAI,KAAK,WAAW,QAAQ,EAAE,GAC5B,KAAK,UAAU,OAAO,KAAK,GAAG,SAAS;QAClC,IAAI,KAAK,WAAW,MAAM,EAAE,GACjC,KAAK,UAAU,MAAM,KAAK,GAAG,wBAAwB;QAChD,IAAI,KAAK,WAAW,aAAa,EAAE,GACxC,KAAK,YAAY,EAAE;QAEnB;EAEJ;CACF;;;;CAKA,MAAM,YAAY,SAAyB;EACzC,MAAM,QAAQ;EACd,OAAO,KAAK,KAAK,QAAQ;GACvB,MAAM,OAAO,KAAK,WAAW,EAAE;GAE/B,IAAI,aAAa,IAAI,KAAK,SAAS,SAAS,SAAS,UAAU,SAAS,IACtE;GAEF;EACF;EACA,IAAI,OAAO,OACT,MAAM,IAAI,cAAc;GAAE,SAAS,cAAc;GAAQ,UAAU;GAAO,OAAO;EAAK,CAAC;EAEzF,OAAO,KAAK,MAAM,OAAO,EAAE;CAC7B;CAEA,MAAM,mBAAyB;EAC7B,OAAO,KAAK,KAAK,UAAU,aAAa,KAAK,WAAW,EAAE,CAAC,GAAG;CAChE;CAEA,MAAM,sBAAsB,MAAc,cAA8B;EACtE,MAAM,QAAQ,KAAK;EAInB,IAAI,UAAU,QAAO,UAAU,KAC7B,MAAM,IAAI,cAAc;GAAE,SAAS,cAAc,KAAK;GAAwB,UAAU;GAAW,OAAO;EAAK,CAAC;EAGlH;EAEA,MAAM,MAAM,KAAK,QAAQ,SAAS,IAAI,EAAE;EAExC,IAAI,QAAQ,IACV,MAAM,IAAI,cAAc;GAAE,SAAS,qCAAqC,KAAK;GAAI,UAAU;GAAI,OAAO;EAAK,CAAC;EAG9G,MAAM,MAAM,KAAK,MAAM,IAAI,GAAG;EAC9B,KAAK,MAAM;EAEX,OAAO,eAAe,GAAG;CAC3B;CAEA,MAAM,qBAA+B;EAInC,MAAM,SAAmC,CAAC;EAC1C,IAAI,gBAAgB;EACpB,SAAS;GACP,WAAW;GACX,IAAI,MAAM,KAAK,QAAQ,MAAM,IAAI,cAAc;IAAE,SAAS;IAA0B,UAAU;IAAI,OAAO;GAAK,CAAC;GAC/G,IAAI,KAAK,WAAW,EAAE,MAAM,IAAI;IAC9B;IACA,OAAO;KAAE;KAAQ,aAAa;KAAO;IAAc;GACrD;GACA,IAAI,KAAK,WAAW,EAAE,MAAM,SAAS,KAAK,KAAK,OAAO,KAAK;IACzD,MAAM;IACN,OAAO;KAAE;KAAQ,aAAa;KAAM;IAAc;GACpD;GACA,MAAM,YAAY;GAClB,MAAM,OAAO,QAAQ,SAAS,gBAAgB,GAAG,aAAa,SAAS;GACvE,WAAW;GACX,IAAI,KAAK,WAAW,EAAE,MAAM,QAAQ,MAAM,IAAI,cAAc;IAAE,SAAS,cAAc,KAAK;IAAe,UAAU;IAAI,OAAO;GAAK,CAAC;GACpI;GACA,WAAW;GACX,OAAA,MAA0B,QAAQ,mBAAmB,MAAM,SAAS;GACpE,gBAAgB;EAClB;CACF;CAEA,MAAM,eAAe,UAA2B;EAC9C,IAAI,QAAQ,SAAS,UACnB,MAAM,IAAI,cAAc;GAAE,SAAS,sCAAsC,SAAS,SAAS;GAAI,UAAU;GAAI,OAAO;EAAK,CAAC;EAC5H,IAAI,KAAK,WAAW,EAAE,MAAM,IAAI,MAAM,IAAI,cAAc;GAAE,SAAS;GAAuB,UAAU;GAAI,OAAO;EAAK,CAAC;EACrH;EAEA,MAAM,OAAO,QAAQ,SAAS,cAAc,GAAG,WAAW,EAAE;EAC5D,MAAM,EAAE,QAAQ,aAAa,kBAAkB,aAAa;EAE5D,IAAI,aAAa,OAAO;GAAE;GAAM,OAAO,cAAc,QAAQ,eAAe,IAAI,KAAK;EAAE;EAMvF,MAAM,UAAU,YAAY,MAAM,QAAQ,KAAK;EAE/C,OAAO;GAAE;GAAM,OAAO,cAAc,QAAQ,eAAe,QAAQ,MAAM,QAAQ,WAAW;EAAE;CAChG;;;;;;;;;;;;CAaA,MAAM,eAAe,MAAc,QAAkC,UAA2B;EAC9F,IAAI,YAAY;EAChB,IAAI,cAAc;EAElB,SACE,QAAQ,gBAAgB,IAAI,GAA5B;GACE,KAAK;IACH,aAAa,YAAY;IACzB;GACF,KAAK;IACH,eAAe,IAAI;IACnB,OAAO;KAAE,MAAM;KAAW;IAAY;GACxC,KAAK;IACH,KAAK,UAAU,OAAO,KAAK,GAAG,SAAS;IACvC;GACF,KAAK;IACH,aAAa,UAAU;IACvB;GACF,KAAK;IACH,KAAK,UAAU,MAAM,KAAK,GAAG,wBAAwB;IACrD;GACF,KAAK;IACH,cAAc;IACd,SAAS,QAAQ,YAAY,QAAQ,CAAC,CAAC;EAG3C;CAEJ;;;;;;;;;;;CAYA,MAAM,mBAAmB,SAA4B;EACnD,IAAI,MAAM,KAAK,QACb,MAAM,IAAI,cAAc;GAAE,SAAS,qBAAqB,KAAK;GAAI,UAAU;GAAI,OAAO;EAAK,CAAC;EAE9F,IAAI,KAAK,WAAW,EAAE,MAAM,IAC1B,OAAO;EAET,IAAI,KAAK,WAAW,MAAM,EAAE,GAC1B,OAAO;EAET,IAAI,KAAK,WAAW,QAAQ,EAAE,GAC5B,OAAO;EAET,IAAI,KAAK,WAAW,aAAa,EAAE,GACjC,OAAO;EAET,IAAI,KAAK,WAAW,MAAM,EAAE,GAC1B,OAAO;EAET,IAAI,KAAK,WAAW,MAAM,EAAE,GAC1B,MAAM,IAAI,cAAc;GAAE,SAAS;GAAkD,UAAU;GAAI,OAAO;EAAK,CAAC;EAElH,OAAO;CACT;;;;;;CAOA,MAAM,kBAAkB,SAAuB;EAC7C,MAAM,aAAa;EACnB,MAAM;EACN,MAAM,UAAU,SAAS,cAAc;EACvC,IAAI,YAAY,MACd,MAAM,IAAI,cAAc;GAAE,SAAS,iBAAiB,QAAQ,oBAAoB,KAAK;GAAI,UAAU;GAAY,OAAO;EAAK,CAAC;EAE9H,WAAW;EACX,IAAI,KAAK,WAAW,EAAE,MAAM,IAC1B,MAAM,IAAI,cAAc;GAAE,SAAS,2BAA2B,QAAQ;GAAI,UAAU;GAAI,OAAO;EAAK,CAAC;EAEvG;CACF;;;;;;CAOA,MAAM,oBAA4B;EAChC,MAAM,OAAO,KAAK,QAAQ,KAAK,EAAE;EACjC,MAAM,MAAM,SAAS,KAAK,KAAK,SAAS;EACxC,MAAM,MAAM,eAAe,KAAK,MAAM,IAAI,GAAG,CAAC;EAC9C,KAAK;EACL,OAAO;CACT;;;;;;;CAQA,MAAM,kBAA0B;EAC9B,MAAM,MAAM,KAAK,QAAQ,OAAO,KAAK,CAAC;EACtC,IAAI,QAAQ,IAAI,MAAM,IAAI,cAAc;GAAE,SAAS;GAA8B,UAAU;GAAI,OAAO;EAAK,CAAC;EAC5G,MAAM,OAAO,KAAK,MAAM,KAAK,GAAG,GAAG;EACnC,KAAK,MAAM;EACX,OAAO;CACT;;;;;;;;CASA,MAAM,YAAY,QAAkC,UAAyB;EAC3E,MAAM,WAAW,OAAO,MAAM;EAC9B,IAAI,aAAa,KAAA,GAAW,OAAO,MAAM,QAAQ,MAAM;OAClD,IAAI,MAAM,QAAQ,QAAQ,GAAG,SAA8B,KAAK,MAAM,KAAK;OAC3E,OAAO,MAAM,QAAQ,CAAC,UAAU,MAAM,KAAK;CAClD;;;;CAKA,MAAM,iBAAiB,QAAkC,eAAwB,MAAc,gBAAmC;EAOhI,MAAM,UAAU,SAAS,qBAAqB,OAAO,KAAK,KAAK;EAE/D,IAAI,CAAC,iBAAiB,CAAC,aAGrB,OAAO;EAKT,IAAI,YAAY,IAAI,OAAO,YAAY;EACvC,OAAO;CACT;CAEA,SAAS;CACT,IAAI,MAAM,KAAK,UAAU,KAAK,WAAW,EAAE,MAAM,IAC/C,MAAM,IAAI,cAAc;EAAE,SAAS;EAAgC,UAAU;EAAI,OAAO;CAAK,CAAC;CAGhG,MAAM,OAAO,YAAY,CAAC;CAE1B,SAAS;CACT,IAAI,KAAK,KAAK,QACZ,MAAM,IAAI,cAAc;EAAE,SAAS;EAA6C,UAAU;EAAI,OAAO;CAAK,CAAC;CAG7G,OAAO;EAAE,MAAM,KAAK;EAAM,OAAO,KAAK;CAAM;AAC9C;;;;;;;;;;AAWA,MAAM,uBAAuB,MAAc,YAAwE;CACjH,IAAI;EACF,OAAO,OAAO,QAAQ,cAAc,MAAM,OAAO,CAAC;CACpD,SAAS,OAAO;EACd,IAAI,iBAAiB,eACnB,OAAO,OAAO,KAAK,KAAK;EAE1B,OAAO,OAAO,KAAK,IAAI,cAAc;GAAE,SAAS,UAAU,QAAQ,KAAK,IAAI,MAAM,UAAU,OAAO,KAAK;GAAG,UAAU;GAAI,OAAO;EAAK,CAAC,CAAC;CACxI;AACF;;;;;;;;;;;;AAaA,MAAM,kBAAkB,QAAwB;CAC9C,IAAI,IAAI,QAAQ,GAAG,MAAM,IAAI,OAAO;CAKpC,OAAO,OAAO,QAAQ,OAAO,cAAc,QAAQ,OAAO,GAAG,SAAS,GAAG,CAAC;AAC5E"}
@@ -0,0 +1,242 @@
1
+ import { TEXT_KEY, isAttributeKey } from "./conventions.js";
2
+ import { Predicate, Result, SchemaAST } from "effect";
3
+ //#region src/plain-value.ts
4
+ /**
5
+ * @description Whether an AST describes a structure - a struct, an array, or a union reachable to one - rather than a plain value. A node that describes a
6
+ * structure keeps a record as a record; a node that does not is where `#text` is read.
7
+ *
8
+ * @param ast - The derived StringTree AST to classify.
9
+ *
10
+ * @returns Whether the node is structural.
11
+ */
12
+ const isStructural = (ast) => {
13
+ if (SchemaAST.isSuspend(ast)) return isStructural(ast.thunk());
14
+ if (SchemaAST.isObjects(ast) || SchemaAST.isArrays(ast)) return true;
15
+ return SchemaAST.isUnion(ast) && ast.types.some(isStructural);
16
+ };
17
+ /**
18
+ * @description The AST a value is read against, with `Suspend` wrappers unwrapped so a recursive schema reaches its node.
19
+ *
20
+ * @param ast - The AST to unwrap.
21
+ *
22
+ * @returns The first node that is not a `Suspend`.
23
+ */
24
+ const resolveNode = (ast) => {
25
+ let node = ast;
26
+ while (SchemaAST.isSuspend(node)) node = node.thunk();
27
+ return node;
28
+ };
29
+ /**
30
+ * @description The AST a record value under one key is read against: the matching property signature, or the first index signature when the object is a record.
31
+ * `undefined` leaves the value alone, which is what an object with neither describes.
32
+ *
33
+ * @param node - The object node.
34
+ * @param key - The record key.
35
+ *
36
+ * @returns The AST for the value, or `undefined`.
37
+ */
38
+ const fieldAst = (node, key) => {
39
+ return node.propertySignatures.find((candidate) => candidate.name === key)?.type ?? node.indexSignatures[0]?.type;
40
+ };
41
+ /**
42
+ * @description The AST one member of an array is read against: the tuple element at that index, or the array's rest element.
43
+ *
44
+ * @param node - The array node.
45
+ * @param index - The member's index.
46
+ *
47
+ * @returns The AST for the member, or `undefined`.
48
+ */
49
+ const memberAst = (node, index) => node.elements[index] ?? node.rest[0];
50
+ /**
51
+ * @description The array node a value is read against. An optional repeated field derives as a union of an array and `undefined`, so an array node can sit behind
52
+ * a union rather than at the top; the first one reachable through the union's members is the one a repeated value belongs to.
53
+ *
54
+ * @param node - The resolved node to search.
55
+ *
56
+ * @returns The array node, or `undefined` when none is reachable.
57
+ */
58
+ const arrayNode = (node) => {
59
+ if (SchemaAST.isArrays(node)) return node;
60
+ if (SchemaAST.isUnion(node)) for (const member of node.types) {
61
+ const found = arrayNode(resolveNode(member));
62
+ if (found !== void 0) return found;
63
+ }
64
+ };
65
+ /**
66
+ * @description Whether an array's member is a structure - a struct, an array, or a union reachable to one - rather than a plain value. An empty element under such
67
+ * an array reads as one empty object, because a structural member cannot be empty character data; a plain-value member reads as an empty array
68
+ * instead.
69
+ *
70
+ * @param node - The array node.
71
+ *
72
+ * @returns Whether the member is structural.
73
+ */
74
+ const arrayMemberIsStructural = (node) => {
75
+ const member = memberAst(node, 0);
76
+ return member !== void 0 && isStructural(member);
77
+ };
78
+ /**
79
+ * @description The value an empty element under an array field reads as: one empty object when the member is structural, so a member the schema requires still
80
+ * reads, and an empty array otherwise.
81
+ *
82
+ * @param node - The array node.
83
+ *
84
+ * @returns The array to decode.
85
+ */
86
+ const emptyArrayElement = (node) => arrayMemberIsStructural(node) ? [{}] : [];
87
+ /**
88
+ * @description Whether a parsed element carries nothing at all: no character data, no attributes and no children. The parser reduces such an element to an empty
89
+ * string, and the codec reduces one whose only attributes were namespace declarations to an empty record. An array field reads either as one empty
90
+ * object or as an empty array, depending on whether its member is structural.
91
+ *
92
+ * @param value - The parsed value.
93
+ *
94
+ * @returns Whether the element is empty.
95
+ */
96
+ const isEmptyElement = (value) => value === "" || Predicate.isReadonlyObject(value) && Object.keys(value).length === 0;
97
+ /**
98
+ * @description The path of a field, for a failure message. The root has no name, so its fields are named on their own.
99
+ *
100
+ * @param parent - The parent's path.
101
+ * @param key - The field's key.
102
+ *
103
+ * @returns The field's path.
104
+ */
105
+ const childPath = (parent, key) => parent === "" ? key : `${parent}.${key}`;
106
+ /**
107
+ * @description Reads the character data of a record that wants a plain value, discarding the attributes around it. A record that also carries a child element is
108
+ * refused, because a plain value has nowhere to put one and dropping it would lose a field the document carried. A record with no `#text` at all is
109
+ * left for the decoder to refuse, which reports the attributes it found instead of a value.
110
+ *
111
+ * @param record - The record to read.
112
+ * @param path - The path of the value, for the failure message.
113
+ *
114
+ * @returns The `#text` value, or the message describing the child element that makes a plain value impossible.
115
+ */
116
+ const readCharacterData = (record, path) => {
117
+ if (!("#text" in record)) return Result.succeed(record);
118
+ const child = Object.keys(record).find((key) => key !== "#text" && !isAttributeKey(key));
119
+ if (child !== void 0) return Result.fail(`the field "${path === "" ? "root" : path}" wants a plain value, but the element also carries the child element "${child}"; only attributes are discarded alongside ${TEXT_KEY}`);
120
+ return Result.succeed(record[TEXT_KEY]);
121
+ };
122
+ /**
123
+ * @description Folds the fields of a record against the object node that describes them, so each field is normalized by its own AST.
124
+ *
125
+ * @param record - The record to fold.
126
+ * @param node - The object node.
127
+ * @param path - The path of the record, for a failure message.
128
+ *
129
+ * @returns The folded record, or the first field's failure.
130
+ */
131
+ const normalizeFields = (record, node, path) => {
132
+ const out = {};
133
+ for (const [key, child] of Object.entries(record)) {
134
+ const field = fieldAst(node, key);
135
+ if (field === void 0 || child === void 0) {
136
+ out[key] = child;
137
+ continue;
138
+ }
139
+ const normalized = normalizePlainValue(child, field, childPath(path, key));
140
+ if (Result.isFailure(normalized)) return normalized;
141
+ out[key] = normalized.success;
142
+ }
143
+ return Result.succeed(out);
144
+ };
145
+ /**
146
+ * @description Folds every member of an array against the array node that describes them. A member that is an empty element - the parser reduces it to an empty
147
+ * string, or to an empty record once the codec drops the namespace declarations it carried - cannot be a structural member, so it is kept as one
148
+ * empty object. That is how the renderer writes an empty array, and how a repeated empty tag reads back as one empty object per element.
149
+ *
150
+ * @param value - The array to fold.
151
+ * @param node - The array node.
152
+ * @param path - The path of the array, for a failure message.
153
+ *
154
+ * @returns The folded array, or the first member's failure.
155
+ */
156
+ const normalizeMembers = (value, node, path) => {
157
+ const out = [];
158
+ for (let index = 0; index < value.length; index++) {
159
+ const element = memberAst(node, index);
160
+ if (element === void 0) {
161
+ out.push(value[index]);
162
+ continue;
163
+ }
164
+ if (isEmptyElement(value[index]) && isStructural(element)) {
165
+ out.push({});
166
+ continue;
167
+ }
168
+ const normalized = normalizePlainValue(value[index], element, `${path}[${index}]`);
169
+ if (Result.isFailure(normalized)) return normalized;
170
+ out.push(normalized.success);
171
+ }
172
+ return Result.succeed(out);
173
+ };
174
+ /**
175
+ * @description Folds a record against every object member of a union, so a plain value nested under any branch is still read from its character data. Which branch
176
+ * the value belongs to is the decoder's job, so the first branch that folds the record cleanly wins; a branch that refuses the record - because a
177
+ * field it wants as a plain value also carries a child element - is skipped. When the union has no object member, or none of them folds the record,
178
+ * the record is left for the decoder to read as a plain value.
179
+ *
180
+ * @param record - The record to fold.
181
+ * @param node - The union node.
182
+ * @param path - The path of the record, for a failure message.
183
+ *
184
+ * @returns The folded value, the first object member's failure, or the record unchanged.
185
+ */
186
+ const normalizeUnion = (record, node, path) => {
187
+ let failure;
188
+ let sawObject = false;
189
+ for (const member of node.types) {
190
+ const resolved = resolveNode(member);
191
+ if (!SchemaAST.isObjects(resolved)) continue;
192
+ sawObject = true;
193
+ const normalized = normalizeFields(record, resolved, path);
194
+ if (Result.isSuccess(normalized)) return normalized;
195
+ failure ??= normalized;
196
+ }
197
+ if (failure !== void 0) return failure;
198
+ return sawObject ? Result.succeed(record) : readCharacterData(record, path);
199
+ };
200
+ /**
201
+ * @description Folds a record against the node that describes it. An object node is folded field by field; a union folds against its object members so a plain
202
+ * value nested under any branch is read; anything else wants a plain value and reads the character data.
203
+ *
204
+ * @param record - The record to fold.
205
+ * @param node - The resolved node.
206
+ * @param path - The path of the record, for a failure message.
207
+ *
208
+ * @returns The folded value, or the failure that makes it impossible.
209
+ */
210
+ const normalizeRecord = (record, node, path) => {
211
+ if (SchemaAST.isObjects(node)) return normalizeFields(record, node, path);
212
+ if (SchemaAST.isUnion(node)) return normalizeUnion(record, node, path);
213
+ if (isStructural(node)) return Result.succeed(record);
214
+ return readCharacterData(record, path);
215
+ };
216
+ /**
217
+ * @description Folds a parsed value against the derived StringTree AST, so a plain value can be read from an element that carries attributes. An empty element
218
+ * under an array field reads as one empty object when the member is structural, and as an empty array otherwise.
219
+ *
220
+ * @param value - The value the parser produced, after namespaces were resolved.
221
+ * @param ast - The derived StringTree AST the decoder will read the value with.
222
+ * @param path - The path of this value, for a failure message.
223
+ *
224
+ * @returns The value to decode, or the message describing why a plain value could not be read.
225
+ */
226
+ const normalizePlainValue = (value, ast, path) => {
227
+ if (Predicate.isUndefined(value)) return Result.succeed(value);
228
+ const node = resolveNode(ast);
229
+ const arrays = arrayNode(node);
230
+ if (Predicate.isString(value)) {
231
+ if (arrays !== void 0 && value === "") return Result.succeed(emptyArrayElement(arrays));
232
+ return Result.succeed(value);
233
+ }
234
+ if (Array.isArray(value)) return arrays !== void 0 ? normalizeMembers(value, arrays, path) : Result.succeed(value);
235
+ if (!Predicate.isReadonlyObject(value)) return Result.succeed(value);
236
+ if (arrays !== void 0 && isEmptyElement(value)) return Result.succeed(emptyArrayElement(arrays));
237
+ return normalizeRecord(value, node, path);
238
+ };
239
+ //#endregion
240
+ export { normalizePlainValue };
241
+
242
+ //# sourceMappingURL=plain-value.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"plain-value.js","names":[],"sources":["../src/plain-value.ts"],"sourcesContent":["// A plain value read from an element that carries more than character data.\n//\n// The parser reduces an element with no attributes and no children to its\n// character data, so a field like `Schema.String` is usually handed a bare\n// string. An element that also carries attributes does not reduce: it arrives\n// as a record holding the attributes, the `#text` character data, and any child\n// elements, because the value model has nowhere else to put them:\n//\n// { \"@lang\": \"en\", \"#text\": \"Dune\" }\n//\n// Effect's `Schema.toCodecStringTree` derivation knows nothing about the `#text`\n// convention, so where a field wants a scalar it sees a record and fails with an\n// `InvalidType`. This module bridges the two. Walking the derived StringTree AST\n// beside the parsed value:\n//\n// - a node that wants character data takes the `#text` key and discards the\n// attributes, because an attribute is not part of the value the schema\n// describes;\n// - a node that wants character data but also finds a child element fails,\n// because a plain value cannot hold one and silently dropping it would lose\n// a field the document actually carried;\n// - a node that describes a struct, array or record keeps its record and\n// recurses, because `@`-prefixed fields and `#text` are ordinary field names\n// to such a schema;\n// - a union rewrites its record against each object member in turn, so a plain\n// value nested under any branch is still read; a repeated field derives as a\n// union of an array and `undefined`, so a record is also read through the\n// first array node reachable behind a union;\n// - an empty element under an array field reads as one empty object when the\n// member is structural, because an empty array renders as a single empty\n// element (see `renderRepeated`) and a member the schema requires must still\n// read; a plain-value member reads as an empty array.\n//\n// The AST is the one `Schema.toCodecStringTree` produces for the source schema,\n// so it is the same shape the decoder is about to read; this pass only rewrites\n// the value, it does not derive a schema.\n\nimport { Predicate, Result, SchemaAST } from 'effect';\n\nimport type { XmlRecord, XmlValue } from './xml-value.ts';\n\nimport { isAttributeKey, TEXT_KEY } from './conventions.ts';\n\n/**\n * @description Whether an AST describes a structure - a struct, an array, or a union reachable to one - rather than a plain value. A node that describes a\n * structure keeps a record as a record; a node that does not is where `#text` is read.\n *\n * @param ast - The derived StringTree AST to classify.\n *\n * @returns Whether the node is structural.\n */\nconst isStructural = (ast: SchemaAST.AST): boolean => {\n if (SchemaAST.isSuspend(ast)) return isStructural(ast.thunk());\n if (SchemaAST.isObjects(ast) || SchemaAST.isArrays(ast)) return true;\n return SchemaAST.isUnion(ast) && ast.types.some(isStructural);\n};\n\n/**\n * @description The AST a value is read against, with `Suspend` wrappers unwrapped so a recursive schema reaches its node.\n *\n * @param ast - The AST to unwrap.\n *\n * @returns The first node that is not a `Suspend`.\n */\nconst resolveNode = (ast: SchemaAST.AST): SchemaAST.AST => {\n let node = ast;\n while (SchemaAST.isSuspend(node)) node = node.thunk();\n return node;\n};\n\n/**\n * @description The AST a record value under one key is read against: the matching property signature, or the first index signature when the object is a record.\n * `undefined` leaves the value alone, which is what an object with neither describes.\n *\n * @param node - The object node.\n * @param key - The record key.\n *\n * @returns The AST for the value, or `undefined`.\n */\nconst fieldAst = (node: SchemaAST.Objects, key: string): SchemaAST.AST | undefined => {\n const property = node.propertySignatures.find(candidate => candidate.name === key);\n return property?.type ?? node.indexSignatures[0]?.type;\n};\n\n/**\n * @description The AST one member of an array is read against: the tuple element at that index, or the array's rest element.\n *\n * @param node - The array node.\n * @param index - The member's index.\n *\n * @returns The AST for the member, or `undefined`.\n */\nconst memberAst = (node: SchemaAST.Arrays, index: number): SchemaAST.AST | undefined => node.elements[index] ?? node.rest[0];\n\n/**\n * @description The array node a value is read against. An optional repeated field derives as a union of an array and `undefined`, so an array node can sit behind\n * a union rather than at the top; the first one reachable through the union's members is the one a repeated value belongs to.\n *\n * @param node - The resolved node to search.\n *\n * @returns The array node, or `undefined` when none is reachable.\n */\nconst arrayNode = (node: SchemaAST.AST): SchemaAST.Arrays | undefined => {\n if (SchemaAST.isArrays(node)) return node;\n if (SchemaAST.isUnion(node)) {\n for (const member of node.types) {\n const found = arrayNode(resolveNode(member));\n if (found !== undefined) return found;\n }\n }\n return undefined;\n};\n\n/**\n * @description Whether an array's member is a structure - a struct, an array, or a union reachable to one - rather than a plain value. An empty element under such\n * an array reads as one empty object, because a structural member cannot be empty character data; a plain-value member reads as an empty array\n * instead.\n *\n * @param node - The array node.\n *\n * @returns Whether the member is structural.\n */\nconst arrayMemberIsStructural = (node: SchemaAST.Arrays): boolean => {\n const member = memberAst(node, 0);\n return member !== undefined && isStructural(member);\n};\n\n/**\n * @description The value an empty element under an array field reads as: one empty object when the member is structural, so a member the schema requires still\n * reads, and an empty array otherwise.\n *\n * @param node - The array node.\n *\n * @returns The array to decode.\n */\nconst emptyArrayElement = (node: SchemaAST.Arrays): XmlValue => (arrayMemberIsStructural(node) ? [{}] : []);\n\n/**\n * @description Whether a parsed element carries nothing at all: no character data, no attributes and no children. The parser reduces such an element to an empty\n * string, and the codec reduces one whose only attributes were namespace declarations to an empty record. An array field reads either as one empty\n * object or as an empty array, depending on whether its member is structural.\n *\n * @param value - The parsed value.\n *\n * @returns Whether the element is empty.\n */\nconst isEmptyElement = (value: XmlValue): boolean => value === '' || (Predicate.isReadonlyObject(value) && Object.keys(value).length === 0);\n\n/**\n * @description The path of a field, for a failure message. The root has no name, so its fields are named on their own.\n *\n * @param parent - The parent's path.\n * @param key - The field's key.\n *\n * @returns The field's path.\n */\nconst childPath = (parent: string, key: string): string => (parent === '' ? key : `${parent}.${key}`);\n\n/**\n * @description Reads the character data of a record that wants a plain value, discarding the attributes around it. A record that also carries a child element is\n * refused, because a plain value has nowhere to put one and dropping it would lose a field the document carried. A record with no `#text` at all is\n * left for the decoder to refuse, which reports the attributes it found instead of a value.\n *\n * @param record - The record to read.\n * @param path - The path of the value, for the failure message.\n *\n * @returns The `#text` value, or the message describing the child element that makes a plain value impossible.\n */\nconst readCharacterData = (record: XmlRecord, path: string): Result.Result<XmlValue, string> => {\n if (!(TEXT_KEY in record)) return Result.succeed(record);\n\n const child = Object.keys(record).find(key => key !== TEXT_KEY && !isAttributeKey(key));\n if (child !== undefined) {\n return Result.fail(\n `the field \"${path === '' ? 'root' : path}\" wants a plain value, but the element also carries the child element \"${child}\"; only attributes are discarded alongside ${TEXT_KEY}`\n );\n }\n\n return Result.succeed(record[TEXT_KEY]);\n};\n\n/**\n * @description Folds the fields of a record against the object node that describes them, so each field is normalized by its own AST.\n *\n * @param record - The record to fold.\n * @param node - The object node.\n * @param path - The path of the record, for a failure message.\n *\n * @returns The folded record, or the first field's failure.\n */\nconst normalizeFields = (record: XmlRecord, node: SchemaAST.Objects, path: string): Result.Result<XmlValue, string> => {\n const out: Record<string, XmlValue> = {};\n for (const [key, child] of Object.entries(record)) {\n const field = fieldAst(node, key);\n if (field === undefined || child === undefined) {\n out[key] = child;\n continue;\n }\n const normalized = normalizePlainValue(child, field, childPath(path, key));\n if (Result.isFailure(normalized)) return normalized;\n out[key] = normalized.success;\n }\n return Result.succeed(out);\n};\n\n/**\n * @description Folds every member of an array against the array node that describes them. A member that is an empty element - the parser reduces it to an empty\n * string, or to an empty record once the codec drops the namespace declarations it carried - cannot be a structural member, so it is kept as one\n * empty object. That is how the renderer writes an empty array, and how a repeated empty tag reads back as one empty object per element.\n *\n * @param value - The array to fold.\n * @param node - The array node.\n * @param path - The path of the array, for a failure message.\n *\n * @returns The folded array, or the first member's failure.\n */\nconst normalizeMembers = (value: ReadonlyArray<XmlValue>, node: SchemaAST.Arrays, path: string): Result.Result<XmlValue, string> => {\n const out: Array<XmlValue> = [];\n for (let index = 0; index < value.length; index++) {\n const element = memberAst(node, index);\n if (element === undefined) {\n out.push(value[index]);\n continue;\n }\n if (isEmptyElement(value[index]) && isStructural(element)) {\n out.push({});\n continue;\n }\n const normalized = normalizePlainValue(value[index], element, `${path}[${index}]`);\n if (Result.isFailure(normalized)) return normalized;\n out.push(normalized.success);\n }\n return Result.succeed(out);\n};\n\n/**\n * @description Folds a record against every object member of a union, so a plain value nested under any branch is still read from its character data. Which branch\n * the value belongs to is the decoder's job, so the first branch that folds the record cleanly wins; a branch that refuses the record - because a\n * field it wants as a plain value also carries a child element - is skipped. When the union has no object member, or none of them folds the record,\n * the record is left for the decoder to read as a plain value.\n *\n * @param record - The record to fold.\n * @param node - The union node.\n * @param path - The path of the record, for a failure message.\n *\n * @returns The folded value, the first object member's failure, or the record unchanged.\n */\nconst normalizeUnion = (record: XmlRecord, node: SchemaAST.Union, path: string): Result.Result<XmlValue, string> => {\n let failure: Result.Result<XmlValue, string> | undefined;\n let sawObject = false;\n for (const member of node.types) {\n const resolved = resolveNode(member);\n if (!SchemaAST.isObjects(resolved)) continue;\n sawObject = true;\n const normalized = normalizeFields(record, resolved, path);\n if (Result.isSuccess(normalized)) return normalized;\n failure ??= normalized;\n }\n if (failure !== undefined) return failure;\n return sawObject ? Result.succeed(record) : readCharacterData(record, path);\n};\n\n/**\n * @description Folds a record against the node that describes it. An object node is folded field by field; a union folds against its object members so a plain\n * value nested under any branch is read; anything else wants a plain value and reads the character data.\n *\n * @param record - The record to fold.\n * @param node - The resolved node.\n * @param path - The path of the record, for a failure message.\n *\n * @returns The folded value, or the failure that makes it impossible.\n */\nconst normalizeRecord = (record: XmlRecord, node: SchemaAST.AST, path: string): Result.Result<XmlValue, string> => {\n if (SchemaAST.isObjects(node)) return normalizeFields(record, node, path);\n if (SchemaAST.isUnion(node)) return normalizeUnion(record, node, path);\n if (isStructural(node)) return Result.succeed(record);\n return readCharacterData(record, path);\n};\n\n/**\n * @description Folds a parsed value against the derived StringTree AST, so a plain value can be read from an element that carries attributes. An empty element\n * under an array field reads as one empty object when the member is structural, and as an empty array otherwise.\n *\n * @param value - The value the parser produced, after namespaces were resolved.\n * @param ast - The derived StringTree AST the decoder will read the value with.\n * @param path - The path of this value, for a failure message.\n *\n * @returns The value to decode, or the message describing why a plain value could not be read.\n */\nexport const normalizePlainValue = (value: XmlValue, ast: SchemaAST.AST, path: string): Result.Result<XmlValue, string> => {\n if (Predicate.isUndefined(value)) return Result.succeed(value);\n\n const node = resolveNode(ast);\n const arrays = arrayNode(node);\n\n if (Predicate.isString(value)) {\n // An empty element under an array field is how the renderer writes an empty array; a structural member still reads as one empty object.\n if (arrays !== undefined && value === '') return Result.succeed(emptyArrayElement(arrays));\n return Result.succeed(value);\n }\n\n if (Array.isArray(value)) {\n return arrays !== undefined ? normalizeMembers(value, arrays, path) : Result.succeed(value);\n }\n\n if (!Predicate.isReadonlyObject(value)) return Result.succeed(value);\n\n // The same empty element, after the codec dropped the namespace declarations it carried, arrives as an empty record.\n if (arrays !== undefined && isEmptyElement(value)) return Result.succeed(emptyArrayElement(arrays));\n\n return normalizeRecord(value as XmlRecord, node, path);\n};\n"],"mappings":";;;;;;;;;;;AAmDA,MAAM,gBAAgB,QAAgC;CACpD,IAAI,UAAU,UAAU,GAAG,GAAG,OAAO,aAAa,IAAI,MAAM,CAAC;CAC7D,IAAI,UAAU,UAAU,GAAG,KAAK,UAAU,SAAS,GAAG,GAAG,OAAO;CAChE,OAAO,UAAU,QAAQ,GAAG,KAAK,IAAI,MAAM,KAAK,YAAY;AAC9D;;;;;;;;AASA,MAAM,eAAe,QAAsC;CACzD,IAAI,OAAO;CACX,OAAO,UAAU,UAAU,IAAI,GAAG,OAAO,KAAK,MAAM;CACpD,OAAO;AACT;;;;;;;;;;AAWA,MAAM,YAAY,MAAyB,QAA2C;CAEpF,OADiB,KAAK,mBAAmB,MAAK,cAAa,UAAU,SAAS,GAChE,CAAC,EAAE,QAAQ,KAAK,gBAAgB,EAAE,EAAE;AACpD;;;;;;;;;AAUA,MAAM,aAAa,MAAwB,UAA6C,KAAK,SAAS,UAAU,KAAK,KAAK;;;;;;;;;AAU1H,MAAM,aAAa,SAAsD;CACvE,IAAI,UAAU,SAAS,IAAI,GAAG,OAAO;CACrC,IAAI,UAAU,QAAQ,IAAI,GACxB,KAAK,MAAM,UAAU,KAAK,OAAO;EAC/B,MAAM,QAAQ,UAAU,YAAY,MAAM,CAAC;EAC3C,IAAI,UAAU,KAAA,GAAW,OAAO;CAClC;AAGJ;;;;;;;;;;AAWA,MAAM,2BAA2B,SAAoC;CACnE,MAAM,SAAS,UAAU,MAAM,CAAC;CAChC,OAAO,WAAW,KAAA,KAAa,aAAa,MAAM;AACpD;;;;;;;;;AAUA,MAAM,qBAAqB,SAAsC,wBAAwB,IAAI,IAAI,CAAC,CAAC,CAAC,IAAI,CAAC;;;;;;;;;;AAWzG,MAAM,kBAAkB,UAA6B,UAAU,MAAO,UAAU,iBAAiB,KAAK,KAAK,OAAO,KAAK,KAAK,CAAC,CAAC,WAAW;;;;;;;;;AAUzI,MAAM,aAAa,QAAgB,QAAyB,WAAW,KAAK,MAAM,GAAG,OAAO,GAAG;;;;;;;;;;;AAY/F,MAAM,qBAAqB,QAAmB,SAAkD;CAC9F,IAAI,EAAA,WAAc,SAAS,OAAO,OAAO,QAAQ,MAAM;CAEvD,MAAM,QAAQ,OAAO,KAAK,MAAM,CAAC,CAAC,MAAK,QAAO,QAAA,WAAoB,CAAC,eAAe,GAAG,CAAC;CACtF,IAAI,UAAU,KAAA,GACZ,OAAO,OAAO,KACZ,cAAc,SAAS,KAAK,SAAS,KAAK,yEAAyE,MAAM,6CAA6C,UACxK;CAGF,OAAO,OAAO,QAAQ,OAAO,SAAS;AACxC;;;;;;;;;;AAWA,MAAM,mBAAmB,QAAmB,MAAyB,SAAkD;CACrH,MAAM,MAAgC,CAAC;CACvC,KAAK,MAAM,CAAC,KAAK,UAAU,OAAO,QAAQ,MAAM,GAAG;EACjD,MAAM,QAAQ,SAAS,MAAM,GAAG;EAChC,IAAI,UAAU,KAAA,KAAa,UAAU,KAAA,GAAW;GAC9C,IAAI,OAAO;GACX;EACF;EACA,MAAM,aAAa,oBAAoB,OAAO,OAAO,UAAU,MAAM,GAAG,CAAC;EACzE,IAAI,OAAO,UAAU,UAAU,GAAG,OAAO;EACzC,IAAI,OAAO,WAAW;CACxB;CACA,OAAO,OAAO,QAAQ,GAAG;AAC3B;;;;;;;;;;;;AAaA,MAAM,oBAAoB,OAAgC,MAAwB,SAAkD;CAClI,MAAM,MAAuB,CAAC;CAC9B,KAAK,IAAI,QAAQ,GAAG,QAAQ,MAAM,QAAQ,SAAS;EACjD,MAAM,UAAU,UAAU,MAAM,KAAK;EACrC,IAAI,YAAY,KAAA,GAAW;GACzB,IAAI,KAAK,MAAM,MAAM;GACrB;EACF;EACA,IAAI,eAAe,MAAM,MAAM,KAAK,aAAa,OAAO,GAAG;GACzD,IAAI,KAAK,CAAC,CAAC;GACX;EACF;EACA,MAAM,aAAa,oBAAoB,MAAM,QAAQ,SAAS,GAAG,KAAK,GAAG,MAAM,EAAE;EACjF,IAAI,OAAO,UAAU,UAAU,GAAG,OAAO;EACzC,IAAI,KAAK,WAAW,OAAO;CAC7B;CACA,OAAO,OAAO,QAAQ,GAAG;AAC3B;;;;;;;;;;;;;AAcA,MAAM,kBAAkB,QAAmB,MAAuB,SAAkD;CAClH,IAAI;CACJ,IAAI,YAAY;CAChB,KAAK,MAAM,UAAU,KAAK,OAAO;EAC/B,MAAM,WAAW,YAAY,MAAM;EACnC,IAAI,CAAC,UAAU,UAAU,QAAQ,GAAG;EACpC,YAAY;EACZ,MAAM,aAAa,gBAAgB,QAAQ,UAAU,IAAI;EACzD,IAAI,OAAO,UAAU,UAAU,GAAG,OAAO;EACzC,YAAY;CACd;CACA,IAAI,YAAY,KAAA,GAAW,OAAO;CAClC,OAAO,YAAY,OAAO,QAAQ,MAAM,IAAI,kBAAkB,QAAQ,IAAI;AAC5E;;;;;;;;;;;AAYA,MAAM,mBAAmB,QAAmB,MAAqB,SAAkD;CACjH,IAAI,UAAU,UAAU,IAAI,GAAG,OAAO,gBAAgB,QAAQ,MAAM,IAAI;CACxE,IAAI,UAAU,QAAQ,IAAI,GAAG,OAAO,eAAe,QAAQ,MAAM,IAAI;CACrE,IAAI,aAAa,IAAI,GAAG,OAAO,OAAO,QAAQ,MAAM;CACpD,OAAO,kBAAkB,QAAQ,IAAI;AACvC;;;;;;;;;;;AAYA,MAAa,uBAAuB,OAAiB,KAAoB,SAAkD;CACzH,IAAI,UAAU,YAAY,KAAK,GAAG,OAAO,OAAO,QAAQ,KAAK;CAE7D,MAAM,OAAO,YAAY,GAAG;CAC5B,MAAM,SAAS,UAAU,IAAI;CAE7B,IAAI,UAAU,SAAS,KAAK,GAAG;EAE7B,IAAI,WAAW,KAAA,KAAa,UAAU,IAAI,OAAO,OAAO,QAAQ,kBAAkB,MAAM,CAAC;EACzF,OAAO,OAAO,QAAQ,KAAK;CAC7B;CAEA,IAAI,MAAM,QAAQ,KAAK,GACrB,OAAO,WAAW,KAAA,IAAY,iBAAiB,OAAO,QAAQ,IAAI,IAAI,OAAO,QAAQ,KAAK;CAG5F,IAAI,CAAC,UAAU,iBAAiB,KAAK,GAAG,OAAO,OAAO,QAAQ,KAAK;CAGnE,IAAI,WAAW,KAAA,KAAa,eAAe,KAAK,GAAG,OAAO,OAAO,QAAQ,kBAAkB,MAAM,CAAC;CAElG,OAAO,gBAAgB,OAAoB,MAAM,IAAI;AACvD"}
package/dist/render.d.ts CHANGED
@@ -84,7 +84,7 @@ export declare const escapeAttribute: (value: string) => string;
84
84
  * A record becomes an element:
85
85
  *
86
86
  * - `@`-prefixed keys become attributes, the reserved `#text` key becomes character data, and every other key becomes a child element.
87
- * - An array repeats its name — a document whose root value is an array wraps it in the root element and names each member `itemName`.
87
+ * - An array repeats its name. A document whose root value is an array wraps it in the root element and names each member `itemName`.
88
88
  * - A string is character data. The walk is synchronous, and what can go wrong is reported by throwing an {@link XmlRenderError}; {@link renderXml}
89
89
  * folds that into the effect's typed error channel. A caller not already in an `Effect` runs it with `Effect.runSync`, which throws the failure it
90
90
  * produced.
@@ -1 +1 @@
1
- {"version":3,"file":"render.d.ts","names":[],"sources":["../src/render.ts"],"mappings":";;;;;;;;;iBA+EiB;;;;;;;WAON;;;;;;WAOA;;;;;;WAOA;;;;WAKA;;;;;;WAOA;;;;;;;WAQA;;;;;;WAOA,OAAO;;;;;;WAOP,aAAa;;;;;;WAOb;;;;;;;;;qBAiHE,aAAU;;;;;;;;qBASV,kBAAe;;;;;;;;;;;;;;;;qBAmDf,YAAS,OAAW,UAAQ,UAAW,qBAAwB,OAAO,eAAe"}
1
+ {"version":3,"file":"render.d.ts","names":[],"sources":["../src/render.ts"],"mappings":";;;;;;;;;iBAgFiB;;;;;;;WAON;;;;;;WAOA;;;;;;WAOA;;;;WAKA;;;;;;WAOA;;;;;;;WAQA;;;;;;WAOA,OAAO;;;;;;WAOP,aAAa;;;;;;WAOb;;;;;;;;;qBAgHE,aAAU;;;;;;;;qBASV,kBAAe;;;;;;;;;;;;;;;;qBAmDf,YAAS,OAAW,UAAQ,UAAW,qBAAwB,OAAO,eAAe"}
package/dist/render.js CHANGED
@@ -42,20 +42,21 @@ const buildTable = (extra) => {
42
42
  const TEXT_TABLE = buildTable({});
43
43
  const ATTRIBUTE_TABLE = buildTable(ATTRIBUTE_WHITESPACE);
44
44
  /**
45
- * @description The characters each table escapes, as a pattern rather than as a set of replacement passes. Finding the first one with a pattern is what makes
46
- * clean text cheap: V8 compiles a single character class into a scan that is several times faster than a JavaScript loop reading the same string a
47
- * code unit at a time, and clean text is most text. `render 20k of clean text` in `bench/codec.bench.ts` is the row that says so — a hand-written
48
- * loop over the same twenty thousand characters is roughly two and a half times slower. Neither pattern is global, so `exec` ignores `lastIndex` and
49
- * always starts at the beginning. One module-level instance of each is therefore safe to reuse, and nothing has to be reset between calls.
45
+ * @description The characters each table escapes, as a pattern rather than as a set of replacement passes. A single pattern finds the first character that needs
46
+ * replacing, which keeps clean text cheap: V8 compiles a single character class into a scan that is several times faster than a JavaScript loop
47
+ * reading the same string a code unit at a time, and clean text is most text. The `render 20k of clean text` benchmark in `bench/codec.bench.ts`
48
+ * measures that difference: a hand-written loop over the same twenty thousand characters is roughly two and a half times slower. Neither pattern is
49
+ * global, so `exec` ignores `lastIndex` and always starts at the beginning. One module-level instance of each is therefore safe to reuse, and nothing
50
+ * has to be reset between calls.
50
51
  */
51
52
  const TEXT_UNSAFE = /[<>&"']/;
52
53
  const ATTRIBUTE_UNSAFE = /[<>&"'\n\r\t]/;
53
54
  /**
54
55
  * @description Builds the name resolver for one render. Every element and every attribute name goes through here, and a document repeats names: a thousand
55
56
  * `<item>` elements, or the same `id` on every row. A validator that runs a regex per occurrence pays that cost a thousand times for one answer, so
56
- * the first result is remembered and the rest are lookups. It also keeps the mode and version in one place, which is what stops a caller from
57
- * resolving a name with different settings than the render it is part of. A cache miss calls {@link resolveNameSync}, which throws an
58
- * {@link XmlParseError} in `'error'` mode; {@link renderXml} catches it and reports it as an {@link XmlRenderError}.
57
+ * the first result is remembered and the rest are lookups. Keeping the mode and version in one place also stops a caller from resolving a name with
58
+ * different settings than the render it is part of. A cache miss calls {@link resolveNameSync}, which throws an {@link XmlParseError} in `'error'`
59
+ * mode; {@link renderXml} catches it and reports it as an {@link XmlRenderError}.
59
60
  *
60
61
  * @param options - Resolved render options.
61
62
  *
@@ -77,8 +78,7 @@ const makeNamer = (options) => {
77
78
  /**
78
79
  * @description A boolean option's value, with an absent one read as the default. The three boolean options are spelled through here rather than through a `??` of
79
80
  * their own, so the table below reads as a list of what each option _is_ instead of a list of nine separate decisions about what an omitted option
80
- * means — and so a reader looking for "which options are on by default" finds three words rather than three mixes of `?? true` and `?? false` to
81
- * read.
81
+ * means, and a reader looking for "which options are on by default" finds three words rather than three mixes of `?? true` and `?? false` to read.
82
82
  *
83
83
  * @param value - The option as the caller wrote it, or `undefined` when the caller left it out.
84
84
  * @param fallback - The value to use when the caller left it out.
@@ -148,10 +148,10 @@ const escapeText = (value) => escape(value, TEXT_UNSAFE, TEXT_TABLE);
148
148
  const escapeAttribute = (value) => escape(value, ATTRIBUTE_UNSAFE, ATTRIBUTE_TABLE);
149
149
  /**
150
150
  * @description Replaces every character the table has an entry for, in one pass over the string. The pattern finds the first character that needs replacing, and a
151
- * string with none is handed straight back — which is the common case, and the one the pattern is there to make fast. From there the rest of the
152
- * string is copied in runs between the replacements rather than a character at a time, so the cost is one pattern scan, one copy, and one
153
- * concatenation per replacement, rather than a whole pass per character class. Only ASCII is looked up. XML carries every other character natively,
154
- * and a code unit above 127 has no entity an XML parser is required to know.
151
+ * string with none is handed straight back. That is the common case, and the pattern exists to keep it fast. From there the rest of the string is
152
+ * copied in runs between the replacements rather than a character at a time, so the cost is one pattern scan, one copy, and one concatenation per
153
+ * replacement, rather than a whole pass per character class. Only ASCII is looked up. XML carries every other character natively, and a code unit
154
+ * above 127 has no entity an XML parser is required to know.
155
155
  *
156
156
  * @param value - The text to escape.
157
157
  * @param pattern - Matches the first character that needs replacing.
@@ -181,7 +181,7 @@ const escape = (value, pattern, table) => {
181
181
  * A record becomes an element:
182
182
  *
183
183
  * - `@`-prefixed keys become attributes, the reserved `#text` key becomes character data, and every other key becomes a child element.
184
- * - An array repeats its name — a document whose root value is an array wraps it in the root element and names each member `itemName`.
184
+ * - An array repeats its name. A document whose root value is an array wraps it in the root element and names each member `itemName`.
185
185
  * - A string is character data. The walk is synchronous, and what can go wrong is reported by throwing an {@link XmlRenderError}; {@link renderXml}
186
186
  * folds that into the effect's typed error channel. A caller not already in an `Effect` runs it with `Effect.runSync`, which throws the failure it
187
187
  * produced.
@@ -245,9 +245,9 @@ const render = (value, options) => {
245
245
  return out.join("");
246
246
  };
247
247
  /**
248
- * @description Renders one named element and its subtree. The value an {@link XmlValue} holds decides which of the four shapes below it takes — a repeated run of
249
- * children, character data, an absent field, or a record — and each of those is written by a function of its own, so this one is the dispatch rather
250
- * than the document.
248
+ * @description Renders one named element and its subtree. The value an {@link XmlValue} holds decides which of the four shapes below it takes: a repeated run of
249
+ * children, character data, an absent field, or a record. Each of those is written by a function of its own, so this one is the dispatch rather than
250
+ * the document.
251
251
  *
252
252
  * @param out - The chunk buffer to append to.
253
253
  * @param name - The element name, not yet resolved.
@@ -272,7 +272,7 @@ const renderElement = (out, name, value, depth, options) => {
272
272
  };
273
273
  /**
274
274
  * @description Refuses to walk deeper than the render allows. A value can nest without end, and every one of those levels costs a stack frame here, so the cap is
275
- * checked on the way down rather than trusted to the caller.
275
+ * checked as the walk descends rather than trusted to the caller.
276
276
  *
277
277
  * @param depth - The depth about to be written.
278
278
  * @param options - Resolved render options.
@@ -304,8 +304,8 @@ const renderRepeated = (out, name, members, depth, options) => {
304
304
  };
305
305
  /**
306
306
  * @description Renders an element whose value is character data, or nothing. An empty string is character data that happens to be empty, and an element holding
307
- * none of it is the same element as one holding nothing at all — as is an `undefined` element, which is an absent one. The renderer is handed values
308
- * that never went through the schema — a caller building a document by hand — so the absent case is reachable, and an empty element is the honest
307
+ * none of it is the same element as one holding nothing at all, as is an `undefined` element, which is an absent one. The renderer is handed values
308
+ * that never went through the schema (a caller building a document by hand), so the absent case is reachable, and an empty element is the correct
309
309
  * rendering of both.
310
310
  *
311
311
  * @param out - The chunk buffer to append to.
@@ -353,9 +353,9 @@ const renderRecord = (out, tag, record, depth, options) => {
353
353
  * @description One pass over a record's keys, collecting all three roles at once: the attributes are rendered as they are found, the child names are set aside for
354
354
  * the pass that writes them, and the text key is left to {@link textOf}. A pass for the attributes, a pass for the children and an index for the text
355
355
  * instead walks the keys three times and allocates the key array twice, which on a document of a few thousand elements is thousands of allocations
356
- * for nothing. Sorting is off by default, and the default path is the one that matters, so the attributes are built as they are found and there is
356
+ * for nothing. Sorting is off by default, and the default path is the important one, so the attributes are built as they are found and there is
357
357
  * nothing to sort. When it is on, the attribute keys are collected instead and rendered afterwards in sorted order, which costs an array per element
358
- * and buys output that does not depend on the order the fields happened to be declared in.
358
+ * and gives output independent of the order the fields happened to be declared in.
359
359
  *
360
360
  * @param record - The element's value.
361
361
  * @param options - Resolved render options.
@@ -447,9 +447,9 @@ const openLine = (out, depth, options) => {
447
447
  };
448
448
  /**
449
449
  * @description Writes an element with no content, in whichever of the two forms the options ask for. Every path that produces an element with nothing in it goes
450
- * through here, so the self-closing decision is made in exactly one place. That matters because "nothing in it" arrives four different ways — an
451
- * empty string, an absent value, an empty array, and a record whose fields are all absent — and four separate decisions are four chances for one of
452
- * them to write the long form by accident.
450
+ * through here, so the self-closing decision is made in exactly one place. That is important because "nothing in it" arrives four different ways: an
451
+ * empty string, an absent value, an empty array, and a record whose fields are all absent. Four separate decisions are four chances for one of them
452
+ * to write the long form by accident.
453
453
  *
454
454
  * @param out - The chunk buffer to append to.
455
455
  * @param tag - The element's name, already resolved.
@@ -487,8 +487,8 @@ const attributeText = (value) => {
487
487
  return renderScalar(value);
488
488
  };
489
489
  /**
490
- * @description Renders a leaf that is not a string as the character data an XML document can hold. A schema-derived value never reaches here —
491
- * `Schema.toCodecStringTree` has already turned every scalar into a string — so this is for values a caller built by hand. A value with no sensible
490
+ * @description Renders a leaf that is not a string as the character data an XML document can hold. A schema-derived value never reaches here, because
491
+ * `Schema.toCodecStringTree` has already turned every scalar into a string, so this is for values a caller built by hand. A value with no sensible
492
492
  * text form is rendered as nothing rather than as `[object Object]`, which would silently write a document that parses back to something else.
493
493
  *
494
494
  * @param value - The leaf to render.