@endevops/effect-codec-xml 0.0.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (61) hide show
  1. package/LICENSE +21 -0
  2. package/LICENSE-is-entities +21 -0
  3. package/LICENSE-is-xml-naming +21 -0
  4. package/README.md +415 -0
  5. package/dist/codec.d.ts +48 -0
  6. package/dist/codec.d.ts.map +1 -0
  7. package/dist/codec.js +63 -0
  8. package/dist/codec.js.map +1 -0
  9. package/dist/conventions.d.ts +88 -0
  10. package/dist/conventions.d.ts.map +1 -0
  11. package/dist/conventions.js +113 -0
  12. package/dist/conventions.js.map +1 -0
  13. package/dist/entities/entity-decoder.d.ts +333 -0
  14. package/dist/entities/entity-decoder.d.ts.map +1 -0
  15. package/dist/entities/entity-decoder.js +841 -0
  16. package/dist/entities/entity-decoder.js.map +1 -0
  17. package/dist/entities/entity-tables.js +16 -0
  18. package/dist/entities/entity-tables.js.map +1 -0
  19. package/dist/errors.d.ts +49 -0
  20. package/dist/errors.d.ts.map +1 -0
  21. package/dist/errors.js +48 -0
  22. package/dist/errors.js.map +1 -0
  23. package/dist/index.d.ts +11 -0
  24. package/dist/index.js +11 -0
  25. package/dist/namespaces.d.ts +101 -0
  26. package/dist/namespaces.d.ts.map +1 -0
  27. package/dist/namespaces.js +663 -0
  28. package/dist/namespaces.js.map +1 -0
  29. package/dist/naming.d.ts +149 -0
  30. package/dist/naming.d.ts.map +1 -0
  31. package/dist/naming.js +296 -0
  32. package/dist/naming.js.map +1 -0
  33. package/dist/parse.d.ts +75 -0
  34. package/dist/parse.d.ts.map +1 -0
  35. package/dist/parse.js +437 -0
  36. package/dist/parse.js.map +1 -0
  37. package/dist/render.d.ts +99 -0
  38. package/dist/render.d.ts.map +1 -0
  39. package/dist/render.js +509 -0
  40. package/dist/render.js.map +1 -0
  41. package/dist/xml-error.d.ts +172 -0
  42. package/dist/xml-error.d.ts.map +1 -0
  43. package/dist/xml-error.js +157 -0
  44. package/dist/xml-error.js.map +1 -0
  45. package/dist/xml-value.d.ts +42 -0
  46. package/dist/xml-value.d.ts.map +1 -0
  47. package/dist/xml-value.js +79 -0
  48. package/dist/xml-value.js.map +1 -0
  49. package/package.json +69 -0
  50. package/src/codec.ts +136 -0
  51. package/src/conventions.ts +145 -0
  52. package/src/entities/entity-decoder.ts +1248 -0
  53. package/src/entities/entity-tables.ts +18 -0
  54. package/src/errors.ts +55 -0
  55. package/src/index.ts +79 -0
  56. package/src/namespaces.ts +968 -0
  57. package/src/naming.ts +519 -0
  58. package/src/parse.ts +597 -0
  59. package/src/render.ts +708 -0
  60. package/src/xml-error.ts +168 -0
  61. package/src/xml-value.ts +108 -0
package/src/render.ts ADDED
@@ -0,0 +1,708 @@
1
+ // Rendering: an `XmlValue` to XML text.
2
+ //
3
+ // This is the hot path for an application that serializes often, so it is built
4
+ // around three things: one pass over each record's keys rather than one per
5
+ // role a key can play, one pass over each character rather than one per
6
+ // character class, and one array of chunks joined once rather than a growing
7
+ // string.
8
+ //
9
+ // The walk itself is plain synchronous functions rather than a chain of
10
+ // `yield*`es. Publicly `renderXml` is still an `Effect` — it suspends the walk so
11
+ // it runs lazily, and folds the one failure the walk can report into the typed
12
+ // error channel — but inside a document there is no effect boundary per element
13
+ // or per attribute. A 500-row report is thousands of elements, and a fiber step
14
+ // for each of them was most of what the `render 500 rows` row measured. The
15
+ // typed failure survives: the walk throws an {@link XmlRenderError} and
16
+ // `renderXml` catches it into `Effect.fail`.
17
+ //
18
+ // Escaping is the part that scales with the size of the document rather than
19
+ // with its structure, and it is written out here rather than delegated, for a
20
+ // measured reason. The entity encoder that used to live beside this package
21
+ // escaped by applying five sequential global replacements, one per character,
22
+ // so a document with a single `&` in twenty thousand characters was scanned
23
+ // five times over to change one byte -- which is what the `render 20k` rows in
24
+ // `bench/codec.bench.ts` measure. The table below covers the same five
25
+ // characters that encoder escaped, and the explicit expectations in
26
+ // `test/render.spec.ts` pin the fast path.
27
+
28
+ import { Effect, Predicate, Result } from 'effect';
29
+
30
+ import type { NameMode } from './conventions.ts';
31
+ import type { XmlVersion } from './naming.ts';
32
+ import type { XmlRecord, XmlValue } from './xml-value.ts';
33
+
34
+ import { attributeName, DEFAULT_ITEM_NAME, DEFAULT_ROOT_NAME, isAttributeKey, isTextKey, resolveNameSync, TEXT_KEY } from './conventions.ts';
35
+ import { XmlRenderError } from './errors.ts';
36
+
37
+ /**
38
+ * @description The five characters XML predefines an entity for, and the names to write for them. Written out rather than referenced from the entity decoder
39
+ * because the table is indexed by character code below.
40
+ */
41
+ const XML_PREDEFINED = { 34: '"', 38: '&', 39: ''', 60: '<', 62: '>' } as const;
42
+
43
+ /**
44
+ * @description The character references for the whitespace XML normalizes inside an attribute value. A parser replaces a literal newline, carriage return or tab
45
+ * in an attribute with a space, so a value that has to survive a round trip has to spell them as references. They are in the attribute table and not
46
+ * the text one: in character data they are content, and only an attribute value is normalized.
47
+ */
48
+ const ATTRIBUTE_WHITESPACE = { 9: '	', 10: '
', 13: '
' } as const;
49
+
50
+ /**
51
+ * @description The replacement for each ASCII character that needs one, and `undefined` for the ones that do not. Indexed by character code and 128 long, so the
52
+ * check is one comparison and one array read with no string search in it.
53
+ *
54
+ * @param extra - Characters to escape in addition to the five predefines.
55
+ *
56
+ * @returns The lookup table.
57
+ */
58
+ const buildTable = (extra: Record<number, string>): ReadonlyArray<string | undefined> => {
59
+ const table = Array.from<string | undefined>({ length: 128 }).fill(undefined);
60
+ for (const [code, entity] of Object.entries({ ...XML_PREDEFINED, ...extra })) table[Number(code)] = entity;
61
+ return table;
62
+ };
63
+
64
+ const TEXT_TABLE = buildTable({});
65
+ const ATTRIBUTE_TABLE = buildTable(ATTRIBUTE_WHITESPACE);
66
+
67
+ /**
68
+ * @description The characters each table escapes, as a pattern rather than as a set of replacement passes. Finding the first one with a pattern is what makes
69
+ * clean text cheap: V8 compiles a single character class into a scan that is several times faster than a JavaScript loop reading the same string a
70
+ * code unit at a time, and clean text is most text. `render 20k of clean text` in `bench/codec.bench.ts` is the row that says so — a hand-written
71
+ * loop over the same twenty thousand characters is roughly two and a half times slower. Neither pattern is global, so `exec` ignores `lastIndex` and
72
+ * always starts at the beginning. One module-level instance of each is therefore safe to reuse, and nothing has to be reset between calls.
73
+ */
74
+ const TEXT_UNSAFE = /[<>&"']/;
75
+ const ATTRIBUTE_UNSAFE = /[<>&"'\n\r\t]/;
76
+
77
+ /**
78
+ * @description Options for {@link renderXml}.
79
+ */
80
+ export interface XmlRenderOptions {
81
+ /**
82
+ * @description Name of the root element.
83
+ *
84
+ * @default 'root'\
85
+ * A codec passes the name it took from the schema's `identifier` annotation when the caller did not set one.
86
+ */
87
+ readonly rootName?: string | undefined;
88
+
89
+ /**
90
+ * @description Element name used for the members of a document whose root value is an array.
91
+ *
92
+ * @default 'item'
93
+ */
94
+ readonly itemName?: string | undefined;
95
+
96
+ /**
97
+ * @description Indent nested elements on their own lines.
98
+ *
99
+ * @default false
100
+ */
101
+ readonly format?: boolean | undefined;
102
+
103
+ /**
104
+ * @description The string one indent level is made of. Defaults to two spaces.
105
+ */
106
+ readonly indent?: string | undefined;
107
+
108
+ /**
109
+ * @description Write an element with no attributes, text or children as `<a/>` rather than `<a></a>`.
110
+ *
111
+ * @default true
112
+ */
113
+ readonly suppressEmptyNode?: boolean | undefined;
114
+
115
+ /**
116
+ * @description Sort an element's keys so the same value always renders to the same bytes.
117
+ *
118
+ * @default false\
119
+ * Which keeps declaration order.\ Worth turning on for snapshot tests, where key order is otherwise the only thing that can make two equal values differ.
120
+ */
121
+ readonly sortKeys?: boolean | undefined;
122
+
123
+ /**
124
+ * @description What to do with a field name that is not a legal XML name.
125
+ *
126
+ * @default 'repair'.
127
+ */
128
+ readonly name?: NameMode | undefined;
129
+
130
+ /**
131
+ * @description XML version to validate names against.
132
+ *
133
+ * @default '1.0'
134
+ */
135
+ readonly xmlVersion?: XmlVersion | undefined;
136
+
137
+ /**
138
+ * @description How deep to nest before giving up. Guards against a value that nests without end taking the stack with it.
139
+ *
140
+ * @default 256
141
+ */
142
+ readonly maxDepth?: number | undefined;
143
+ }
144
+
145
+ /**
146
+ * @description Options every render call needs, with the defaults already applied.
147
+ */
148
+ interface ResolvedOptions {
149
+ readonly rootName: string;
150
+ readonly itemName: string;
151
+ readonly format: boolean;
152
+ readonly indent: string;
153
+ readonly suppressEmptyNode: boolean;
154
+ readonly sortKeys: boolean;
155
+ readonly name: NameMode;
156
+ readonly xmlVersion: XmlVersion;
157
+ readonly maxDepth: number;
158
+
159
+ /**
160
+ * @description Resolves a field name to a legal XML name, in the mode this render was configured with. Synchronous and memoized; see {@link makeNamer}.
161
+ */
162
+ readonly namer: (name: string) => string;
163
+
164
+ /**
165
+ * @description The indent for a given depth, when pretty-printing. Built on first use at each depth and kept, so an indented document builds one string per
166
+ * level rather than one per line.
167
+ */
168
+ readonly lineAt: (depth: number) => string;
169
+ }
170
+
171
+ /**
172
+ * @description Builds the name resolver for one render. Every element and every attribute name goes through here, and a document repeats names: a thousand
173
+ * `<item>` elements, or the same `id` on every row. A validator that runs a regex per occurrence pays that cost a thousand times for one answer, so
174
+ * the first result is remembered and the rest are lookups. It also keeps the mode and version in one place, which is what stops a caller from
175
+ * resolving a name with different settings than the render it is part of. A cache miss calls {@link resolveNameSync}, which throws an
176
+ * {@link XmlParseError} in `'error'` mode; {@link renderXml} catches it and reports it as an {@link XmlRenderError}.
177
+ *
178
+ * @param options - Resolved render options.
179
+ *
180
+ * @returns A function from field name to the legal XML name.
181
+ */
182
+ const makeNamer = (options: Omit<ResolvedOptions, 'namer' | 'lineAt'>): ((name: string) => string) => {
183
+ const cache = new Map<string, string>();
184
+ return (name: string): string => {
185
+ const hit = cache.get(name);
186
+ if (hit !== undefined) return hit;
187
+ const resolved = resolveNameSync(name, { mode: options.name, xmlVersion: options.xmlVersion });
188
+ cache.set(name, resolved);
189
+ return resolved;
190
+ };
191
+ };
192
+
193
+ /**
194
+ * @description A boolean option's value, with an absent one read as the default. The three boolean options are spelled through here rather than through a `??` of
195
+ * their own, so the table below reads as a list of what each option _is_ instead of a list of nine separate decisions about what an omitted option
196
+ * means — and so a reader looking for "which options are on by default" finds three words rather than three mixes of `?? true` and `?? false` to
197
+ * read.
198
+ *
199
+ * @param value - The option as the caller wrote it, or `undefined` when the caller left it out.
200
+ * @param fallback - The value to use when the caller left it out.
201
+ *
202
+ * @returns The option's value.
203
+ */
204
+ const flag = (value: boolean | undefined, fallback: boolean): boolean => value ?? fallback;
205
+
206
+ /**
207
+ * @description The indent for a given depth, built the first time a render reaches that depth and kept. A document of a few thousand elements on several lines
208
+ * each would otherwise call `repeat` once per line and allocate the same handful of strings thousands of times over.
209
+ *
210
+ * @param indent - The string one level of indentation is made of.
211
+ *
212
+ * @returns A function from depth to the indent for that depth.
213
+ */
214
+ const makeLineAt = (indent: string): ((depth: number) => string) => {
215
+ const lines: Array<string> = [''];
216
+ return depth => {
217
+ const line = lines[depth];
218
+ if (line !== undefined) return line;
219
+ const built = indent.repeat(depth);
220
+ lines[depth] = built;
221
+ return built;
222
+ };
223
+ };
224
+
225
+ /**
226
+ * @description Applies the defaults to one call's options, and builds the two things a render needs that are not options: the memoized name resolver and the
227
+ * memoized indent lines.
228
+ *
229
+ * @param options - The options as the caller wrote them.
230
+ *
231
+ * @returns Every option a render reads, defaulted, with the resolver and the indent lines attached.
232
+ */
233
+ const resolveOptions = (options: XmlRenderOptions): ResolvedOptions => {
234
+ const resolved = {
235
+ rootName: options.rootName ?? DEFAULT_ROOT_NAME,
236
+ itemName: options.itemName ?? DEFAULT_ITEM_NAME,
237
+ format: flag(options.format, false),
238
+ indent: options.indent ?? ' ',
239
+ suppressEmptyNode: flag(options.suppressEmptyNode, true),
240
+ sortKeys: flag(options.sortKeys, false),
241
+ name: options.name ?? ('repair' as NameMode),
242
+ xmlVersion: options.xmlVersion ?? ('1.0' as XmlVersion),
243
+ maxDepth: options.maxDepth ?? 256,
244
+ };
245
+ return { ...resolved, namer: makeNamer(resolved), lineAt: makeLineAt(resolved.indent) };
246
+ };
247
+
248
+ /**
249
+ * @description Escapes a value for use as character data.
250
+ *
251
+ * @param value - The text to escape.
252
+ *
253
+ * @returns The text with the XML-unsafe characters replaced by predefined entities, or the very same string when there is nothing to escape.
254
+ */
255
+ export const escapeText = (value: string): string => escape(value, TEXT_UNSAFE, TEXT_TABLE);
256
+
257
+ /**
258
+ * @description Escapes a value for use inside a double-quoted attribute.
259
+ *
260
+ * @param value - The text to escape.
261
+ *
262
+ * @returns The escaped text, with the whitespace that XML would otherwise normalize spelled as character references.
263
+ */
264
+ export const escapeAttribute = (value: string): string => escape(value, ATTRIBUTE_UNSAFE, ATTRIBUTE_TABLE);
265
+
266
+ /**
267
+ * @description Replaces every character the table has an entry for, in one pass over the string. The pattern finds the first character that needs replacing, and a
268
+ * string with none is handed straight back — which is the common case, and the one the pattern is there to make fast. From there the rest of the
269
+ * string is copied in runs between the replacements rather than a character at a time, so the cost is one pattern scan, one copy, and one
270
+ * concatenation per replacement, rather than a whole pass per character class. Only ASCII is looked up. XML carries every other character natively,
271
+ * and a code unit above 127 has no entity an XML parser is required to know.
272
+ *
273
+ * @param value - The text to escape.
274
+ * @param pattern - Matches the first character that needs replacing.
275
+ * @param table - The replacement for each ASCII character that needs one.
276
+ *
277
+ * @returns The escaped text, or `value` itself when there is nothing to escape.
278
+ */
279
+ const escape = (value: string, pattern: RegExp, table: ReadonlyArray<string | undefined>): string => {
280
+ const found = pattern.exec(value);
281
+ if (found === null) return value; // nothing to escape: hand back the same string
282
+
283
+ const length = value.length;
284
+ const start = found.index;
285
+ let out = value.slice(0, start);
286
+ let copied = start;
287
+
288
+ for (let index = start; index < length; index++) {
289
+ const code = value.charCodeAt(index);
290
+ const entity = code < 128 ? table[code] : undefined;
291
+ if (entity !== undefined) {
292
+ out += value.slice(copied, index) + entity;
293
+ copied = index + 1;
294
+ }
295
+ }
296
+
297
+ return copied === length ? out : out + value.slice(copied);
298
+ };
299
+
300
+ /**
301
+ * @description Renders an {@link XmlValue} as an XML document.\
302
+ * A record becomes an element:
303
+ *
304
+ * - `@`-prefixed keys become attributes, the reserved `#text` key becomes character data, and every other key becomes a child element.
305
+ * - An array repeats its name — a document whose root value is an array wraps it in the root element and names each member `itemName`.
306
+ * - A string is character data. The walk is synchronous, and what can go wrong is reported by throwing an {@link XmlRenderError}; {@link renderXml}
307
+ * folds that into the effect's typed error channel. A caller not already in an `Effect` runs it with `Effect.runSync`, which throws the failure it
308
+ * produced.
309
+ *
310
+ * @param value - The value to render.
311
+ * @param options - Root name, formatting, empty-element and name-resolution settings.
312
+ *
313
+ * @returns An effect producing the XML document as a string.
314
+ */
315
+ export const renderXml = (value: XmlValue, options: XmlRenderOptions = {}): Effect.Effect<string, XmlRenderError> =>
316
+ Effect.suspend(() => Effect.fromResult(renderResult(value, options)));
317
+
318
+ /**
319
+ * @description Runs the synchronous walk and folds the one failure it reports into a {@link Result}, which {@link renderXml} turns back into an `Effect`. Kept
320
+ * separate so the walk itself can throw without the public API ever throwing.
321
+ *
322
+ * @param value - The value to render.
323
+ * @param options - The options as the caller wrote them.
324
+ *
325
+ * @returns The document, or the failure to report.
326
+ */
327
+ const renderResult = (value: XmlValue, options: XmlRenderOptions): Result.Result<string, XmlRenderError> => {
328
+ try {
329
+ return Result.succeed(render(value, options));
330
+ } catch (cause) {
331
+ return Result.fail(toRenderError(cause));
332
+ }
333
+ };
334
+
335
+ /**
336
+ * @description Reports a failure the synchronous walk threw in the render's own error type. The walk only throws an {@link XmlRenderError} of its own or an
337
+ * {@link XmlParseError} from the name resolver; the latter carries the message the spec asserts on, so it is carried across rather than replaced.
338
+ *
339
+ * @param cause - Whatever was thrown.
340
+ *
341
+ * @returns The failure to report.
342
+ */
343
+ const toRenderError = (cause: unknown): XmlRenderError => {
344
+ if (cause instanceof XmlRenderError) return cause;
345
+ if (Predicate.isError(cause)) return new XmlRenderError({ message: cause.message });
346
+ return new XmlRenderError({ message: String(cause) });
347
+ };
348
+
349
+ /**
350
+ * @description The synchronous walk behind {@link renderXml}.
351
+ *
352
+ * @param value - The value to render.
353
+ * @param options - The options as the caller wrote them.
354
+ *
355
+ * @returns The XML document as a string.
356
+ *
357
+ * @throws {XmlRenderError} When the value nests past `maxDepth`, or the name resolver refuses a field name.
358
+ */
359
+ const render = (value: XmlValue, options: XmlRenderOptions): string => {
360
+ const resolved = resolveOptions(options);
361
+ const out: Array<string> = [];
362
+
363
+ // A document has exactly one root element, so a root value that is an array
364
+ // is wrapped rather than emitted as several roots. Inside a named element an
365
+ // array repeats that element's own name, so this wrapping is the only place
366
+ // `itemName` is ever used.
367
+ if (Array.isArray(value)) {
368
+ const tag = resolved.namer(resolved.rootName);
369
+ out.push('<', tag, '>');
370
+ for (const member of value) renderElement(out, resolved.itemName, member, 1, resolved);
371
+ if (resolved.format) out.push('\n');
372
+ out.push('</', tag, '>');
373
+ } else {
374
+ renderElement(out, resolved.rootName, value, 0, resolved);
375
+ }
376
+
377
+ if (resolved.format) out.push('\n');
378
+ return out.join('');
379
+ };
380
+
381
+ /**
382
+ * @description Renders one named element and its subtree. The value an {@link XmlValue} holds decides which of the four shapes below it takes — a repeated run of
383
+ * children, character data, an absent field, or a record — and each of those is written by a function of its own, so this one is the dispatch rather
384
+ * than the document.
385
+ *
386
+ * @param out - The chunk buffer to append to.
387
+ * @param name - The element name, not yet resolved.
388
+ * @param value - The element's value.
389
+ * @param depth - Current nesting depth, for indentation and the depth cap.
390
+ * @param options - Resolved render options.
391
+ */
392
+ const renderElement = (out: Array<string>, name: string, value: XmlValue, depth: number, options: ResolvedOptions): void => {
393
+ assertWithinDepth(depth, options);
394
+
395
+ if (Array.isArray(value)) {
396
+ renderRepeated(out, name, value, depth, options);
397
+ return;
398
+ }
399
+
400
+ // An element opens its own line rather than having its caller do it, which is
401
+ // what keeps a repeated run of children on separate lines. The root is the
402
+ // one element that has nothing in front of it.
403
+ if (options.format && depth > 0) openLine(out, depth, options);
404
+
405
+ const tag = options.namer(name);
406
+
407
+ if (Predicate.isString(value) || Predicate.isUndefined(value)) {
408
+ renderLeaf(out, tag, value, options);
409
+ return;
410
+ }
411
+
412
+ // The array case returned above; `Predicate.isObject` narrows what is left to
413
+ // a record, since `Array.isArray` alone leaves a `ReadonlyArray` in the union.
414
+ if (!Predicate.isObject(value)) return;
415
+ renderRecord(out, tag, value as XmlRecord, depth, options);
416
+ };
417
+
418
+ /**
419
+ * @description Refuses to walk deeper than the render allows. A value can nest without end, and every one of those levels costs a stack frame here, so the cap is
420
+ * checked on the way down rather than trusted to the caller.
421
+ *
422
+ * @param depth - The depth about to be written.
423
+ * @param options - Resolved render options.
424
+ *
425
+ * @throws {XmlRenderError} When the depth is past the cap.
426
+ */
427
+ const assertWithinDepth = (depth: number, options: ResolvedOptions): void => {
428
+ if (depth > options.maxDepth) {
429
+ throw new XmlRenderError({
430
+ message: `XML nesting exceeded maxDepth (${options.maxDepth}). Raise the limit if the document is legitimately this deep.`,
431
+ });
432
+ }
433
+ };
434
+
435
+ /**
436
+ * @description Renders a repeated run of children under one name: `tags: ['a', 'b']` renders `<tags>a</tags><tags>b</tags>`, not one element wrapping both. The
437
+ * name is already the element's own, so a name only has to be supplied where no name is available. Checked before the line break and the name are
438
+ * taken, because the array itself is not an element: opening a line for it as well as for each of its members would leave a blank line where the
439
+ * array was.
440
+ *
441
+ * @param out - The chunk buffer to append to.
442
+ * @param name - The element name, not yet resolved.
443
+ * @param members - The children to write, one after another.
444
+ * @param depth - The depth the run sits at.
445
+ * @param options - Resolved render options.
446
+ */
447
+ const renderRepeated = (out: Array<string>, name: string, members: ReadonlyArray<XmlValue>, depth: number, options: ResolvedOptions): void => {
448
+ // An empty array still gets an element. Writing nothing would make a field
449
+ // that was present and empty indistinguishable from one that was never
450
+ // there, and a document that came from a schema is easier to trust when the
451
+ // element it describes is actually in the output.
452
+ if (members.length === 0) {
453
+ if (options.format && depth > 0) openLine(out, depth, options);
454
+ writeEmpty(out, options.namer(name), options);
455
+ return;
456
+ }
457
+ for (const member of members) renderElement(out, name, member, depth, options);
458
+ };
459
+
460
+ /**
461
+ * @description Renders an element whose value is character data, or nothing. An empty string is character data that happens to be empty, and an element holding
462
+ * none of it is the same element as one holding nothing at all — as is an `undefined` element, which is an absent one. The renderer is handed values
463
+ * that never went through the schema — a caller building a document by hand — so the absent case is reachable, and an empty element is the honest
464
+ * rendering of both.
465
+ *
466
+ * @param out - The chunk buffer to append to.
467
+ * @param tag - The element's name, already resolved.
468
+ * @param value - The element's character data, or `undefined` for an absent element.
469
+ * @param options - Resolved render options.
470
+ */
471
+ const renderLeaf = (out: Array<string>, tag: string, value: string | undefined, options: ResolvedOptions): void => {
472
+ if (Predicate.isUndefined(value) || value === '') {
473
+ writeEmpty(out, tag, options);
474
+ return;
475
+ }
476
+ out.push('<', tag, '>', escapeText(value), '</', tag, '>');
477
+ };
478
+
479
+ /**
480
+ * @description Renders an element holding a record: the attributes gathered from its `@` keys, the character data from its `#text` key, and its remaining keys as
481
+ * child elements.
482
+ *
483
+ * @param out - The chunk buffer to append to.
484
+ * @param tag - The element's name, already resolved.
485
+ * @param record - The element's value.
486
+ * @param depth - The depth the element sits at.
487
+ * @param options - Resolved render options.
488
+ */
489
+ const renderRecord = (out: Array<string>, tag: string, record: XmlRecord, depth: number, options: ResolvedOptions): void => {
490
+ const fields = collectFields(record, options);
491
+ const text = textOf(record);
492
+ const children = fields.children;
493
+
494
+ // Self-closing is decided by whether the element has any *content*, not by
495
+ // whether it has attributes: `<a id="1"/>` is the same element as
496
+ // `<a id="1">` with nothing in it, and the short form is what every XML
497
+ // writer produces.
498
+ if (children === undefined && text === '') {
499
+ writeEmpty(out, tag, options, fields.attributes);
500
+ return;
501
+ }
502
+
503
+ out.push('<', tag, fields.attributes, '>');
504
+
505
+ // Character data sits inline when it is all an element has, and on its own
506
+ // line when the element also has children, so an indented document does not
507
+ // end up with its first line of text glued to its opening tag.
508
+ if (text !== '') {
509
+ if (children !== undefined && options.format) openLine(out, depth + 1, options);
510
+ out.push(escapeText(text));
511
+ }
512
+
513
+ if (children !== undefined) {
514
+ writeChildren(out, record, children, depth, options);
515
+ if (options.format) openLine(out, depth, options);
516
+ }
517
+
518
+ out.push('</', tag, '>');
519
+ };
520
+
521
+ /**
522
+ * @description An element's keys resolved into the roles they play.
523
+ */
524
+ interface Fields {
525
+ /**
526
+ * @description The element's rendered attributes, each with its leading space, or `''` when it has none.
527
+ */
528
+ readonly attributes: string;
529
+
530
+ /**
531
+ * @description The names of the child elements, in the order they will be written, or `undefined` when the element has none. `undefined` rather than an empty
532
+ * array because "has children" is one of the two things the self-closing decision turns on, and the other is the text.
533
+ */
534
+ readonly children: Array<string> | undefined;
535
+ }
536
+
537
+ /**
538
+ * @description One pass over a record's keys, collecting all three roles at once: the attributes are rendered as they are found, the child names are set aside for
539
+ * the pass that writes them, and the text key is left to {@link textOf}. A pass for the attributes, a pass for the children and an index for the text
540
+ * instead walks the keys three times and allocates the key array twice, which on a document of a few thousand elements is thousands of allocations
541
+ * for nothing. Sorting is off by default, and the default path is the one that matters, so the attributes are built as they are found and there is
542
+ * nothing to sort. When it is on, the attribute keys are collected instead and rendered afterwards in sorted order, which costs an array per element
543
+ * and buys output that does not depend on the order the fields happened to be declared in.
544
+ *
545
+ * @param record - The element's value.
546
+ * @param options - Resolved render options.
547
+ *
548
+ * @returns The element's rendered attributes and its child names.
549
+ */
550
+ const collectFields = (record: XmlRecord, options: ResolvedOptions): Fields => {
551
+ const keys = Object.keys(record);
552
+ let attributes = '';
553
+ let children: Array<string> | undefined;
554
+ const sortAttributes = options.sortKeys ? ([] as Array<string>) : undefined;
555
+
556
+ for (let i = 0; i < keys.length; i++) {
557
+ const key = keys[i] as string;
558
+ const child = record[key];
559
+
560
+ // An absent field is not written at all, which is what keeps an unset
561
+ // optional attribute out of the document rather than in it as `a=""`, and
562
+ // an absent child out of it rather than in it as `<a/>`. The text key is
563
+ // read by `textOf` either way, so skipping it here costs nothing.
564
+ if (child === undefined) continue;
565
+
566
+ if (isAttributeKey(key)) {
567
+ // Only keys with a value are collected, so every name in `sortAttributes`
568
+ // has one to read back out of.
569
+ if (sortAttributes === undefined) attributes += renderAttribute(key, child, options);
570
+ else sortAttributes.push(key);
571
+ continue;
572
+ }
573
+
574
+ if (!isTextKey(key)) (children ??= []).push(key);
575
+ }
576
+
577
+ if (sortAttributes !== undefined) {
578
+ sortAttributes.sort();
579
+ attributes += sortedAttributes(sortAttributes, record, options);
580
+ children?.sort();
581
+ }
582
+
583
+ return { attributes, children };
584
+ };
585
+
586
+ /**
587
+ * @description The attributes named by `keys`, rendered in the order given. Only reached when the render was asked to sort keys, where the names are collected
588
+ * during the key pass and written here so their order does not follow the order the fields were declared in.
589
+ *
590
+ * @param keys - The attribute keys to write, in the order to write them.
591
+ * @param record - The element's value, to read the attribute values out of.
592
+ * @param options - Resolved render options.
593
+ *
594
+ * @returns The rendered attributes, each with its leading space.
595
+ */
596
+ const sortedAttributes = (keys: ReadonlyArray<string>, record: XmlRecord, options: ResolvedOptions): string => {
597
+ let attributes = '';
598
+ for (const key of keys) attributes += renderAttribute(key, record[key], options);
599
+ return attributes;
600
+ };
601
+
602
+ /**
603
+ * @description One attribute, written whole. The leading space is part of it so the caller can concatenate attributes and the opening tag without a separator of
604
+ * its own.
605
+ *
606
+ * @param key - The attribute's key, with or without its `@` prefix.
607
+ * @param value - The attribute's value.
608
+ * @param options - Resolved render options.
609
+ *
610
+ * @returns The attribute, ready to write inside the opening tag.
611
+ */
612
+ const renderAttribute = (key: string, value: XmlValue, options: ResolvedOptions): string => {
613
+ const name = options.namer(attributeName(key));
614
+ return ' ' + name + '="' + escapeAttribute(attributeText(value)) + '"';
615
+ };
616
+
617
+ /**
618
+ * @description Writes an element's children, by name and in the order their keys were found. The names are what the key pass kept; the values are read back out of
619
+ * the record here, because keeping both would mean a second array per element.
620
+ *
621
+ * @param out - The chunk buffer to append to.
622
+ * @param record - The element's value.
623
+ * @param children - The child names, in the order to write them.
624
+ * @param depth - The depth the parent sits at; its children are one deeper.
625
+ * @param options - Resolved render options.
626
+ */
627
+ const writeChildren = (out: Array<string>, record: XmlRecord, children: ReadonlyArray<string>, depth: number, options: ResolvedOptions): void => {
628
+ for (let i = 0; i < children.length; i++) {
629
+ const key = children[i] as string;
630
+ const child = record[key];
631
+ if (child === undefined) continue;
632
+ renderElement(out, key, child, depth + 1, options);
633
+ }
634
+ };
635
+
636
+ /**
637
+ * @description Starts a new line at the given depth, when pretty-printing.
638
+ *
639
+ * @param out - The chunk buffer to append to.
640
+ * @param depth - The depth the line sits at.
641
+ * @param options - Resolved render options.
642
+ */
643
+ const openLine = (out: Array<string>, depth: number, options: ResolvedOptions): void => {
644
+ out.push('\n');
645
+ out.push(options.lineAt(depth));
646
+ };
647
+
648
+ /**
649
+ * @description Writes an element with no content, in whichever of the two forms the options ask for. Every path that produces an element with nothing in it goes
650
+ * through here, so the self-closing decision is made in exactly one place. That matters because "nothing in it" arrives four different ways — an
651
+ * empty string, an absent value, an empty array, and a record whose fields are all absent — and four separate decisions are four chances for one of
652
+ * them to write the long form by accident.
653
+ *
654
+ * @param out - The chunk buffer to append to.
655
+ * @param tag - The element's name, already resolved.
656
+ * @param options - Resolved render options.
657
+ * @param attributes - The element's rendered attributes, if it has any. Defaults to none.
658
+ */
659
+ const writeEmpty = (out: Array<string>, tag: string, options: ResolvedOptions, attributes = ''): void => {
660
+ if (options.suppressEmptyNode) out.push('<', tag, attributes, '/>');
661
+ else out.push('<', tag, attributes, '></', tag, '>');
662
+ };
663
+
664
+ /**
665
+ * @description An element's character data, with the reserved text key read off its value.
666
+ *
667
+ * @param record - The element's value.
668
+ *
669
+ * @returns The text to write between the tags, or `''` when the element has none.
670
+ */
671
+ const textOf = (record: XmlRecord): string => {
672
+ const text = record[TEXT_KEY];
673
+ if (Predicate.isUndefined(text)) return '';
674
+ return Predicate.isString(text) ? text : renderScalar(text);
675
+ };
676
+
677
+ /**
678
+ * @description An attribute value as the character data it is written as. A bare string is the only sensible shape, since an attribute holds nothing else. A
679
+ * non-string is stringified rather than rejected: the schema is what enforces the field's type, and rejecting here would duplicate that check with a
680
+ * different error and a message that names neither the field nor the document.
681
+ *
682
+ * @param value - The attribute's value.
683
+ *
684
+ * @returns The text to escape and write between the quotes.
685
+ */
686
+ const attributeText = (value: XmlValue): string => {
687
+ if (Predicate.isString(value)) return value;
688
+ if (Predicate.isUndefined(value)) return '';
689
+ return renderScalar(value);
690
+ };
691
+
692
+ /**
693
+ * @description Renders a leaf that is not a string as the character data an XML document can hold. A schema-derived value never reaches here —
694
+ * `Schema.toCodecStringTree` has already turned every scalar into a string — so this is for values a caller built by hand. A value with no sensible
695
+ * text form is rendered as nothing rather than as `[object Object]`, which would silently write a document that parses back to something else.
696
+ *
697
+ * @param value - The leaf to render.
698
+ *
699
+ * @returns The leaf's textual form.
700
+ */
701
+ const renderScalar = (value: Exclude<XmlValue, string | undefined>): string => {
702
+ if (Predicate.isNull(value)) return 'null';
703
+ try {
704
+ return JSON.stringify(value) ?? '';
705
+ } catch {
706
+ return ''; // circular or otherwise not representable
707
+ }
708
+ };