@tradik/xslt-processor 1.0.3 → 1.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (131) hide show
  1. package/LICENSE.md +1 -1
  2. package/README.md +110 -520
  3. package/bin/lib/decode.js +15 -0
  4. package/bin/lib/dom.js +177 -0
  5. package/bin/lib/loaders.js +127 -0
  6. package/bin/lib/options.js +131 -0
  7. package/bin/lib/output.js +114 -0
  8. package/bin/lib/paths.js +186 -0
  9. package/bin/lib/transform.js +206 -0
  10. package/bin/xslt.js +73 -168
  11. package/dist/xslt-processor.browser.js +9564 -1585
  12. package/dist/xslt-processor.browser.js.map +4 -4
  13. package/dist/xslt-processor.browser.min.js +13 -2
  14. package/dist/xslt-processor.browser.min.js.map +4 -4
  15. package/dist/xslt-processor.cjs +9572 -1586
  16. package/dist/xslt-processor.cjs.map +4 -4
  17. package/dist/xslt-processor.d.cts +658 -0
  18. package/dist/xslt-processor.d.ts +459 -12
  19. package/dist/xslt-processor.js +9546 -1582
  20. package/dist/xslt-processor.js.map +4 -4
  21. package/package.json +71 -20
  22. package/src/XSLTProcessor.js +494 -48
  23. package/src/async/abort.js +63 -0
  24. package/src/async/documentUris.js +128 -0
  25. package/src/async/loaders.js +134 -0
  26. package/src/async/preload.js +159 -0
  27. package/src/async/processor.js +206 -0
  28. package/src/async/stream.js +125 -0
  29. package/src/bridge/engine.js +221 -0
  30. package/src/bridge/loader.js +78 -0
  31. package/src/bridge/results.js +75 -0
  32. package/src/bridge/version.js +63 -0
  33. package/src/index.js +26 -8
  34. package/src/io/decode.js +140 -0
  35. package/src/io/readSource.js +167 -0
  36. package/src/xpath/axes.js +562 -0
  37. package/src/xpath/documentOrder.js +270 -0
  38. package/src/xpath/evaluator.js +518 -357
  39. package/src/xpath/index.js +8 -2
  40. package/src/xpath/namespaceNodes.js +172 -0
  41. package/src/xpath/nodeSetFunctions.js +169 -0
  42. package/src/xpath/parser.js +30 -5
  43. package/src/xpath/strings.js +183 -0
  44. package/src/xpath/tokenizer.js +37 -23
  45. package/src/xslt/attributeSets.js +95 -0
  46. package/src/xslt/avt.js +103 -0
  47. package/src/xslt/computedNames.js +91 -0
  48. package/src/xslt/copying.js +212 -0
  49. package/src/xslt/declarationNames.js +80 -0
  50. package/src/xslt/domParsing.js +95 -0
  51. package/src/xslt/elements.js +57 -0
  52. package/src/xslt/engine/bindings.js +195 -0
  53. package/src/xslt/engine/context.js +105 -0
  54. package/src/xslt/engine/controlFlow.js +145 -0
  55. package/src/xslt/engine/copyInstructions.js +133 -0
  56. package/src/xslt/engine/declarations.js +233 -0
  57. package/src/xslt/engine/functionSupport.js +103 -0
  58. package/src/xslt/engine/methods.js +33 -0
  59. package/src/xslt/engine/nodeConstruction.js +187 -0
  60. package/src/xslt/engine/numbering.js +104 -0
  61. package/src/xslt/engine/outputDeclaration.js +77 -0
  62. package/src/xslt/engine/sequenceConstructor.js +228 -0
  63. package/src/xslt/engine/stylesheetLoading.js +208 -0
  64. package/src/xslt/engine/templateInvocation.js +253 -0
  65. package/src/xslt/engine/templateRules.js +243 -0
  66. package/src/xslt/engine/textInstructions.js +171 -0
  67. package/src/xslt/engine/topLevel.js +130 -0
  68. package/src/xslt/engine/transformation.js +263 -0
  69. package/src/xslt/engine/workStack.js +245 -0
  70. package/src/xslt/engine.js +184 -1736
  71. package/src/xslt/exslt/arguments.js +99 -0
  72. package/src/xslt/exslt/calendar.js +120 -0
  73. package/src/xslt/exslt/common.js +44 -0
  74. package/src/xslt/exslt/dateCalc.js +261 -0
  75. package/src/xslt/exslt/dateFormat.js +150 -0
  76. package/src/xslt/exslt/dateParse.js +265 -0
  77. package/src/xslt/exslt/dates.js +259 -0
  78. package/src/xslt/exslt/duration.js +207 -0
  79. package/src/xslt/exslt/dynamic.js +59 -0
  80. package/src/xslt/exslt/index.js +59 -0
  81. package/src/xslt/exslt/math.js +177 -0
  82. package/src/xslt/exslt/sets.js +96 -0
  83. package/src/xslt/exslt/stringOps.js +163 -0
  84. package/src/xslt/exslt/strings.js +147 -0
  85. package/src/xslt/exslt/uri.js +92 -0
  86. package/src/xslt/formatNumber.js +233 -0
  87. package/src/xslt/forwardsCompatible.js +75 -0
  88. package/src/xslt/functions.js +270 -0
  89. package/src/xslt/index.js +38 -1
  90. package/src/xslt/keys.js +164 -0
  91. package/src/xslt/literalResult.js +223 -0
  92. package/src/xslt/matchScope.js +116 -0
  93. package/src/xslt/number.js +271 -0
  94. package/src/xslt/numberFormat.js +253 -0
  95. package/src/xslt/outputNames.js +58 -0
  96. package/src/xslt/patternCompiler.js +175 -0
  97. package/src/xslt/patterns.js +324 -0
  98. package/src/xslt/qname.js +90 -0
  99. package/src/xslt/resultDocument.js +98 -0
  100. package/src/xslt/resultNamespaces.js +219 -0
  101. package/src/xslt/resultTree.js +211 -0
  102. package/src/xslt/serializer/baseWriter.js +390 -0
  103. package/src/xslt/serializer/chunks.js +120 -0
  104. package/src/xslt/serializer/constants.js +92 -0
  105. package/src/xslt/serializer/encoding.js +327 -0
  106. package/src/xslt/serializer/escape.js +135 -0
  107. package/src/xslt/serializer/frames.js +168 -0
  108. package/src/xslt/serializer/htmlDoctype.js +102 -0
  109. package/src/xslt/serializer/htmlEntities.js +77 -0
  110. package/src/xslt/serializer/htmlSerializer.js +239 -0
  111. package/src/xslt/serializer/indent.js +51 -0
  112. package/src/xslt/serializer/namespaces.js +68 -0
  113. package/src/xslt/serializer/rawText.js +41 -0
  114. package/src/xslt/serializer/settings.js +179 -0
  115. package/src/xslt/serializer/textSerializer.js +77 -0
  116. package/src/xslt/serializer/xhtmlDocument.js +103 -0
  117. package/src/xslt/serializer/xmlSerializer.js +227 -0
  118. package/src/xslt/serializer.js +90 -0
  119. package/src/xslt/sort.js +151 -0
  120. package/src/xslt/spaceNameTests.js +115 -0
  121. package/src/xslt/stylesheetChecks.js +206 -0
  122. package/src/xslt/stylesheetNamespaces.js +266 -0
  123. package/src/xslt/templatePriority.js +45 -0
  124. package/src/xslt/uri.js +68 -0
  125. package/src/xslt/variables.js +152 -0
  126. package/src/xslt/whitespace.js +200 -0
  127. package/LICENSE +0 -29
  128. package/src/XSLTProcessor.test.js +0 -930
  129. package/src/xpath/evaluator.test.js +0 -1852
  130. package/src/xpath/tokenizer.test.js +0 -224
  131. package/src/xslt/engine.test.js +0 -3130
@@ -0,0 +1,327 @@
1
+ /**
2
+ * Output Encodings
3
+ *
4
+ * The `encoding` attribute of `xsl:output` names the character encoding of the
5
+ * serialized result. A character the encoding cannot represent is written as
6
+ * a character reference (XSLT 1.0 section 16.1: `€` in ISO-8859-1), so
7
+ * the serialized string only holds characters the declared encoding has, and
8
+ * turning it into bytes (see {@link encodeOutput}, used by the CLI) is
9
+ * lossless.
10
+ *
11
+ * Supported encodings:
12
+ * - UTF-8 and UTF-16 (every label TextDecoder maps to them) represent every
13
+ * character; UTF-16 output starts with a byte order mark;
14
+ * - ISO-8859-1 / latin1 (up to U+00FF) and US-ASCII (up to U+007F), handled
15
+ * here because the WHATWG Encoding Standard aliases both to windows-1252;
16
+ * - every other single-byte encoding known to TextDecoder (windows-125x,
17
+ * ISO-8859-x, KOI8-R/U, IBM866, macintosh, ...): its repertoire is read
18
+ * lazily by decoding the 256 byte values once;
19
+ * - multi-byte encodings other than UTF (Shift_JIS, EUC-JP, GBK, Big5, ...) and
20
+ * unknown labels are treated as able to represent every character: no
21
+ * character reference is written, and {@link encodeOutput} falls back to
22
+ * UTF-8 bytes (`isExact` is false so callers can warn).
23
+ *
24
+ * @module xslt/serializer/encoding
25
+ */
26
+
27
+ /** Labels of US-ASCII (IANA names and aliases). */
28
+ const ASCII_LABELS = new Set([
29
+ "ascii",
30
+ "us-ascii",
31
+ "ansi_x3.4-1968",
32
+ "iso646-us",
33
+ "iso-ir-6",
34
+ "csascii",
35
+ "us",
36
+ ]);
37
+
38
+ /** Labels of ISO-8859-1 (IANA names and aliases). */
39
+ const LATIN1_LABELS = new Set([
40
+ "iso-8859-1",
41
+ "iso8859-1",
42
+ "iso88591",
43
+ "iso_8859-1",
44
+ "iso_8859-1:1987",
45
+ "latin1",
46
+ "l1",
47
+ "cp819",
48
+ "ibm819",
49
+ "csisolatin1",
50
+ "iso-ir-100",
51
+ ]);
52
+
53
+ /** Canonical WHATWG names of the single-byte encodings. */
54
+ const SINGLE_BYTE_ENCODINGS = new Set([
55
+ "ibm866",
56
+ "iso-8859-2",
57
+ "iso-8859-3",
58
+ "iso-8859-4",
59
+ "iso-8859-5",
60
+ "iso-8859-6",
61
+ "iso-8859-7",
62
+ "iso-8859-8",
63
+ "iso-8859-8-i",
64
+ "iso-8859-10",
65
+ "iso-8859-13",
66
+ "iso-8859-14",
67
+ "iso-8859-15",
68
+ "iso-8859-16",
69
+ "koi8-r",
70
+ "koi8-u",
71
+ "macintosh",
72
+ "windows-874",
73
+ "windows-1250",
74
+ "windows-1251",
75
+ "windows-1252",
76
+ "windows-1253",
77
+ "windows-1254",
78
+ "windows-1255",
79
+ "windows-1256",
80
+ "windows-1257",
81
+ "windows-1258",
82
+ "x-mac-cyrillic",
83
+ ]);
84
+
85
+ /**
86
+ * @typedef {Object} OutputEncoding
87
+ * @property {string} name - Canonical name ("utf-8", "utf-16le", "iso-8859-1"...)
88
+ * @property {boolean} isUnicode - True when every character is representable
89
+ * @property {boolean} isExact - False when bytes fall back to UTF-8
90
+ * @property {Map<number, number>|null} bytes - Byte of every representable
91
+ * code point, for single-byte encodings
92
+ */
93
+
94
+ /** Encodings already resolved, by lower-cased label. */
95
+ const cache = new Map();
96
+
97
+ /**
98
+ * Build a single-byte encoding whose bytes map to the same code points.
99
+ *
100
+ * @param {string} name - Canonical name
101
+ * @param {number} last - Highest byte value (0x7F or 0xFF)
102
+ * @returns {OutputEncoding} The encoding
103
+ */
104
+ function identityEncoding(name, last) {
105
+ const bytes = new Map();
106
+ for (let byte = 0; byte <= last; byte++) bytes.set(byte, byte);
107
+ return { name, isUnicode: false, isExact: true, bytes };
108
+ }
109
+
110
+ /**
111
+ * Build a single-byte encoding from the repertoire TextDecoder reports.
112
+ *
113
+ * @param {TextDecoder} decoder - Decoder of the encoding
114
+ * @returns {OutputEncoding} The encoding
115
+ */
116
+ function tableEncoding(decoder) {
117
+ const bytes = new Map();
118
+ for (let byte = 0; byte <= 0xff; byte++) {
119
+ const codePoint = decoder.decode(Uint8Array.of(byte)).codePointAt(0);
120
+ if (codePoint !== 0xfffd && !bytes.has(codePoint)) {
121
+ bytes.set(codePoint, byte);
122
+ }
123
+ }
124
+ return { name: decoder.encoding, isUnicode: false, isExact: true, bytes };
125
+ }
126
+
127
+ /**
128
+ * Resolve an encoding label that is not ASCII or ISO-8859-1.
129
+ *
130
+ * @param {string} label - Lower-cased, trimmed label
131
+ * @returns {OutputEncoding} The encoding
132
+ */
133
+ function resolveWithDecoder(label) {
134
+ let decoder;
135
+ try {
136
+ decoder = new globalThis.TextDecoder(label);
137
+ } catch {
138
+ return { name: label, isUnicode: true, isExact: false, bytes: null };
139
+ }
140
+ const name = decoder.encoding;
141
+ if (SINGLE_BYTE_ENCODINGS.has(name)) return tableEncoding(decoder);
142
+ const isExact = name === "utf-8" || name.startsWith("utf-16");
143
+ return { name, isUnicode: true, isExact, bytes: null };
144
+ }
145
+
146
+ /**
147
+ * Resolve the `encoding` of `xsl:output`.
148
+ *
149
+ * @param {string} [label] - Encoding label, UTF-8 when absent
150
+ * @returns {OutputEncoding} The encoding
151
+ *
152
+ * @example
153
+ * getOutputEncoding("ISO-8859-1").bytes.has(0x20ac); // false
154
+ */
155
+ export function getOutputEncoding(label = "UTF-8") {
156
+ const key = String(label).trim().toLowerCase();
157
+ let encoding = cache.get(key);
158
+ if (!encoding) {
159
+ if (ASCII_LABELS.has(key)) encoding = identityEncoding("us-ascii", 0x7f);
160
+ else if (LATIN1_LABELS.has(key)) {
161
+ encoding = identityEncoding("iso-8859-1", 0xff);
162
+ } else encoding = resolveWithDecoder(key);
163
+ cache.set(key, encoding);
164
+ }
165
+ return encoding;
166
+ }
167
+
168
+ /**
169
+ * Numeric character reference of a code point.
170
+ *
171
+ * @param {number} codePoint - The code point
172
+ * @returns {string} `&#N;`
173
+ */
174
+ export function characterReference(codePoint) {
175
+ return `&#${codePoint};`;
176
+ }
177
+
178
+ /** Characters outside ASCII, as whole code points. */
179
+ const NON_ASCII = /[\u{80}-\u{10FFFF}]/gu;
180
+
181
+ /**
182
+ * Replace every character an encoding cannot represent.
183
+ *
184
+ * Every supported encoding is a superset of ASCII, so only non-ASCII
185
+ * characters are checked.
186
+ *
187
+ * @param {string} text - Text to write
188
+ * @param {OutputEncoding} encoding - The output encoding
189
+ * @param {(codePoint: number) => string} [reference] - Replacement of an
190
+ * unrepresentable code point; a numeric character reference by default
191
+ * @returns {string} Text holding only representable characters
192
+ *
193
+ * @example
194
+ * replaceUnencodable("€é", getOutputEncoding("latin1")); // "&#8364;é"
195
+ */
196
+ export function replaceUnencodable(
197
+ text,
198
+ encoding,
199
+ reference = characterReference,
200
+ ) {
201
+ if (encoding.isUnicode) return text;
202
+ return text.replace(NON_ASCII, (character) => {
203
+ const codePoint = character.codePointAt(0);
204
+ return encoding.bytes.has(codePoint) ? character : reference(codePoint);
205
+ });
206
+ }
207
+
208
+ /**
209
+ * Split text into runs an encoding can and cannot represent, as needed to
210
+ * write a CDATA section, which cannot hold character references.
211
+ *
212
+ * @param {string} text - Text to write
213
+ * @param {OutputEncoding} encoding - The output encoding
214
+ * @returns {Array<{text: string, representable: boolean}>} The runs, in order;
215
+ * an unrepresentable run holds one code point
216
+ */
217
+ export function splitUnencodable(text, encoding) {
218
+ if (encoding.isUnicode) return [{ text, representable: true }];
219
+ const runs = [];
220
+ let start = 0;
221
+ for (const match of text.matchAll(NON_ASCII)) {
222
+ if (encoding.bytes.has(match[0].codePointAt(0))) continue;
223
+ if (match.index > start) {
224
+ runs.push({ text: text.slice(start, match.index), representable: true });
225
+ }
226
+ runs.push({ text: match[0], representable: false });
227
+ start = match.index + match[0].length;
228
+ }
229
+ if (start < text.length || runs.length === 0) {
230
+ runs.push({ text: text.slice(start), representable: true });
231
+ }
232
+ return runs;
233
+ }
234
+
235
+ /**
236
+ * Encode UTF-16 code units, after a byte order mark when requested.
237
+ *
238
+ * @param {string} text - Text to encode
239
+ * @param {boolean} bigEndian - Byte order
240
+ * @param {boolean} bom - Whether to start with a byte order mark
241
+ * @returns {Uint8Array} The bytes
242
+ */
243
+ function encodeUtf16(text, bigEndian, bom) {
244
+ const offset = bom ? 2 : 0;
245
+ const bytes = new Uint8Array(offset + text.length * 2);
246
+ const view = new DataView(bytes.buffer);
247
+ if (bom) view.setUint16(0, 0xfeff, !bigEndian);
248
+ for (let i = 0; i < text.length; i++) {
249
+ view.setUint16(offset + i * 2, text.charCodeAt(i), !bigEndian);
250
+ }
251
+ return bytes;
252
+ }
253
+
254
+ /**
255
+ * Encode text with a single-byte encoding; a character without a byte
256
+ * (possible only in comments and processing instructions, which cannot hold
257
+ * references) becomes a character reference, as libxml2 does.
258
+ *
259
+ * @param {string} text - Text to encode
260
+ * @param {Map<number, number>} table - Byte of every representable code point
261
+ * @returns {Uint8Array} The bytes
262
+ */
263
+ function encodeSingleByte(text, table) {
264
+ const bytes = [];
265
+ for (const character of text) {
266
+ const codePoint = character.codePointAt(0);
267
+ const byte = table.get(codePoint);
268
+ if (byte === undefined) {
269
+ for (const unit of characterReference(codePoint)) {
270
+ bytes.push(unit.charCodeAt(0));
271
+ }
272
+ } else {
273
+ bytes.push(byte);
274
+ }
275
+ }
276
+ return Uint8Array.from(bytes);
277
+ }
278
+
279
+ /**
280
+ * Create an encoder turning serialized output, chunk after chunk, into the
281
+ * bytes of its declared encoding: the concatenated bytes equal
282
+ * {@link encodeOutput} of the concatenated text, since only the first chunk
283
+ * gets the UTF-16 byte order mark (chunks never split a surrogate pair, see
284
+ * chunks.js).
285
+ *
286
+ * @param {string} [label] - The `xsl:output` encoding, UTF-8 when absent
287
+ * @returns {(text: string) => Uint8Array} The stateful encoder
288
+ *
289
+ * @example
290
+ * const encode = createOutputEncoder("UTF-16");
291
+ * encode("a"); // Uint8Array [0xff, 0xfe, 0x61, 0x00]
292
+ * encode("b"); // Uint8Array [0x62, 0x00]
293
+ */
294
+ export function createOutputEncoder(label) {
295
+ const encoding = getOutputEncoding(label);
296
+ if (encoding.bytes) return (text) => encodeSingleByte(text, encoding.bytes);
297
+ if (encoding.name.startsWith("utf-16")) {
298
+ const bigEndian = encoding.name === "utf-16be";
299
+ let bom = true;
300
+ return (text) => {
301
+ const bytes = encodeUtf16(text, bigEndian, bom);
302
+ bom = false;
303
+ return bytes;
304
+ };
305
+ }
306
+ const encoder = new globalThis.TextEncoder();
307
+ return (text) => encoder.encode(text);
308
+ }
309
+
310
+ /**
311
+ * Turn serialized output into the bytes of its declared encoding.
312
+ *
313
+ * UTF-8 is written without a byte order mark, UTF-16 with one. Encodings
314
+ * without an encoder here (multi-byte encodings other than UTF, unknown
315
+ * labels) are written as UTF-8; `getOutputEncoding(label).isExact` is false
316
+ * for them.
317
+ *
318
+ * @param {string} text - Serialized output
319
+ * @param {string} [label] - The `xsl:output` encoding, UTF-8 when absent
320
+ * @returns {Uint8Array} The encoded bytes
321
+ *
322
+ * @example
323
+ * encodeOutput("é", "ISO-8859-1"); // Uint8Array [0xe9]
324
+ */
325
+ export function encodeOutput(text, label) {
326
+ return createOutputEncoder(label)(text);
327
+ }
@@ -0,0 +1,135 @@
1
+ /**
2
+ * Output Escaping Helpers
3
+ *
4
+ * Character escaping rules for the xml, xhtml and html output methods
5
+ * (XSLT 1.0 section 16).
6
+ */
7
+
8
+ const XML_TEXT_ESCAPES = { "&": "&amp;", "<": "&lt;", "\r": "&#13;" };
9
+
10
+ const HTML_TEXT_ESCAPES = { "&": "&amp;", "<": "&lt;", ">": "&gt;" };
11
+
12
+ const XML_ATTRIBUTE_ESCAPES = {
13
+ "&": "&amp;",
14
+ "<": "&lt;",
15
+ ">": "&gt;",
16
+ '"': "&quot;",
17
+ "\t": "&#9;",
18
+ "\n": "&#10;",
19
+ "\r": "&#13;",
20
+ };
21
+
22
+ const HTML_ATTRIBUTE_ESCAPES = { "&": "&amp;", '"': "&quot;" };
23
+
24
+ /**
25
+ * Replace every character matched by a pattern using a lookup table.
26
+ *
27
+ * @param {string} value - Text to escape
28
+ * @param {RegExp} pattern - Global pattern selecting the characters to replace
29
+ * @param {Record<string, string>} escapes - Character to replacement mapping
30
+ * @returns {string} Escaped text
31
+ */
32
+ function escapeWith(value, pattern, escapes) {
33
+ return String(value).replaceAll(pattern, (character) => escapes[character]);
34
+ }
35
+
36
+ /**
37
+ * Escape character data for the xml output method.
38
+ *
39
+ * `>` is only escaped where it would close a CDATA section, matching the
40
+ * "minimal escaping" rule of XSLT 1.0 section 16.1. A carriage return is
41
+ * written as `&#13;`, as libxml2 does, since an XML parser would turn a
42
+ * literal one into a line feed.
43
+ *
44
+ * @param {string} value - Text content
45
+ * @returns {string} Escaped text
46
+ *
47
+ * @example
48
+ * escapeXmlText("a<b\r"); // "a&lt;b&#13;"
49
+ */
50
+ export function escapeXmlText(value) {
51
+ return escapeWith(value, /[&<\r]/g, XML_TEXT_ESCAPES).replaceAll(
52
+ "]]>",
53
+ "]]&gt;",
54
+ );
55
+ }
56
+
57
+ /**
58
+ * Escape an attribute value for the xml output method.
59
+ *
60
+ * @param {string} value - Attribute value
61
+ * @returns {string} Escaped value
62
+ */
63
+ export function escapeXmlAttribute(value) {
64
+ return escapeWith(value, /[&<>"\t\n\r]/g, XML_ATTRIBUTE_ESCAPES);
65
+ }
66
+
67
+ /**
68
+ * Escape character data for the html output method.
69
+ *
70
+ * @param {string} value - Text content
71
+ * @returns {string} Escaped text
72
+ */
73
+ export function escapeHtmlText(value) {
74
+ return escapeWith(value, /[&<>]/g, HTML_TEXT_ESCAPES);
75
+ }
76
+
77
+ /**
78
+ * Escape an attribute value for the html output method.
79
+ *
80
+ * Only `&` and `"` are escaped: XSLT 1.0 section 16.2 keeps `<` unescaped
81
+ * and `&` unescaped before `{` (HTML 4.0 script macros), as libxslt does.
82
+ *
83
+ * @param {string} value - Attribute value
84
+ * @returns {string} Escaped value
85
+ *
86
+ * @example
87
+ * escapeHtmlAttribute('&{x} & <b> "'); // '&{x} &amp; <b> &quot;'
88
+ */
89
+ export function escapeHtmlAttribute(value) {
90
+ return escapeWith(value, /&(?!\{)|"/g, HTML_ATTRIBUTE_ESCAPES);
91
+ }
92
+
93
+ /** Leading HTML whitespace, which libxml2 writes unescaped. */
94
+ const LEADING_HTML_SPACE = /^[ \t\n\f\r]*/;
95
+
96
+ /** Runs of spaces, control characters, DEL and characters outside ASCII. */
97
+ const URI_UNSAFE_RUN = /[^!-~]+/gu;
98
+
99
+ /**
100
+ * %-escape a URI attribute value as libxml2's HTML serializer does (and so
101
+ * Chrome's XSLTProcessor): after any leading HTML whitespace, which is kept,
102
+ * spaces, control characters, DEL and non-ASCII characters become the
103
+ * %-escaped bytes of their UTF-8 encoding (XSLT 1.0 section 16.2, HTML 4.01
104
+ * appendix B.2.1). Printable ASCII, `%` included, is kept as it is.
105
+ *
106
+ * @param {string} value - Attribute value
107
+ * @returns {string} The escaped value
108
+ *
109
+ * @example
110
+ * escapeUriAttribute("/café?q=a b"); // "/caf%C3%A9?q=a%20b"
111
+ */
112
+ export function escapeUriAttribute(value) {
113
+ const text = String(value);
114
+ const leading = LEADING_HTML_SPACE.exec(text)[0];
115
+ const encoder = new globalThis.TextEncoder();
116
+ const rest = text
117
+ .slice(leading.length)
118
+ .replaceAll(URI_UNSAFE_RUN, (run) =>
119
+ Array.from(
120
+ encoder.encode(run),
121
+ (byte) => `%${byte.toString(16).toUpperCase().padStart(2, "0")}`,
122
+ ).join(""),
123
+ );
124
+ return leading + rest;
125
+ }
126
+
127
+ /**
128
+ * Wrap text in a CDATA section, splitting it around any `]]>` terminator.
129
+ *
130
+ * @param {string} value - Text content
131
+ * @returns {string} One or more CDATA sections
132
+ */
133
+ export function wrapCdata(value) {
134
+ return `<![CDATA[${String(value).replaceAll("]]>", "]]]]><![CDATA[>")}]]>`;
135
+ }
@@ -0,0 +1,168 @@
1
+ /**
2
+ * Open Node Frames
3
+ *
4
+ * The writer walks the result tree without recursion: every element (and
5
+ * the result root) whose children are being written is a frame on an
6
+ * explicit stack, so the depth of the tree is bounded by memory, not by the
7
+ * JavaScript call stack. A frame hands out its children one at a time
8
+ * ({@link ContentFrame#nextChild}) and knows the markup that closes it
9
+ * (`end`), written when it has no child left.
10
+ *
11
+ * Three kinds of content exist, as in XSLT 1.0 section 16 output:
12
+ * - {@link ContentFrame}: children written as they are, adjacent character
13
+ * data joined into one CDATA section in `cdata-section-elements`;
14
+ * - {@link TopLevelFrame}: the children of the result root (document or
15
+ * fragment), with libxslt's line break after a top-level comment;
16
+ * - {@link IndentFrame}: element-only content, each child on its own
17
+ * indented line (`indent="yes"`).
18
+ *
19
+ * @module xslt/serializer/frames
20
+ */
21
+
22
+ import { INDENT_UNIT, NODE_TYPE, TEXT_MODE } from "./constants.js";
23
+ import { isRawText } from "./rawText.js";
24
+
25
+ /**
26
+ * Children of a node written without adding whitespace.
27
+ *
28
+ * In CDATA mode (`cdata-section-elements`), adjacent character data nodes
29
+ * are one text node of the XPath data model, so they are written as one
30
+ * CDATA section, as libxslt does, not one section per DOM node; text
31
+ * written with `disable-output-escaping` ends the run.
32
+ */
33
+ export class ContentFrame {
34
+ /**
35
+ * @param {Node} parent - Node whose children are written
36
+ * @param {Map<string, string>} scope - Namespace scope for the children
37
+ * @param {number} depth - Indentation depth of the children
38
+ * @param {string} textMode - {@link TEXT_MODE} for character data children
39
+ * @param {string} end - Markup written once every child is written
40
+ */
41
+ constructor(parent, scope, depth, textMode, end) {
42
+ this.scope = scope;
43
+ this.depth = depth;
44
+ this.textMode = textMode;
45
+ this.end = end;
46
+ this.next = parent.firstChild;
47
+ this.joinsText = textMode === TEXT_MODE.CDATA;
48
+ }
49
+
50
+ /**
51
+ * Whether a child joins the pending CDATA run instead of being written.
52
+ *
53
+ * @param {Node} child - Child node
54
+ * @returns {boolean} True for escaped character data in CDATA mode
55
+ */
56
+ joins(child) {
57
+ const type = child.nodeType;
58
+ return (
59
+ (type === NODE_TYPE.TEXT || type === NODE_TYPE.CDATA_SECTION) &&
60
+ !isRawText(child)
61
+ );
62
+ }
63
+
64
+ /**
65
+ * The next child to write. In CDATA mode the character data before it is
66
+ * written first, as one CDATA section.
67
+ *
68
+ * @param {import('./baseWriter.js').BaseWriter} writer - The writer
69
+ * @returns {Node|null} The child, or null when every child is written
70
+ */
71
+ nextChild(writer) {
72
+ let child = this.next;
73
+ if (this.joinsText) {
74
+ let run = "";
75
+ while (child && this.joins(child)) {
76
+ run += child.nodeValue;
77
+ child = child.nextSibling;
78
+ }
79
+ if (run) writer.write(writer.cdataMarkup(run));
80
+ }
81
+ this.next = child ? child.nextSibling : null;
82
+ return child;
83
+ }
84
+ }
85
+
86
+ /**
87
+ * Children of the result root (document or fragment). A line break follows
88
+ * a comment that another node follows when the dialect asks for it
89
+ * (`topLevelLineBreaks`); comments are leaves, so the break is written just
90
+ * before that next node.
91
+ */
92
+ export class TopLevelFrame extends ContentFrame {
93
+ /**
94
+ * @param {Node} parent - The result root
95
+ * @param {Map<string, string>} scope - Namespace scope for the children
96
+ * @param {number} depth - Indentation depth of the children
97
+ * @param {string} textMode - {@link TEXT_MODE} for character data children
98
+ * @param {boolean} lineBreaks - Whether a top-level comment ends a line
99
+ */
100
+ constructor(parent, scope, depth, textMode, lineBreaks) {
101
+ super(parent, scope, depth, textMode, "");
102
+ this.lineBreaks = lineBreaks;
103
+ this.afterComment = false;
104
+ }
105
+
106
+ /**
107
+ * The next top-level node, preceded by a line break after a comment.
108
+ *
109
+ * @param {import('./baseWriter.js').BaseWriter} writer - The writer
110
+ * @returns {Node|null} The child, or null when every child is written
111
+ */
112
+ nextChild(writer) {
113
+ const child = super.nextChild(writer);
114
+ if (child && this.afterComment) writer.write("\n");
115
+ this.afterComment =
116
+ this.lineBreaks && child?.nodeType === NODE_TYPE.COMMENT;
117
+ return child;
118
+ }
119
+ }
120
+
121
+ /**
122
+ * Element-only content written with `indent="yes"`: each child starts a new
123
+ * line indented one level deeper than the parent, and the end tag of the
124
+ * parent starts a line at the parent's own indentation.
125
+ */
126
+ export class IndentFrame {
127
+ /**
128
+ * @param {Node[]} children - Children to indent (see indent.js)
129
+ * @param {Map<string, string>} scope - Namespace scope for the children
130
+ * @param {number} depth - Indentation depth of the parent element
131
+ * @param {string} textMode - {@link TEXT_MODE} for character data children
132
+ * @param {string} endTag - End tag of the parent element
133
+ */
134
+ constructor(children, scope, depth, textMode, endTag) {
135
+ this.children = children;
136
+ this.index = 0;
137
+ this.scope = scope;
138
+ this.depth = depth + 1;
139
+ this.textMode = textMode;
140
+ this.lineStart = IndentFrame.lineStart(this.depth);
141
+ this.end = IndentFrame.lineStart(depth) + endTag;
142
+ }
143
+
144
+ /**
145
+ * A line break followed by the indentation of a depth.
146
+ *
147
+ * @param {number} depth - Indentation depth
148
+ * @returns {string} The line start
149
+ *
150
+ * @example
151
+ * IndentFrame.lineStart(2); // "\n "
152
+ */
153
+ static lineStart(depth) {
154
+ return `\n${INDENT_UNIT.repeat(depth)}`;
155
+ }
156
+
157
+ /**
158
+ * The next child to write, on a new indented line.
159
+ *
160
+ * @param {import('./baseWriter.js').BaseWriter} writer - The writer
161
+ * @returns {Node|null} The child, or null when every child is written
162
+ */
163
+ nextChild(writer) {
164
+ if (this.index === this.children.length) return null;
165
+ writer.write(this.lineStart);
166
+ return this.children[this.index++];
167
+ }
168
+ }