@tradik/xslt-processor 1.0.3 → 1.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE.md +1 -1
- package/README.md +110 -520
- package/bin/lib/decode.js +15 -0
- package/bin/lib/dom.js +177 -0
- package/bin/lib/loaders.js +127 -0
- package/bin/lib/options.js +131 -0
- package/bin/lib/output.js +114 -0
- package/bin/lib/paths.js +186 -0
- package/bin/lib/transform.js +206 -0
- package/bin/xslt.js +73 -168
- package/dist/xslt-processor.browser.js +9564 -1585
- package/dist/xslt-processor.browser.js.map +4 -4
- package/dist/xslt-processor.browser.min.js +13 -2
- package/dist/xslt-processor.browser.min.js.map +4 -4
- package/dist/xslt-processor.cjs +9572 -1586
- package/dist/xslt-processor.cjs.map +4 -4
- package/dist/xslt-processor.d.cts +658 -0
- package/dist/xslt-processor.d.ts +459 -12
- package/dist/xslt-processor.js +9546 -1582
- package/dist/xslt-processor.js.map +4 -4
- package/package.json +71 -20
- package/src/XSLTProcessor.js +494 -48
- package/src/async/abort.js +63 -0
- package/src/async/documentUris.js +128 -0
- package/src/async/loaders.js +134 -0
- package/src/async/preload.js +159 -0
- package/src/async/processor.js +206 -0
- package/src/async/stream.js +125 -0
- package/src/bridge/engine.js +221 -0
- package/src/bridge/loader.js +78 -0
- package/src/bridge/results.js +75 -0
- package/src/bridge/version.js +63 -0
- package/src/index.js +26 -8
- package/src/io/decode.js +140 -0
- package/src/io/readSource.js +167 -0
- package/src/xpath/axes.js +562 -0
- package/src/xpath/documentOrder.js +270 -0
- package/src/xpath/evaluator.js +518 -357
- package/src/xpath/index.js +8 -2
- package/src/xpath/namespaceNodes.js +172 -0
- package/src/xpath/nodeSetFunctions.js +169 -0
- package/src/xpath/parser.js +30 -5
- package/src/xpath/strings.js +183 -0
- package/src/xpath/tokenizer.js +37 -23
- package/src/xslt/attributeSets.js +95 -0
- package/src/xslt/avt.js +103 -0
- package/src/xslt/computedNames.js +91 -0
- package/src/xslt/copying.js +212 -0
- package/src/xslt/declarationNames.js +80 -0
- package/src/xslt/domParsing.js +95 -0
- package/src/xslt/elements.js +57 -0
- package/src/xslt/engine/bindings.js +195 -0
- package/src/xslt/engine/context.js +105 -0
- package/src/xslt/engine/controlFlow.js +145 -0
- package/src/xslt/engine/copyInstructions.js +133 -0
- package/src/xslt/engine/declarations.js +233 -0
- package/src/xslt/engine/functionSupport.js +103 -0
- package/src/xslt/engine/methods.js +33 -0
- package/src/xslt/engine/nodeConstruction.js +187 -0
- package/src/xslt/engine/numbering.js +104 -0
- package/src/xslt/engine/outputDeclaration.js +77 -0
- package/src/xslt/engine/sequenceConstructor.js +228 -0
- package/src/xslt/engine/stylesheetLoading.js +208 -0
- package/src/xslt/engine/templateInvocation.js +253 -0
- package/src/xslt/engine/templateRules.js +243 -0
- package/src/xslt/engine/textInstructions.js +171 -0
- package/src/xslt/engine/topLevel.js +130 -0
- package/src/xslt/engine/transformation.js +263 -0
- package/src/xslt/engine/workStack.js +245 -0
- package/src/xslt/engine.js +184 -1736
- package/src/xslt/exslt/arguments.js +99 -0
- package/src/xslt/exslt/calendar.js +120 -0
- package/src/xslt/exslt/common.js +44 -0
- package/src/xslt/exslt/dateCalc.js +261 -0
- package/src/xslt/exslt/dateFormat.js +150 -0
- package/src/xslt/exslt/dateParse.js +265 -0
- package/src/xslt/exslt/dates.js +259 -0
- package/src/xslt/exslt/duration.js +207 -0
- package/src/xslt/exslt/dynamic.js +59 -0
- package/src/xslt/exslt/index.js +59 -0
- package/src/xslt/exslt/math.js +177 -0
- package/src/xslt/exslt/sets.js +96 -0
- package/src/xslt/exslt/stringOps.js +163 -0
- package/src/xslt/exslt/strings.js +147 -0
- package/src/xslt/exslt/uri.js +92 -0
- package/src/xslt/formatNumber.js +233 -0
- package/src/xslt/forwardsCompatible.js +75 -0
- package/src/xslt/functions.js +270 -0
- package/src/xslt/index.js +38 -1
- package/src/xslt/keys.js +164 -0
- package/src/xslt/literalResult.js +223 -0
- package/src/xslt/matchScope.js +116 -0
- package/src/xslt/number.js +271 -0
- package/src/xslt/numberFormat.js +253 -0
- package/src/xslt/outputNames.js +58 -0
- package/src/xslt/patternCompiler.js +175 -0
- package/src/xslt/patterns.js +324 -0
- package/src/xslt/qname.js +90 -0
- package/src/xslt/resultDocument.js +98 -0
- package/src/xslt/resultNamespaces.js +219 -0
- package/src/xslt/resultTree.js +211 -0
- package/src/xslt/serializer/baseWriter.js +390 -0
- package/src/xslt/serializer/chunks.js +120 -0
- package/src/xslt/serializer/constants.js +92 -0
- package/src/xslt/serializer/encoding.js +327 -0
- package/src/xslt/serializer/escape.js +135 -0
- package/src/xslt/serializer/frames.js +168 -0
- package/src/xslt/serializer/htmlDoctype.js +102 -0
- package/src/xslt/serializer/htmlEntities.js +77 -0
- package/src/xslt/serializer/htmlSerializer.js +239 -0
- package/src/xslt/serializer/indent.js +51 -0
- package/src/xslt/serializer/namespaces.js +68 -0
- package/src/xslt/serializer/rawText.js +41 -0
- package/src/xslt/serializer/settings.js +179 -0
- package/src/xslt/serializer/textSerializer.js +77 -0
- package/src/xslt/serializer/xhtmlDocument.js +103 -0
- package/src/xslt/serializer/xmlSerializer.js +227 -0
- package/src/xslt/serializer.js +90 -0
- package/src/xslt/sort.js +151 -0
- package/src/xslt/spaceNameTests.js +115 -0
- package/src/xslt/stylesheetChecks.js +206 -0
- package/src/xslt/stylesheetNamespaces.js +266 -0
- package/src/xslt/templatePriority.js +45 -0
- package/src/xslt/uri.js +68 -0
- package/src/xslt/variables.js +152 -0
- package/src/xslt/whitespace.js +200 -0
- package/LICENSE +0 -29
- package/src/XSLTProcessor.test.js +0 -930
- package/src/xpath/evaluator.test.js +0 -1852
- package/src/xpath/tokenizer.test.js +0 -224
- package/src/xslt/engine.test.js +0 -3130
|
@@ -0,0 +1,327 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Output Encodings
|
|
3
|
+
*
|
|
4
|
+
* The `encoding` attribute of `xsl:output` names the character encoding of the
|
|
5
|
+
* serialized result. A character the encoding cannot represent is written as
|
|
6
|
+
* a character reference (XSLT 1.0 section 16.1: `€` in ISO-8859-1), so
|
|
7
|
+
* the serialized string only holds characters the declared encoding has, and
|
|
8
|
+
* turning it into bytes (see {@link encodeOutput}, used by the CLI) is
|
|
9
|
+
* lossless.
|
|
10
|
+
*
|
|
11
|
+
* Supported encodings:
|
|
12
|
+
* - UTF-8 and UTF-16 (every label TextDecoder maps to them) represent every
|
|
13
|
+
* character; UTF-16 output starts with a byte order mark;
|
|
14
|
+
* - ISO-8859-1 / latin1 (up to U+00FF) and US-ASCII (up to U+007F), handled
|
|
15
|
+
* here because the WHATWG Encoding Standard aliases both to windows-1252;
|
|
16
|
+
* - every other single-byte encoding known to TextDecoder (windows-125x,
|
|
17
|
+
* ISO-8859-x, KOI8-R/U, IBM866, macintosh, ...): its repertoire is read
|
|
18
|
+
* lazily by decoding the 256 byte values once;
|
|
19
|
+
* - multi-byte encodings other than UTF (Shift_JIS, EUC-JP, GBK, Big5, ...) and
|
|
20
|
+
* unknown labels are treated as able to represent every character: no
|
|
21
|
+
* character reference is written, and {@link encodeOutput} falls back to
|
|
22
|
+
* UTF-8 bytes (`isExact` is false so callers can warn).
|
|
23
|
+
*
|
|
24
|
+
* @module xslt/serializer/encoding
|
|
25
|
+
*/
|
|
26
|
+
|
|
27
|
+
/** Labels of US-ASCII (IANA names and aliases). */
|
|
28
|
+
const ASCII_LABELS = new Set([
|
|
29
|
+
"ascii",
|
|
30
|
+
"us-ascii",
|
|
31
|
+
"ansi_x3.4-1968",
|
|
32
|
+
"iso646-us",
|
|
33
|
+
"iso-ir-6",
|
|
34
|
+
"csascii",
|
|
35
|
+
"us",
|
|
36
|
+
]);
|
|
37
|
+
|
|
38
|
+
/** Labels of ISO-8859-1 (IANA names and aliases). */
|
|
39
|
+
const LATIN1_LABELS = new Set([
|
|
40
|
+
"iso-8859-1",
|
|
41
|
+
"iso8859-1",
|
|
42
|
+
"iso88591",
|
|
43
|
+
"iso_8859-1",
|
|
44
|
+
"iso_8859-1:1987",
|
|
45
|
+
"latin1",
|
|
46
|
+
"l1",
|
|
47
|
+
"cp819",
|
|
48
|
+
"ibm819",
|
|
49
|
+
"csisolatin1",
|
|
50
|
+
"iso-ir-100",
|
|
51
|
+
]);
|
|
52
|
+
|
|
53
|
+
/** Canonical WHATWG names of the single-byte encodings. */
|
|
54
|
+
const SINGLE_BYTE_ENCODINGS = new Set([
|
|
55
|
+
"ibm866",
|
|
56
|
+
"iso-8859-2",
|
|
57
|
+
"iso-8859-3",
|
|
58
|
+
"iso-8859-4",
|
|
59
|
+
"iso-8859-5",
|
|
60
|
+
"iso-8859-6",
|
|
61
|
+
"iso-8859-7",
|
|
62
|
+
"iso-8859-8",
|
|
63
|
+
"iso-8859-8-i",
|
|
64
|
+
"iso-8859-10",
|
|
65
|
+
"iso-8859-13",
|
|
66
|
+
"iso-8859-14",
|
|
67
|
+
"iso-8859-15",
|
|
68
|
+
"iso-8859-16",
|
|
69
|
+
"koi8-r",
|
|
70
|
+
"koi8-u",
|
|
71
|
+
"macintosh",
|
|
72
|
+
"windows-874",
|
|
73
|
+
"windows-1250",
|
|
74
|
+
"windows-1251",
|
|
75
|
+
"windows-1252",
|
|
76
|
+
"windows-1253",
|
|
77
|
+
"windows-1254",
|
|
78
|
+
"windows-1255",
|
|
79
|
+
"windows-1256",
|
|
80
|
+
"windows-1257",
|
|
81
|
+
"windows-1258",
|
|
82
|
+
"x-mac-cyrillic",
|
|
83
|
+
]);
|
|
84
|
+
|
|
85
|
+
/**
|
|
86
|
+
* @typedef {Object} OutputEncoding
|
|
87
|
+
* @property {string} name - Canonical name ("utf-8", "utf-16le", "iso-8859-1"...)
|
|
88
|
+
* @property {boolean} isUnicode - True when every character is representable
|
|
89
|
+
* @property {boolean} isExact - False when bytes fall back to UTF-8
|
|
90
|
+
* @property {Map<number, number>|null} bytes - Byte of every representable
|
|
91
|
+
* code point, for single-byte encodings
|
|
92
|
+
*/
|
|
93
|
+
|
|
94
|
+
/** Encodings already resolved, by lower-cased label. */
|
|
95
|
+
const cache = new Map();
|
|
96
|
+
|
|
97
|
+
/**
|
|
98
|
+
* Build a single-byte encoding whose bytes map to the same code points.
|
|
99
|
+
*
|
|
100
|
+
* @param {string} name - Canonical name
|
|
101
|
+
* @param {number} last - Highest byte value (0x7F or 0xFF)
|
|
102
|
+
* @returns {OutputEncoding} The encoding
|
|
103
|
+
*/
|
|
104
|
+
function identityEncoding(name, last) {
|
|
105
|
+
const bytes = new Map();
|
|
106
|
+
for (let byte = 0; byte <= last; byte++) bytes.set(byte, byte);
|
|
107
|
+
return { name, isUnicode: false, isExact: true, bytes };
|
|
108
|
+
}
|
|
109
|
+
|
|
110
|
+
/**
|
|
111
|
+
* Build a single-byte encoding from the repertoire TextDecoder reports.
|
|
112
|
+
*
|
|
113
|
+
* @param {TextDecoder} decoder - Decoder of the encoding
|
|
114
|
+
* @returns {OutputEncoding} The encoding
|
|
115
|
+
*/
|
|
116
|
+
function tableEncoding(decoder) {
|
|
117
|
+
const bytes = new Map();
|
|
118
|
+
for (let byte = 0; byte <= 0xff; byte++) {
|
|
119
|
+
const codePoint = decoder.decode(Uint8Array.of(byte)).codePointAt(0);
|
|
120
|
+
if (codePoint !== 0xfffd && !bytes.has(codePoint)) {
|
|
121
|
+
bytes.set(codePoint, byte);
|
|
122
|
+
}
|
|
123
|
+
}
|
|
124
|
+
return { name: decoder.encoding, isUnicode: false, isExact: true, bytes };
|
|
125
|
+
}
|
|
126
|
+
|
|
127
|
+
/**
|
|
128
|
+
* Resolve an encoding label that is not ASCII or ISO-8859-1.
|
|
129
|
+
*
|
|
130
|
+
* @param {string} label - Lower-cased, trimmed label
|
|
131
|
+
* @returns {OutputEncoding} The encoding
|
|
132
|
+
*/
|
|
133
|
+
function resolveWithDecoder(label) {
|
|
134
|
+
let decoder;
|
|
135
|
+
try {
|
|
136
|
+
decoder = new globalThis.TextDecoder(label);
|
|
137
|
+
} catch {
|
|
138
|
+
return { name: label, isUnicode: true, isExact: false, bytes: null };
|
|
139
|
+
}
|
|
140
|
+
const name = decoder.encoding;
|
|
141
|
+
if (SINGLE_BYTE_ENCODINGS.has(name)) return tableEncoding(decoder);
|
|
142
|
+
const isExact = name === "utf-8" || name.startsWith("utf-16");
|
|
143
|
+
return { name, isUnicode: true, isExact, bytes: null };
|
|
144
|
+
}
|
|
145
|
+
|
|
146
|
+
/**
|
|
147
|
+
* Resolve the `encoding` of `xsl:output`.
|
|
148
|
+
*
|
|
149
|
+
* @param {string} [label] - Encoding label, UTF-8 when absent
|
|
150
|
+
* @returns {OutputEncoding} The encoding
|
|
151
|
+
*
|
|
152
|
+
* @example
|
|
153
|
+
* getOutputEncoding("ISO-8859-1").bytes.has(0x20ac); // false
|
|
154
|
+
*/
|
|
155
|
+
export function getOutputEncoding(label = "UTF-8") {
|
|
156
|
+
const key = String(label).trim().toLowerCase();
|
|
157
|
+
let encoding = cache.get(key);
|
|
158
|
+
if (!encoding) {
|
|
159
|
+
if (ASCII_LABELS.has(key)) encoding = identityEncoding("us-ascii", 0x7f);
|
|
160
|
+
else if (LATIN1_LABELS.has(key)) {
|
|
161
|
+
encoding = identityEncoding("iso-8859-1", 0xff);
|
|
162
|
+
} else encoding = resolveWithDecoder(key);
|
|
163
|
+
cache.set(key, encoding);
|
|
164
|
+
}
|
|
165
|
+
return encoding;
|
|
166
|
+
}
|
|
167
|
+
|
|
168
|
+
/**
|
|
169
|
+
* Numeric character reference of a code point.
|
|
170
|
+
*
|
|
171
|
+
* @param {number} codePoint - The code point
|
|
172
|
+
* @returns {string} `&#N;`
|
|
173
|
+
*/
|
|
174
|
+
export function characterReference(codePoint) {
|
|
175
|
+
return `&#${codePoint};`;
|
|
176
|
+
}
|
|
177
|
+
|
|
178
|
+
/** Characters outside ASCII, as whole code points. */
|
|
179
|
+
const NON_ASCII = /[\u{80}-\u{10FFFF}]/gu;
|
|
180
|
+
|
|
181
|
+
/**
|
|
182
|
+
* Replace every character an encoding cannot represent.
|
|
183
|
+
*
|
|
184
|
+
* Every supported encoding is a superset of ASCII, so only non-ASCII
|
|
185
|
+
* characters are checked.
|
|
186
|
+
*
|
|
187
|
+
* @param {string} text - Text to write
|
|
188
|
+
* @param {OutputEncoding} encoding - The output encoding
|
|
189
|
+
* @param {(codePoint: number) => string} [reference] - Replacement of an
|
|
190
|
+
* unrepresentable code point; a numeric character reference by default
|
|
191
|
+
* @returns {string} Text holding only representable characters
|
|
192
|
+
*
|
|
193
|
+
* @example
|
|
194
|
+
* replaceUnencodable("ێ", getOutputEncoding("latin1")); // "ێ"
|
|
195
|
+
*/
|
|
196
|
+
export function replaceUnencodable(
|
|
197
|
+
text,
|
|
198
|
+
encoding,
|
|
199
|
+
reference = characterReference,
|
|
200
|
+
) {
|
|
201
|
+
if (encoding.isUnicode) return text;
|
|
202
|
+
return text.replace(NON_ASCII, (character) => {
|
|
203
|
+
const codePoint = character.codePointAt(0);
|
|
204
|
+
return encoding.bytes.has(codePoint) ? character : reference(codePoint);
|
|
205
|
+
});
|
|
206
|
+
}
|
|
207
|
+
|
|
208
|
+
/**
|
|
209
|
+
* Split text into runs an encoding can and cannot represent, as needed to
|
|
210
|
+
* write a CDATA section, which cannot hold character references.
|
|
211
|
+
*
|
|
212
|
+
* @param {string} text - Text to write
|
|
213
|
+
* @param {OutputEncoding} encoding - The output encoding
|
|
214
|
+
* @returns {Array<{text: string, representable: boolean}>} The runs, in order;
|
|
215
|
+
* an unrepresentable run holds one code point
|
|
216
|
+
*/
|
|
217
|
+
export function splitUnencodable(text, encoding) {
|
|
218
|
+
if (encoding.isUnicode) return [{ text, representable: true }];
|
|
219
|
+
const runs = [];
|
|
220
|
+
let start = 0;
|
|
221
|
+
for (const match of text.matchAll(NON_ASCII)) {
|
|
222
|
+
if (encoding.bytes.has(match[0].codePointAt(0))) continue;
|
|
223
|
+
if (match.index > start) {
|
|
224
|
+
runs.push({ text: text.slice(start, match.index), representable: true });
|
|
225
|
+
}
|
|
226
|
+
runs.push({ text: match[0], representable: false });
|
|
227
|
+
start = match.index + match[0].length;
|
|
228
|
+
}
|
|
229
|
+
if (start < text.length || runs.length === 0) {
|
|
230
|
+
runs.push({ text: text.slice(start), representable: true });
|
|
231
|
+
}
|
|
232
|
+
return runs;
|
|
233
|
+
}
|
|
234
|
+
|
|
235
|
+
/**
|
|
236
|
+
* Encode UTF-16 code units, after a byte order mark when requested.
|
|
237
|
+
*
|
|
238
|
+
* @param {string} text - Text to encode
|
|
239
|
+
* @param {boolean} bigEndian - Byte order
|
|
240
|
+
* @param {boolean} bom - Whether to start with a byte order mark
|
|
241
|
+
* @returns {Uint8Array} The bytes
|
|
242
|
+
*/
|
|
243
|
+
function encodeUtf16(text, bigEndian, bom) {
|
|
244
|
+
const offset = bom ? 2 : 0;
|
|
245
|
+
const bytes = new Uint8Array(offset + text.length * 2);
|
|
246
|
+
const view = new DataView(bytes.buffer);
|
|
247
|
+
if (bom) view.setUint16(0, 0xfeff, !bigEndian);
|
|
248
|
+
for (let i = 0; i < text.length; i++) {
|
|
249
|
+
view.setUint16(offset + i * 2, text.charCodeAt(i), !bigEndian);
|
|
250
|
+
}
|
|
251
|
+
return bytes;
|
|
252
|
+
}
|
|
253
|
+
|
|
254
|
+
/**
|
|
255
|
+
* Encode text with a single-byte encoding; a character without a byte
|
|
256
|
+
* (possible only in comments and processing instructions, which cannot hold
|
|
257
|
+
* references) becomes a character reference, as libxml2 does.
|
|
258
|
+
*
|
|
259
|
+
* @param {string} text - Text to encode
|
|
260
|
+
* @param {Map<number, number>} table - Byte of every representable code point
|
|
261
|
+
* @returns {Uint8Array} The bytes
|
|
262
|
+
*/
|
|
263
|
+
function encodeSingleByte(text, table) {
|
|
264
|
+
const bytes = [];
|
|
265
|
+
for (const character of text) {
|
|
266
|
+
const codePoint = character.codePointAt(0);
|
|
267
|
+
const byte = table.get(codePoint);
|
|
268
|
+
if (byte === undefined) {
|
|
269
|
+
for (const unit of characterReference(codePoint)) {
|
|
270
|
+
bytes.push(unit.charCodeAt(0));
|
|
271
|
+
}
|
|
272
|
+
} else {
|
|
273
|
+
bytes.push(byte);
|
|
274
|
+
}
|
|
275
|
+
}
|
|
276
|
+
return Uint8Array.from(bytes);
|
|
277
|
+
}
|
|
278
|
+
|
|
279
|
+
/**
|
|
280
|
+
* Create an encoder turning serialized output, chunk after chunk, into the
|
|
281
|
+
* bytes of its declared encoding: the concatenated bytes equal
|
|
282
|
+
* {@link encodeOutput} of the concatenated text, since only the first chunk
|
|
283
|
+
* gets the UTF-16 byte order mark (chunks never split a surrogate pair, see
|
|
284
|
+
* chunks.js).
|
|
285
|
+
*
|
|
286
|
+
* @param {string} [label] - The `xsl:output` encoding, UTF-8 when absent
|
|
287
|
+
* @returns {(text: string) => Uint8Array} The stateful encoder
|
|
288
|
+
*
|
|
289
|
+
* @example
|
|
290
|
+
* const encode = createOutputEncoder("UTF-16");
|
|
291
|
+
* encode("a"); // Uint8Array [0xff, 0xfe, 0x61, 0x00]
|
|
292
|
+
* encode("b"); // Uint8Array [0x62, 0x00]
|
|
293
|
+
*/
|
|
294
|
+
export function createOutputEncoder(label) {
|
|
295
|
+
const encoding = getOutputEncoding(label);
|
|
296
|
+
if (encoding.bytes) return (text) => encodeSingleByte(text, encoding.bytes);
|
|
297
|
+
if (encoding.name.startsWith("utf-16")) {
|
|
298
|
+
const bigEndian = encoding.name === "utf-16be";
|
|
299
|
+
let bom = true;
|
|
300
|
+
return (text) => {
|
|
301
|
+
const bytes = encodeUtf16(text, bigEndian, bom);
|
|
302
|
+
bom = false;
|
|
303
|
+
return bytes;
|
|
304
|
+
};
|
|
305
|
+
}
|
|
306
|
+
const encoder = new globalThis.TextEncoder();
|
|
307
|
+
return (text) => encoder.encode(text);
|
|
308
|
+
}
|
|
309
|
+
|
|
310
|
+
/**
|
|
311
|
+
* Turn serialized output into the bytes of its declared encoding.
|
|
312
|
+
*
|
|
313
|
+
* UTF-8 is written without a byte order mark, UTF-16 with one. Encodings
|
|
314
|
+
* without an encoder here (multi-byte encodings other than UTF, unknown
|
|
315
|
+
* labels) are written as UTF-8; `getOutputEncoding(label).isExact` is false
|
|
316
|
+
* for them.
|
|
317
|
+
*
|
|
318
|
+
* @param {string} text - Serialized output
|
|
319
|
+
* @param {string} [label] - The `xsl:output` encoding, UTF-8 when absent
|
|
320
|
+
* @returns {Uint8Array} The encoded bytes
|
|
321
|
+
*
|
|
322
|
+
* @example
|
|
323
|
+
* encodeOutput("é", "ISO-8859-1"); // Uint8Array [0xe9]
|
|
324
|
+
*/
|
|
325
|
+
export function encodeOutput(text, label) {
|
|
326
|
+
return createOutputEncoder(label)(text);
|
|
327
|
+
}
|
|
@@ -0,0 +1,135 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Output Escaping Helpers
|
|
3
|
+
*
|
|
4
|
+
* Character escaping rules for the xml, xhtml and html output methods
|
|
5
|
+
* (XSLT 1.0 section 16).
|
|
6
|
+
*/
|
|
7
|
+
|
|
8
|
+
const XML_TEXT_ESCAPES = { "&": "&", "<": "<", "\r": " " };
|
|
9
|
+
|
|
10
|
+
const HTML_TEXT_ESCAPES = { "&": "&", "<": "<", ">": ">" };
|
|
11
|
+
|
|
12
|
+
const XML_ATTRIBUTE_ESCAPES = {
|
|
13
|
+
"&": "&",
|
|
14
|
+
"<": "<",
|
|
15
|
+
">": ">",
|
|
16
|
+
'"': """,
|
|
17
|
+
"\t": "	",
|
|
18
|
+
"\n": " ",
|
|
19
|
+
"\r": " ",
|
|
20
|
+
};
|
|
21
|
+
|
|
22
|
+
const HTML_ATTRIBUTE_ESCAPES = { "&": "&", '"': """ };
|
|
23
|
+
|
|
24
|
+
/**
|
|
25
|
+
* Replace every character matched by a pattern using a lookup table.
|
|
26
|
+
*
|
|
27
|
+
* @param {string} value - Text to escape
|
|
28
|
+
* @param {RegExp} pattern - Global pattern selecting the characters to replace
|
|
29
|
+
* @param {Record<string, string>} escapes - Character to replacement mapping
|
|
30
|
+
* @returns {string} Escaped text
|
|
31
|
+
*/
|
|
32
|
+
function escapeWith(value, pattern, escapes) {
|
|
33
|
+
return String(value).replaceAll(pattern, (character) => escapes[character]);
|
|
34
|
+
}
|
|
35
|
+
|
|
36
|
+
/**
|
|
37
|
+
* Escape character data for the xml output method.
|
|
38
|
+
*
|
|
39
|
+
* `>` is only escaped where it would close a CDATA section, matching the
|
|
40
|
+
* "minimal escaping" rule of XSLT 1.0 section 16.1. A carriage return is
|
|
41
|
+
* written as ` `, as libxml2 does, since an XML parser would turn a
|
|
42
|
+
* literal one into a line feed.
|
|
43
|
+
*
|
|
44
|
+
* @param {string} value - Text content
|
|
45
|
+
* @returns {string} Escaped text
|
|
46
|
+
*
|
|
47
|
+
* @example
|
|
48
|
+
* escapeXmlText("a<b\r"); // "a<b "
|
|
49
|
+
*/
|
|
50
|
+
export function escapeXmlText(value) {
|
|
51
|
+
return escapeWith(value, /[&<\r]/g, XML_TEXT_ESCAPES).replaceAll(
|
|
52
|
+
"]]>",
|
|
53
|
+
"]]>",
|
|
54
|
+
);
|
|
55
|
+
}
|
|
56
|
+
|
|
57
|
+
/**
|
|
58
|
+
* Escape an attribute value for the xml output method.
|
|
59
|
+
*
|
|
60
|
+
* @param {string} value - Attribute value
|
|
61
|
+
* @returns {string} Escaped value
|
|
62
|
+
*/
|
|
63
|
+
export function escapeXmlAttribute(value) {
|
|
64
|
+
return escapeWith(value, /[&<>"\t\n\r]/g, XML_ATTRIBUTE_ESCAPES);
|
|
65
|
+
}
|
|
66
|
+
|
|
67
|
+
/**
|
|
68
|
+
* Escape character data for the html output method.
|
|
69
|
+
*
|
|
70
|
+
* @param {string} value - Text content
|
|
71
|
+
* @returns {string} Escaped text
|
|
72
|
+
*/
|
|
73
|
+
export function escapeHtmlText(value) {
|
|
74
|
+
return escapeWith(value, /[&<>]/g, HTML_TEXT_ESCAPES);
|
|
75
|
+
}
|
|
76
|
+
|
|
77
|
+
/**
|
|
78
|
+
* Escape an attribute value for the html output method.
|
|
79
|
+
*
|
|
80
|
+
* Only `&` and `"` are escaped: XSLT 1.0 section 16.2 keeps `<` unescaped
|
|
81
|
+
* and `&` unescaped before `{` (HTML 4.0 script macros), as libxslt does.
|
|
82
|
+
*
|
|
83
|
+
* @param {string} value - Attribute value
|
|
84
|
+
* @returns {string} Escaped value
|
|
85
|
+
*
|
|
86
|
+
* @example
|
|
87
|
+
* escapeHtmlAttribute('&{x} & <b> "'); // '&{x} & <b> "'
|
|
88
|
+
*/
|
|
89
|
+
export function escapeHtmlAttribute(value) {
|
|
90
|
+
return escapeWith(value, /&(?!\{)|"/g, HTML_ATTRIBUTE_ESCAPES);
|
|
91
|
+
}
|
|
92
|
+
|
|
93
|
+
/** Leading HTML whitespace, which libxml2 writes unescaped. */
|
|
94
|
+
const LEADING_HTML_SPACE = /^[ \t\n\f\r]*/;
|
|
95
|
+
|
|
96
|
+
/** Runs of spaces, control characters, DEL and characters outside ASCII. */
|
|
97
|
+
const URI_UNSAFE_RUN = /[^!-~]+/gu;
|
|
98
|
+
|
|
99
|
+
/**
|
|
100
|
+
* %-escape a URI attribute value as libxml2's HTML serializer does (and so
|
|
101
|
+
* Chrome's XSLTProcessor): after any leading HTML whitespace, which is kept,
|
|
102
|
+
* spaces, control characters, DEL and non-ASCII characters become the
|
|
103
|
+
* %-escaped bytes of their UTF-8 encoding (XSLT 1.0 section 16.2, HTML 4.01
|
|
104
|
+
* appendix B.2.1). Printable ASCII, `%` included, is kept as it is.
|
|
105
|
+
*
|
|
106
|
+
* @param {string} value - Attribute value
|
|
107
|
+
* @returns {string} The escaped value
|
|
108
|
+
*
|
|
109
|
+
* @example
|
|
110
|
+
* escapeUriAttribute("/café?q=a b"); // "/caf%C3%A9?q=a%20b"
|
|
111
|
+
*/
|
|
112
|
+
export function escapeUriAttribute(value) {
|
|
113
|
+
const text = String(value);
|
|
114
|
+
const leading = LEADING_HTML_SPACE.exec(text)[0];
|
|
115
|
+
const encoder = new globalThis.TextEncoder();
|
|
116
|
+
const rest = text
|
|
117
|
+
.slice(leading.length)
|
|
118
|
+
.replaceAll(URI_UNSAFE_RUN, (run) =>
|
|
119
|
+
Array.from(
|
|
120
|
+
encoder.encode(run),
|
|
121
|
+
(byte) => `%${byte.toString(16).toUpperCase().padStart(2, "0")}`,
|
|
122
|
+
).join(""),
|
|
123
|
+
);
|
|
124
|
+
return leading + rest;
|
|
125
|
+
}
|
|
126
|
+
|
|
127
|
+
/**
|
|
128
|
+
* Wrap text in a CDATA section, splitting it around any `]]>` terminator.
|
|
129
|
+
*
|
|
130
|
+
* @param {string} value - Text content
|
|
131
|
+
* @returns {string} One or more CDATA sections
|
|
132
|
+
*/
|
|
133
|
+
export function wrapCdata(value) {
|
|
134
|
+
return `<![CDATA[${String(value).replaceAll("]]>", "]]]]><![CDATA[>")}]]>`;
|
|
135
|
+
}
|
|
@@ -0,0 +1,168 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Open Node Frames
|
|
3
|
+
*
|
|
4
|
+
* The writer walks the result tree without recursion: every element (and
|
|
5
|
+
* the result root) whose children are being written is a frame on an
|
|
6
|
+
* explicit stack, so the depth of the tree is bounded by memory, not by the
|
|
7
|
+
* JavaScript call stack. A frame hands out its children one at a time
|
|
8
|
+
* ({@link ContentFrame#nextChild}) and knows the markup that closes it
|
|
9
|
+
* (`end`), written when it has no child left.
|
|
10
|
+
*
|
|
11
|
+
* Three kinds of content exist, as in XSLT 1.0 section 16 output:
|
|
12
|
+
* - {@link ContentFrame}: children written as they are, adjacent character
|
|
13
|
+
* data joined into one CDATA section in `cdata-section-elements`;
|
|
14
|
+
* - {@link TopLevelFrame}: the children of the result root (document or
|
|
15
|
+
* fragment), with libxslt's line break after a top-level comment;
|
|
16
|
+
* - {@link IndentFrame}: element-only content, each child on its own
|
|
17
|
+
* indented line (`indent="yes"`).
|
|
18
|
+
*
|
|
19
|
+
* @module xslt/serializer/frames
|
|
20
|
+
*/
|
|
21
|
+
|
|
22
|
+
import { INDENT_UNIT, NODE_TYPE, TEXT_MODE } from "./constants.js";
|
|
23
|
+
import { isRawText } from "./rawText.js";
|
|
24
|
+
|
|
25
|
+
/**
|
|
26
|
+
* Children of a node written without adding whitespace.
|
|
27
|
+
*
|
|
28
|
+
* In CDATA mode (`cdata-section-elements`), adjacent character data nodes
|
|
29
|
+
* are one text node of the XPath data model, so they are written as one
|
|
30
|
+
* CDATA section, as libxslt does, not one section per DOM node; text
|
|
31
|
+
* written with `disable-output-escaping` ends the run.
|
|
32
|
+
*/
|
|
33
|
+
export class ContentFrame {
|
|
34
|
+
/**
|
|
35
|
+
* @param {Node} parent - Node whose children are written
|
|
36
|
+
* @param {Map<string, string>} scope - Namespace scope for the children
|
|
37
|
+
* @param {number} depth - Indentation depth of the children
|
|
38
|
+
* @param {string} textMode - {@link TEXT_MODE} for character data children
|
|
39
|
+
* @param {string} end - Markup written once every child is written
|
|
40
|
+
*/
|
|
41
|
+
constructor(parent, scope, depth, textMode, end) {
|
|
42
|
+
this.scope = scope;
|
|
43
|
+
this.depth = depth;
|
|
44
|
+
this.textMode = textMode;
|
|
45
|
+
this.end = end;
|
|
46
|
+
this.next = parent.firstChild;
|
|
47
|
+
this.joinsText = textMode === TEXT_MODE.CDATA;
|
|
48
|
+
}
|
|
49
|
+
|
|
50
|
+
/**
|
|
51
|
+
* Whether a child joins the pending CDATA run instead of being written.
|
|
52
|
+
*
|
|
53
|
+
* @param {Node} child - Child node
|
|
54
|
+
* @returns {boolean} True for escaped character data in CDATA mode
|
|
55
|
+
*/
|
|
56
|
+
joins(child) {
|
|
57
|
+
const type = child.nodeType;
|
|
58
|
+
return (
|
|
59
|
+
(type === NODE_TYPE.TEXT || type === NODE_TYPE.CDATA_SECTION) &&
|
|
60
|
+
!isRawText(child)
|
|
61
|
+
);
|
|
62
|
+
}
|
|
63
|
+
|
|
64
|
+
/**
|
|
65
|
+
* The next child to write. In CDATA mode the character data before it is
|
|
66
|
+
* written first, as one CDATA section.
|
|
67
|
+
*
|
|
68
|
+
* @param {import('./baseWriter.js').BaseWriter} writer - The writer
|
|
69
|
+
* @returns {Node|null} The child, or null when every child is written
|
|
70
|
+
*/
|
|
71
|
+
nextChild(writer) {
|
|
72
|
+
let child = this.next;
|
|
73
|
+
if (this.joinsText) {
|
|
74
|
+
let run = "";
|
|
75
|
+
while (child && this.joins(child)) {
|
|
76
|
+
run += child.nodeValue;
|
|
77
|
+
child = child.nextSibling;
|
|
78
|
+
}
|
|
79
|
+
if (run) writer.write(writer.cdataMarkup(run));
|
|
80
|
+
}
|
|
81
|
+
this.next = child ? child.nextSibling : null;
|
|
82
|
+
return child;
|
|
83
|
+
}
|
|
84
|
+
}
|
|
85
|
+
|
|
86
|
+
/**
|
|
87
|
+
* Children of the result root (document or fragment). A line break follows
|
|
88
|
+
* a comment that another node follows when the dialect asks for it
|
|
89
|
+
* (`topLevelLineBreaks`); comments are leaves, so the break is written just
|
|
90
|
+
* before that next node.
|
|
91
|
+
*/
|
|
92
|
+
export class TopLevelFrame extends ContentFrame {
|
|
93
|
+
/**
|
|
94
|
+
* @param {Node} parent - The result root
|
|
95
|
+
* @param {Map<string, string>} scope - Namespace scope for the children
|
|
96
|
+
* @param {number} depth - Indentation depth of the children
|
|
97
|
+
* @param {string} textMode - {@link TEXT_MODE} for character data children
|
|
98
|
+
* @param {boolean} lineBreaks - Whether a top-level comment ends a line
|
|
99
|
+
*/
|
|
100
|
+
constructor(parent, scope, depth, textMode, lineBreaks) {
|
|
101
|
+
super(parent, scope, depth, textMode, "");
|
|
102
|
+
this.lineBreaks = lineBreaks;
|
|
103
|
+
this.afterComment = false;
|
|
104
|
+
}
|
|
105
|
+
|
|
106
|
+
/**
|
|
107
|
+
* The next top-level node, preceded by a line break after a comment.
|
|
108
|
+
*
|
|
109
|
+
* @param {import('./baseWriter.js').BaseWriter} writer - The writer
|
|
110
|
+
* @returns {Node|null} The child, or null when every child is written
|
|
111
|
+
*/
|
|
112
|
+
nextChild(writer) {
|
|
113
|
+
const child = super.nextChild(writer);
|
|
114
|
+
if (child && this.afterComment) writer.write("\n");
|
|
115
|
+
this.afterComment =
|
|
116
|
+
this.lineBreaks && child?.nodeType === NODE_TYPE.COMMENT;
|
|
117
|
+
return child;
|
|
118
|
+
}
|
|
119
|
+
}
|
|
120
|
+
|
|
121
|
+
/**
|
|
122
|
+
* Element-only content written with `indent="yes"`: each child starts a new
|
|
123
|
+
* line indented one level deeper than the parent, and the end tag of the
|
|
124
|
+
* parent starts a line at the parent's own indentation.
|
|
125
|
+
*/
|
|
126
|
+
export class IndentFrame {
|
|
127
|
+
/**
|
|
128
|
+
* @param {Node[]} children - Children to indent (see indent.js)
|
|
129
|
+
* @param {Map<string, string>} scope - Namespace scope for the children
|
|
130
|
+
* @param {number} depth - Indentation depth of the parent element
|
|
131
|
+
* @param {string} textMode - {@link TEXT_MODE} for character data children
|
|
132
|
+
* @param {string} endTag - End tag of the parent element
|
|
133
|
+
*/
|
|
134
|
+
constructor(children, scope, depth, textMode, endTag) {
|
|
135
|
+
this.children = children;
|
|
136
|
+
this.index = 0;
|
|
137
|
+
this.scope = scope;
|
|
138
|
+
this.depth = depth + 1;
|
|
139
|
+
this.textMode = textMode;
|
|
140
|
+
this.lineStart = IndentFrame.lineStart(this.depth);
|
|
141
|
+
this.end = IndentFrame.lineStart(depth) + endTag;
|
|
142
|
+
}
|
|
143
|
+
|
|
144
|
+
/**
|
|
145
|
+
* A line break followed by the indentation of a depth.
|
|
146
|
+
*
|
|
147
|
+
* @param {number} depth - Indentation depth
|
|
148
|
+
* @returns {string} The line start
|
|
149
|
+
*
|
|
150
|
+
* @example
|
|
151
|
+
* IndentFrame.lineStart(2); // "\n "
|
|
152
|
+
*/
|
|
153
|
+
static lineStart(depth) {
|
|
154
|
+
return `\n${INDENT_UNIT.repeat(depth)}`;
|
|
155
|
+
}
|
|
156
|
+
|
|
157
|
+
/**
|
|
158
|
+
* The next child to write, on a new indented line.
|
|
159
|
+
*
|
|
160
|
+
* @param {import('./baseWriter.js').BaseWriter} writer - The writer
|
|
161
|
+
* @returns {Node|null} The child, or null when every child is written
|
|
162
|
+
*/
|
|
163
|
+
nextChild(writer) {
|
|
164
|
+
if (this.index === this.children.length) return null;
|
|
165
|
+
writer.write(this.lineStart);
|
|
166
|
+
return this.children[this.index++];
|
|
167
|
+
}
|
|
168
|
+
}
|