@tradik/xslt-processor 1.1.1 → 1.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (122) hide show
  1. package/LICENSE.md +1 -1
  2. package/README.md +102 -757
  3. package/bin/lib/decode.js +15 -0
  4. package/bin/lib/dom.js +177 -0
  5. package/bin/lib/loaders.js +127 -0
  6. package/bin/lib/options.js +17 -0
  7. package/bin/lib/output.js +114 -0
  8. package/bin/lib/paths.js +3 -3
  9. package/bin/lib/transform.js +124 -33
  10. package/bin/xslt.js +26 -27
  11. package/dist/xslt-processor.browser.js +8784 -2720
  12. package/dist/xslt-processor.browser.js.map +4 -4
  13. package/dist/xslt-processor.browser.min.js +13 -6
  14. package/dist/xslt-processor.browser.min.js.map +4 -4
  15. package/dist/xslt-processor.cjs +8789 -2723
  16. package/dist/xslt-processor.cjs.map +4 -4
  17. package/dist/xslt-processor.d.cts +380 -21
  18. package/dist/xslt-processor.d.ts +380 -21
  19. package/dist/xslt-processor.js +8770 -2722
  20. package/dist/xslt-processor.js.map +4 -4
  21. package/package.json +51 -11
  22. package/src/XSLTProcessor.js +343 -66
  23. package/src/async/abort.js +63 -0
  24. package/src/async/documentUris.js +128 -0
  25. package/src/async/loaders.js +134 -0
  26. package/src/async/preload.js +159 -0
  27. package/src/async/processor.js +206 -0
  28. package/src/async/stream.js +125 -0
  29. package/src/bridge/engine.js +221 -0
  30. package/src/bridge/loader.js +78 -0
  31. package/src/bridge/results.js +75 -0
  32. package/src/bridge/version.js +63 -0
  33. package/src/index.js +16 -4
  34. package/src/io/decode.js +140 -0
  35. package/src/io/readSource.js +167 -0
  36. package/src/xpath/axes.js +562 -0
  37. package/src/xpath/documentOrder.js +270 -0
  38. package/src/xpath/evaluator.js +475 -355
  39. package/src/xpath/index.js +8 -2
  40. package/src/xpath/namespaceNodes.js +172 -0
  41. package/src/xpath/nodeSetFunctions.js +169 -0
  42. package/src/xpath/parser.js +30 -5
  43. package/src/xpath/strings.js +183 -0
  44. package/src/xpath/tokenizer.js +37 -23
  45. package/src/xslt/attributeSets.js +95 -0
  46. package/src/xslt/avt.js +103 -0
  47. package/src/xslt/computedNames.js +91 -0
  48. package/src/xslt/copying.js +212 -0
  49. package/src/xslt/declarationNames.js +80 -0
  50. package/src/xslt/domParsing.js +95 -0
  51. package/src/xslt/elements.js +1 -1
  52. package/src/xslt/engine/bindings.js +195 -0
  53. package/src/xslt/engine/context.js +105 -0
  54. package/src/xslt/engine/controlFlow.js +145 -0
  55. package/src/xslt/engine/copyInstructions.js +133 -0
  56. package/src/xslt/engine/declarations.js +233 -0
  57. package/src/xslt/engine/functionSupport.js +103 -0
  58. package/src/xslt/engine/methods.js +33 -0
  59. package/src/xslt/engine/nodeConstruction.js +187 -0
  60. package/src/xslt/engine/numbering.js +104 -0
  61. package/src/xslt/engine/outputDeclaration.js +77 -0
  62. package/src/xslt/engine/sequenceConstructor.js +228 -0
  63. package/src/xslt/engine/stylesheetLoading.js +208 -0
  64. package/src/xslt/engine/templateInvocation.js +253 -0
  65. package/src/xslt/engine/templateRules.js +243 -0
  66. package/src/xslt/engine/textInstructions.js +171 -0
  67. package/src/xslt/engine/topLevel.js +130 -0
  68. package/src/xslt/engine/transformation.js +263 -0
  69. package/src/xslt/engine/workStack.js +245 -0
  70. package/src/xslt/engine.js +176 -2020
  71. package/src/xslt/exslt/arguments.js +99 -0
  72. package/src/xslt/exslt/calendar.js +120 -0
  73. package/src/xslt/exslt/common.js +44 -0
  74. package/src/xslt/exslt/dateCalc.js +261 -0
  75. package/src/xslt/exslt/dateFormat.js +150 -0
  76. package/src/xslt/exslt/dateParse.js +265 -0
  77. package/src/xslt/exslt/dates.js +259 -0
  78. package/src/xslt/exslt/duration.js +207 -0
  79. package/src/xslt/exslt/dynamic.js +59 -0
  80. package/src/xslt/exslt/index.js +59 -0
  81. package/src/xslt/exslt/math.js +177 -0
  82. package/src/xslt/exslt/sets.js +96 -0
  83. package/src/xslt/exslt/stringOps.js +163 -0
  84. package/src/xslt/exslt/strings.js +147 -0
  85. package/src/xslt/exslt/uri.js +92 -0
  86. package/src/xslt/formatNumber.js +22 -9
  87. package/src/xslt/forwardsCompatible.js +75 -0
  88. package/src/xslt/functions.js +94 -15
  89. package/src/xslt/index.js +7 -1
  90. package/src/xslt/keys.js +51 -28
  91. package/src/xslt/literalResult.js +63 -7
  92. package/src/xslt/matchScope.js +116 -0
  93. package/src/xslt/number.js +171 -78
  94. package/src/xslt/numberFormat.js +124 -26
  95. package/src/xslt/outputNames.js +58 -0
  96. package/src/xslt/patternCompiler.js +175 -0
  97. package/src/xslt/patterns.js +324 -0
  98. package/src/xslt/qname.js +90 -0
  99. package/src/xslt/resultDocument.js +98 -0
  100. package/src/xslt/resultNamespaces.js +219 -0
  101. package/src/xslt/resultTree.js +143 -6
  102. package/src/xslt/serializer/baseWriter.js +173 -66
  103. package/src/xslt/serializer/chunks.js +120 -0
  104. package/src/xslt/serializer/constants.js +14 -0
  105. package/src/xslt/serializer/encoding.js +327 -0
  106. package/src/xslt/serializer/escape.js +49 -12
  107. package/src/xslt/serializer/frames.js +168 -0
  108. package/src/xslt/serializer/htmlDoctype.js +102 -0
  109. package/src/xslt/serializer/htmlEntities.js +77 -0
  110. package/src/xslt/serializer/htmlSerializer.js +123 -25
  111. package/src/xslt/serializer/settings.js +89 -13
  112. package/src/xslt/serializer/textSerializer.js +58 -10
  113. package/src/xslt/serializer/xhtmlDocument.js +103 -0
  114. package/src/xslt/serializer/xmlSerializer.js +113 -13
  115. package/src/xslt/serializer.js +50 -17
  116. package/src/xslt/sort.js +151 -0
  117. package/src/xslt/spaceNameTests.js +115 -0
  118. package/src/xslt/stylesheetChecks.js +206 -0
  119. package/src/xslt/stylesheetNamespaces.js +266 -0
  120. package/src/xslt/variables.js +152 -0
  121. package/src/xslt/whitespace.js +43 -27
  122. package/LICENSE +0 -29
@@ -0,0 +1,77 @@
1
+ /**
2
+ * HTML 4.01 Character Entity References
3
+ *
4
+ * The html output method writes a character the output encoding cannot
5
+ * represent as an entity reference when HTML 4.01 defines one (`€` in
6
+ * ISO-8859-1 output) and as a numeric character reference otherwise, as
7
+ * libxslt does (XSLT 1.0 section 16.2).
8
+ *
9
+ * @module xslt/serializer/htmlEntities
10
+ */
11
+
12
+ import { characterReference } from "./encoding.js";
13
+
14
+ /** Entity names of U+00A0 to U+00FF, in code point order. */
15
+ const LATIN1_NAMES =
16
+ "nbsp iexcl cent pound curren yen brvbar sect uml copy ordf laquo not shy " +
17
+ "reg macr deg plusmn sup2 sup3 acute micro para middot cedil sup1 ordm " +
18
+ "raquo frac14 frac12 frac34 iquest Agrave Aacute Acirc Atilde Auml Aring " +
19
+ "AElig Ccedil Egrave Eacute Ecirc Euml Igrave Iacute Icirc Iuml ETH Ntilde " +
20
+ "Ograve Oacute Ocirc Otilde Ouml times Oslash Ugrave Uacute Ucirc Uuml " +
21
+ "Yacute THORN szlig agrave aacute acirc atilde auml aring aelig ccedil " +
22
+ "egrave eacute ecirc euml igrave iacute icirc iuml eth ntilde ograve " +
23
+ "oacute ocirc otilde ouml divide oslash ugrave uacute ucirc uuml yacute " +
24
+ "thorn yuml";
25
+
26
+ /** The other entities, as `name codePoint` pairs. */
27
+ const OTHER_ENTITIES =
28
+ "OElig 338 oelig 339 Scaron 352 scaron 353 Yuml 376 fnof 402 circ 710 " +
29
+ "tilde 732 Alpha 913 Beta 914 Gamma 915 Delta 916 Epsilon 917 Zeta 918 " +
30
+ "Eta 919 Theta 920 Iota 921 Kappa 922 Lambda 923 Mu 924 Nu 925 Xi 926 " +
31
+ "Omicron 927 Pi 928 Rho 929 Sigma 931 Tau 932 Upsilon 933 Phi 934 Chi 935 " +
32
+ "Psi 936 Omega 937 alpha 945 beta 946 gamma 947 delta 948 epsilon 949 " +
33
+ "zeta 950 eta 951 theta 952 iota 953 kappa 954 lambda 955 mu 956 nu 957 " +
34
+ "xi 958 omicron 959 pi 960 rho 961 sigmaf 962 sigma 963 tau 964 " +
35
+ "upsilon 965 phi 966 chi 967 psi 968 omega 969 thetasym 977 upsih 978 " +
36
+ "piv 982 ensp 8194 emsp 8195 thinsp 8201 zwnj 8204 zwj 8205 lrm 8206 " +
37
+ "rlm 8207 ndash 8211 mdash 8212 lsquo 8216 rsquo 8217 sbquo 8218 " +
38
+ "ldquo 8220 rdquo 8221 bdquo 8222 dagger 8224 Dagger 8225 bull 8226 " +
39
+ "hellip 8230 permil 8240 prime 8242 Prime 8243 lsaquo 8249 rsaquo 8250 " +
40
+ "oline 8254 frasl 8260 euro 8364 image 8465 weierp 8472 real 8476 " +
41
+ "trade 8482 alefsym 8501 larr 8592 uarr 8593 rarr 8594 darr 8595 " +
42
+ "harr 8596 crarr 8629 lArr 8656 uArr 8657 rArr 8658 dArr 8659 hArr 8660 " +
43
+ "forall 8704 part 8706 exist 8707 empty 8709 nabla 8711 isin 8712 " +
44
+ "notin 8713 ni 8715 prod 8719 sum 8721 minus 8722 lowast 8727 radic 8730 " +
45
+ "prop 8733 infin 8734 ang 8736 and 8743 or 8744 cap 8745 cup 8746 " +
46
+ "int 8747 there4 8756 sim 8764 cong 8773 asymp 8776 ne 8800 equiv 8801 " +
47
+ "le 8804 ge 8805 sub 8834 sup 8835 nsub 8836 sube 8838 supe 8839 " +
48
+ "oplus 8853 otimes 8855 perp 8869 sdot 8901 lceil 8968 rceil 8969 " +
49
+ "lfloor 8970 rfloor 8971 lang 9001 rang 9002 loz 9674 spades 9824 " +
50
+ "clubs 9827 hearts 9829 diams 9830";
51
+
52
+ /**
53
+ * Entity name by code point.
54
+ * @type {Map<number, string>}
55
+ */
56
+ const ENTITY_NAMES = new Map(
57
+ LATIN1_NAMES.split(" ").map((name, index) => [0xa0 + index, name]),
58
+ );
59
+ const otherParts = OTHER_ENTITIES.split(" ");
60
+ for (let i = 0; i < otherParts.length; i += 2) {
61
+ ENTITY_NAMES.set(Number(otherParts[i + 1]), otherParts[i]);
62
+ }
63
+
64
+ /**
65
+ * Reference to a character the output encoding cannot represent: the HTML
66
+ * 4.01 entity reference when there is one, a numeric reference otherwise.
67
+ *
68
+ * @param {number} codePoint - The code point
69
+ * @returns {string} `&name;` or `&#N;`
70
+ *
71
+ * @example
72
+ * htmlCharacterReference(0x20ac); // "&euro;"
73
+ */
74
+ export function htmlCharacterReference(codePoint) {
75
+ const name = ENTITY_NAMES.get(codePoint);
76
+ return name ? `&${name};` : characterReference(codePoint);
77
+ }
@@ -2,19 +2,79 @@
2
2
  * HTML Output Serializer
3
3
  *
4
4
  * Implements the `html` output method of XSLT 1.0 section 16.2 on top of the
5
- * XML writer: no XML declaration, no namespace declarations, void elements
6
- * without a closing slash, minimized boolean attributes and unescaped
7
- * script/style content.
5
+ * XML writer: no XML declaration, void elements without a closing slash,
6
+ * minimized boolean attributes and unescaped script/style content. Like
7
+ * libxml2 (and so Chrome), namespace declarations of the result tree are
8
+ * written, and an element in a namespace other than XHTML is not an HTML
9
+ * element (`<s:img></s:img>`). Unlike libxml2, XHTML elements keep the HTML
10
+ * rules, so `<br/>` in the XHTML namespace is not written as `<br></br>`
11
+ * (which HTML parsers read as two line breaks).
8
12
  */
9
13
 
10
14
  import {
15
+ NODE_TYPE,
11
16
  PRESERVE_SPACE_ELEMENTS,
12
17
  RAW_TEXT_ELEMENTS,
13
18
  TEXT_MODE,
19
+ URI_ATTRIBUTES,
20
+ URI_ATTRIBUTES_OF_A,
21
+ XHTML_NAMESPACE,
14
22
  } from "./constants.js";
15
- import { escapeHtmlAttribute, escapeHtmlText } from "./escape.js";
23
+ import {
24
+ escapeHtmlAttribute,
25
+ escapeHtmlText,
26
+ escapeUriAttribute,
27
+ } from "./escape.js";
28
+ import { htmlCharacterReference } from "./htmlEntities.js";
29
+ import { htmlDoctypeMarkup } from "./htmlDoctype.js";
16
30
  import { XmlWriter } from "./xmlSerializer.js";
17
31
 
32
+ /**
33
+ * Whether a `head` element already declares the content type or character
34
+ * set: a `meta` child with `http-equiv="Content-Type"` or a `charset`
35
+ * attribute (names and the http-equiv value compared case-insensitively).
36
+ *
37
+ * @param {Element} head - The head element
38
+ * @returns {boolean} True when no meta element has to be added
39
+ */
40
+ function hasContentTypeMeta(head) {
41
+ for (const child of head.childNodes) {
42
+ if (
43
+ child.nodeType !== NODE_TYPE.ELEMENT ||
44
+ child.localName.toLowerCase() !== "meta"
45
+ ) {
46
+ continue;
47
+ }
48
+ for (const { name, value } of child.attributes) {
49
+ const lower = name.toLowerCase();
50
+ if (lower === "charset") return true;
51
+ if (lower === "http-equiv" && value.toLowerCase() === "content-type") {
52
+ return true;
53
+ }
54
+ }
55
+ }
56
+ return false;
57
+ }
58
+
59
+ /**
60
+ * Whether the html output method %-escapes an attribute, as libxml2 does:
61
+ * `href`, `action` and `src`, and `name` on `a` (names compared
62
+ * case-insensitively), when neither the attribute nor its element is in a
63
+ * namespace.
64
+ *
65
+ * @param {Attr} attribute - Attribute being written
66
+ * @returns {boolean} True for URI attributes
67
+ */
68
+ function isUriAttribute(attribute) {
69
+ const element = attribute.ownerElement;
70
+ if (attribute.namespaceURI || element.namespaceURI) return false;
71
+ const name = attribute.localName.toLowerCase();
72
+ return (
73
+ URI_ATTRIBUTES.has(name) ||
74
+ (URI_ATTRIBUTES_OF_A.has(name) && element.localName.toLowerCase() === "a")
75
+ );
76
+ }
77
+
18
78
  export class HtmlWriter extends XmlWriter {
19
79
  /**
20
80
  * The html output method never writes an XML declaration.
@@ -25,10 +85,18 @@ export class HtmlWriter extends XmlWriter {
25
85
  }
26
86
 
27
87
  /**
28
- * The html output method never writes namespace declarations.
88
+ * No line breaks are added between the top-level nodes of html output.
89
+ * @returns {boolean} Always false
90
+ */
91
+ get topLevelLineBreaks() {
92
+ return false;
93
+ }
94
+
95
+ /**
96
+ * HTML output is never an XHTML document.
29
97
  * @returns {boolean} Always false
30
98
  */
31
- get emitsNamespaces() {
99
+ get isXhtmlDocument() {
32
100
  return false;
33
101
  }
34
102
 
@@ -49,25 +117,17 @@ export class HtmlWriter extends XmlWriter {
49
117
  }
50
118
 
51
119
  /**
52
- * Build the document type declaration for the html output method.
120
+ * Build the document type declaration for the html output method (see
121
+ * htmlDoctype.js: declared identifiers, else derived from `version`).
53
122
  *
54
123
  * @param {Element|null} rootElement - Result document element
55
124
  * @returns {string} Doctype markup, or an empty string when not applicable
56
125
  */
57
126
  doctypeMarkup(rootElement) {
58
- const { doctypePublic, doctypeSystem } = this.settings;
59
- if (!doctypePublic && !doctypeSystem) {
60
- return "";
61
- }
62
-
63
- const name = rootElement ? rootElement.nodeName : "html";
64
- if (doctypePublic && doctypeSystem) {
65
- return `<!DOCTYPE ${name} PUBLIC "${doctypePublic}" "${doctypeSystem}">`;
66
- }
67
- if (doctypePublic) {
68
- return `<!DOCTYPE ${name} PUBLIC "${doctypePublic}">`;
69
- }
70
- return `<!DOCTYPE ${name} SYSTEM "${doctypeSystem}">`;
127
+ return htmlDoctypeMarkup(
128
+ this.settings,
129
+ rootElement ? rootElement.nodeName : "html",
130
+ );
71
131
  }
72
132
 
73
133
  /**
@@ -102,11 +162,14 @@ export class HtmlWriter extends XmlWriter {
102
162
  * @returns {string} Markup terminating the start tag
103
163
  */
104
164
  emptyElementMarkup(element, name) {
105
- return this.isVoidElement(element) ? ">" : `></${name}>`;
165
+ const namespace = element.namespaceURI;
166
+ const isHtml = !namespace || namespace === XHTML_NAMESPACE;
167
+ return isHtml && this.isVoidElement(element) ? ">" : `></${name}>`;
106
168
  }
107
169
 
108
170
  /**
109
- * Boolean attributes are minimized to their name alone.
171
+ * Boolean attributes are minimized to their name alone; URI attributes
172
+ * are %-escaped (see isUriAttribute).
110
173
  *
111
174
  * @param {Attr} attribute - Attribute to write
112
175
  * @returns {string} Attribute markup, starting with a space
@@ -116,7 +179,31 @@ export class HtmlWriter extends XmlWriter {
116
179
  if (String(value).toLowerCase() === name.toLowerCase()) {
117
180
  return ` ${name}`;
118
181
  }
119
- return ` ${name}="${this.escapeAttribute(value)}"`;
182
+ const written = isUriAttribute(attribute)
183
+ ? escapeUriAttribute(value)
184
+ : value;
185
+ return ` ${name}="${this.escapeAttribute(written)}"`;
186
+ }
187
+
188
+ /**
189
+ * The content type `meta` element libxslt (and so Chrome) writes as the
190
+ * first child of an HTML `head` element that does not already have one
191
+ * (XSLT 1.0 section 16.2 recommends it). Firefox does not add it.
192
+ *
193
+ * @param {Element} element - Element being written
194
+ * @returns {string} The meta element markup, or an empty string
195
+ */
196
+ leadingChildMarkup(element) {
197
+ // libxml2 finds the head element by name, whatever its namespace
198
+ if (
199
+ element.localName.toLowerCase() !== "head" ||
200
+ hasContentTypeMeta(element)
201
+ ) {
202
+ return "";
203
+ }
204
+ const { mediaType, encoding } = this.settings;
205
+ const content = `${mediaType || "text/html"}; charset=${encoding}`;
206
+ return `<meta http-equiv="Content-Type" content="${this.escapeAttribute(content)}">`;
120
207
  }
121
208
 
122
209
  /**
@@ -126,7 +213,7 @@ export class HtmlWriter extends XmlWriter {
126
213
  * @returns {string} Escaped text
127
214
  */
128
215
  escapeText(value) {
129
- return escapeHtmlText(value);
216
+ return this.encodeReferences(escapeHtmlText(value));
130
217
  }
131
218
 
132
219
  /**
@@ -136,6 +223,17 @@ export class HtmlWriter extends XmlWriter {
136
223
  * @returns {string} Escaped value
137
224
  */
138
225
  escapeAttribute(value) {
139
- return escapeHtmlAttribute(value);
226
+ return this.encodeReferences(escapeHtmlAttribute(value));
227
+ }
228
+
229
+ /**
230
+ * Reference to a character the output encoding cannot represent: an HTML
231
+ * entity reference where HTML 4.01 has one, as libxslt writes.
232
+ *
233
+ * @param {number} codePoint - The code point
234
+ * @returns {string} An entity or numeric character reference
235
+ */
236
+ characterReference(codePoint) {
237
+ return htmlCharacterReference(codePoint);
140
238
  }
141
239
  }
@@ -18,19 +18,63 @@ function isYes(value) {
18
18
  }
19
19
 
20
20
  /**
21
- * Convert a `cdata-section-elements` value into a lookup set.
21
+ * Lookup key of an expanded name, in Clark notation: `{uri}local`, or the
22
+ * bare local name for a name in no namespace.
22
23
  *
23
- * @param {string|string[]|undefined} value - Whitespace separated names or array
24
- * @returns {Set<string>} Element names requiring CDATA sections
24
+ * @param {string|null|undefined} namespaceUri - Namespace URI, empty for none
25
+ * @param {string} localName - Local name
26
+ * @returns {string} The key
27
+ *
28
+ * @example
29
+ * expandedNameKey("urn:p", "c"); // "{urn:p}c"
30
+ * expandedNameKey(null, "c"); // "c"
25
31
  */
26
- function toNameSet(value) {
27
- if (Array.isArray(value)) {
28
- return new Set(value);
29
- }
30
- if (typeof value === "string") {
31
- return new Set(value.split(/\s+/).filter(Boolean));
32
+ export function expandedNameKey(namespaceUri, localName) {
33
+ return namespaceUri ? `{${namespaceUri}}${localName}` : localName;
34
+ }
35
+
36
+ /**
37
+ * @typedef {Object} CdataNames
38
+ * @property {Set<string>} expanded - Expanded name keys ({@link expandedNameKey})
39
+ * @property {Array<{prefix: string, localName: string}>} qnames - Prefixed
40
+ * names whose prefix is resolved against the result element
41
+ */
42
+
43
+ /**
44
+ * Normalize a `cdata-section-elements` value.
45
+ *
46
+ * Entries resolved by the engine, `{namespaceUri, localName}`, are exact
47
+ * expanded names. A plain string QName carries no namespace bindings: an
48
+ * unprefixed name is taken to be in no namespace, and a prefixed one is kept
49
+ * aside to be resolved with the in-scope namespaces of each result element.
50
+ *
51
+ * @param {string|Array<string|{namespaceUri: ?string, localName: string}>|undefined} value -
52
+ * Whitespace separated QNames, or an array of QNames and expanded names
53
+ * @returns {CdataNames} The names
54
+ */
55
+ function toCdataNames(value) {
56
+ const entries =
57
+ typeof value === "string" ? value.split(/\s+/) : [value ?? []].flat();
58
+ const expanded = new Set();
59
+ const qnames = [];
60
+
61
+ for (const entry of entries) {
62
+ if (typeof entry !== "string") {
63
+ expanded.add(expandedNameKey(entry.namespaceUri, entry.localName));
64
+ continue;
65
+ }
66
+ const colon = entry.indexOf(":");
67
+ if (colon === -1) {
68
+ if (entry) expanded.add(entry);
69
+ } else {
70
+ qnames.push({
71
+ prefix: entry.slice(0, colon),
72
+ localName: entry.slice(colon + 1),
73
+ });
74
+ }
32
75
  }
33
- return new Set();
76
+
77
+ return { expanded, qnames };
34
78
  }
35
79
 
36
80
  /**
@@ -54,11 +98,35 @@ export function findRootElement(node) {
54
98
  return null;
55
99
  }
56
100
 
101
+ /** Text made only of XML whitespace (#x20 #x9 #xD #xA). */
102
+ const XML_WHITESPACE_ONLY = /^[ \t\r\n]*$/;
103
+
104
+ /**
105
+ * Whether text other than XML whitespace precedes the first element child.
106
+ *
107
+ * @param {Node} node - Document or fragment that has an element child
108
+ * @returns {boolean} True when a non-whitespace text node comes first
109
+ */
110
+ function hasLeadingText(node) {
111
+ for (
112
+ let child = node.firstChild;
113
+ child.nodeType !== NODE_TYPE.ELEMENT;
114
+ child = child.nextSibling
115
+ ) {
116
+ const isText =
117
+ child.nodeType === NODE_TYPE.TEXT ||
118
+ child.nodeType === NODE_TYPE.CDATA_SECTION;
119
+ if (isText && !XML_WHITESPACE_ONLY.test(child.nodeValue)) return true;
120
+ }
121
+ return false;
122
+ }
123
+
57
124
  /**
58
125
  * Derive the default output method from the result tree.
59
126
  *
60
127
  * XSLT 1.0 section 16 defaults to `html` when the document element is `html`
61
- * in no namespace, and to `xml` otherwise.
128
+ * in no namespace and no text other than whitespace precedes it, and to
129
+ * `xml` otherwise.
62
130
  *
63
131
  * @param {Node|null} node - Result tree root
64
132
  * @returns {string} Either "html" or "xml"
@@ -66,7 +134,10 @@ export function findRootElement(node) {
66
134
  export function detectOutputMethod(node) {
67
135
  const root = findRootElement(node);
68
136
  const isHtmlRoot =
69
- root && !root.namespaceURI && root.localName.toLowerCase() === "html";
137
+ root &&
138
+ !root.namespaceURI &&
139
+ root.localName.toLowerCase() === "html" &&
140
+ (root === node || !hasLeadingText(node));
70
141
  return isHtmlRoot ? "html" : "xml";
71
142
  }
72
143
 
@@ -87,6 +158,7 @@ export function resolveOutputSettings(outputSettings, node) {
87
158
  declared && declared !== "auto"
88
159
  ? declared.toLowerCase()
89
160
  : detectOutputMethod(node);
161
+ const cdata = toCdataNames(raw.cdataSectionElements);
90
162
 
91
163
  return {
92
164
  method,
@@ -94,10 +166,14 @@ export function resolveOutputSettings(outputSettings, node) {
94
166
  encoding: raw.encoding || "UTF-8",
95
167
  standalone: raw.standalone || null,
96
168
  indent: isYes(raw.indent),
169
+ // libxslt writes a line break after a top-level comment followed by
170
+ // another node unless indent="no" is declared (xsltSaveResultTo)
171
+ topLevelLineBreaks: raw.indent == null || isYes(raw.indent),
97
172
  omitXmlDeclaration: isYes(raw.omitXmlDeclaration),
98
173
  doctypePublic: raw.doctypePublic || null,
99
174
  doctypeSystem: raw.doctypeSystem || null,
100
175
  mediaType: raw.mediaType || null,
101
- cdataSectionElements: toNameSet(raw.cdataSectionElements),
176
+ cdataSectionElements: cdata.expanded,
177
+ cdataSectionQNames: cdata.qnames,
102
178
  };
103
179
  }
@@ -6,24 +6,72 @@
6
6
  */
7
7
 
8
8
  import { NODE_TYPE } from "./constants.js";
9
+ import { ChunkBuffer } from "./chunks.js";
9
10
 
10
11
  /**
11
- * Serialize a result tree with the text output method.
12
+ * Whether a node is character data (text or CDATA section).
12
13
  *
13
- * @param {Node} node - Document, fragment, element or character data node
14
- * @returns {string} Concatenated character data
14
+ * @param {Node} node - Node to test
15
+ * @returns {boolean} True for character data
15
16
  */
16
- export function serializeText(node) {
17
- if (
17
+ function isCharacterData(node) {
18
+ return (
18
19
  node.nodeType === NODE_TYPE.TEXT ||
19
20
  node.nodeType === NODE_TYPE.CDATA_SECTION
20
- ) {
21
- return node.nodeValue || "";
21
+ );
22
+ }
23
+
24
+ /**
25
+ * The character data nodes under `root` in document order, found without
26
+ * recursion (firstChild/nextSibling walk), so deep trees cannot overflow the
27
+ * stack.
28
+ *
29
+ * @param {Node} root - Document, fragment, element or character data node
30
+ * @yields {Node} Character data nodes
31
+ * @returns {Generator<Node, void, void>} The nodes
32
+ */
33
+ function* characterDataNodes(root) {
34
+ let node = root;
35
+ while (node) {
36
+ if (isCharacterData(node)) yield node;
37
+ if (node.firstChild) {
38
+ node = node.firstChild;
39
+ continue;
40
+ }
41
+ while (node !== root && !node.nextSibling) node = node.parentNode;
42
+ node = node === root ? null : node.nextSibling;
22
43
  }
44
+ }
23
45
 
24
- let text = "";
25
- for (const child of node.childNodes || []) {
26
- text += serializeText(child);
46
+ /**
47
+ * Serialize a result tree with the text output method, in chunks.
48
+ *
49
+ * @param {Node} node - Document, fragment, element or character data node
50
+ * @param {number} [chunkSize] - Chunk size in UTF-16 code units; Infinity
51
+ * yields the whole text as one chunk
52
+ * @yields {string} Non-empty chunks of at most `chunkSize` code units
53
+ * @returns {Generator<string, void, void>} The chunks, in order
54
+ *
55
+ * @example
56
+ * [...textChunks(fragment, 16384)].join("");
57
+ */
58
+ export function* textChunks(node, chunkSize) {
59
+ const buffer = new ChunkBuffer(chunkSize);
60
+ for (const text of characterDataNodes(node)) {
61
+ buffer.write(text.nodeValue || "");
62
+ if (buffer.full) yield* buffer.take();
27
63
  }
64
+ yield* buffer.take(true);
65
+ }
66
+
67
+ /**
68
+ * Serialize a result tree with the text output method.
69
+ *
70
+ * @param {Node} node - Document, fragment, element or character data node
71
+ * @returns {string} Concatenated character data
72
+ */
73
+ export function serializeText(node) {
74
+ let text = "";
75
+ for (const chunk of textChunks(node, Infinity)) text += chunk;
28
76
  return text;
29
77
  }
@@ -0,0 +1,103 @@
1
+ /**
2
+ * XHTML 1.0 documents in xml output, as libxml2 (and so Chrome's
3
+ * XSLTProcessor) writes them.
4
+ *
5
+ * When the xml output declares one of the XHTML 1.0 document types
6
+ * (`doctype-public` or `doctype-system` of XHTML 1.0 Strict, Transitional
7
+ * or Frameset), libxml2 switches to its XHTML serializer, which follows the
8
+ * compatibility guidelines of XHTML 1.0 appendix C:
9
+ * - an `html` element in no namespace without namespace declarations gets
10
+ * `xmlns="http://www.w3.org/1999/xhtml"` (A.3.1.1);
11
+ * - a `head` child of the `html` document element without a Content-Type
12
+ * `meta` gets `<meta http-equiv="Content-Type" content="text/html;
13
+ * charset=..." />` as its first child (C.9);
14
+ * - elements in no namespace follow the empty element rules of XHTML
15
+ * elements (C.2, C.3).
16
+ *
17
+ * @module xslt/serializer/xhtmlDocument
18
+ */
19
+
20
+ import { NODE_TYPE, XHTML_NAMESPACE } from "./constants.js";
21
+
22
+ /** Public identifiers of the XHTML 1.0 document types. */
23
+ const XHTML1_PUBLIC_IDS = new Set([
24
+ "-//W3C//DTD XHTML 1.0 Strict//EN",
25
+ "-//W3C//DTD XHTML 1.0 Transitional//EN",
26
+ "-//W3C//DTD XHTML 1.0 Frameset//EN",
27
+ ]);
28
+
29
+ /** System identifiers of the XHTML 1.0 document types. */
30
+ const XHTML1_SYSTEM_IDS = new Set([
31
+ "http://www.w3.org/TR/xhtml1/DTD/xhtml1-strict.dtd",
32
+ "http://www.w3.org/TR/xhtml1/DTD/xhtml1-transitional.dtd",
33
+ "http://www.w3.org/TR/xhtml1/DTD/xhtml1-frameset.dtd",
34
+ ]);
35
+
36
+ /**
37
+ * Whether the output settings declare an XHTML 1.0 document type.
38
+ *
39
+ * @param {{doctypePublic?: string|null, doctypeSystem?: string|null}} settings -
40
+ * Normalized output settings
41
+ * @returns {boolean} True for the XHTML 1.0 public or system identifiers
42
+ *
43
+ * @example
44
+ * isXhtml1Doctype({ doctypePublic: "-//W3C//DTD XHTML 1.0 Strict//EN" }); // true
45
+ */
46
+ export function isXhtml1Doctype(settings) {
47
+ return (
48
+ XHTML1_PUBLIC_IDS.has(settings.doctypePublic) ||
49
+ XHTML1_SYSTEM_IDS.has(settings.doctypeSystem)
50
+ );
51
+ }
52
+
53
+ /**
54
+ * The namespace declaration libxml2 adds to an `html` element in no
55
+ * namespace that declares no namespace itself.
56
+ *
57
+ * @param {Element} element - Element being written
58
+ * @param {Array<object>} declarations - Declarations written on it
59
+ * @returns {string} ` xmlns="http://www.w3.org/1999/xhtml"`, or ""
60
+ */
61
+ export function xhtmlRootNamespace(element, declarations) {
62
+ const needed =
63
+ element.localName === "html" &&
64
+ !element.namespaceURI &&
65
+ declarations.length === 0;
66
+ return needed ? ` xmlns="${XHTML_NAMESPACE}"` : "";
67
+ }
68
+
69
+ /**
70
+ * Whether a `head` element has a `meta` child with
71
+ * `http-equiv="Content-Type"` (compared case-insensitively).
72
+ *
73
+ * @param {Element} head - The head element
74
+ * @returns {boolean} True when libxml2 adds no meta element
75
+ */
76
+ function hasContentTypeMeta(head) {
77
+ for (const child of head.childNodes) {
78
+ if (child.nodeType !== NODE_TYPE.ELEMENT || child.localName !== "meta") {
79
+ continue;
80
+ }
81
+ const value = child.getAttribute("http-equiv");
82
+ if (value?.toLowerCase() === "content-type") return true;
83
+ }
84
+ return false;
85
+ }
86
+
87
+ /**
88
+ * The Content-Type `meta` element libxml2 writes as the first child of the
89
+ * `head` child of an `html` document element.
90
+ *
91
+ * @param {Element} element - Element being written
92
+ * @param {string} encoding - The output encoding, as declared
93
+ * @returns {string} The meta element markup, or ""
94
+ */
95
+ export function xhtmlHeadMeta(element, encoding) {
96
+ const parent = element.parentNode;
97
+ const isHead =
98
+ element.localName === "head" &&
99
+ parent?.localName === "html" &&
100
+ parent.parentNode?.nodeType !== NODE_TYPE.ELEMENT;
101
+ if (!isHead || hasContentTypeMeta(element)) return "";
102
+ return `<meta http-equiv="Content-Type" content="text/html; charset=${encoding}" />`;
103
+ }