@tradik/xslt-processor 1.0.3 → 1.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (131) hide show
  1. package/LICENSE.md +1 -1
  2. package/README.md +110 -520
  3. package/bin/lib/decode.js +15 -0
  4. package/bin/lib/dom.js +177 -0
  5. package/bin/lib/loaders.js +127 -0
  6. package/bin/lib/options.js +131 -0
  7. package/bin/lib/output.js +114 -0
  8. package/bin/lib/paths.js +186 -0
  9. package/bin/lib/transform.js +206 -0
  10. package/bin/xslt.js +73 -168
  11. package/dist/xslt-processor.browser.js +9564 -1585
  12. package/dist/xslt-processor.browser.js.map +4 -4
  13. package/dist/xslt-processor.browser.min.js +13 -2
  14. package/dist/xslt-processor.browser.min.js.map +4 -4
  15. package/dist/xslt-processor.cjs +9572 -1586
  16. package/dist/xslt-processor.cjs.map +4 -4
  17. package/dist/xslt-processor.d.cts +658 -0
  18. package/dist/xslt-processor.d.ts +459 -12
  19. package/dist/xslt-processor.js +9546 -1582
  20. package/dist/xslt-processor.js.map +4 -4
  21. package/package.json +71 -20
  22. package/src/XSLTProcessor.js +494 -48
  23. package/src/async/abort.js +63 -0
  24. package/src/async/documentUris.js +128 -0
  25. package/src/async/loaders.js +134 -0
  26. package/src/async/preload.js +159 -0
  27. package/src/async/processor.js +206 -0
  28. package/src/async/stream.js +125 -0
  29. package/src/bridge/engine.js +221 -0
  30. package/src/bridge/loader.js +78 -0
  31. package/src/bridge/results.js +75 -0
  32. package/src/bridge/version.js +63 -0
  33. package/src/index.js +26 -8
  34. package/src/io/decode.js +140 -0
  35. package/src/io/readSource.js +167 -0
  36. package/src/xpath/axes.js +562 -0
  37. package/src/xpath/documentOrder.js +270 -0
  38. package/src/xpath/evaluator.js +518 -357
  39. package/src/xpath/index.js +8 -2
  40. package/src/xpath/namespaceNodes.js +172 -0
  41. package/src/xpath/nodeSetFunctions.js +169 -0
  42. package/src/xpath/parser.js +30 -5
  43. package/src/xpath/strings.js +183 -0
  44. package/src/xpath/tokenizer.js +37 -23
  45. package/src/xslt/attributeSets.js +95 -0
  46. package/src/xslt/avt.js +103 -0
  47. package/src/xslt/computedNames.js +91 -0
  48. package/src/xslt/copying.js +212 -0
  49. package/src/xslt/declarationNames.js +80 -0
  50. package/src/xslt/domParsing.js +95 -0
  51. package/src/xslt/elements.js +57 -0
  52. package/src/xslt/engine/bindings.js +195 -0
  53. package/src/xslt/engine/context.js +105 -0
  54. package/src/xslt/engine/controlFlow.js +145 -0
  55. package/src/xslt/engine/copyInstructions.js +133 -0
  56. package/src/xslt/engine/declarations.js +233 -0
  57. package/src/xslt/engine/functionSupport.js +103 -0
  58. package/src/xslt/engine/methods.js +33 -0
  59. package/src/xslt/engine/nodeConstruction.js +187 -0
  60. package/src/xslt/engine/numbering.js +104 -0
  61. package/src/xslt/engine/outputDeclaration.js +77 -0
  62. package/src/xslt/engine/sequenceConstructor.js +228 -0
  63. package/src/xslt/engine/stylesheetLoading.js +208 -0
  64. package/src/xslt/engine/templateInvocation.js +253 -0
  65. package/src/xslt/engine/templateRules.js +243 -0
  66. package/src/xslt/engine/textInstructions.js +171 -0
  67. package/src/xslt/engine/topLevel.js +130 -0
  68. package/src/xslt/engine/transformation.js +263 -0
  69. package/src/xslt/engine/workStack.js +245 -0
  70. package/src/xslt/engine.js +184 -1736
  71. package/src/xslt/exslt/arguments.js +99 -0
  72. package/src/xslt/exslt/calendar.js +120 -0
  73. package/src/xslt/exslt/common.js +44 -0
  74. package/src/xslt/exslt/dateCalc.js +261 -0
  75. package/src/xslt/exslt/dateFormat.js +150 -0
  76. package/src/xslt/exslt/dateParse.js +265 -0
  77. package/src/xslt/exslt/dates.js +259 -0
  78. package/src/xslt/exslt/duration.js +207 -0
  79. package/src/xslt/exslt/dynamic.js +59 -0
  80. package/src/xslt/exslt/index.js +59 -0
  81. package/src/xslt/exslt/math.js +177 -0
  82. package/src/xslt/exslt/sets.js +96 -0
  83. package/src/xslt/exslt/stringOps.js +163 -0
  84. package/src/xslt/exslt/strings.js +147 -0
  85. package/src/xslt/exslt/uri.js +92 -0
  86. package/src/xslt/formatNumber.js +233 -0
  87. package/src/xslt/forwardsCompatible.js +75 -0
  88. package/src/xslt/functions.js +270 -0
  89. package/src/xslt/index.js +38 -1
  90. package/src/xslt/keys.js +164 -0
  91. package/src/xslt/literalResult.js +223 -0
  92. package/src/xslt/matchScope.js +116 -0
  93. package/src/xslt/number.js +271 -0
  94. package/src/xslt/numberFormat.js +253 -0
  95. package/src/xslt/outputNames.js +58 -0
  96. package/src/xslt/patternCompiler.js +175 -0
  97. package/src/xslt/patterns.js +324 -0
  98. package/src/xslt/qname.js +90 -0
  99. package/src/xslt/resultDocument.js +98 -0
  100. package/src/xslt/resultNamespaces.js +219 -0
  101. package/src/xslt/resultTree.js +211 -0
  102. package/src/xslt/serializer/baseWriter.js +390 -0
  103. package/src/xslt/serializer/chunks.js +120 -0
  104. package/src/xslt/serializer/constants.js +92 -0
  105. package/src/xslt/serializer/encoding.js +327 -0
  106. package/src/xslt/serializer/escape.js +135 -0
  107. package/src/xslt/serializer/frames.js +168 -0
  108. package/src/xslt/serializer/htmlDoctype.js +102 -0
  109. package/src/xslt/serializer/htmlEntities.js +77 -0
  110. package/src/xslt/serializer/htmlSerializer.js +239 -0
  111. package/src/xslt/serializer/indent.js +51 -0
  112. package/src/xslt/serializer/namespaces.js +68 -0
  113. package/src/xslt/serializer/rawText.js +41 -0
  114. package/src/xslt/serializer/settings.js +179 -0
  115. package/src/xslt/serializer/textSerializer.js +77 -0
  116. package/src/xslt/serializer/xhtmlDocument.js +103 -0
  117. package/src/xslt/serializer/xmlSerializer.js +227 -0
  118. package/src/xslt/serializer.js +90 -0
  119. package/src/xslt/sort.js +151 -0
  120. package/src/xslt/spaceNameTests.js +115 -0
  121. package/src/xslt/stylesheetChecks.js +206 -0
  122. package/src/xslt/stylesheetNamespaces.js +266 -0
  123. package/src/xslt/templatePriority.js +45 -0
  124. package/src/xslt/uri.js +68 -0
  125. package/src/xslt/variables.js +152 -0
  126. package/src/xslt/whitespace.js +200 -0
  127. package/LICENSE +0 -29
  128. package/src/XSLTProcessor.test.js +0 -930
  129. package/src/xpath/evaluator.test.js +0 -1852
  130. package/src/xpath/tokenizer.test.js +0 -224
  131. package/src/xslt/engine.test.js +0 -3130
@@ -22,12 +22,18 @@ import { XPathEvaluator, XPathContext } from "./evaluator.js";
22
22
  *
23
23
  * @param {string} expression - XPath expression
24
24
  * @param {Node} contextNode - Context node
25
- * @param {Object} options - Options (variables, namespaces)
25
+ * @param {Object} options - Options: variables, namespaces, and the
26
+ * evaluator limits maxResultSize, maxRecursionDepth and maxStringLength
27
+ * (defaults in XPathLimits)
26
28
  * @returns {*} Evaluation result
27
29
  */
28
30
  export function evaluate(expression, contextNode, options = {}) {
29
31
  const ast = parse(expression);
30
- const evaluator = new XPathEvaluator();
32
+ const evaluator = new XPathEvaluator({
33
+ maxResultSize: options.maxResultSize,
34
+ maxRecursionDepth: options.maxRecursionDepth,
35
+ maxStringLength: options.maxStringLength,
36
+ });
31
37
  const context = new XPathContext(
32
38
  contextNode,
33
39
  1,
@@ -0,0 +1,172 @@
1
+ /**
2
+ * Namespace nodes and the `namespace` axis (XPath 1.0 sections 2.2, 5.4).
3
+ *
4
+ * The DOM has no namespace nodes, so they are synthesized: an element has one
5
+ * namespace node for every namespace binding in scope on it, the implicit
6
+ * `xml` binding included, and an undeclared default namespace (`xmlns=""`)
7
+ * has none. Bindings come from `xmlns` attributes of the element and its
8
+ * ancestors and, for trees built with `createElementNS`, from the prefixes of
9
+ * the element and attribute names themselves. HTML documents are treated as
10
+ * libxslt sees them after re-parsing their markup: their elements are in no
11
+ * namespace, so only explicit `xmlns` attributes (and `xml`) bind prefixes.
12
+ *
13
+ * A namespace node looks like a DOM node where the processor reads one:
14
+ * `nodeType` 13, `localName`/`nodeName` = the prefix ("" for the default
15
+ * namespace), `namespaceURI` = null (the expanded-name of a namespace node
16
+ * has a null URI), `nodeValue`/`textContent` = the namespace URI and
17
+ * `ownerElement` = its parent element. Nodes are created once per element and
18
+ * cache, so the same binding is the same object: unions deduplicate it and
19
+ * `generate-id()` is stable. In document order the namespace nodes of an
20
+ * element follow the element and precede its attributes (section 5).
21
+ *
22
+ * @module xpath/namespaceNodes
23
+ */
24
+
25
+ "use strict";
26
+
27
+ /** Node type of a namespace node (the DOM XPath XPathNamespace type). */
28
+ export const NAMESPACE_NODE = 13;
29
+
30
+ /** Namespace bound to the `xml` prefix (Namespaces in XML 1.0). */
31
+ const XML_NAMESPACE = "http://www.w3.org/XML/1998/namespace";
32
+
33
+ /** Namespace of `xmlns` and `xmlns:*` attributes. */
34
+ const XMLNS_NAMESPACE = "http://www.w3.org/2000/xmlns/";
35
+
36
+ /**
37
+ * A synthesized XPath namespace node.
38
+ */
39
+ export class NamespaceNode {
40
+ /**
41
+ * @param {Element} element - The element the node belongs to (its parent)
42
+ * @param {string} prefix - The bound prefix, "" for the default namespace
43
+ * @param {string} uri - The namespace URI
44
+ * @param {number} index - Position among the element's namespace nodes
45
+ */
46
+ constructor(element, prefix, uri, index) {
47
+ this.nodeType = NAMESPACE_NODE;
48
+ this.ownerElement = element;
49
+ this.ownerDocument = element.ownerDocument;
50
+ this.localName = prefix;
51
+ this.nodeName = prefix;
52
+ this.namespaceURI = null;
53
+ this.prefix = null;
54
+ this.nodeValue = uri;
55
+ this.textContent = uri;
56
+ this.index = index;
57
+ this.parentNode = null;
58
+ this.firstChild = null;
59
+ this.lastChild = null;
60
+ this.previousSibling = null;
61
+ this.nextSibling = null;
62
+ this.childNodes = [];
63
+ }
64
+ }
65
+
66
+ /**
67
+ * Whether a node is a synthesized namespace node.
68
+ *
69
+ * @param {*} node - Any value
70
+ * @returns {boolean} True for namespace nodes
71
+ */
72
+ export function isNamespaceNode(node) {
73
+ return node?.nodeType === NAMESPACE_NODE;
74
+ }
75
+
76
+ /**
77
+ * The prefix declared by an `xmlns` or `xmlns:*` attribute.
78
+ *
79
+ * @param {Attr} attr - Any attribute
80
+ * @returns {string|null} The prefix ("" for `xmlns`), or null for any other attribute
81
+ */
82
+ function declaredPrefix(attr) {
83
+ const name = attr.name;
84
+ if (name === "xmlns") return "";
85
+ return name.startsWith("xmlns:") ? name.slice(6) : null;
86
+ }
87
+
88
+ /**
89
+ * Record the bindings one element contributes, nearest binding first: an
90
+ * ancestor cannot override what a descendant already bound.
91
+ *
92
+ * @param {Element} element - An element on the ancestor-or-self axis
93
+ * @param {boolean} fromNames - Whether prefixes of element and attribute names count
94
+ * @param {Map<string, string>} bindings - URIs by prefix, "" marking an undeclared prefix
95
+ * @returns {void}
96
+ */
97
+ function addBindings(element, fromNames, bindings) {
98
+ const bind = (prefix, uri) => {
99
+ if (prefix !== "xml" && !bindings.has(prefix)) bindings.set(prefix, uri);
100
+ };
101
+ for (const attr of element.attributes) {
102
+ const prefix = declaredPrefix(attr);
103
+ if (prefix !== null) bind(prefix, attr.value);
104
+ }
105
+ if (!fromNames) return;
106
+ bind(element.prefix ?? "", element.namespaceURI ?? "");
107
+ for (const attr of element.attributes) {
108
+ if (attr.prefix && attr.namespaceURI !== XMLNS_NAMESPACE) {
109
+ bind(attr.prefix, attr.namespaceURI);
110
+ }
111
+ }
112
+ }
113
+
114
+ /**
115
+ * The namespace bindings in scope on an element, `xml` first.
116
+ *
117
+ * @param {Element} element - The element
118
+ * @returns {Array<[string, string]>} `[prefix, uri]` pairs
119
+ *
120
+ * @example
121
+ * // <r xmlns:a="u"/>
122
+ * inScopeBindings(r); // [["xml", "http://www.w3.org/XML/1998/namespace"], ["a", "u"]]
123
+ */
124
+ export function inScopeBindings(element) {
125
+ const fromNames = element.ownerDocument?.contentType !== "text/html";
126
+ const bindings = new Map();
127
+ for (let node = element; node?.nodeType === 1; node = node.parentNode) {
128
+ addBindings(node, fromNames, bindings);
129
+ }
130
+ const result = [["xml", XML_NAMESPACE]];
131
+ for (const [prefix, uri] of bindings) {
132
+ if (uri !== "") result.push([prefix, uri]);
133
+ }
134
+ return result;
135
+ }
136
+
137
+ /**
138
+ * The namespace axis of a node: the namespace nodes of an element, nothing
139
+ * for any other node. Nodes are cached per element in `cache`, so repeated
140
+ * evaluations return identical objects.
141
+ *
142
+ * @param {Node} node - Context node
143
+ * @param {WeakMap<Element, NamespaceNode[]>} cache - Namespace nodes by element
144
+ * @returns {NamespaceNode[]} The namespace nodes, in document order
145
+ *
146
+ * @example
147
+ * namespaceAxis(element, new WeakMap()).map((ns) => ns.localName); // ["xml", "a"]
148
+ */
149
+ export function namespaceAxis(node, cache) {
150
+ if (node.nodeType !== 1) return [];
151
+ let nodes = cache.get(node);
152
+ if (!nodes) {
153
+ nodes = inScopeBindings(node).map(
154
+ ([prefix, uri], index) => new NamespaceNode(node, prefix, uri, index),
155
+ );
156
+ cache.set(node, nodes);
157
+ }
158
+ return nodes.slice();
159
+ }
160
+
161
+ /**
162
+ * Match a name test against a namespace node: its name is the prefix, with a
163
+ * null namespace URI, so only unprefixed tests can match (section 2.3).
164
+ *
165
+ * @param {{name: string, prefix: (string|null)}} nodeTest - Name test AST node
166
+ * @param {NamespaceNode} node - The namespace node
167
+ * @returns {boolean} Whether the node matches
168
+ */
169
+ export function matchNamespaceNameTest(nodeTest, node) {
170
+ if (nodeTest.prefix) return false;
171
+ return nodeTest.name === "*" || nodeTest.name === node.localName;
172
+ }
@@ -0,0 +1,169 @@
1
+ /**
2
+ * XPath 1.0 node-set functions (section 4.1), plus `sum()` (4.4) and `lang()`
3
+ * (4.3), which also take node-sets or walk the tree.
4
+ *
5
+ * Functions whose argument must be a node-set raise a type error for any other
6
+ * value, as libxslt does ("count() expects a node-set"). A single node, which
7
+ * is how this processor represents a result tree fragment, is accepted as a
8
+ * node-set of that node: XSLT 1.0 forbids it, but libxslt accepts
9
+ * `count($rtf)` and stylesheets rely on it.
10
+ *
11
+ * @module xpath/nodeSetFunctions
12
+ */
13
+
14
+ "use strict";
15
+
16
+ import { splitXmlSpace } from "./strings.js";
17
+ import { parentOf } from "./axes.js";
18
+ import { NAMESPACE_NODE } from "./namespaceNodes.js";
19
+
20
+ /** Qualified name of the attribute read by `lang()`. */
21
+ const XML_LANG = "xml:lang";
22
+
23
+ /**
24
+ * Whether a node has an expanded name made of a namespace URI and a local
25
+ * part: elements and attributes (XPath 5.2, 5.3).
26
+ *
27
+ * @param {Node} node - Any node
28
+ * @returns {boolean} True for element and attribute nodes
29
+ */
30
+ function hasQualifiedName(node) {
31
+ const type = node.nodeType;
32
+ return type === 1 || type === 2;
33
+ }
34
+
35
+ /**
36
+ * Whether a node has a name without a namespace part: processing
37
+ * instructions (their target) and namespace nodes (their prefix).
38
+ *
39
+ * @param {Node} node - Any node
40
+ * @returns {boolean} True for processing instruction and namespace nodes
41
+ */
42
+ function hasLocalNameOnly(node) {
43
+ const type = node.nodeType;
44
+ return type === 7 || type === NAMESPACE_NODE;
45
+ }
46
+
47
+ /**
48
+ * The element whose `xml:lang` decides the language of a node: the node
49
+ * itself for an element, the element of an attribute or namespace node, the
50
+ * parent of any other node.
51
+ *
52
+ * @param {Node} node - Context node
53
+ * @returns {Node|null} The first node to inspect
54
+ */
55
+ function languageStart(node) {
56
+ return node.nodeType === 1 ? node : parentOf(node);
57
+ }
58
+
59
+ /**
60
+ * Build the node-set functions for an evaluator.
61
+ *
62
+ * @param {import('./evaluator.js').XPathEvaluator} evaluator - The evaluator
63
+ * @returns {Object<string, Function>} Functions by name
64
+ */
65
+ export function createNodeSetFunctions(evaluator) {
66
+ /**
67
+ * Evaluate an argument that must be a node-set.
68
+ *
69
+ * @param {string} name - Function name, for the error message
70
+ * @param {object} arg - Argument expression
71
+ * @param {import('./evaluator.js').XPathContext} ctx - Evaluation context
72
+ * @returns {Node[]} The node-set
73
+ * @throws {TypeError} When the argument is not a node-set
74
+ */
75
+ const nodeSetArgument = (name, arg, ctx) => {
76
+ const value = evaluator.evaluate(arg, ctx);
77
+ if (Array.isArray(value)) return value;
78
+ if (value?.nodeType) return [value];
79
+ throw new TypeError(`${name}() expects a node-set`);
80
+ };
81
+
82
+ /**
83
+ * The node a name function applies to: the context node without argument,
84
+ * otherwise the first node of the node-set argument.
85
+ *
86
+ * @param {string} name - Function name, for the error message
87
+ * @param {object[]} args - Argument expressions
88
+ * @param {import('./evaluator.js').XPathContext} ctx - Evaluation context
89
+ * @returns {Node|undefined} The node, undefined for an empty node-set
90
+ */
91
+ const nameTarget = (name, args, ctx) =>
92
+ args.length === 0 ? ctx.node : nodeSetArgument(name, args[0], ctx)[0];
93
+
94
+ return {
95
+ count: (args, ctx) => nodeSetArgument("count", args[0], ctx).length,
96
+
97
+ /**
98
+ * `id(object)`: every whitespace separated token of the argument (of the
99
+ * string value of every node, for a node-set) names an ID; the result is
100
+ * in document order without duplicates.
101
+ */
102
+ id: (args, ctx) => {
103
+ const value = evaluator.evaluate(args[0], ctx);
104
+ const strings = Array.isArray(value)
105
+ ? value.map((node) => evaluator.getStringValue(node))
106
+ : [evaluator.toString(value)];
107
+ const doc = ctx.node.ownerDocument || ctx.node;
108
+ const found = new Set();
109
+ for (const string of strings) {
110
+ for (const token of splitXmlSpace(string)) {
111
+ const element = doc.getElementById(token);
112
+ if (element) found.add(element);
113
+ }
114
+ }
115
+ return evaluator.sortByDocumentOrder([...found]);
116
+ },
117
+
118
+ "local-name": (args, ctx) => {
119
+ const node = nameTarget("local-name", args, ctx);
120
+ if (!node) return "";
121
+ if (hasQualifiedName(node)) return node.localName;
122
+ return hasLocalNameOnly(node) ? node.nodeName : "";
123
+ },
124
+
125
+ "namespace-uri": (args, ctx) => {
126
+ const node = nameTarget("namespace-uri", args, ctx);
127
+ return node && hasQualifiedName(node) ? node.namespaceURI || "" : "";
128
+ },
129
+
130
+ name: (args, ctx) => {
131
+ const node = nameTarget("name", args, ctx);
132
+ if (!node) return "";
133
+ return hasQualifiedName(node) || hasLocalNameOnly(node)
134
+ ? node.nodeName
135
+ : "";
136
+ },
137
+
138
+ sum: (args, ctx) =>
139
+ nodeSetArgument("sum", args[0], ctx).reduce(
140
+ (total, node) =>
141
+ total + evaluator.toNumber(evaluator.getStringValue(node)),
142
+ 0,
143
+ ),
144
+
145
+ /**
146
+ * `lang(string)`: whether the `xml:lang` of the nearest element at or
147
+ * above the context node (an attribute or text node looks from its
148
+ * element) is the language or one of its sublanguages. A plain `lang`
149
+ * attribute is not `xml:lang` and is ignored.
150
+ */
151
+ lang: (args, ctx) => {
152
+ const wanted = evaluator.toString(evaluator.evaluate(args[0], ctx));
153
+ const lang = wanted.toLowerCase();
154
+ for (
155
+ let node = languageStart(ctx.node);
156
+ node?.nodeType === 1;
157
+ node = node.parentNode
158
+ ) {
159
+ // The xml prefix is always bound to one namespace, so the qualified
160
+ // name also finds attributes created without a namespace URI
161
+ if (node.hasAttribute(XML_LANG)) {
162
+ const actual = node.getAttribute(XML_LANG).toLowerCase();
163
+ return actual === lang || actual.startsWith(`${lang}-`);
164
+ }
165
+ }
166
+ return false;
167
+ },
168
+ };
169
+ }
@@ -248,7 +248,7 @@ export class XPathParser {
248
248
  }
249
249
 
250
250
  // Check if this looks like a location path starting with step
251
- if (this.isStepStart()) {
251
+ if (this.isStepStart() && !this.isPrefixedFunctionCall()) {
252
252
  return this.parseLocationPath();
253
253
  }
254
254
 
@@ -505,16 +505,21 @@ export class XPathParser {
505
505
  return this.parseFunctionCall();
506
506
  }
507
507
 
508
+ // Prefixed function call: the tokenizer emits NAME ':' FUNCTION
509
+ if (this.isPrefixedFunctionCall()) {
510
+ const prefix = this.advance().value;
511
+ this.advance(); // ':'
512
+ return this.parseFunctionCallArgs(this.advance().value, prefix);
513
+ }
514
+
508
515
  throw new Error(
509
516
  `Unexpected token ${this.peek().type} at position ${this.peek().position}`,
510
517
  );
511
518
  }
512
519
 
513
520
  // FunctionCall ::= FunctionName '(' ( Argument ( ',' Argument )* )? ')'
514
- // Note: Prefixed function calls (prefix:fn()) are not supported because
515
- // the tokenizer identifies functions by NAME followed by '(' - prefixed
516
- // names like 'fn:name()' are tokenized as NAME:NAME() which is parsed
517
- // as a location path, not a function call.
521
+ // FunctionName is a QName: an unprefixed name arrives as one FUNCTION
522
+ // token, a prefixed one as NAME ':' FUNCTION (see isPrefixedFunctionCall).
518
523
  parseFunctionCall() {
519
524
  const name = this.advance().value;
520
525
  return this.parseFunctionCallArgs(name, null);
@@ -537,6 +542,26 @@ export class XPathParser {
537
542
  }
538
543
 
539
544
  // Helper methods
545
+
546
+ /**
547
+ * Whether the next tokens form a prefixed function name, `prefix:name(`.
548
+ *
549
+ * The tokenizer classifies the local part as a FUNCTION token (or as a
550
+ * NODE_TYPE token when it is spelled like one, as in `f:node()`), so
551
+ * `NAME ':' (FUNCTION | NODE_TYPE)` can only start a function call; a
552
+ * prefixed name test is `NAME ':' (NAME | '*')`.
553
+ *
554
+ * @returns {boolean} True when a prefixed function call follows
555
+ */
556
+ isPrefixedFunctionCall() {
557
+ const next = this.tokens[this.position + 2];
558
+ return (
559
+ this.check(TokenType.NAME) &&
560
+ this.tokens[this.position + 1].type === TokenType.COLON &&
561
+ (next.type === TokenType.FUNCTION || next.type === TokenType.NODE_TYPE)
562
+ );
563
+ }
564
+
540
565
  isStepStart() {
541
566
  const type = this.peek().type;
542
567
  return (
@@ -0,0 +1,183 @@
1
+ /**
2
+ * String and number conversions of the XPath 1.0 data model.
3
+ *
4
+ * JavaScript's own conversions are close to XPath's but not equal: `\s` and
5
+ * `trim()` know Unicode spaces, `Number()` reads exponents, hexadecimal and
6
+ * `Infinity`, `String()` writes exponents, and string indexes count UTF-16
7
+ * code units. The helpers here follow the recommendation instead: XML
8
+ * whitespace only (#x20 #x9 #xD #xA), the Number grammar of section 4.4, the
9
+ * decimal form of section 4.2 and characters counted as code points.
10
+ *
11
+ * @module xpath/strings
12
+ */
13
+
14
+ /** A run of XML whitespace characters. */
15
+ const XML_WHITESPACE_RUN = /[ \t\r\n]+/g;
16
+
17
+ /**
18
+ * Whether a character is XML whitespace (#x20, #x9, #xD, #xA).
19
+ *
20
+ * @param {string} char - A single character
21
+ * @returns {boolean} True for XML whitespace
22
+ */
23
+ function isXmlSpace(char) {
24
+ return char === " " || char === "\t" || char === "\r" || char === "\n";
25
+ }
26
+
27
+ /**
28
+ * Strip leading and trailing XML whitespace in linear time (a regular
29
+ * expression such as `/\s+$/` backtracks quadratically on long runs).
30
+ *
31
+ * @param {string} str - Any string
32
+ * @returns {string} The trimmed string
33
+ */
34
+ export function trimXmlSpace(str) {
35
+ let start = 0;
36
+ let end = str.length;
37
+ while (start < end && isXmlSpace(str[start])) start++;
38
+ while (end > start && isXmlSpace(str[end - 1])) end--;
39
+ return str.slice(start, end);
40
+ }
41
+
42
+ /** `S? '-'? (Digits ('.' Digits?)? | '.' Digits) S?` (XPath 4.4 number()). */
43
+ const XPATH_NUMBER = /^-?(?:\d+(?:\.\d*)?|\.\d+)$/;
44
+
45
+ /** Any UTF-16 surrogate code unit. */
46
+ const SURROGATE = /[\uD800-\uDFFF]/;
47
+
48
+ /**
49
+ * Strip leading and trailing XML whitespace and collapse inner runs of it to
50
+ * one space, as `normalize-space()` does.
51
+ *
52
+ * @param {string} str - Any string
53
+ * @returns {string} The normalized string
54
+ *
55
+ * @example
56
+ * normalizeXmlSpace(" a \n b "); // "a b" - a no-break space is kept
57
+ */
58
+ export function normalizeXmlSpace(str) {
59
+ return trimXmlSpace(str).replace(XML_WHITESPACE_RUN, " ");
60
+ }
61
+
62
+ /**
63
+ * Split a string on XML whitespace, dropping empty tokens.
64
+ *
65
+ * @param {string} str - Whitespace separated tokens
66
+ * @returns {string[]} The tokens
67
+ */
68
+ export function splitXmlSpace(str) {
69
+ return str.split(XML_WHITESPACE_RUN).filter((token) => token !== "");
70
+ }
71
+
72
+ /**
73
+ * Convert a string to a number following the XPath Number grammar: optional
74
+ * XML whitespace, an optional minus sign, digits with an optional decimal
75
+ * point. Anything else, including the empty string, is NaN.
76
+ *
77
+ * @param {string} str - The string to convert
78
+ * @returns {number} The number, or NaN
79
+ *
80
+ * @example
81
+ * parseXPathNumber(" -3.5 "); // -3.5
82
+ * parseXPathNumber("1e3"); // NaN
83
+ */
84
+ export function parseXPathNumber(str) {
85
+ const trimmed = trimXmlSpace(str);
86
+ if (!XPATH_NUMBER.test(trimmed)) return Number.NaN;
87
+ return Number(trimmed);
88
+ }
89
+
90
+ /**
91
+ * Convert a number to its XPath string form: NaN, Infinity and -Infinity by
92
+ * name, zero as "0", every other number in decimal notation without an
93
+ * exponent. The digits are those of `Number#toString`; exponent forms are
94
+ * rewritten by moving the decimal point in the string, so no floating point
95
+ * arithmetic can alter them.
96
+ *
97
+ * @param {number} value - The number
98
+ * @returns {string} The string form
99
+ *
100
+ * @example
101
+ * formatXPathNumber(1e21); // "1000000000000000000000"
102
+ * formatXPathNumber(1e-7); // "0.0000001"
103
+ */
104
+ export function formatXPathNumber(value) {
105
+ if (Number.isNaN(value)) return "NaN";
106
+ if (value === Infinity) return "Infinity";
107
+ if (value === -Infinity) return "-Infinity";
108
+ if (value === 0) return "0";
109
+
110
+ const text = String(value);
111
+ const exponentAt = text.indexOf("e");
112
+ if (exponentAt === -1) return text;
113
+
114
+ // Exponent forms are "d.ddde+NN" (|value| >= 1e21) or "d.ddde-N" (< 1e-6)
115
+ const sign = value < 0 ? "-" : "";
116
+ const exponent = Number(text.substring(exponentAt + 1));
117
+ const digits = text.substring(sign.length, exponentAt).replace(".", "");
118
+
119
+ if (exponent < 0) {
120
+ return `${sign}0.${"0".repeat(-exponent - 1)}${digits}`;
121
+ }
122
+ return sign + digits + "0".repeat(exponent + 1 - digits.length);
123
+ }
124
+
125
+ /**
126
+ * Number of characters (Unicode code points) in a string.
127
+ *
128
+ * @param {string} str - Any string
129
+ * @returns {number} The number of code points
130
+ */
131
+ export function codePointLength(str) {
132
+ if (!SURROGATE.test(str)) return str.length;
133
+ return Array.from(str).length;
134
+ }
135
+
136
+ /**
137
+ * XPath `substring()` on code points: the characters whose 1-based position
138
+ * p satisfies `p >= round(start)` and, with a length, `p < round(start) +
139
+ * round(length)`. NaN and infinite arguments follow from those comparisons.
140
+ *
141
+ * @param {string} str - The string
142
+ * @param {number} start - Start position, not yet rounded
143
+ * @param {number} [length] - Number of characters, not yet rounded
144
+ * @returns {string} The substring
145
+ *
146
+ * @example
147
+ * xpathSubstring("12345", 1.5, 2.6); // "234"
148
+ */
149
+ export function xpathSubstring(str, start, length) {
150
+ const first = Math.round(start);
151
+ const end = length === undefined ? Infinity : first + Math.round(length);
152
+ const from = Math.max(first, 1);
153
+ if (Number.isNaN(end) || Number.isNaN(from) || end <= from) return "";
154
+
155
+ const chars = SURROGATE.test(str) ? Array.from(str) : null;
156
+ if (chars === null) return str.slice(from - 1, end - 1);
157
+ return chars.slice(from - 1, end - 1).join("");
158
+ }
159
+
160
+ /**
161
+ * XPath `translate()` on code points: each character of `str` found in
162
+ * `from` is replaced by the character at the same position in `to`, or
163
+ * removed when `to` is shorter. The first occurrence in `from` wins.
164
+ *
165
+ * @param {string} str - The string to translate
166
+ * @param {string} from - Characters to replace
167
+ * @param {string} to - Replacement characters
168
+ * @returns {string} The translated string
169
+ */
170
+ export function xpathTranslate(str, from, to) {
171
+ const fromChars = Array.from(from);
172
+ const toChars = Array.from(to);
173
+ let result = "";
174
+ for (const char of str) {
175
+ const index = fromChars.indexOf(char);
176
+ if (index === -1) {
177
+ result += char;
178
+ } else if (index < toChars.length) {
179
+ result += toChars[index];
180
+ }
181
+ }
182
+ return result;
183
+ }