jq79 0.7.2 → 0.7.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/html.ts ADDED
@@ -0,0 +1,289 @@
1
+ // ---------------------------------------------------------------------------
2
+ // an HTML tree without a DOM
3
+ //
4
+ // precompile() has to read a component where there is no DOMParser - on node,
5
+ // for the Vite plugin, and in a service worker - and find every attribute and
6
+ // every text node the runtime's own parse finds. This builds that tree:
7
+ // elements with their attributes and children, and text decoded the way the
8
+ // HTML parser decodes it.
9
+ //
10
+ // It is not a conforming HTML parser and doesn't try to be. What precompile
11
+ // reads is *which* attributes and texts exist, and wherever the real parser
12
+ // would put one somewhere else - an implied end tag, a table's foster parent -
13
+ // it still exists. The places where it would read a different *value* are the
14
+ // ones that matter, and those it follows:
15
+ // - newlines normalized first (CRLF and a lone CR are LF), as the input stream is
16
+ // - character references decoded in text and attribute values, with the
17
+ // attribute rule for a legacy reference with no semicolon
18
+ // - <html>, <head> and <body> dropped, as a template drops them
19
+ // - attribute names lowercased, and the first of two duplicates kept
20
+ // - <script>/<style> (and the other raw text elements) read verbatim to their
21
+ // end tag, <textarea>/<title> decoded but never parsed for tags
22
+ // - comments dropped, and the text on either side kept as two texts
23
+ //
24
+ // tests/precompile.test.ts holds it against DOMParser, tree for tree
25
+ // ---------------------------------------------------------------------------
26
+
27
+ export type HTMLElementNode = { tag: string; attrs: Record<string, string>; children: HTMLNode[] }
28
+ export type HTMLNode = HTMLElementNode | string
29
+
30
+ const VOID_TAGS = new Set([
31
+ "area", "base", "br", "col", "embed", "hr", "img", "input",
32
+ "link", "meta", "param", "source", "track", "wbr",
33
+ ])
34
+
35
+ // the document's own structure, which a template can't hold: inside one the
36
+ // parser drops these tags, start and end, and keeps what they wrapped
37
+ const DOCUMENT_TAGS = new Set(["html", "head", "body"])
38
+
39
+ // read verbatim up to their end tag - and, for the escapable two, decoded
40
+ const RAW_TEXT_TAGS = new Set(["script", "style", "xmp", "iframe", "noembed", "noframes"])
41
+ const ESCAPABLE_RAW_TEXT_TAGS = new Set(["textarea", "title"])
42
+
43
+ // the named references a component plausibly writes. The HTML table has over
44
+ // two thousand; one missing here leaves its text undecoded, and that is a
45
+ // precompiled function under a key the runtime never asks for - which safe
46
+ // mode reports by name, rather than a wrong function running
47
+ const NAMED_REFERENCES: Record<string, string> = {
48
+ apos: "'", trade: "\u2122", hellip: "\u2026", mdash: "\u2014", ndash: "\u2013",
49
+ lsquo: "\u2018", rsquo: "\u2019", ldquo: "\u201c", rdquo: "\u201d", bull: "\u2022", euro: "\u20ac",
50
+ larr: "\u2190", rarr: "\u2192", uarr: "\u2191", darr: "\u2193", check: "\u2713",
51
+ lbrace: "{", rbrace: "}", lcub: "{", rcub: "}", lpar: "(", rpar: ")", lsqb: "[", rsqb: "]",
52
+ colon: ":", comma: ",", period: ".", semi: ";", excl: "!", quest: "?", num: "#", dollar: "$",
53
+ percnt: "%", ast: "*", plus: "+", equals: "=", sol: "/", bsol: "\\", verbar: "|", vert: "|",
54
+ grave: "`", Hat: "^", lowbar: "_", commat: "@", Tab: "\t", NewLine: "\n",
55
+ }
56
+
57
+ // the references that also decode without their semicolon - the web before
58
+ // HTML required one - which is the whole of this list, from the standard: the
59
+ // five markup ones in both cases, and Latin-1's 96, U+00A0 to U+00FF in order.
60
+ // A name that isn't a reference is read as the longest of these it starts
61
+ // with (`&notareference;` is `¬areference;`), so all of them are needed to
62
+ // read what the parser reads
63
+ const LATIN_1 = (
64
+ "nbsp iexcl cent pound curren yen brvbar sect uml copy ordf laquo not shy reg macr deg plusmn sup2 sup3 " +
65
+ "acute micro para middot cedil sup1 ordm raquo frac14 frac12 frac34 iquest Agrave Aacute Acirc Atilde Auml " +
66
+ "Aring AElig Ccedil Egrave Eacute Ecirc Euml Igrave Iacute Icirc Iuml ETH Ntilde Ograve Oacute Ocirc Otilde " +
67
+ "Ouml times Oslash Ugrave Uacute Ucirc Uuml Yacute THORN szlig agrave aacute acirc atilde auml aring aelig " +
68
+ "ccedil egrave eacute ecirc euml igrave iacute icirc iuml eth ntilde ograve oacute ocirc otilde ouml divide " +
69
+ "oslash ugrave uacute ucirc uuml yacute thorn yuml"
70
+ ).split(" ")
71
+ const LEGACY_REFERENCES: Record<string, string> = {
72
+ amp: "&", AMP: "&", lt: "<", LT: "<", gt: ">", GT: ">", quot: "\"", QUOT: "\"", COPY: "\u00a9", REG: "\u00ae",
73
+ ...Object.fromEntries(LATIN_1.map((name, i) => [name, String.fromCharCode(0xa0 + i)])),
74
+ }
75
+ // longest first, so `&sup2` is ² and not "sup" plus a 2
76
+ const LEGACY_NAMES = Object.keys(LEGACY_REFERENCES).sort((a, b) => b.length - a.length)
77
+
78
+ const has = (table: Record<string, string>, name: string) => Object.prototype.hasOwnProperty.call(table, name)
79
+
80
+ const fromCodePoint = (code: number): string =>
81
+ code === 0 || code > 0x10ffff || (code >= 0xd800 && code <= 0xdfff) ? "�" : String.fromCodePoint(code)
82
+
83
+ export const CHARACTER_REFERENCE_RE = /&(#[0-9]+;?|#[xX][0-9a-fA-F]+;?|[A-Za-z][A-Za-z0-9]*;?)/g
84
+ const ALPHANUMERIC_RE = /[A-Za-z0-9]/
85
+
86
+ // what one match of CHARACTER_REFERENCE_RE in `text`, at `offset`, decodes to.
87
+ // In an attribute a legacy reference with no semicolon stays literal when the
88
+ // next character is `=` or alphanumeric - `?a=1&copy=2` is a query string, not
89
+ // a copyright sign. In text it decodes, and what follows the name stays.
90
+ // Exported for the editor extension, which decodes an attribute value while
91
+ // keeping where each character came from (editors/vscode/src/entities.ts)
92
+ export const decodeReference = (text: string, whole: string, ref: string, offset: number, inAttribute: boolean): string => {
93
+ if (ref[0] === "#") {
94
+ const hex = ref[1] === "x" || ref[1] === "X"
95
+ const digits = ref.slice(hex ? 2 : 1).replace(/;$/, "")
96
+ return fromCodePoint(parseInt(digits, hex ? 16 : 10))
97
+ }
98
+ const semicolon = ref.endsWith(";")
99
+ const name = semicolon ? ref.slice(0, -1) : ref
100
+ if (semicolon && has(LEGACY_REFERENCES, name)) return LEGACY_REFERENCES[name]
101
+ if (semicolon && has(NAMED_REFERENCES, name)) return NAMED_REFERENCES[name]
102
+ const legacy = LEGACY_NAMES.find(candidate => name.startsWith(candidate))
103
+ if (legacy === undefined) return whole
104
+ const after = name.length > legacy.length ? name[legacy.length] : semicolon ? ";" : text[offset + whole.length]
105
+ if (inAttribute && after !== undefined && (after === "=" || ALPHANUMERIC_RE.test(after))) return whole
106
+ return LEGACY_REFERENCES[legacy] + ref.slice(legacy.length)
107
+ }
108
+
109
+ const decode = (text: string, inAttribute: boolean): string => {
110
+ if (!text.includes("&")) return text
111
+ return text.replace(CHARACTER_REFERENCE_RE, (whole: string, ref: string, offset: number) =>
112
+ decodeReference(text, whole, ref, offset, inAttribute))
113
+ }
114
+
115
+ const TAG_NAME_RE = /[^\s/>]*/y
116
+ const ATTRIBUTE_NAME_RE = /[^\s/>][^\s/>=]*/y
117
+ const WHITESPACE_RE = /\s*/y
118
+ const UNQUOTED_VALUE_RE = /[^\s>]*/y
119
+ const END_TAG_RE = /<\/([A-Za-z][^\s/>]*)[^>]*>/y
120
+
121
+ // runs a sticky pattern at `at` and answers with the match (or "")
122
+ const readAt = (re: RegExp, src: string, at: number): string => {
123
+ re.lastIndex = at
124
+ return re.exec(src)?.[0] ?? ""
125
+ }
126
+
127
+ // where a raw text element's content ends: its own end tag, whatever the case,
128
+ // followed by what can end a tag name
129
+ const rawTextEnd = (src: string, from: number, tag: string): number => {
130
+ const re = new RegExp(`</${tag}(?=[\\s/>])`, "gi")
131
+ re.lastIndex = from
132
+ const match = re.exec(src)
133
+ return match ? match.index : src.length
134
+ }
135
+
136
+ // the element children of a comment are nothing, and the texts on either side
137
+ // of it stay two texts - a comment is a node in the DOM, which is what keeps
138
+ // them apart. A marker holds its place while the tree is built
139
+ const COMMENT = null
140
+
141
+ type BuildNode = { tag: string; attrs: Record<string, string>; children: (BuildNode | string | null)[] }
142
+
143
+ const finish = (node: BuildNode): HTMLElementNode => ({
144
+ tag: node.tag,
145
+ attrs: node.attrs,
146
+ children: node.children.flatMap((child): HTMLNode[] =>
147
+ child === COMMENT ? [] : typeof child === "string" ? (child ? [child] : []) : [finish(child)]
148
+ ),
149
+ })
150
+
151
+ export const parseHTML = (input: string): HTMLNode[] => {
152
+ const src = input.replace(/\r\n?/g, "\n")
153
+ const root: BuildNode = { tag: "#root", attrs: {}, children: [] }
154
+ const stack: BuildNode[] = [root]
155
+ const current = () => stack[stack.length - 1]
156
+ // inside <svg> or <math> a `/>` closes the element - in HTML it is ignored
157
+ let foreignDepth = 0
158
+
159
+ const appendText = (text: string) => {
160
+ if (!text) return
161
+ const children = current().children
162
+ const last = children[children.length - 1]
163
+ if (typeof last === "string") children[children.length - 1] = last + text
164
+ else children.push(text)
165
+ }
166
+
167
+ const close = (tag: string) => {
168
+ for (let depth = stack.length - 1; depth > 0; depth--) {
169
+ if (stack[depth].tag !== tag) continue
170
+ for (let popped = stack.length - 1; popped >= depth; popped--) {
171
+ if (stack[popped].tag === "svg" || stack[popped].tag === "math") foreignDepth--
172
+ }
173
+ stack.length = depth
174
+ return
175
+ }
176
+ }
177
+
178
+ let at = 0
179
+ while (at < src.length) {
180
+ const lt = src.indexOf("<", at)
181
+ if (lt === -1) {
182
+ appendText(decode(src.slice(at), false))
183
+ break
184
+ }
185
+ appendText(decode(src.slice(at, lt), false))
186
+ at = lt
187
+
188
+ // a comment - `<!-->` and `<!--->` are complete, empty ones
189
+ if (src.startsWith("<!--", at)) {
190
+ current().children.push(COMMENT)
191
+ if (src.startsWith(">", at + 4)) at += 5
192
+ else if (src.startsWith("->", at + 4)) at += 6
193
+ else {
194
+ const end = src.indexOf("-->", at + 4)
195
+ at = end === -1 ? src.length : end + 3
196
+ }
197
+ continue
198
+ }
199
+ // a doctype leaves no node at all - so the texts around it are one text -
200
+ // and a CDATA section or a processing instruction is a (bogus) comment
201
+ if (src[at + 1] === "!" || src[at + 1] === "?") {
202
+ const end = src.indexOf(">", at)
203
+ if (!/^<!doctype/i.test(src.slice(at, at + 9))) current().children.push(COMMENT)
204
+ at = end === -1 ? src.length : end + 1
205
+ continue
206
+ }
207
+ if (src[at + 1] === "/") {
208
+ END_TAG_RE.lastIndex = at
209
+ const end = END_TAG_RE.exec(src)
210
+ if (end) {
211
+ if (!DOCUMENT_TAGS.has(end[1].toLowerCase())) close(end[1].toLowerCase())
212
+ at = END_TAG_RE.lastIndex
213
+ } else {
214
+ // `</>` is dropped, and `</ anything>` is a bogus comment
215
+ const gt = src.indexOf(">", at)
216
+ at = gt === -1 ? src.length : gt + 1
217
+ }
218
+ continue
219
+ }
220
+ if (!/[A-Za-z]/.test(src[at + 1] ?? "")) {
221
+ appendText("<")
222
+ at++
223
+ continue
224
+ }
225
+
226
+ // a start tag
227
+ const tag = readAt(TAG_NAME_RE, src, at + 1).toLowerCase()
228
+ const ignored = DOCUMENT_TAGS.has(tag)
229
+ let cursor = at + 1 + tag.length
230
+ const attrs: Record<string, string> = {}
231
+ let selfClosing = false
232
+ let complete = false
233
+ while (cursor < src.length) {
234
+ cursor += readAt(WHITESPACE_RE, src, cursor).length
235
+ const char = src[cursor]
236
+ if (char === ">") { cursor++; complete = true; break }
237
+ if (char === "/") {
238
+ if (src[cursor + 1] === ">") { selfClosing = true; cursor += 2; complete = true; break }
239
+ cursor++
240
+ continue
241
+ }
242
+ if (char === undefined) break
243
+ const name = readAt(ATTRIBUTE_NAME_RE, src, cursor)
244
+ cursor += name.length
245
+ let value = ""
246
+ const afterName = cursor + readAt(WHITESPACE_RE, src, cursor).length
247
+ if (src[afterName] === "=") {
248
+ cursor = afterName + 1
249
+ cursor += readAt(WHITESPACE_RE, src, cursor).length
250
+ const quote = src[cursor]
251
+ if (quote === "\"" || quote === "'") {
252
+ const close = src.indexOf(quote, cursor + 1)
253
+ const end = close === -1 ? src.length : close
254
+ value = src.slice(cursor + 1, end)
255
+ cursor = end + 1
256
+ } else {
257
+ value = readAt(UNQUOTED_VALUE_RE, src, cursor)
258
+ cursor += value.length
259
+ }
260
+ value = decode(value, true)
261
+ }
262
+ const key = name.toLowerCase()
263
+ if (!(key in attrs)) attrs[key] = value
264
+ }
265
+ // a tag cut off by the end of the input is dropped, as the parser drops it
266
+ if (!complete) break
267
+ at = cursor
268
+ if (ignored) continue
269
+
270
+ const node: BuildNode = { tag, attrs, children: [] }
271
+ current().children.push(node)
272
+ if (VOID_TAGS.has(tag)) continue
273
+ if (selfClosing && foreignDepth > 0) continue
274
+
275
+ if (foreignDepth === 0 && (RAW_TEXT_TAGS.has(tag) || ESCAPABLE_RAW_TEXT_TAGS.has(tag))) {
276
+ const end = rawTextEnd(src, at, tag)
277
+ const text = src.slice(at, end)
278
+ node.children.push(ESCAPABLE_RAW_TEXT_TAGS.has(tag) ? decode(text, false) : text)
279
+ const gt = src.indexOf(">", end)
280
+ at = end === src.length || gt === -1 ? src.length : gt + 1
281
+ continue
282
+ }
283
+
284
+ if (tag === "svg" || tag === "math") foreignDepth++
285
+ stack.push(node)
286
+ }
287
+
288
+ return finish(root).children
289
+ }