@mavogel/cdk-vscode-server 0.0.123 → 0.0.124
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.jsii +3 -3
- package/lib/idle-monitor/idle-monitor.js +1 -1
- package/lib/vscode-server.js +1 -1
- package/node_modules/node-html-parser/CHANGELOG.md +79 -0
- package/node_modules/node-html-parser/README.md +39 -39
- package/node_modules/node-html-parser/dist/index.cjs +5079 -0
- package/node_modules/node-html-parser/dist/index.d.ts +10 -10
- package/node_modules/node-html-parser/dist/index.mjs +5068 -0
- package/node_modules/node-html-parser/dist/index.umd.js +5083 -0
- package/node_modules/node-html-parser/dist/matcher.d.ts +22 -6
- package/node_modules/node-html-parser/dist/nodes/comment.d.ts +1 -1
- package/node_modules/node-html-parser/dist/nodes/html.d.ts +2 -2
- package/node_modules/node-html-parser/dist/nodes/node.d.ts +2 -2
- package/node_modules/node-html-parser/dist/nodes/text.d.ts +1 -1
- package/node_modules/node-html-parser/dist/valid.d.ts +1 -1
- package/node_modules/node-html-parser/node_modules/entities/LICENSE +11 -0
- package/node_modules/node-html-parser/node_modules/entities/dist/decode-codepoint.d.ts +28 -0
- package/node_modules/node-html-parser/node_modules/entities/dist/decode-codepoint.d.ts.map +1 -0
- package/node_modules/node-html-parser/node_modules/entities/dist/decode-codepoint.js +61 -0
- package/node_modules/node-html-parser/node_modules/entities/dist/decode-codepoint.js.map +1 -0
- package/node_modules/node-html-parser/node_modules/entities/dist/decode.d.ts +227 -0
- package/node_modules/node-html-parser/node_modules/entities/dist/decode.d.ts.map +1 -0
- package/node_modules/node-html-parser/node_modules/entities/dist/decode.js +1132 -0
- package/node_modules/node-html-parser/node_modules/entities/dist/decode.js.map +1 -0
- package/node_modules/node-html-parser/node_modules/entities/dist/encode.d.ts +26 -0
- package/node_modules/node-html-parser/node_modules/entities/dist/encode.d.ts.map +1 -0
- package/node_modules/node-html-parser/node_modules/entities/dist/encode.js +295 -0
- package/node_modules/node-html-parser/node_modules/entities/dist/encode.js.map +1 -0
- package/node_modules/node-html-parser/node_modules/entities/dist/escape.d.ts +60 -0
- package/node_modules/node-html-parser/node_modules/entities/dist/escape.d.ts.map +1 -0
- package/node_modules/node-html-parser/node_modules/entities/dist/escape.js +172 -0
- package/node_modules/node-html-parser/node_modules/entities/dist/escape.js.map +1 -0
- package/node_modules/node-html-parser/node_modules/entities/dist/generated/decode-data-html.d.ts +3 -0
- package/node_modules/node-html-parser/node_modules/entities/dist/generated/decode-data-html.d.ts.map +1 -0
- package/node_modules/node-html-parser/node_modules/entities/dist/generated/decode-data-html.js +5 -0
- package/node_modules/node-html-parser/node_modules/entities/dist/generated/decode-data-html.js.map +1 -0
- package/node_modules/node-html-parser/node_modules/entities/dist/generated/decode-data-xml.d.ts +3 -0
- package/node_modules/node-html-parser/node_modules/entities/dist/generated/decode-data-xml.d.ts.map +1 -0
- package/node_modules/node-html-parser/node_modules/entities/dist/generated/decode-data-xml.js +7 -0
- package/node_modules/node-html-parser/node_modules/entities/dist/generated/decode-data-xml.js.map +1 -0
- package/node_modules/node-html-parser/node_modules/entities/dist/generated/encode-html.d.ts +3 -0
- package/node_modules/node-html-parser/node_modules/entities/dist/generated/encode-html.d.ts.map +1 -0
- package/node_modules/node-html-parser/node_modules/entities/dist/generated/encode-html.js +3 -0
- package/node_modules/node-html-parser/node_modules/entities/dist/generated/encode-html.js.map +1 -0
- package/node_modules/node-html-parser/node_modules/entities/dist/index.d.ts +89 -0
- package/node_modules/node-html-parser/node_modules/entities/dist/index.d.ts.map +1 -0
- package/node_modules/node-html-parser/node_modules/entities/dist/index.js +87 -0
- package/node_modules/node-html-parser/node_modules/entities/dist/index.js.map +1 -0
- package/node_modules/node-html-parser/node_modules/entities/dist/internal/bin-trie-flags.d.ts +48 -0
- package/node_modules/node-html-parser/node_modules/entities/dist/internal/bin-trie-flags.d.ts.map +1 -0
- package/node_modules/node-html-parser/node_modules/entities/dist/internal/bin-trie-flags.js +49 -0
- package/node_modules/node-html-parser/node_modules/entities/dist/internal/bin-trie-flags.js.map +1 -0
- package/node_modules/node-html-parser/node_modules/entities/dist/internal/decode-shared.d.ts +38 -0
- package/node_modules/node-html-parser/node_modules/entities/dist/internal/decode-shared.d.ts.map +1 -0
- package/node_modules/node-html-parser/node_modules/entities/dist/internal/decode-shared.js +195 -0
- package/node_modules/node-html-parser/node_modules/entities/dist/internal/decode-shared.js.map +1 -0
- package/node_modules/node-html-parser/node_modules/entities/package.json +83 -0
- package/node_modules/node-html-parser/node_modules/entities/readme.md +131 -0
- package/node_modules/node-html-parser/node_modules/entities/src/decode-codepoint.ts +69 -0
- package/node_modules/node-html-parser/node_modules/entities/src/decode.ts +1368 -0
- package/node_modules/node-html-parser/node_modules/entities/src/encode.ts +337 -0
- package/node_modules/node-html-parser/node_modules/entities/src/escape.ts +187 -0
- package/node_modules/node-html-parser/node_modules/entities/src/generated/decode-data-html.ts +12 -0
- package/node_modules/node-html-parser/node_modules/entities/src/generated/decode-data-xml.ts +7 -0
- package/node_modules/node-html-parser/node_modules/entities/src/generated/encode-html.ts +3 -0
- package/node_modules/node-html-parser/node_modules/entities/src/index.ts +160 -0
- package/node_modules/node-html-parser/node_modules/entities/src/internal/bin-trie-flags.ts +47 -0
- package/node_modules/node-html-parser/node_modules/entities/src/internal/decode-shared.ts +209 -0
- package/node_modules/node-html-parser/package.json +35 -30
- package/package.json +3 -3
- package/node_modules/he/LICENSE-MIT.txt +0 -20
- package/node_modules/he/README.md +0 -379
- package/node_modules/he/bin/he +0 -148
- package/node_modules/he/he.js +0 -345
- package/node_modules/he/man/he.1 +0 -78
- package/node_modules/he/package.json +0 -58
- package/node_modules/node-html-parser/dist/back.js +0 -6
- package/node_modules/node-html-parser/dist/index.js +0 -31
- package/node_modules/node-html-parser/dist/main.js +0 -1804
- package/node_modules/node-html-parser/dist/matcher.js +0 -106
- package/node_modules/node-html-parser/dist/nodes/comment.js +0 -33
- package/node_modules/node-html-parser/dist/nodes/html.js +0 -1186
- package/node_modules/node-html-parser/dist/nodes/node.js +0 -41
- package/node_modules/node-html-parser/dist/nodes/text.js +0 -105
- package/node_modules/node-html-parser/dist/nodes/type.js +0 -9
- package/node_modules/node-html-parser/dist/parse.js +0 -5
- package/node_modules/node-html-parser/dist/valid.js +0 -12
- package/node_modules/node-html-parser/dist/void-tag.js +0 -27
|
@@ -0,0 +1,160 @@
|
|
|
1
|
+
import { type DecodingMode, decodeHTML, decodeXML } from "./decode.js";
|
|
2
|
+
import { encodeHTML, encodeNonAsciiHTML } from "./encode.js";
|
|
3
|
+
import {
|
|
4
|
+
encodeXML,
|
|
5
|
+
escapeAttribute,
|
|
6
|
+
escapeText,
|
|
7
|
+
escapeUTF8,
|
|
8
|
+
} from "./escape.js";
|
|
9
|
+
|
|
10
|
+
/** The level of entities to support. */
|
|
11
|
+
export enum EntityLevel {
|
|
12
|
+
/** Support only XML entities. */
|
|
13
|
+
XML = 0,
|
|
14
|
+
/** Support HTML entities, which are a superset of XML entities. */
|
|
15
|
+
HTML = 1,
|
|
16
|
+
}
|
|
17
|
+
|
|
18
|
+
/**
|
|
19
|
+
* Encoding strategy used by `encode`.
|
|
20
|
+
*/
|
|
21
|
+
export enum EncodingMode {
|
|
22
|
+
/**
|
|
23
|
+
* The output is UTF-8 encoded. Only characters that need escaping within
|
|
24
|
+
* XML will be escaped.
|
|
25
|
+
*/
|
|
26
|
+
UTF8,
|
|
27
|
+
/**
|
|
28
|
+
* The output consists only of ASCII characters. Characters that need
|
|
29
|
+
* escaping within HTML, and characters that aren't ASCII characters will
|
|
30
|
+
* be escaped.
|
|
31
|
+
*/
|
|
32
|
+
ASCII,
|
|
33
|
+
/**
|
|
34
|
+
* Encode all characters that have an equivalent entity, as well as all
|
|
35
|
+
* characters that are not ASCII characters.
|
|
36
|
+
*/
|
|
37
|
+
Extensive,
|
|
38
|
+
/**
|
|
39
|
+
* Encode all characters that have to be escaped in HTML attributes,
|
|
40
|
+
* following {@link https://html.spec.whatwg.org/multipage/parsing.html#escapingString}.
|
|
41
|
+
*/
|
|
42
|
+
Attribute,
|
|
43
|
+
/**
|
|
44
|
+
* Encode all characters that have to be escaped in HTML text,
|
|
45
|
+
* following {@link https://html.spec.whatwg.org/multipage/parsing.html#escapingString}.
|
|
46
|
+
*/
|
|
47
|
+
Text,
|
|
48
|
+
}
|
|
49
|
+
|
|
50
|
+
/**
|
|
51
|
+
* Options for `decode`.
|
|
52
|
+
*/
|
|
53
|
+
export interface DecodingOptions {
|
|
54
|
+
/**
|
|
55
|
+
* The level of entities to support.
|
|
56
|
+
* @default {@link EntityLevel.XML}
|
|
57
|
+
*/
|
|
58
|
+
level?: EntityLevel;
|
|
59
|
+
/**
|
|
60
|
+
* Decoding mode. If `Legacy`, will support legacy entities not terminated
|
|
61
|
+
* with a semicolon (`;`).
|
|
62
|
+
*
|
|
63
|
+
* Always `Strict` for XML. For HTML, set this to
|
|
64
|
+
* {@link DecodingMode.Attribute} if you are parsing an attribute value.
|
|
65
|
+
* @default {@link DecodingMode.Legacy}
|
|
66
|
+
*/
|
|
67
|
+
mode?: DecodingMode | undefined;
|
|
68
|
+
}
|
|
69
|
+
|
|
70
|
+
/**
|
|
71
|
+
* Decodes a string with entities.
|
|
72
|
+
* @param input String to decode.
|
|
73
|
+
* @param options Decoding options.
|
|
74
|
+
*/
|
|
75
|
+
export function decode(
|
|
76
|
+
input: string,
|
|
77
|
+
options: DecodingOptions | EntityLevel = EntityLevel.XML,
|
|
78
|
+
): string {
|
|
79
|
+
const level = typeof options === "number" ? options : options.level;
|
|
80
|
+
|
|
81
|
+
if (level === EntityLevel.HTML) {
|
|
82
|
+
const mode = typeof options === "object" ? options.mode : undefined;
|
|
83
|
+
return decodeHTML(input, mode);
|
|
84
|
+
}
|
|
85
|
+
|
|
86
|
+
return decodeXML(input);
|
|
87
|
+
}
|
|
88
|
+
|
|
89
|
+
/**
|
|
90
|
+
* Options for `encode`.
|
|
91
|
+
*/
|
|
92
|
+
export interface EncodingOptions {
|
|
93
|
+
/**
|
|
94
|
+
* The level of entities to support.
|
|
95
|
+
* @default {@link EntityLevel.XML}
|
|
96
|
+
*/
|
|
97
|
+
level?: EntityLevel;
|
|
98
|
+
/**
|
|
99
|
+
* Output format.
|
|
100
|
+
* @default {@link EncodingMode.Extensive}
|
|
101
|
+
*/
|
|
102
|
+
mode?: EncodingMode;
|
|
103
|
+
}
|
|
104
|
+
|
|
105
|
+
/**
|
|
106
|
+
* Encodes a string with entities.
|
|
107
|
+
* @param input String to encode.
|
|
108
|
+
* @param options Encoding options.
|
|
109
|
+
*/
|
|
110
|
+
export function encode(
|
|
111
|
+
input: string,
|
|
112
|
+
options: EncodingOptions | EntityLevel = EntityLevel.XML,
|
|
113
|
+
): string {
|
|
114
|
+
const { mode = EncodingMode.Extensive, level = EntityLevel.XML } =
|
|
115
|
+
typeof options === "number" ? { level: options } : options;
|
|
116
|
+
|
|
117
|
+
switch (mode) {
|
|
118
|
+
case EncodingMode.UTF8: {
|
|
119
|
+
return escapeUTF8(input);
|
|
120
|
+
}
|
|
121
|
+
case EncodingMode.Attribute: {
|
|
122
|
+
return escapeAttribute(input);
|
|
123
|
+
}
|
|
124
|
+
case EncodingMode.Text: {
|
|
125
|
+
return escapeText(input);
|
|
126
|
+
}
|
|
127
|
+
case EncodingMode.ASCII: {
|
|
128
|
+
return (
|
|
129
|
+
level === EntityLevel.HTML ? encodeNonAsciiHTML : encodeXML
|
|
130
|
+
)(input);
|
|
131
|
+
}
|
|
132
|
+
// biome-ignore lint/complexity/noUselessSwitchCase: we get an error for the switch not being exhaustive
|
|
133
|
+
case EncodingMode.Extensive:
|
|
134
|
+
default: {
|
|
135
|
+
return (level === EntityLevel.HTML ? encodeHTML : encodeXML)(input);
|
|
136
|
+
}
|
|
137
|
+
}
|
|
138
|
+
}
|
|
139
|
+
|
|
140
|
+
export {
|
|
141
|
+
DecodingMode,
|
|
142
|
+
decodeHTML,
|
|
143
|
+
decodeHTMLAttribute,
|
|
144
|
+
decodeHTMLStrict,
|
|
145
|
+
decodeXML,
|
|
146
|
+
decodeXML as decodeXMLStrict,
|
|
147
|
+
EntityDecoder,
|
|
148
|
+
} from "./decode.js";
|
|
149
|
+
|
|
150
|
+
export {
|
|
151
|
+
encodeHTML,
|
|
152
|
+
encodeNonAsciiHTML,
|
|
153
|
+
} from "./encode.js";
|
|
154
|
+
export {
|
|
155
|
+
encodeXML,
|
|
156
|
+
escape,
|
|
157
|
+
escapeAttribute,
|
|
158
|
+
escapeText,
|
|
159
|
+
escapeUTF8,
|
|
160
|
+
} from "./escape.js";
|
|
@@ -0,0 +1,47 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Bit flags & masks for the binary trie encoding used for entity decoding.
|
|
3
|
+
*
|
|
4
|
+
* The trie is a flat `Uint16Array`. Every node starts with one header word:
|
|
5
|
+
*
|
|
6
|
+
* 15..14 VALUE_LENGTH Number of words the value occupies, +1.
|
|
7
|
+
* 0 = no value; 1 = value inline in bits 12..0;
|
|
8
|
+
* 2/3 = value in the 1/2 words after the header.
|
|
9
|
+
* 13 FLAG13 If VALUE_LENGTH > 0: semicolon required ("strict"
|
|
10
|
+
* entity; `;` is never stored as a branch).
|
|
11
|
+
* If VALUE_LENGTH == 0: this node is a compact run.
|
|
12
|
+
* 12..7 BRANCH_LENGTH Number of branches (or run length for runs).
|
|
13
|
+
* 6..0 JUMP_TABLE Jump-table offset / single-branch char / first
|
|
14
|
+
* run char (see below).
|
|
15
|
+
*
|
|
16
|
+
* Branch data follows the header and any value words. Its shape is selected
|
|
17
|
+
* by (JUMP_TABLE, BRANCH_LENGTH) in the header:
|
|
18
|
+
*
|
|
19
|
+
* Single branch JUMP_TABLE = the only child's char, BRANCH_LENGTH = 0.
|
|
20
|
+
* No branch words; the child node follows immediately.
|
|
21
|
+
* Jump table JUMP_TABLE = first covered char (> 0), BRANCH_LENGTH =
|
|
22
|
+
* table length. One word per covered char: 0 = no branch,
|
|
23
|
+
* otherwise the child's offset from the END of the table,
|
|
24
|
+
* +1 (so 0 stays the no-branch sentinel).
|
|
25
|
+
* Dictionary JUMP_TABLE = 0, BRANCH_LENGTH = number of branches.
|
|
26
|
+
* ceil(n/2) words of sorted keys packed two per word
|
|
27
|
+
* (low byte first), then n pointer words storing the
|
|
28
|
+
* child's offset from the END of the branch data.
|
|
29
|
+
* Compact run VALUE_LENGTH = 0, FLAG13 set. BRANCH_LENGTH = run
|
|
30
|
+
* length (3..63), JUMP_TABLE = first char; remaining run
|
|
31
|
+
* chars packed two per word after the header. The target
|
|
32
|
+
* node follows the packed words immediately.
|
|
33
|
+
*
|
|
34
|
+
* Pointers are end-relative (rather than relative to the pointer's own
|
|
35
|
+
* position) because that makes the common "child encoded right after the
|
|
36
|
+
* branch data" case a small constant, which compresses far better. Offsets
|
|
37
|
+
* to already-encoded (shared) nodes wrap via uint16 modulo arithmetic; the
|
|
38
|
+
* decoder masks navigation results with `& 0xff_ff` to match.
|
|
39
|
+
*/
|
|
40
|
+
export const enum BinTrieFlags {
|
|
41
|
+
VALUE_LENGTH = 0b1100_0000_0000_0000,
|
|
42
|
+
FLAG13 = 0b0010_0000_0000_0000,
|
|
43
|
+
BRANCH_LENGTH = 0b0001_1111_1000_0000,
|
|
44
|
+
JUMP_TABLE = 0b0000_0000_0111_1111,
|
|
45
|
+
/** Bits 12..0: the inline value of a VALUE_LENGTH = 1 header word. */
|
|
46
|
+
VALUE_MASK = 0b0001_1111_1111_1111,
|
|
47
|
+
}
|
|
@@ -0,0 +1,209 @@
|
|
|
1
|
+
/*
|
|
2
|
+
* Inverse of the encoder's SAFE alphabet (0x21..0x7E minus 0x22, 0x24, 0x5C),
|
|
3
|
+
* precomputed once at module load. Entries for excluded chars stay 0 but
|
|
4
|
+
* are never read.
|
|
5
|
+
*/
|
|
6
|
+
const BASE91_INVERSE = /* #__PURE__ */ (() => {
|
|
7
|
+
const table = new Uint8Array(127);
|
|
8
|
+
let code = 0;
|
|
9
|
+
for (let char = 0x21; char <= 0x7e; char++) {
|
|
10
|
+
if (char !== 0x22 && char !== 0x24 && char !== 0x5c) {
|
|
11
|
+
table[char] = code++;
|
|
12
|
+
}
|
|
13
|
+
}
|
|
14
|
+
return table;
|
|
15
|
+
})();
|
|
16
|
+
|
|
17
|
+
/**
|
|
18
|
+
* Decode a dictionary-encoded trie string back into its Uint16Array.
|
|
19
|
+
*
|
|
20
|
+
* Stream layout (consumed in this order):
|
|
21
|
+
* 1. dict1 atoms — `dict1AtomCount` uint16 values, delta+RLE encoded.
|
|
22
|
+
* 2. dict2 atoms — `atomCount - dict1AtomCount` values, delta+RLE.
|
|
23
|
+
* 3. dict2 ngrams — `ngramCount - (dictSize - dict1AtomCount)` entries,
|
|
24
|
+
* each a pair of slot codes that resolve to earlier slots.
|
|
25
|
+
* 4. dict1 ngrams — `dictSize - dict1AtomCount` entries, same shape.
|
|
26
|
+
* 5. data — slot codes, each expanding to one or more uint16 values.
|
|
27
|
+
*
|
|
28
|
+
* Codes use a 91-char base (printable ASCII minus `"`, `$`, `\`):
|
|
29
|
+
* - char1 < dictSize → 1-char code, slot = char1
|
|
30
|
+
* - char1 ≥ dictSize → 2-char code, slot = dictSize + (char1 - dictSize)*91 + char2
|
|
31
|
+
*
|
|
32
|
+
* Slot index → token kind:
|
|
33
|
+
* [0, A) dict1 atoms (1-char codes)
|
|
34
|
+
* [A, dictSize) dict1 ngrams (1-char codes)
|
|
35
|
+
* [dictSize, dictSize+D) dict2 atoms (2-char codes)
|
|
36
|
+
* [dictSize+D, end) dict2 ngrams (2-char codes)
|
|
37
|
+
*
|
|
38
|
+
* Both atom dicts decode before any ngram, and dict2 ngrams decode before
|
|
39
|
+
* dict1 ngrams. So every ngram entry references slots whose contents are
|
|
40
|
+
* already filled — no forward references to handle.
|
|
41
|
+
*
|
|
42
|
+
* This runs on library import. Flat typed arrays store each slot as either
|
|
43
|
+
* a plain value (`single`, covering every atom) or a range in a shared
|
|
44
|
+
* `pool` (ngrams).
|
|
45
|
+
* @param input Packed trie string.
|
|
46
|
+
* @param resultLength Expected number of uint16 values in the output.
|
|
47
|
+
* @param atomCount Total number of distinct uint16 values in the trie.
|
|
48
|
+
* @param dict1AtomCount Atoms in the 1-char range (`A` above).
|
|
49
|
+
* @param ngramCount Total number of ngram entries (dict1 + dict2).
|
|
50
|
+
* @param dictSize Number of 1-char code slots; the rest of `BASE - dictSize`
|
|
51
|
+
* first-byte values are 2-char codes.
|
|
52
|
+
*/
|
|
53
|
+
export function decodeTrieDict(
|
|
54
|
+
input: string,
|
|
55
|
+
resultLength: number,
|
|
56
|
+
atomCount: number,
|
|
57
|
+
dict1AtomCount: number,
|
|
58
|
+
ngramCount: number,
|
|
59
|
+
dictSize: number,
|
|
60
|
+
): Uint16Array {
|
|
61
|
+
const base = 91;
|
|
62
|
+
const inputLength = input.length;
|
|
63
|
+
// For 2-char codes, slot = char1 * base - twoCharBias + char2.
|
|
64
|
+
const twoCharBias = dictSize * (base - 1);
|
|
65
|
+
|
|
66
|
+
let pos = 0;
|
|
67
|
+
|
|
68
|
+
/** Read one slot code at `pos` and return its slot index, advancing pos. */
|
|
69
|
+
const readSlotCode = (): number => {
|
|
70
|
+
const c1 = BASE91_INVERSE[input.charCodeAt(pos++)];
|
|
71
|
+
return c1 < dictSize
|
|
72
|
+
? c1
|
|
73
|
+
: c1 * base - twoCharBias + BASE91_INVERSE[input.charCodeAt(pos++)];
|
|
74
|
+
};
|
|
75
|
+
|
|
76
|
+
const dict2AtomCount = atomCount - dict1AtomCount;
|
|
77
|
+
const slotCount = atomCount + ngramCount;
|
|
78
|
+
|
|
79
|
+
/*
|
|
80
|
+
* Per-slot contents: atoms (always a single value) live directly in
|
|
81
|
+
* `single`; ngram slots hold -1 there and expand to
|
|
82
|
+
* `pool[start[slot] .. start[slot] + length[slot])`.
|
|
83
|
+
*/
|
|
84
|
+
const single = new Int32Array(slotCount);
|
|
85
|
+
single.fill(-1, dict1AtomCount, dictSize);
|
|
86
|
+
single.fill(-1, dictSize + dict2AtomCount, slotCount);
|
|
87
|
+
const start = new Int32Array(slotCount);
|
|
88
|
+
const length = new Int32Array(slotCount);
|
|
89
|
+
|
|
90
|
+
/**
|
|
91
|
+
* Decode `count` ascending uint16 values from a delta+RLE stream into
|
|
92
|
+
* `single[off..off+count)`.
|
|
93
|
+
*
|
|
94
|
+
* code < 89 → delta = code
|
|
95
|
+
* code == 89 → run-length: next char encodes runLength-2; emit `runLength` consecutive +1 values
|
|
96
|
+
* code == 90, next < 90 → escape: delta = 89 + next * BASE + after-next
|
|
97
|
+
* code == 90, next == 90 → double-escape: extra char for very large deltas
|
|
98
|
+
* @param count
|
|
99
|
+
* @param off
|
|
100
|
+
*/
|
|
101
|
+
function decodeDelta(count: number, off: number): void {
|
|
102
|
+
let previous = 0;
|
|
103
|
+
let slot = off;
|
|
104
|
+
const end = off + count;
|
|
105
|
+
while (slot < end) {
|
|
106
|
+
const code = BASE91_INVERSE[input.charCodeAt(pos++)];
|
|
107
|
+
if (code < 89) {
|
|
108
|
+
previous += code;
|
|
109
|
+
single[slot++] = previous;
|
|
110
|
+
} else if (code === 89) {
|
|
111
|
+
let runLength = BASE91_INVERSE[input.charCodeAt(pos++)] + 2;
|
|
112
|
+
while (runLength--) single[slot++] = ++previous;
|
|
113
|
+
} else {
|
|
114
|
+
const next = BASE91_INVERSE[input.charCodeAt(pos++)];
|
|
115
|
+
previous +=
|
|
116
|
+
89 +
|
|
117
|
+
// eslint-disable-next-line unicorn/prefer-minimal-ternary -- branches read a different number of side-effecting input bytes
|
|
118
|
+
(next < 90
|
|
119
|
+
? next * base + BASE91_INVERSE[input.charCodeAt(pos++)]
|
|
120
|
+
: BASE91_INVERSE[input.charCodeAt(pos++)] * 8281 +
|
|
121
|
+
BASE91_INVERSE[input.charCodeAt(pos++)] * base +
|
|
122
|
+
BASE91_INVERSE[input.charCodeAt(pos++)]);
|
|
123
|
+
single[slot++] = previous;
|
|
124
|
+
}
|
|
125
|
+
}
|
|
126
|
+
}
|
|
127
|
+
|
|
128
|
+
// Streams 1 & 2: atoms decoded into their slot ranges.
|
|
129
|
+
decodeDelta(dict1AtomCount, 0);
|
|
130
|
+
decodeDelta(dict2AtomCount, dictSize);
|
|
131
|
+
|
|
132
|
+
/*
|
|
133
|
+
* Streams 3 & 4 are read in two passes: first collect every ngram's two
|
|
134
|
+
* references and derive its expanded length (each ref resolves to an earlier
|
|
135
|
+
* slot, so lengths are already known), which sizes the shared pool.
|
|
136
|
+
* Pool ranges are handed out in decode order, so the second pass fills
|
|
137
|
+
* the pool contiguously with a single write cursor.
|
|
138
|
+
*/
|
|
139
|
+
const references = new Int32Array(ngramCount * 2);
|
|
140
|
+
let poolSize = 0;
|
|
141
|
+
let ngramIndex = 0;
|
|
142
|
+
|
|
143
|
+
/**
|
|
144
|
+
* Read `count` ngram entries (each = 2 slot-code references) for the slots
|
|
145
|
+
* starting at `startSlot`, recording references and assigning pool ranges.
|
|
146
|
+
* @param count
|
|
147
|
+
* @param startSlot
|
|
148
|
+
*/
|
|
149
|
+
function readNgramReferences(count: number, startSlot: number): void {
|
|
150
|
+
for (let index = 0; index < count; index++) {
|
|
151
|
+
const slot = startSlot + index;
|
|
152
|
+
const a = readSlotCode();
|
|
153
|
+
const b = readSlotCode();
|
|
154
|
+
references[ngramIndex * 2] = a;
|
|
155
|
+
references[ngramIndex * 2 + 1] = b;
|
|
156
|
+
ngramIndex += 1;
|
|
157
|
+
start[slot] = poolSize;
|
|
158
|
+
const entryLength =
|
|
159
|
+
(single[a] < 0 ? length[a] : 1) +
|
|
160
|
+
(single[b] < 0 ? length[b] : 1);
|
|
161
|
+
length[slot] = entryLength;
|
|
162
|
+
poolSize += entryLength;
|
|
163
|
+
}
|
|
164
|
+
}
|
|
165
|
+
readNgramReferences(
|
|
166
|
+
ngramCount - dictSize + dict1AtomCount,
|
|
167
|
+
dictSize + dict2AtomCount,
|
|
168
|
+
);
|
|
169
|
+
readNgramReferences(dictSize - dict1AtomCount, dict1AtomCount);
|
|
170
|
+
|
|
171
|
+
// Second pass: concatenate each ngram's two halves into the pool.
|
|
172
|
+
const pool = new Uint16Array(poolSize);
|
|
173
|
+
let write = 0;
|
|
174
|
+
for (let index = 0; index < ngramIndex; index++) {
|
|
175
|
+
for (let half = 0; half < 2; half++) {
|
|
176
|
+
const source = references[index * 2 + half];
|
|
177
|
+
const value = single[source];
|
|
178
|
+
if (value < 0) {
|
|
179
|
+
let read = start[source];
|
|
180
|
+
const readEnd = read + length[source];
|
|
181
|
+
while (read < readEnd) pool[write++] = pool[read++];
|
|
182
|
+
} else {
|
|
183
|
+
pool[write++] = value;
|
|
184
|
+
}
|
|
185
|
+
}
|
|
186
|
+
}
|
|
187
|
+
|
|
188
|
+
// Stream 5: data. Each code expands to its slot's stored values.
|
|
189
|
+
const out = new Uint16Array(resultLength);
|
|
190
|
+
let outIndex = 0;
|
|
191
|
+
while (pos < inputLength) {
|
|
192
|
+
let slot = BASE91_INVERSE[input.charCodeAt(pos++)];
|
|
193
|
+
if (slot >= dictSize) {
|
|
194
|
+
slot =
|
|
195
|
+
slot * base -
|
|
196
|
+
twoCharBias +
|
|
197
|
+
BASE91_INVERSE[input.charCodeAt(pos++)];
|
|
198
|
+
}
|
|
199
|
+
const value = single[slot];
|
|
200
|
+
if (value < 0) {
|
|
201
|
+
let read = start[slot];
|
|
202
|
+
const readEnd = read + length[slot];
|
|
203
|
+
while (read < readEnd) out[outIndex++] = pool[read++];
|
|
204
|
+
} else {
|
|
205
|
+
out[outIndex++] = value;
|
|
206
|
+
}
|
|
207
|
+
}
|
|
208
|
+
return out;
|
|
209
|
+
}
|
|
@@ -1,31 +1,37 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "node-html-parser",
|
|
3
|
-
"version": "
|
|
3
|
+
"version": "9.0.4",
|
|
4
4
|
"description": "A very fast HTML parser, generating a simplified DOM, with basic element query support.",
|
|
5
|
-
"main": "dist/index.
|
|
5
|
+
"main": "dist/index.cjs",
|
|
6
6
|
"types": "dist/index.d.ts",
|
|
7
|
+
"module": "dist/index.mjs",
|
|
8
|
+
"exports": {
|
|
9
|
+
".": {
|
|
10
|
+
"types": "./dist/index.d.ts",
|
|
11
|
+
"import": "./dist/index.mjs",
|
|
12
|
+
"require": "./dist/index.cjs"
|
|
13
|
+
}
|
|
14
|
+
},
|
|
7
15
|
"scripts": {
|
|
8
|
-
"compile": "tsc",
|
|
9
|
-
"build": "
|
|
10
|
-
"
|
|
11
|
-
"watch": "npx tsc -m commonjs --watch --preserveWatchOutput",
|
|
12
|
-
"compile:amd": "tsc -t es5 -m amd -d false --outFile ./dist/main.js",
|
|
16
|
+
"compile": "tsdown && tsc --emitDeclarationOnly --outDir dist",
|
|
17
|
+
"build": "bun run compile",
|
|
18
|
+
"watch": "tsdown --watch",
|
|
13
19
|
"lint": "eslint ./src/*.ts ./src/**/*.ts",
|
|
14
20
|
"---------------": "",
|
|
15
|
-
"pretest": "cd ./test
|
|
16
|
-
"test": "
|
|
17
|
-
"test:src": "cross-env TEST_TARGET=src
|
|
18
|
-
"test:dist": "cross-env TEST_TARGET=dist
|
|
21
|
+
"pretest": "cd ./test && bun install && cd ..",
|
|
22
|
+
"test": "bun run test:dist",
|
|
23
|
+
"test:src": "cross-env TEST_TARGET=src mocha --recursive \"./test/tests\" --require ts-node/register --require blanket --require should --require spec",
|
|
24
|
+
"test:dist": "cross-env TEST_TARGET=dist mocha --recursive \"./test/tests\" --require blanket --require should --require spec",
|
|
19
25
|
"benchmark": "node ./test/benchmark/compare.mjs",
|
|
20
26
|
"--------------- ": "",
|
|
21
27
|
"clean": "npx rimraf ./dist/",
|
|
22
|
-
"clean:global": "
|
|
23
|
-
"reset": "
|
|
28
|
+
"clean:global": "bun run clean && rm -rf test/node_modules node_modules",
|
|
29
|
+
"reset": "bun run clean:global && bun install && bun run build",
|
|
24
30
|
"--------------- ": "",
|
|
25
|
-
"test:target": "mocha --recursive \"./test/tests\"",
|
|
26
|
-
"test:ci": "cd test &&
|
|
27
|
-
"posttest": "
|
|
28
|
-
"prepare": "
|
|
31
|
+
"test:target": "cross-env TEST_TARGET=dist mocha --recursive \"./test/tests\" --require blanket --require should --require spec",
|
|
32
|
+
"test:ci": "cd test && bun install && cd .. && cross-env TEST_TARGET=dist bun run test:target",
|
|
33
|
+
"posttest": "bun run benchmark",
|
|
34
|
+
"prepare": "bun run compile",
|
|
29
35
|
"release": "standard-version && git push --follow-tags origin main"
|
|
30
36
|
},
|
|
31
37
|
"keywords": [
|
|
@@ -51,35 +57,32 @@
|
|
|
51
57
|
},
|
|
52
58
|
"dependencies": {
|
|
53
59
|
"css-select": "^5.1.0",
|
|
54
|
-
"
|
|
60
|
+
"entities": "^8.0.0"
|
|
55
61
|
},
|
|
56
62
|
"devDependencies": {
|
|
57
|
-
"@types/entities": "latest",
|
|
58
|
-
"@types/he": "latest",
|
|
59
63
|
"@types/node": "latest",
|
|
60
64
|
"@typescript-eslint/eslint-plugin": "latest",
|
|
61
|
-
"@typescript-eslint/eslint-plugin-tslint": "latest",
|
|
62
65
|
"@typescript-eslint/parser": "latest",
|
|
63
66
|
"blanket": "latest",
|
|
64
67
|
"boolbase": "^1.0.0",
|
|
65
|
-
"cheerio": "^1.
|
|
68
|
+
"cheerio": "^1.2.0",
|
|
66
69
|
"cross-env": "^7.0.3",
|
|
67
70
|
"eslint": "^8.23.1",
|
|
68
71
|
"eslint-config-prettier": "latest",
|
|
69
72
|
"eslint-plugin-import": "latest",
|
|
70
73
|
"high5": "^1.0.0",
|
|
71
|
-
"html-dom-parser": "^
|
|
74
|
+
"html-dom-parser": "^8.0.0",
|
|
72
75
|
"html-parser": "^0.11.0",
|
|
73
|
-
"html5parser": "^
|
|
74
|
-
"htmljs-parser": "^5.
|
|
76
|
+
"html5parser": "^3.0.0",
|
|
77
|
+
"htmljs-parser": "^5.10.2",
|
|
75
78
|
"htmlparser": "^1.7.7",
|
|
76
79
|
"htmlparser-benchmark": "^1.1.3",
|
|
77
80
|
"htmlparser2": "^8.0.1",
|
|
78
|
-
"mocha": "
|
|
81
|
+
"mocha": "^11.7.6",
|
|
79
82
|
"mocha-each": "^2.0.1",
|
|
80
83
|
"neutron-html5parser": "^0.2.0",
|
|
81
84
|
"np": "latest",
|
|
82
|
-
"parse5": "^
|
|
85
|
+
"parse5": "^8.0.1",
|
|
83
86
|
"rimraf": "^3.0.2",
|
|
84
87
|
"saxes": "^6.0.0",
|
|
85
88
|
"should": "latest",
|
|
@@ -87,8 +90,9 @@
|
|
|
87
90
|
"standard-version": "^9.5.0",
|
|
88
91
|
"travis-cov": "latest",
|
|
89
92
|
"ts-node": "^10.9.1",
|
|
90
|
-
"
|
|
91
|
-
"
|
|
93
|
+
"tsdown": "^0.22.3",
|
|
94
|
+
"@typescript/native": "npm:typescript@^7.0.2",
|
|
95
|
+
"typescript": "npm:@typescript/typescript6@^6.0.2"
|
|
92
96
|
},
|
|
93
97
|
"config": {
|
|
94
98
|
"blanket": {
|
|
@@ -112,5 +116,6 @@
|
|
|
112
116
|
"url": "https://github.com/taoqf/node-fast-html-parser/issues"
|
|
113
117
|
},
|
|
114
118
|
"homepage": "https://github.com/taoqf/node-fast-html-parser",
|
|
115
|
-
"sideEffects": false
|
|
119
|
+
"sideEffects": false,
|
|
120
|
+
"packageManager": "bun@1.3.14"
|
|
116
121
|
}
|
package/package.json
CHANGED
|
@@ -63,7 +63,7 @@
|
|
|
63
63
|
"@stylistic/eslint-plugin": "^2",
|
|
64
64
|
"@types/aws-lambda": "^8.10.164",
|
|
65
65
|
"@types/jest": "^30.0.0",
|
|
66
|
-
"@types/jsdom": "^
|
|
66
|
+
"@types/jsdom": "^30.0.0",
|
|
67
67
|
"@types/node": "^24",
|
|
68
68
|
"@typescript-eslint/eslint-plugin": "^8",
|
|
69
69
|
"@typescript-eslint/parser": "^8",
|
|
@@ -95,7 +95,7 @@
|
|
|
95
95
|
"@mavogel/mvc-projen": "^0.0.50",
|
|
96
96
|
"cdk-nag": "^3.0.1",
|
|
97
97
|
"constructs": "^10.5.1",
|
|
98
|
-
"node-html-parser": "^
|
|
98
|
+
"node-html-parser": "^9.0.0"
|
|
99
99
|
},
|
|
100
100
|
"bundledDependencies": [
|
|
101
101
|
"node-html-parser"
|
|
@@ -122,7 +122,7 @@
|
|
|
122
122
|
"publishConfig": {
|
|
123
123
|
"access": "public"
|
|
124
124
|
},
|
|
125
|
-
"version": "0.0.
|
|
125
|
+
"version": "0.0.124",
|
|
126
126
|
"jest": {
|
|
127
127
|
"coverageProvider": "v8",
|
|
128
128
|
"testMatch": [
|
|
@@ -1,20 +0,0 @@
|
|
|
1
|
-
Copyright Mathias Bynens <https://mathiasbynens.be/>
|
|
2
|
-
|
|
3
|
-
Permission is hereby granted, free of charge, to any person obtaining
|
|
4
|
-
a copy of this software and associated documentation files (the
|
|
5
|
-
"Software"), to deal in the Software without restriction, including
|
|
6
|
-
without limitation the rights to use, copy, modify, merge, publish,
|
|
7
|
-
distribute, sublicense, and/or sell copies of the Software, and to
|
|
8
|
-
permit persons to whom the Software is furnished to do so, subject to
|
|
9
|
-
the following conditions:
|
|
10
|
-
|
|
11
|
-
The above copyright notice and this permission notice shall be
|
|
12
|
-
included in all copies or substantial portions of the Software.
|
|
13
|
-
|
|
14
|
-
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND,
|
|
15
|
-
EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF
|
|
16
|
-
MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND
|
|
17
|
-
NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT HOLDERS BE
|
|
18
|
-
LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN AN ACTION
|
|
19
|
-
OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION
|
|
20
|
-
WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
|