@telorun/cel 0.0.0-stage → 0.107.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +17 -0
- package/README.md +281 -2
- package/dist/activation.d.ts +27 -0
- package/dist/activation.d.ts.map +1 -0
- package/dist/activation.js +24 -0
- package/dist/backend-runtime.d.ts +146 -0
- package/dist/backend-runtime.d.ts.map +1 -0
- package/dist/backend-runtime.js +322 -0
- package/dist/bounded-cache.d.ts +21 -0
- package/dist/bounded-cache.d.ts.map +1 -0
- package/dist/bounded-cache.js +42 -0
- package/dist/catalog-runtime.d.ts +59 -0
- package/dist/catalog-runtime.d.ts.map +1 -0
- package/dist/catalog-runtime.js +785 -0
- package/dist/cel-expression.d.ts +32 -0
- package/dist/cel-expression.d.ts.map +1 -0
- package/dist/cel-expression.js +29 -0
- package/dist/cel-map-value.d.ts +34 -0
- package/dist/cel-map-value.d.ts.map +1 -0
- package/dist/cel-map-value.js +74 -0
- package/dist/cel-program.d.ts +44 -0
- package/dist/cel-program.d.ts.map +1 -0
- package/dist/cel-program.js +72 -0
- package/dist/cel-type.d.ts +131 -0
- package/dist/cel-type.d.ts.map +1 -0
- package/dist/cel-type.js +293 -0
- package/dist/cel-value.d.ts +159 -0
- package/dist/cel-value.d.ts.map +1 -0
- package/dist/cel-value.js +225 -0
- package/dist/check-diagnostic.d.ts +53 -0
- package/dist/check-diagnostic.d.ts.map +1 -0
- package/dist/check-diagnostic.js +66 -0
- package/dist/checker.d.ts +77 -0
- package/dist/checker.d.ts.map +1 -0
- package/dist/checker.js +721 -0
- package/dist/closure-backend.d.ts +21 -0
- package/dist/closure-backend.d.ts.map +1 -0
- package/dist/closure-backend.js +436 -0
- package/dist/comprehension-bindings.d.ts +33 -0
- package/dist/comprehension-bindings.d.ts.map +1 -0
- package/dist/comprehension-bindings.js +50 -0
- package/dist/comprehension-runtime.d.ts +44 -0
- package/dist/comprehension-runtime.d.ts.map +1 -0
- package/dist/comprehension-runtime.js +137 -0
- package/dist/declared-chain.d.ts +35 -0
- package/dist/declared-chain.d.ts.map +1 -0
- package/dist/declared-chain.js +36 -0
- package/dist/duration-value.d.ts +59 -0
- package/dist/duration-value.d.ts.map +1 -0
- package/dist/duration-value.js +135 -0
- package/dist/emitted-module.d.ts +207 -0
- package/dist/emitted-module.d.ts.map +1 -0
- package/dist/emitted-module.js +359 -0
- package/dist/engine-version.d.ts +3 -0
- package/dist/engine-version.d.ts.map +1 -0
- package/dist/engine-version.js +8 -0
- package/dist/environment-digest.d.ts +44 -0
- package/dist/environment-digest.d.ts.map +1 -0
- package/dist/environment-digest.js +98 -0
- package/dist/environment.d.ts +283 -0
- package/dist/environment.d.ts.map +1 -0
- package/dist/environment.js +459 -0
- package/dist/function-catalog.d.ts +66 -0
- package/dist/function-catalog.d.ts.map +1 -0
- package/dist/function-catalog.js +77 -0
- package/dist/function-registry.d.ts +78 -0
- package/dist/function-registry.d.ts.map +1 -0
- package/dist/function-registry.js +189 -0
- package/dist/index.d.ts +91 -0
- package/dist/index.d.ts.map +1 -0
- package/dist/index.js +59 -0
- package/dist/integer-arithmetic.d.ts +27 -0
- package/dist/integer-arithmetic.d.ts.map +1 -0
- package/dist/integer-arithmetic.js +58 -0
- package/dist/js-emitter.d.ts +132 -0
- package/dist/js-emitter.d.ts.map +1 -0
- package/dist/js-emitter.js +562 -0
- package/dist/json-schema-type.d.ts +182 -0
- package/dist/json-schema-type.d.ts.map +1 -0
- package/dist/json-schema-type.js +487 -0
- package/dist/json-text-scan.d.ts +28 -0
- package/dist/json-text-scan.d.ts.map +1 -0
- package/dist/json-text-scan.js +159 -0
- package/dist/lexer.d.ts +103 -0
- package/dist/lexer.d.ts.map +1 -0
- package/dist/lexer.js +458 -0
- package/dist/macro-check.d.ts +33 -0
- package/dist/macro-check.d.ts.map +1 -0
- package/dist/macro-check.js +162 -0
- package/dist/macro-shape.d.ts +24 -0
- package/dist/macro-shape.d.ts.map +1 -0
- package/dist/macro-shape.js +55 -0
- package/dist/member-read.d.ts +52 -0
- package/dist/member-read.d.ts.map +1 -0
- package/dist/member-read.js +125 -0
- package/dist/namespace-resolution.d.ts +35 -0
- package/dist/namespace-resolution.d.ts.map +1 -0
- package/dist/namespace-resolution.js +160 -0
- package/dist/nominal-type.d.ts +63 -0
- package/dist/nominal-type.d.ts.map +1 -0
- package/dist/nominal-type.js +98 -0
- package/dist/nullable-access.d.ts +38 -0
- package/dist/nullable-access.d.ts.map +1 -0
- package/dist/nullable-access.js +93 -0
- package/dist/parse-limits.d.ts +26 -0
- package/dist/parse-limits.d.ts.map +1 -0
- package/dist/parse-limits.js +21 -0
- package/dist/parser.d.ts +48 -0
- package/dist/parser.d.ts.map +1 -0
- package/dist/parser.js +503 -0
- package/dist/qualified-calls.d.ts +22 -0
- package/dist/qualified-calls.d.ts.map +1 -0
- package/dist/qualified-calls.js +27 -0
- package/dist/regular-expression.d.ts +48 -0
- package/dist/regular-expression.d.ts.map +1 -0
- package/dist/regular-expression.js +77 -0
- package/dist/reserved-words.d.ts +52 -0
- package/dist/reserved-words.d.ts.map +1 -0
- package/dist/reserved-words.js +77 -0
- package/dist/resolved-call.d.ts +33 -0
- package/dist/resolved-call.d.ts.map +1 -0
- package/dist/resolved-call.js +15 -0
- package/dist/root-references.d.ts +23 -0
- package/dist/root-references.d.ts.map +1 -0
- package/dist/root-references.js +120 -0
- package/dist/runtime-library.d.ts +56 -0
- package/dist/runtime-library.d.ts.map +1 -0
- package/dist/runtime-library.js +545 -0
- package/dist/serializer.d.ts +24 -0
- package/dist/serializer.d.ts.map +1 -0
- package/dist/serializer.js +240 -0
- package/dist/sha256.d.ts +20 -0
- package/dist/sha256.d.ts.map +1 -0
- package/dist/sha256.js +103 -0
- package/dist/signature.d.ts +72 -0
- package/dist/signature.d.ts.map +1 -0
- package/dist/signature.js +61 -0
- package/dist/signatures/function-catalog.json +788 -0
- package/dist/signatures/standard-library.json +229 -0
- package/dist/standard-library.d.ts +41 -0
- package/dist/standard-library.d.ts.map +1 -0
- package/dist/standard-library.js +85 -0
- package/dist/syntax-diagnostic.d.ts +61 -0
- package/dist/syntax-diagnostic.d.ts.map +1 -0
- package/dist/syntax-diagnostic.js +28 -0
- package/dist/syntax-tree.d.ts +160 -0
- package/dist/syntax-tree.d.ts.map +1 -0
- package/dist/syntax-tree.js +58 -0
- package/dist/timestamp-value.d.ts +53 -0
- package/dist/timestamp-value.d.ts.map +1 -0
- package/dist/timestamp-value.js +223 -0
- package/dist/tree-equality.d.ts +15 -0
- package/dist/tree-equality.d.ts.map +1 -0
- package/dist/tree-equality.js +105 -0
- package/dist/type-expression.d.ts +26 -0
- package/dist/type-expression.d.ts.map +1 -0
- package/dist/type-expression.js +134 -0
- package/dist/value-equality.d.ts +38 -0
- package/dist/value-equality.d.ts.map +1 -0
- package/dist/value-equality.js +196 -0
- package/dist/value-text.d.ts +25 -0
- package/dist/value-text.d.ts.map +1 -0
- package/dist/value-text.js +44 -0
- package/dist/zoned-calendar.d.ts +61 -0
- package/dist/zoned-calendar.d.ts.map +1 -0
- package/dist/zoned-calendar.js +143 -0
- package/package.json +56 -3
- package/src/activation.ts +32 -0
- package/src/backend-runtime.ts +454 -0
- package/src/bounded-cache.ts +45 -0
- package/src/catalog-runtime.ts +921 -0
- package/src/cel-expression.ts +53 -0
- package/src/cel-map-value.ts +86 -0
- package/src/cel-program.ts +103 -0
- package/src/cel-type.ts +359 -0
- package/src/cel-value.ts +353 -0
- package/src/check-diagnostic.ts +102 -0
- package/src/checker.ts +932 -0
- package/src/closure-backend.ts +502 -0
- package/src/comprehension-bindings.ts +66 -0
- package/src/comprehension-runtime.ts +157 -0
- package/src/declared-chain.ts +45 -0
- package/src/duration-value.ts +145 -0
- package/src/emitted-module.ts +494 -0
- package/src/engine-version.ts +9 -0
- package/src/environment-digest.ts +111 -0
- package/src/environment.ts +740 -0
- package/src/function-catalog.ts +140 -0
- package/src/function-registry.ts +229 -0
- package/src/index.ts +386 -0
- package/src/integer-arithmetic.ts +64 -0
- package/src/js-emitter.ts +713 -0
- package/src/json-schema-type.ts +664 -0
- package/src/json-text-scan.ts +163 -0
- package/src/lexer.ts +562 -0
- package/src/macro-check.ts +191 -0
- package/src/macro-shape.ts +66 -0
- package/src/member-read.ts +126 -0
- package/src/namespace-resolution.ts +167 -0
- package/src/nominal-type.ts +149 -0
- package/src/nullable-access.ts +95 -0
- package/src/parse-limits.ts +36 -0
- package/src/parser.ts +554 -0
- package/src/qualified-calls.ts +39 -0
- package/src/regular-expression.ts +101 -0
- package/src/reserved-words.ts +94 -0
- package/src/resolved-call.ts +34 -0
- package/src/root-references.ts +126 -0
- package/src/runtime-library.ts +615 -0
- package/src/serializer.ts +246 -0
- package/src/sha256.ts +112 -0
- package/src/signature.ts +127 -0
- package/src/signatures/function-catalog.json +788 -0
- package/src/signatures/standard-library.json +235 -0
- package/src/standard-library.ts +149 -0
- package/src/syntax-diagnostic.ts +72 -0
- package/src/syntax-tree.ts +229 -0
- package/src/timestamp-value.ts +294 -0
- package/src/tree-equality.ts +130 -0
- package/src/type-expression.ts +160 -0
- package/src/value-equality.ts +201 -0
- package/src/value-text.ts +45 -0
- package/src/zoned-calendar.ts +178 -0
package/src/lexer.ts
ADDED
|
@@ -0,0 +1,562 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* CEL's tokens.
|
|
3
|
+
*
|
|
4
|
+
* The lexer decodes literals as it reads them — a string token carries its text
|
|
5
|
+
* with every escape resolved, a bytes token its bytes, an integer its magnitude as
|
|
6
|
+
* a `bigint` — so nothing downstream re-reads source text to learn a value.
|
|
7
|
+
*
|
|
8
|
+
* It stops at the first thing it cannot read, reporting one diagnostic and ending
|
|
9
|
+
* the token stream there. The parser then builds whatever the tokens before it
|
|
10
|
+
* support, which is what leaves a half-typed expression usable.
|
|
11
|
+
*
|
|
12
|
+
* Three readings are deliberate, each pinned by the conformance vectors:
|
|
13
|
+
*
|
|
14
|
+
* - **Only a single-letter prefix** introduces a string: `r`, `R`, `b`, `B`. `br'x'`
|
|
15
|
+
* is the identifier `br` followed by a string, which no expression admits.
|
|
16
|
+
* - **A raw string still lets a backslash take the next character with it**, keeping
|
|
17
|
+
* both as written, so `r'\''` is a backslash and a quote rather than an unclosed
|
|
18
|
+
* string.
|
|
19
|
+
* - **A bytes literal holds the UTF-8 of its text.** `b'ÿ'` is `0xC3 0xBF`, two bytes,
|
|
20
|
+
* and an escape (`\xff`, `\303`) is the one byte it names. cel-spec's own answer, and
|
|
21
|
+
* the only reading under which `b'ÿ' == b'\303\277'` — a row — holds.
|
|
22
|
+
*
|
|
23
|
+
* A number never begins with `.`: `.5` is a dot and a number, and no expression
|
|
24
|
+
* admits that either.
|
|
25
|
+
*/
|
|
26
|
+
|
|
27
|
+
import { wordReading } from "./reserved-words.js";
|
|
28
|
+
import { FirstSyntaxDiagnostic } from "./syntax-diagnostic.js";
|
|
29
|
+
|
|
30
|
+
export const MAX_INT = 9223372036854775807n;
|
|
31
|
+
export const MIN_INT = -9223372036854775808n;
|
|
32
|
+
export const MAX_UINT = 18446744073709551615n;
|
|
33
|
+
|
|
34
|
+
export type TokenType =
|
|
35
|
+
/** A name, and a word read as a literal (`true`, `false`, `null`). */
|
|
36
|
+
| "ident"
|
|
37
|
+
/** A member name written between backticks, whatever it spells. */
|
|
38
|
+
| "quotedIdent"
|
|
39
|
+
/** A reserved word with no other reading: refused wherever a name is read. */
|
|
40
|
+
| "reserved"
|
|
41
|
+
/** An operator written as a word (`in`). */
|
|
42
|
+
| "keyword"
|
|
43
|
+
| "int"
|
|
44
|
+
| "uint"
|
|
45
|
+
| "double"
|
|
46
|
+
| "string"
|
|
47
|
+
| "bytes"
|
|
48
|
+
| "punct"
|
|
49
|
+
| "eof";
|
|
50
|
+
|
|
51
|
+
export interface Token {
|
|
52
|
+
readonly type: TokenType;
|
|
53
|
+
readonly start: number;
|
|
54
|
+
readonly end: number;
|
|
55
|
+
/** The lexeme, for `punct`; the name, for `ident` and `reserved`. */
|
|
56
|
+
readonly text: string;
|
|
57
|
+
readonly int?: bigint;
|
|
58
|
+
readonly double?: number;
|
|
59
|
+
readonly string?: string;
|
|
60
|
+
readonly bytes?: Uint8Array;
|
|
61
|
+
/**
|
|
62
|
+
* An integer magnitude of exactly 2^63 — legal only directly under a unary
|
|
63
|
+
* minus, which is how the int64 minimum is written.
|
|
64
|
+
*/
|
|
65
|
+
readonly atIntBoundary?: boolean;
|
|
66
|
+
}
|
|
67
|
+
|
|
68
|
+
const PUNCTUATION = new Set(["(", ")", "[", "]", "{", "}", ".", ",", ":", "?", "!", "+", "-", "*", "/", "%"]);
|
|
69
|
+
|
|
70
|
+
const SIMPLE_ESCAPES = new Map<string, number>([
|
|
71
|
+
["\\", 0x5c],
|
|
72
|
+
["'", 0x27],
|
|
73
|
+
['"', 0x22],
|
|
74
|
+
["`", 0x60],
|
|
75
|
+
["?", 0x3f],
|
|
76
|
+
["a", 0x07],
|
|
77
|
+
["b", 0x08],
|
|
78
|
+
["f", 0x0c],
|
|
79
|
+
["n", 0x0a],
|
|
80
|
+
["r", 0x0d],
|
|
81
|
+
["t", 0x09],
|
|
82
|
+
["v", 0x0b],
|
|
83
|
+
]);
|
|
84
|
+
|
|
85
|
+
function isDigit(ch: string): boolean {
|
|
86
|
+
return ch >= "0" && ch <= "9";
|
|
87
|
+
}
|
|
88
|
+
|
|
89
|
+
function isHexDigit(ch: string): boolean {
|
|
90
|
+
return isDigit(ch) || (ch >= "a" && ch <= "f") || (ch >= "A" && ch <= "F");
|
|
91
|
+
}
|
|
92
|
+
|
|
93
|
+
function isOctalDigit(ch: string): boolean {
|
|
94
|
+
return ch >= "0" && ch <= "7";
|
|
95
|
+
}
|
|
96
|
+
|
|
97
|
+
function isIdentStart(ch: string): boolean {
|
|
98
|
+
return ch === "_" || (ch >= "a" && ch <= "z") || (ch >= "A" && ch <= "Z");
|
|
99
|
+
}
|
|
100
|
+
|
|
101
|
+
function isIdentPart(ch: string): boolean {
|
|
102
|
+
return isIdentStart(ch) || isDigit(ch);
|
|
103
|
+
}
|
|
104
|
+
|
|
105
|
+
/**
|
|
106
|
+
* What a word before a quote makes of the literal.
|
|
107
|
+
*
|
|
108
|
+
* cel-spec nests the two markers — `BYTES_LIT: [bB] STRING_LIT` over
|
|
109
|
+
* `STRING_LIT: [rR]? (…)` — so the bytes marker comes first and `rb'…'` is not a
|
|
110
|
+
* literal at all: it is the name `rb` beside a string, which no expression admits.
|
|
111
|
+
*/
|
|
112
|
+
function stringPrefix(text: string): { raw: boolean; bytes: boolean } | undefined {
|
|
113
|
+
const lower = text.toLowerCase();
|
|
114
|
+
if (lower === "r") return { raw: true, bytes: false };
|
|
115
|
+
if (lower === "b") return { raw: false, bytes: true };
|
|
116
|
+
if (lower === "br") return { raw: true, bytes: true };
|
|
117
|
+
return undefined;
|
|
118
|
+
}
|
|
119
|
+
|
|
120
|
+
function isSpace(ch: string): boolean {
|
|
121
|
+
return ch === " " || ch === "\t" || ch === "\n" || ch === "\r" || ch === "\f" || ch === "\v";
|
|
122
|
+
}
|
|
123
|
+
|
|
124
|
+
/** A decoded string or bytes literal, or nothing when it could not be read. */
|
|
125
|
+
interface LiteralText {
|
|
126
|
+
readonly units: number[];
|
|
127
|
+
readonly end: number;
|
|
128
|
+
}
|
|
129
|
+
|
|
130
|
+
export class Lexer {
|
|
131
|
+
private at = 0;
|
|
132
|
+
|
|
133
|
+
constructor(
|
|
134
|
+
private readonly source: string,
|
|
135
|
+
private readonly diagnostics: FirstSyntaxDiagnostic,
|
|
136
|
+
) {}
|
|
137
|
+
|
|
138
|
+
/** Every token up to the end of the source, or up to the first unreadable text. */
|
|
139
|
+
tokenize(): Token[] {
|
|
140
|
+
const tokens: Token[] = [];
|
|
141
|
+
for (;;) {
|
|
142
|
+
this.skipIgnored();
|
|
143
|
+
if (this.at >= this.source.length) {
|
|
144
|
+
tokens.push(this.eof());
|
|
145
|
+
return tokens;
|
|
146
|
+
}
|
|
147
|
+
const token = this.next();
|
|
148
|
+
if (!token) {
|
|
149
|
+
tokens.push(this.eof());
|
|
150
|
+
return tokens;
|
|
151
|
+
}
|
|
152
|
+
tokens.push(token);
|
|
153
|
+
}
|
|
154
|
+
}
|
|
155
|
+
|
|
156
|
+
private eof(): Token {
|
|
157
|
+
return { type: "eof", start: this.at, end: this.at, text: "" };
|
|
158
|
+
}
|
|
159
|
+
|
|
160
|
+
private skipIgnored(): void {
|
|
161
|
+
while (this.at < this.source.length) {
|
|
162
|
+
const ch = this.source[this.at]!;
|
|
163
|
+
if (isSpace(ch)) {
|
|
164
|
+
this.at += 1;
|
|
165
|
+
continue;
|
|
166
|
+
}
|
|
167
|
+
if (ch === "/" && this.source[this.at + 1] === "/") {
|
|
168
|
+
while (this.at < this.source.length && this.source[this.at] !== "\n") this.at += 1;
|
|
169
|
+
continue;
|
|
170
|
+
}
|
|
171
|
+
return;
|
|
172
|
+
}
|
|
173
|
+
}
|
|
174
|
+
|
|
175
|
+
private next(): Token | undefined {
|
|
176
|
+
const start = this.at;
|
|
177
|
+
const ch = this.source[start]!;
|
|
178
|
+
if (ch === "'" || ch === '"') return this.readQuoted(start, start, false, false);
|
|
179
|
+
if (isIdentStart(ch)) return this.readWord(start);
|
|
180
|
+
if (isDigit(ch)) return this.readNumber(start);
|
|
181
|
+
// A double may begin with its point: cel-spec's FLOAT_LIT is `DIGIT* . DIGIT+`.
|
|
182
|
+
if (ch === "." && isDigit(this.source[start + 1] ?? "")) return this.readNumber(start);
|
|
183
|
+
if (ch === "`") return this.readQuotedIdentifier(start);
|
|
184
|
+
return this.readPunctuation(start, ch);
|
|
185
|
+
}
|
|
186
|
+
|
|
187
|
+
private readPunctuation(start: number, ch: string): Token | undefined {
|
|
188
|
+
const pair = this.source.slice(start, start + 2);
|
|
189
|
+
if (pair === "==" || pair === "!=" || pair === "<=" || pair === ">=" || pair === "&&" || pair === "||") {
|
|
190
|
+
this.at = start + 2;
|
|
191
|
+
return { type: "punct", start, end: this.at, text: pair };
|
|
192
|
+
}
|
|
193
|
+
if (ch === "<" || ch === ">" || PUNCTUATION.has(ch)) {
|
|
194
|
+
this.at = start + 1;
|
|
195
|
+
return { type: "punct", start, end: this.at, text: ch };
|
|
196
|
+
}
|
|
197
|
+
this.diagnostics.report(
|
|
198
|
+
"unexpected_character",
|
|
199
|
+
`unexpected character ${JSON.stringify(ch)}`,
|
|
200
|
+
start,
|
|
201
|
+
start + 1,
|
|
202
|
+
);
|
|
203
|
+
this.at = start;
|
|
204
|
+
return undefined;
|
|
205
|
+
}
|
|
206
|
+
|
|
207
|
+
/**
|
|
208
|
+
* A member name between backticks. cel-spec's `ESCAPED_IDENTIFIER` takes no escapes
|
|
209
|
+
* and cannot hold a backtick, so the text between them is the name as written.
|
|
210
|
+
*/
|
|
211
|
+
private readQuotedIdentifier(start: number): Token | undefined {
|
|
212
|
+
const close = this.source.indexOf("`", start + 1);
|
|
213
|
+
if (close === -1 || this.source.slice(start + 1, close).includes("\n")) {
|
|
214
|
+
this.diagnostics.report(
|
|
215
|
+
"unterminated_string",
|
|
216
|
+
"a quoted member name ends at its closing backtick",
|
|
217
|
+
start,
|
|
218
|
+
close === -1 ? this.source.length : close,
|
|
219
|
+
);
|
|
220
|
+
return undefined;
|
|
221
|
+
}
|
|
222
|
+
this.at = close + 1;
|
|
223
|
+
return { type: "quotedIdent", start, end: this.at, text: this.source.slice(start + 1, close) };
|
|
224
|
+
}
|
|
225
|
+
|
|
226
|
+
/** An identifier, a literal word, or the prefix of a string literal. */
|
|
227
|
+
private readWord(start: number): Token | undefined {
|
|
228
|
+
let at = start;
|
|
229
|
+
while (at < this.source.length && isIdentPart(this.source[at]!)) at += 1;
|
|
230
|
+
const text = this.source.slice(start, at);
|
|
231
|
+
const prefix = stringPrefix(text);
|
|
232
|
+
if (prefix && (this.source[at] === "'" || this.source[at] === '"')) {
|
|
233
|
+
return this.readQuoted(start, at, prefix.raw, prefix.bytes);
|
|
234
|
+
}
|
|
235
|
+
this.at = at;
|
|
236
|
+
const reading = wordReading(text);
|
|
237
|
+
const type = reading === "operator" ? "keyword" : reading === "refused" ? "reserved" : "ident";
|
|
238
|
+
return { type, start, end: at, text };
|
|
239
|
+
}
|
|
240
|
+
|
|
241
|
+
private readNumber(start: number): Token | undefined {
|
|
242
|
+
const source = this.source;
|
|
243
|
+
let at = start;
|
|
244
|
+
let kind: "int" | "double" = "int";
|
|
245
|
+
let digits: string;
|
|
246
|
+
let radix = 10;
|
|
247
|
+
if (source[at] === "0" && (source[at + 1] === "x" || source[at + 1] === "X")) {
|
|
248
|
+
at += 2;
|
|
249
|
+
const from = at;
|
|
250
|
+
while (at < source.length && isHexDigit(source[at]!)) at += 1;
|
|
251
|
+
if (at === from) return this.invalidNumber(start, at);
|
|
252
|
+
digits = source.slice(from, at);
|
|
253
|
+
radix = 16;
|
|
254
|
+
} else {
|
|
255
|
+
const from = at;
|
|
256
|
+
while (at < source.length && isDigit(source[at]!)) at += 1;
|
|
257
|
+
if (source[at] === "." && isDigit(source[at + 1] ?? "")) {
|
|
258
|
+
kind = "double";
|
|
259
|
+
at += 1;
|
|
260
|
+
while (at < source.length && isDigit(source[at]!)) at += 1;
|
|
261
|
+
}
|
|
262
|
+
if (source[at] === "e" || source[at] === "E") {
|
|
263
|
+
let exponent = at + 1;
|
|
264
|
+
if (source[exponent] === "+" || source[exponent] === "-") exponent += 1;
|
|
265
|
+
if (isDigit(source[exponent] ?? "")) {
|
|
266
|
+
kind = "double";
|
|
267
|
+
at = exponent;
|
|
268
|
+
while (at < source.length && isDigit(source[at]!)) at += 1;
|
|
269
|
+
}
|
|
270
|
+
}
|
|
271
|
+
digits = source.slice(from, at);
|
|
272
|
+
}
|
|
273
|
+
|
|
274
|
+
const unsigned = source[at] === "u" || source[at] === "U";
|
|
275
|
+
if (unsigned) at += 1;
|
|
276
|
+
if (at < source.length && isIdentPart(source[at]!)) return this.invalidNumber(start, at);
|
|
277
|
+
|
|
278
|
+
this.at = at;
|
|
279
|
+
if (kind === "double") {
|
|
280
|
+
if (unsigned) return this.invalidNumber(start, at);
|
|
281
|
+
return { type: "double", start, end: at, text: source.slice(start, at), double: Number(digits) };
|
|
282
|
+
}
|
|
283
|
+
const magnitude = radix === 16 ? BigInt(`0x${digits}`) : BigInt(digits);
|
|
284
|
+
if (unsigned) {
|
|
285
|
+
if (magnitude > MAX_UINT) {
|
|
286
|
+
this.diagnostics.report(
|
|
287
|
+
"invalid_unsigned_integer",
|
|
288
|
+
`${source.slice(start, at)} is outside the range of an unsigned 64-bit integer`,
|
|
289
|
+
start,
|
|
290
|
+
at,
|
|
291
|
+
);
|
|
292
|
+
return undefined;
|
|
293
|
+
}
|
|
294
|
+
return { type: "uint", start, end: at, text: source.slice(start, at), int: magnitude };
|
|
295
|
+
}
|
|
296
|
+
if (magnitude > -MIN_INT) {
|
|
297
|
+
this.diagnostics.report(
|
|
298
|
+
"invalid_integer",
|
|
299
|
+
`${source.slice(start, at)} is outside the range of a 64-bit integer`,
|
|
300
|
+
start,
|
|
301
|
+
at,
|
|
302
|
+
);
|
|
303
|
+
return undefined;
|
|
304
|
+
}
|
|
305
|
+
return {
|
|
306
|
+
type: "int",
|
|
307
|
+
start,
|
|
308
|
+
end: at,
|
|
309
|
+
text: source.slice(start, at),
|
|
310
|
+
int: magnitude,
|
|
311
|
+
...(magnitude === -MIN_INT ? { atIntBoundary: true } : {}),
|
|
312
|
+
};
|
|
313
|
+
}
|
|
314
|
+
|
|
315
|
+
private invalidNumber(start: number, at: number): undefined {
|
|
316
|
+
let end = at;
|
|
317
|
+
while (end < this.source.length && isIdentPart(this.source[end]!)) end += 1;
|
|
318
|
+
this.diagnostics.report(
|
|
319
|
+
"invalid_number",
|
|
320
|
+
`${this.source.slice(start, end)} is not a number`,
|
|
321
|
+
start,
|
|
322
|
+
end,
|
|
323
|
+
);
|
|
324
|
+
return undefined;
|
|
325
|
+
}
|
|
326
|
+
|
|
327
|
+
/**
|
|
328
|
+
* A string or bytes literal. `start` is the literal's own start (its prefix, when
|
|
329
|
+
* it has one) and `quoteAt` the opening quote.
|
|
330
|
+
*/
|
|
331
|
+
private readQuoted(start: number, quoteAt: number, raw: boolean, bytes: boolean): Token | undefined {
|
|
332
|
+
const quote = this.source[quoteAt]!;
|
|
333
|
+
const triple = this.source.slice(quoteAt, quoteAt + 3) === quote.repeat(3);
|
|
334
|
+
const terminator = triple ? quote.repeat(3) : quote;
|
|
335
|
+
const decoded = raw
|
|
336
|
+
? this.scanRaw(quoteAt + terminator.length, terminator, triple, bytes)
|
|
337
|
+
: this.scanEscaped(quoteAt + terminator.length, terminator, triple, bytes);
|
|
338
|
+
if (!decoded) return undefined;
|
|
339
|
+
this.at = decoded.end;
|
|
340
|
+
const text = this.source.slice(start, decoded.end);
|
|
341
|
+
if (bytes) {
|
|
342
|
+
return { type: "bytes", start, end: decoded.end, text, bytes: Uint8Array.from(decoded.units) };
|
|
343
|
+
}
|
|
344
|
+
return { type: "string", start, end: decoded.end, text, string: textOf(decoded.units) };
|
|
345
|
+
}
|
|
346
|
+
|
|
347
|
+
private unterminated(start: number, at: number): undefined {
|
|
348
|
+
this.diagnostics.report("unterminated_string", "unterminated string", start, at);
|
|
349
|
+
return undefined;
|
|
350
|
+
}
|
|
351
|
+
|
|
352
|
+
/** A raw literal: a backslash takes the next character with it, both kept as written. */
|
|
353
|
+
private scanRaw(
|
|
354
|
+
from: number,
|
|
355
|
+
terminator: string,
|
|
356
|
+
triple: boolean,
|
|
357
|
+
bytes: boolean,
|
|
358
|
+
): LiteralText | undefined {
|
|
359
|
+
const source = this.source;
|
|
360
|
+
const units: number[] = [];
|
|
361
|
+
let at = from;
|
|
362
|
+
while (at < source.length) {
|
|
363
|
+
if (source.startsWith(terminator, at)) return { units, end: at + terminator.length };
|
|
364
|
+
const ch = source[at]!;
|
|
365
|
+
if (!triple && ch === "\n") break;
|
|
366
|
+
if (ch === "\\" && at + 1 < source.length) {
|
|
367
|
+
units.push(0x5c);
|
|
368
|
+
at += 1 + this.literalCharacter(at + 1, bytes, units);
|
|
369
|
+
continue;
|
|
370
|
+
}
|
|
371
|
+
at += this.literalCharacter(at, bytes, units);
|
|
372
|
+
}
|
|
373
|
+
return this.unterminated(from - terminator.length, at);
|
|
374
|
+
}
|
|
375
|
+
|
|
376
|
+
/**
|
|
377
|
+
* One character of a literal's text, as the literal holds it: a code unit in a string,
|
|
378
|
+
* the character's UTF-8 bytes in a bytes literal. Answers how many code units it read,
|
|
379
|
+
* since a character outside the basic plane is written as a surrogate pair.
|
|
380
|
+
*/
|
|
381
|
+
private literalCharacter(at: number, bytes: boolean, units: number[]): number {
|
|
382
|
+
if (!bytes) {
|
|
383
|
+
units.push(this.source.charCodeAt(at));
|
|
384
|
+
return 1;
|
|
385
|
+
}
|
|
386
|
+
const point = this.source.codePointAt(at)!;
|
|
387
|
+
pushUtf8(units, point);
|
|
388
|
+
return point > 0xffff ? 2 : 1;
|
|
389
|
+
}
|
|
390
|
+
|
|
391
|
+
private scanEscaped(
|
|
392
|
+
from: number,
|
|
393
|
+
terminator: string,
|
|
394
|
+
triple: boolean,
|
|
395
|
+
bytes: boolean,
|
|
396
|
+
): LiteralText | undefined {
|
|
397
|
+
const source = this.source;
|
|
398
|
+
const units: number[] = [];
|
|
399
|
+
let at = from;
|
|
400
|
+
while (at < source.length) {
|
|
401
|
+
if (source.startsWith(terminator, at)) return { units, end: at + terminator.length };
|
|
402
|
+
const ch = source[at]!;
|
|
403
|
+
if (!triple && ch === "\n") break;
|
|
404
|
+
if (ch !== "\\") {
|
|
405
|
+
at += this.literalCharacter(at, bytes, units);
|
|
406
|
+
continue;
|
|
407
|
+
}
|
|
408
|
+
const next = this.readEscape(at, bytes, units);
|
|
409
|
+
if (next === undefined) return undefined;
|
|
410
|
+
at = next;
|
|
411
|
+
}
|
|
412
|
+
return this.unterminated(from - terminator.length, at);
|
|
413
|
+
}
|
|
414
|
+
|
|
415
|
+
/** Decodes one escape sequence onto `units`, answering where it ends. */
|
|
416
|
+
private readEscape(at: number, bytes: boolean, units: number[]): number | undefined {
|
|
417
|
+
const source = this.source;
|
|
418
|
+
const ch = source[at + 1];
|
|
419
|
+
if (ch === undefined) {
|
|
420
|
+
this.diagnostics.report("invalid_escape_sequence", "the escape has no character", at, at + 1);
|
|
421
|
+
return undefined;
|
|
422
|
+
}
|
|
423
|
+
const simple = SIMPLE_ESCAPES.get(ch);
|
|
424
|
+
if (simple !== undefined) {
|
|
425
|
+
units.push(simple);
|
|
426
|
+
return at + 2;
|
|
427
|
+
}
|
|
428
|
+
if (ch === "x" || ch === "X") return this.readHexEscape(at, units);
|
|
429
|
+
if (ch === "u" || ch === "U") {
|
|
430
|
+
if (bytes) {
|
|
431
|
+
this.diagnostics.report(
|
|
432
|
+
"bytes_unicode_escape",
|
|
433
|
+
`\\${ch} names text, which a bytes literal cannot hold — write the bytes with \\x`,
|
|
434
|
+
at,
|
|
435
|
+
at + 2,
|
|
436
|
+
);
|
|
437
|
+
return undefined;
|
|
438
|
+
}
|
|
439
|
+
return this.readUnicodeEscape(at, ch === "u" ? 4 : 8, units);
|
|
440
|
+
}
|
|
441
|
+
if (isOctalDigit(ch)) return this.readOctalEscape(at, units);
|
|
442
|
+
this.diagnostics.report(
|
|
443
|
+
"invalid_escape_sequence",
|
|
444
|
+
`\\${ch} is not an escape sequence`,
|
|
445
|
+
at,
|
|
446
|
+
at + 2,
|
|
447
|
+
);
|
|
448
|
+
return undefined;
|
|
449
|
+
}
|
|
450
|
+
|
|
451
|
+
private readHexEscape(at: number, units: number[]): number | undefined {
|
|
452
|
+
const digits = this.source.slice(at + 2, at + 4);
|
|
453
|
+
if (digits.length < 2 || !isHexDigit(digits[0]!) || !isHexDigit(digits[1]!)) {
|
|
454
|
+
this.diagnostics.report("invalid_hex_escape", "a \\x escape takes two hexadecimal digits", at, at + 4);
|
|
455
|
+
return undefined;
|
|
456
|
+
}
|
|
457
|
+
units.push(Number.parseInt(digits, 16));
|
|
458
|
+
return at + 4;
|
|
459
|
+
}
|
|
460
|
+
|
|
461
|
+
private readOctalEscape(at: number, units: number[]): number | undefined {
|
|
462
|
+
const digits = this.source.slice(at + 1, at + 4);
|
|
463
|
+
if (digits.length < 3 || ![...digits].every(isOctalDigit)) {
|
|
464
|
+
this.diagnostics.report("invalid_octal_escape", "an octal escape takes three octal digits", at, at + 4);
|
|
465
|
+
return undefined;
|
|
466
|
+
}
|
|
467
|
+
const value = Number.parseInt(digits, 8);
|
|
468
|
+
if (value > 0xff) {
|
|
469
|
+
this.diagnostics.report("octal_escape_out_of_range", `\\${digits} is above 255`, at, at + 4);
|
|
470
|
+
return undefined;
|
|
471
|
+
}
|
|
472
|
+
units.push(value);
|
|
473
|
+
return at + 4;
|
|
474
|
+
}
|
|
475
|
+
|
|
476
|
+
private readUnicodeEscape(at: number, width: number, units: number[]): number | undefined {
|
|
477
|
+
const digits = this.source.slice(at + 2, at + 2 + width);
|
|
478
|
+
if (digits.length < width || ![...digits].every(isHexDigit)) {
|
|
479
|
+
this.diagnostics.report(
|
|
480
|
+
"invalid_unicode_escape",
|
|
481
|
+
`a \\${width === 4 ? "u" : "U"} escape takes ${width} hexadecimal digits`,
|
|
482
|
+
at,
|
|
483
|
+
at + 2 + width,
|
|
484
|
+
);
|
|
485
|
+
return undefined;
|
|
486
|
+
}
|
|
487
|
+
const point = Number.parseInt(digits, 16);
|
|
488
|
+
const end = at + 2 + width;
|
|
489
|
+
if (point > 0x10ffff) {
|
|
490
|
+
this.diagnostics.report("invalid_unicode_escape", `U+${digits} is not a code point`, at, end);
|
|
491
|
+
return undefined;
|
|
492
|
+
}
|
|
493
|
+
if (point >= 0xdc00 && point <= 0xdfff) {
|
|
494
|
+
this.diagnostics.report("invalid_unicode_surrogate", `U+${digits} is a trailing surrogate`, at, end);
|
|
495
|
+
return undefined;
|
|
496
|
+
}
|
|
497
|
+
if (point >= 0xd800 && point <= 0xdbff) return this.readSurrogatePair(at, end, point, units);
|
|
498
|
+
if (point > 0xffff) {
|
|
499
|
+
units.push(0xd800 + ((point - 0x10000) >> 10), 0xdc00 + ((point - 0x10000) & 0x3ff));
|
|
500
|
+
return end;
|
|
501
|
+
}
|
|
502
|
+
units.push(point);
|
|
503
|
+
return end;
|
|
504
|
+
}
|
|
505
|
+
|
|
506
|
+
/** A leading surrogate stands only beside the trailing one that completes it. */
|
|
507
|
+
private readSurrogatePair(at: number, end: number, lead: number, units: number[]): number | undefined {
|
|
508
|
+
const follows = this.source.slice(end, end + 6);
|
|
509
|
+
const trail = /^\\u([0-9a-fA-F]{4})$/.exec(follows);
|
|
510
|
+
const point = trail ? Number.parseInt(trail[1]!, 16) : 0;
|
|
511
|
+
if (!trail || point < 0xdc00 || point > 0xdfff) {
|
|
512
|
+
this.diagnostics.report(
|
|
513
|
+
"invalid_unicode_surrogate",
|
|
514
|
+
"a leading surrogate must be followed by a trailing surrogate escape",
|
|
515
|
+
at,
|
|
516
|
+
end,
|
|
517
|
+
);
|
|
518
|
+
return undefined;
|
|
519
|
+
}
|
|
520
|
+
units.push(lead, point);
|
|
521
|
+
return end + 6;
|
|
522
|
+
}
|
|
523
|
+
}
|
|
524
|
+
|
|
525
|
+
/** The UTF-8 bytes of one code point, which is what a bytes literal holds of its text. */
|
|
526
|
+
function pushUtf8(units: number[], point: number): void {
|
|
527
|
+
if (point < 0x80) {
|
|
528
|
+
units.push(point);
|
|
529
|
+
return;
|
|
530
|
+
}
|
|
531
|
+
if (point < 0x800) {
|
|
532
|
+
units.push(0xc0 | (point >> 6), 0x80 | (point & 0x3f));
|
|
533
|
+
return;
|
|
534
|
+
}
|
|
535
|
+
if (point < 0x10000) {
|
|
536
|
+
units.push(0xe0 | (point >> 12), 0x80 | ((point >> 6) & 0x3f), 0x80 | (point & 0x3f));
|
|
537
|
+
return;
|
|
538
|
+
}
|
|
539
|
+
units.push(
|
|
540
|
+
0xf0 | (point >> 18),
|
|
541
|
+
0x80 | ((point >> 12) & 0x3f),
|
|
542
|
+
0x80 | ((point >> 6) & 0x3f),
|
|
543
|
+
0x80 | (point & 0x3f),
|
|
544
|
+
);
|
|
545
|
+
}
|
|
546
|
+
|
|
547
|
+
/** Code units to text, in chunks so that a long literal does not exhaust the stack. */
|
|
548
|
+
function textOf(units: readonly number[]): string {
|
|
549
|
+
if (units.length <= 4096) return String.fromCharCode(...units);
|
|
550
|
+
let text = "";
|
|
551
|
+
for (let at = 0; at < units.length; at += 4096) {
|
|
552
|
+
text += String.fromCharCode(...units.slice(at, at + 4096));
|
|
553
|
+
}
|
|
554
|
+
return text;
|
|
555
|
+
}
|
|
556
|
+
|
|
557
|
+
/** Every token of the source, with at most one diagnostic for where it stopped. */
|
|
558
|
+
export function tokenize(source: string): { tokens: Token[]; diagnostics: FirstSyntaxDiagnostic } {
|
|
559
|
+
const diagnostics = new FirstSyntaxDiagnostic();
|
|
560
|
+
const tokens = new Lexer(source, diagnostics).tokenize();
|
|
561
|
+
return { tokens, diagnostics };
|
|
562
|
+
}
|