xbintsc 0.3.35 → 0.3.49
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/AGENTS.md +95 -0
- package/README.md +31 -3
- package/README.zh-CN.md +23 -0
- package/dist/src/cli/hints.d.ts +54 -0
- package/dist/src/cli/hints.js +165 -0
- package/dist/src/cli/hints.js.map +1 -0
- package/dist/src/cli/main.js +73 -9
- package/dist/src/cli/main.js.map +1 -1
- package/dist/src/codegen/generator/module.js +10 -0
- package/dist/src/codegen/generator/module.js.map +1 -1
- package/dist/src/codegen/generator/state.js +4 -0
- package/dist/src/codegen/generator/state.js.map +1 -1
- package/dist/src/codegen/generator/tables.d.ts +26 -0
- package/dist/src/codegen/generator/tables.js +64 -12
- package/dist/src/codegen/generator/tables.js.map +1 -1
- package/dist/src/diagnostics/source-text.d.ts +22 -0
- package/dist/src/diagnostics/source-text.js +76 -0
- package/dist/src/diagnostics/source-text.js.map +1 -0
- package/dist/src/driver/bundler/graph.js +2 -1
- package/dist/src/driver/bundler/graph.js.map +1 -1
- package/dist/src/driver/compiler.js +3 -2
- package/dist/src/driver/compiler.js.map +1 -1
- package/dist/src/lexer/scanner/strings.js +16 -3
- package/dist/src/lexer/scanner/strings.js.map +1 -1
- package/dist/tests/cli/hints.test.d.ts +9 -0
- package/dist/tests/cli/hints.test.js +143 -0
- package/dist/tests/cli/hints.test.js.map +1 -0
- package/dist/tests/cli/main.test.js +6 -4
- package/dist/tests/cli/main.test.js.map +1 -1
- package/dist/tests/codegen/llvm.test.js +17 -2
- package/dist/tests/codegen/llvm.test.js.map +1 -1
- package/dist/tests/e2e/gc.test.d.ts +1 -0
- package/dist/tests/e2e/gc.test.js +168 -0
- package/dist/tests/e2e/gc.test.js.map +1 -0
- package/dist/tests/e2e/harness.d.ts +2 -0
- package/dist/tests/e2e/harness.js +1 -0
- package/dist/tests/e2e/harness.js.map +1 -1
- package/dist/tests/helpers.js +3 -2
- package/dist/tests/helpers.js.map +1 -1
- package/dist/tests/lexer/strings.test.js +14 -2
- package/dist/tests/lexer/strings.test.js.map +1 -1
- package/doc/DESIGN.md +117 -0
- package/doc/ai/README.md +63 -0
- package/doc/ai/build-recipe.md +137 -0
- package/doc/ai/cli.md +142 -0
- package/doc/ai/contributing.md +196 -0
- package/doc/ai/extensions.md +148 -0
- package/doc/ai/language-support.md +152 -0
- package/doc/ai/troubleshooting.md +163 -0
- package/doc/ai/zh-CN/README.md +56 -0
- package/doc/ai/zh-CN/build-recipe.md +132 -0
- package/doc/ai/zh-CN/cli.md +127 -0
- package/doc/ai/zh-CN/contributing.md +173 -0
- package/doc/ai/zh-CN/extensions.md +139 -0
- package/doc/ai/zh-CN/language-support.md +147 -0
- package/doc/ai/zh-CN/troubleshooting.md +150 -0
- package/doc/gui-scripts.md +350 -0
- package/doc/gui.md +646 -0
- package/doc/icon.md +265 -0
- package/doc/implemented.md +373 -0
- package/doc/node-implemented.md +588 -0
- package/doc/node-unimplemented.md +167 -0
- package/doc/post/announce.md +43 -0
- package/doc/requirements.md +145 -0
- package/doc/unimplemented.md +286 -0
- package/doc/xbintsc.config.schema.json +67 -0
- package/doc/zh-CN/DESIGN.md +104 -0
- package/doc/zh-CN/gui-scripts.md +329 -0
- package/doc/zh-CN/gui.md +588 -0
- package/doc/zh-CN/icon.md +241 -0
- package/doc/zh-CN/implemented.md +365 -0
- package/doc/zh-CN/node-implemented.md +533 -0
- package/doc/zh-CN/node-unimplemented.md +141 -0
- package/doc/zh-CN/plan-require-node-modules.md +284 -0
- package/doc/zh-CN/post/announce.md +47 -0
- package/doc/zh-CN/requirements.md +134 -0
- package/doc/zh-CN/unimplemented.md +247 -0
- package/llms.txt +45 -0
- package/package.json +6 -2
- package/runtime/ext_gui/dom_api_proto.cpp +5 -0
- package/runtime/ext_gui/gui.cpp +3 -1
- package/runtime/ext_gui/renderer.cpp +13 -11
- package/runtime/ext_gui/renderer_image.cpp +12 -8
- package/runtime/ext_gui/renderer_shaders.h +131 -4
- package/runtime/ext_gui/renderer_shaders_data.h +1809 -0
- package/runtime/ext_gui/renderer_text.cpp +12 -8
- package/runtime/ext_gui/shaders.hlsl +98 -0
- package/runtime/ext_gui/spirv/fill.frag +19 -0
- package/runtime/ext_gui/spirv/fill.vert +42 -0
- package/runtime/ext_gui/spirv/image.frag +16 -0
- package/runtime/ext_gui/spirv/quad.vert +30 -0
- package/runtime/ext_gui/spirv/text.frag +16 -0
- package/runtime/ext_gui/window.cpp +1 -0
- package/runtime/ext_node/buffer/parts/prototype.inc +1 -0
- package/runtime/ext_node/dgram/dgram.c +1 -0
- package/runtime/ext_node/events/events.c +1 -0
- package/runtime/ext_node/fs/fs_ops.c +10 -25
- package/runtime/ext_node/fs/glob.c +13 -30
- package/runtime/ext_node/fs/promises.c +1 -0
- package/runtime/ext_node/http/parts/prototypes.inc +5 -0
- package/runtime/ext_node/net/parts/prototypes.inc +2 -0
- package/runtime/ext_node/process/process.c +9 -7
- package/runtime/ext_node/stream/stream.c +1 -0
- package/runtime/ext_node/util/util.c +6 -6
- package/runtime/rt.h +10 -0
- package/runtime/rt_internal.h +63 -2
- package/runtime/xt_alloc.c +442 -5
- package/runtime/xt_generator.c +95 -1
- package/runtime/xt_loop.c +35 -1
- package/runtime/xt_promise.c +46 -0
- package/runtime/xt_stdlib2/error.inc +1 -0
- package/runtime/xt_stdlib2/regexp-match.inc +8 -7
- package/runtime/xt_symbol.c +2 -0
- package/runtime/xt_typed_array/construction.inc +142 -0
- package/runtime/xt_typed_array/elements.inc +92 -0
- package/runtime/xt_typed_array/methods.inc +329 -0
- package/runtime/xt_typed_array.c +6 -548
- package/runtime/xt_values/number-format.inc +26 -0
- package/scripts/build-gui-shaders.mjs +204 -0
- package/scripts/build-gui.ts +35 -0
- package/scripts/check-file-length.ts +89 -0
- package/src/cli/hints.ts +194 -0
- package/src/cli/main.ts +82 -9
- package/src/codegen/generator/module.ts +10 -0
- package/src/codegen/generator/state.ts +4 -0
- package/src/codegen/generator/tables.ts +60 -14
- package/src/diagnostics/source-text.ts +78 -0
- package/src/driver/bundler/graph.ts +2 -1
- package/src/driver/compiler.ts +3 -2
- package/src/lexer/scanner/strings.ts +16 -3
package/src/cli/main.ts
CHANGED
|
@@ -22,7 +22,16 @@ import { build, compileEntry, COMPILER_VERSION, type EmitKind } from "../driver/
|
|
|
22
22
|
import { findRuntimeDir, platformSlug } from "../driver/paths.js";
|
|
23
23
|
import { runtimeLibDir } from "../driver/runtime-lib.js";
|
|
24
24
|
import { resolveToolchain } from "../driver/toolchain-provider.js";
|
|
25
|
-
import { realRunner } from "../driver/toolchain.js";
|
|
25
|
+
import { realRunner, ToolchainError } from "../driver/toolchain.js";
|
|
26
|
+
import {
|
|
27
|
+
DOC_CLI,
|
|
28
|
+
DOC_TROUBLESHOOTING,
|
|
29
|
+
documentationHints,
|
|
30
|
+
thrownFailure,
|
|
31
|
+
toolchainPointer,
|
|
32
|
+
UNEXPECTED_FAILURE_HINT,
|
|
33
|
+
type DocumentationPointer,
|
|
34
|
+
} from "./hints.js";
|
|
26
35
|
import { createDefaultRegistry, type ExtensionRegistry } from "../extensions/registry.js";
|
|
27
36
|
import { bundledExtensions } from "../extensions/catalog.js";
|
|
28
37
|
import { nativeExtensionFromManifest } from "../extensions/native.js";
|
|
@@ -46,6 +55,13 @@ const defaultIo: CliIo = {
|
|
|
46
55
|
stderr: (text) => process.stderr.write(text),
|
|
47
56
|
};
|
|
48
57
|
|
|
58
|
+
/**
|
|
59
|
+
* `--no-hints` state for the current {@link run} call. A module-level flag (and
|
|
60
|
+
* not a parameter) keeps the hint layer out of the signatures of every helper
|
|
61
|
+
* that can fail.
|
|
62
|
+
*/
|
|
63
|
+
let hintsDisabled = false;
|
|
64
|
+
|
|
49
65
|
interface ParsedArgs {
|
|
50
66
|
readonly command?: string;
|
|
51
67
|
readonly positionals: string[];
|
|
@@ -152,6 +168,46 @@ function printDiagnostics(diagnostics: readonly Diagnostic[], fileName: string,
|
|
|
152
168
|
}
|
|
153
169
|
io.stderr(formatDiagnostic(diagnostic, source) + "\n");
|
|
154
170
|
}
|
|
171
|
+
for (const hint of documentationHints(diagnostics)) emitHint(io, hint);
|
|
172
|
+
}
|
|
173
|
+
|
|
174
|
+
/** Print one `hint:` line unless the caller asked for plain output. */
|
|
175
|
+
function emitHint(io: CliIo, line: string): void {
|
|
176
|
+
if (hintsDisabled) return;
|
|
177
|
+
io.stderr(`${line}\n`);
|
|
178
|
+
}
|
|
179
|
+
|
|
180
|
+
/**
|
|
181
|
+
* Report a failure that threw instead of becoming a diagnostic, ending with the
|
|
182
|
+
* documentation pointer that explains it. Without this a toolchain failure is a
|
|
183
|
+
* raw stack trace, which tells a reader nothing about which document answers
|
|
184
|
+
* the question.
|
|
185
|
+
*/
|
|
186
|
+
function reportThrown(error: unknown, io: CliIo): number {
|
|
187
|
+
const message =
|
|
188
|
+
error instanceof ToolchainError
|
|
189
|
+
? `xbintsc: ${error.message}`
|
|
190
|
+
: `xbintsc: ${error instanceof Error ? error.message : String(error)}`;
|
|
191
|
+
io.stderr(`${message.trimEnd()}\n`);
|
|
192
|
+
const text = error instanceof ToolchainError ? error.message : message;
|
|
193
|
+
const pointer = error instanceof ToolchainError ? toolchainPointer(text) : thrownPointer(text);
|
|
194
|
+
emitHint(io, `hint: ${pointer.text} — see ${pointer.path}`);
|
|
195
|
+
return 1;
|
|
196
|
+
}
|
|
197
|
+
|
|
198
|
+
/** The document that explains a failure the CLI reports as a plain message. */
|
|
199
|
+
function thrownPointer(message: string): DocumentationPointer {
|
|
200
|
+
return thrownFailure(message)?.pointer ?? UNEXPECTED_FAILURE_HINT;
|
|
201
|
+
}
|
|
202
|
+
|
|
203
|
+
/** Report a project-config failure, which never becomes a diagnostic bag. */
|
|
204
|
+
function reportConfigError(error: string, io: CliIo): number {
|
|
205
|
+
io.stderr(`xbintsc: ${error}\n`);
|
|
206
|
+
const pointer = /JSON/i.test(error)
|
|
207
|
+
? `hint: fix the JSON; an unreadable or malformed config fails the build — see ${DOC_TROUBLESHOOTING}`
|
|
208
|
+
: `hint: check the config path and its fields — see ${DOC_CLI}`;
|
|
209
|
+
emitHint(io, pointer);
|
|
210
|
+
return 1;
|
|
155
211
|
}
|
|
156
212
|
|
|
157
213
|
const HELP = `xbintsc ${COMPILER_VERSION} - TypeScript binary compiler
|
|
@@ -181,6 +237,14 @@ Options:
|
|
|
181
237
|
--app-id <id> macOS bundle identifier (e.g. com.example.demo)
|
|
182
238
|
--force Ignore the incremental cache
|
|
183
239
|
--verbose Print progress information
|
|
240
|
+
--no-hints Do not append the "hint: … see <doc>" line to failures
|
|
241
|
+
|
|
242
|
+
xbintsc compiles a subset of TypeScript to a native binary. Types are erased
|
|
243
|
+
(never checked), Node modules need --ext node, third-party npm imports are not
|
|
244
|
+
supported, and clang 16+ is required to link.
|
|
245
|
+
|
|
246
|
+
Docs: doc/ai/README.md (task guide) - doc/ai/language-support.md (subset) -
|
|
247
|
+
doc/ai/troubleshooting.md (failures) - llms.txt (index of every document)
|
|
184
248
|
`;
|
|
185
249
|
|
|
186
250
|
function firstLine(text: string): string {
|
|
@@ -281,6 +345,19 @@ function listFlag(flags: Map<string, string | boolean>, name: string): string[]
|
|
|
281
345
|
}
|
|
282
346
|
|
|
283
347
|
export function run(argv: readonly string[], io: CliIo = defaultIo): number {
|
|
348
|
+
const disableHints = parseArgs(argv).flags.has("no-hints");
|
|
349
|
+
const previous = hintsDisabled;
|
|
350
|
+
hintsDisabled = disableHints;
|
|
351
|
+
try {
|
|
352
|
+
return dispatch(argv, io);
|
|
353
|
+
} catch (error) {
|
|
354
|
+
return reportThrown(error, io);
|
|
355
|
+
} finally {
|
|
356
|
+
hintsDisabled = previous;
|
|
357
|
+
}
|
|
358
|
+
}
|
|
359
|
+
|
|
360
|
+
function dispatch(argv: readonly string[], io: CliIo): number {
|
|
284
361
|
const args = parseArgs(argv);
|
|
285
362
|
const command = args.command;
|
|
286
363
|
|
|
@@ -301,13 +378,11 @@ export function run(argv: readonly string[], io: CliIo = defaultIo): number {
|
|
|
301
378
|
if (command === "emit") {
|
|
302
379
|
const cliEntry = args.positionals[0];
|
|
303
380
|
const resolved = resolveProjectConfig(args.flags, cliEntry);
|
|
304
|
-
if (!resolved.ok)
|
|
305
|
-
io.stderr(`xbintsc: ${resolved.error}\n`);
|
|
306
|
-
return 1;
|
|
307
|
-
}
|
|
381
|
+
if (!resolved.ok) return reportConfigError(resolved.error, io);
|
|
308
382
|
const entry = cliEntry ?? resolved.config.entry;
|
|
309
383
|
if (!entry) {
|
|
310
384
|
io.stderr(`xbintsc: emit requires a source file (or 'entry' in ${CONFIG_FILE_NAME})\n`);
|
|
385
|
+
emitHint(io, `hint: pass the entry file, or set 'entry' in ${CONFIG_FILE_NAME} — see ${DOC_CLI}`);
|
|
311
386
|
return 1;
|
|
312
387
|
}
|
|
313
388
|
const extNames = listFlag(args.flags, "ext") ?? resolved.config.extensions ?? [];
|
|
@@ -325,14 +400,12 @@ export function run(argv: readonly string[], io: CliIo = defaultIo): number {
|
|
|
325
400
|
if (command === "build" || command === "run") {
|
|
326
401
|
const cliEntry = args.positionals[0];
|
|
327
402
|
const resolved = resolveProjectConfig(args.flags, cliEntry);
|
|
328
|
-
if (!resolved.ok)
|
|
329
|
-
io.stderr(`xbintsc: ${resolved.error}\n`);
|
|
330
|
-
return 1;
|
|
331
|
-
}
|
|
403
|
+
if (!resolved.ok) return reportConfigError(resolved.error, io);
|
|
332
404
|
const config = resolved.config;
|
|
333
405
|
const entry = cliEntry ?? config.entry;
|
|
334
406
|
if (!entry) {
|
|
335
407
|
io.stderr(`xbintsc: ${command} requires a source file (or 'entry' in ${CONFIG_FILE_NAME})\n`);
|
|
408
|
+
emitHint(io, `hint: pass the entry file, or set 'entry' in ${CONFIG_FILE_NAME} — see ${DOC_CLI}`);
|
|
336
409
|
return 1;
|
|
337
410
|
}
|
|
338
411
|
const emit = (args.flags.get("emit") as EmitKind | undefined) ?? "exe";
|
|
@@ -227,9 +227,19 @@ export const moduleMethods: ModuleMethods = {
|
|
|
227
227
|
|
|
228
228
|
emitMain(): void {
|
|
229
229
|
const moduleName = this.functionName(this.binding.moduleFunction);
|
|
230
|
+
// Module bindings and class constructors live in globals; register them as
|
|
231
|
+
// GC roots before the module body runs so a collection cannot free them.
|
|
232
|
+
const roots = [...this.classGlobals.values(), ...this.moduleGlobals.values()];
|
|
233
|
+
const rootLines = roots.map((name) => ` call void @xt_gc_add_root(i64* ${name})`);
|
|
230
234
|
this.functions.push(
|
|
231
235
|
[
|
|
232
236
|
"define i32 @main(i32 %argc, i8** %argv) {",
|
|
237
|
+
" %stackbase = alloca i64",
|
|
238
|
+
" %stackbase.ptr = bitcast i64* %stackbase to i8*",
|
|
239
|
+
" call void @xt_gc_set_stack_base(i8* %stackbase.ptr)",
|
|
240
|
+
" call void @xt_gc_init()",
|
|
241
|
+
...rootLines,
|
|
242
|
+
" call void @xt_gc_arm()",
|
|
233
243
|
" call void @xt_set_program_args(i32 %argc, i8** %argv)",
|
|
234
244
|
` %result = call i64 @${moduleName}(i64 ${i64(XT_UNDEFINED)}, i64 ${i64(XT_UNDEFINED)}, i32 0, i64* null)`,
|
|
235
245
|
" call void @xt_drain_microtasks()",
|
|
@@ -103,6 +103,10 @@ export const RUNTIME_DECLARATIONS: readonly string[] = [
|
|
|
103
103
|
"declare i64 @xt_await(i64)",
|
|
104
104
|
"declare void @xt_drain_microtasks()",
|
|
105
105
|
"declare void @xt_run_event_loop()",
|
|
106
|
+
"declare void @xt_gc_init()",
|
|
107
|
+
"declare void @xt_gc_set_stack_base(i8*)",
|
|
108
|
+
"declare void @xt_gc_arm()",
|
|
109
|
+
"declare void @xt_gc_add_root(i64*)",
|
|
106
110
|
"declare i64 @xt_promise_resolve(i64)",
|
|
107
111
|
"declare i64 @xt_promise_reject(i64)",
|
|
108
112
|
"declare i64 @xt_promise_ctor(i32, i64*)",
|
|
@@ -327,27 +327,73 @@ export const KIND_NAMES = new Map<number, string>([
|
|
|
327
327
|
[SyntaxKind.SwitchStatement, "switch statement"],
|
|
328
328
|
]);
|
|
329
329
|
|
|
330
|
+
/**
|
|
331
|
+
* The UTF-8 bytes of one code point, which must be an ASCII byte or a Unicode
|
|
332
|
+
* scalar (`0..0x10ffff`, not a surrogate). Encoded by hand rather than with
|
|
333
|
+
* `String.fromCharCode`/`String.fromCodePoint`, which a self-hosted build would
|
|
334
|
+
* run through the runtime and encode a second time.
|
|
335
|
+
*/
|
|
336
|
+
export function codePointBytes(code: number): number[] {
|
|
337
|
+
if (code < 0x80) return [code];
|
|
338
|
+
if (code < 0x800) return [0xc0 | (code >> 6), 0x80 | (code & 0x3f)];
|
|
339
|
+
if (code < 0x10000) return [0xe0 | (code >> 12), 0x80 | ((code >> 6) & 0x3f), 0x80 | (code & 0x3f)];
|
|
340
|
+
return [
|
|
341
|
+
0xf0 | (code >> 18),
|
|
342
|
+
0x80 | ((code >> 12) & 0x3f),
|
|
343
|
+
0x80 | ((code >> 6) & 0x3f),
|
|
344
|
+
0x80 | (code & 0x3f),
|
|
345
|
+
];
|
|
346
|
+
}
|
|
347
|
+
|
|
348
|
+
/**
|
|
349
|
+
* The bytes a string literal stands for.
|
|
350
|
+
*
|
|
351
|
+
* A compiled string is a sequence of UTF-8 bytes (`"é".length` is 1 and
|
|
352
|
+
* `"😀".length` is 4 - see doc/ai/language-support.md). The text reaching this
|
|
353
|
+
* point is whatever a host's strings are, and both shapes have to produce the
|
|
354
|
+
* same bytes:
|
|
355
|
+
*
|
|
356
|
+
* - a UTF-16 host decodes a file into characters, so `é` is one code unit
|
|
357
|
+
* (0xe9) and has to be encoded here;
|
|
358
|
+
* - the self-hosted runtime has no UTF-8 decoder, so `readFileSync` hands over
|
|
359
|
+
* a file's bytes, and `é` is already `c3 a9`. Encoding that again is what
|
|
360
|
+
* turned the three bytes of `—` into six and broke the self-hosting fixpoint.
|
|
361
|
+
*
|
|
362
|
+
* A code unit above 0xff can only be text, so its presence settles which shape
|
|
363
|
+
* this is: text is encoded (once), and text that is already a byte sequence is
|
|
364
|
+
* emitted as it is. A source file read on either host therefore lands on the
|
|
365
|
+
* same bytes, which is what the fixpoint needs.
|
|
366
|
+
*/
|
|
330
367
|
export function utf8Bytes(text: string): number[] {
|
|
368
|
+
if (!hasMultiByteUnit(text)) {
|
|
369
|
+
const bytes: number[] = [];
|
|
370
|
+
for (let index = 0; index < text.length; index++) bytes.push(text.charCodeAt(index) & 0xff);
|
|
371
|
+
return bytes;
|
|
372
|
+
}
|
|
331
373
|
const bytes: number[] = [];
|
|
332
|
-
|
|
333
|
-
|
|
334
|
-
|
|
335
|
-
|
|
336
|
-
|
|
337
|
-
|
|
338
|
-
|
|
339
|
-
|
|
340
|
-
|
|
341
|
-
0xf0 | (code >> 18),
|
|
342
|
-
0x80 | ((code >> 12) & 0x3f),
|
|
343
|
-
0x80 | ((code >> 6) & 0x3f),
|
|
344
|
-
0x80 | (code & 0x3f),
|
|
345
|
-
);
|
|
374
|
+
const length = text.length;
|
|
375
|
+
for (let index = 0; index < length; index++) {
|
|
376
|
+
let code = text.charCodeAt(index);
|
|
377
|
+
if (code >= 0xd800 && code <= 0xdbff && index + 1 < length) {
|
|
378
|
+
const low = text.charCodeAt(index + 1);
|
|
379
|
+
if (low >= 0xdc00 && low <= 0xdfff) {
|
|
380
|
+
code = 0x10000 + ((code - 0xd800) << 10) + (low - 0xdc00);
|
|
381
|
+
index++;
|
|
382
|
+
}
|
|
346
383
|
}
|
|
384
|
+
for (const byte of codePointBytes(code)) bytes.push(byte);
|
|
347
385
|
}
|
|
348
386
|
return bytes;
|
|
349
387
|
}
|
|
350
388
|
|
|
389
|
+
/** Whether any code unit is above one byte, which means text, not bytes. */
|
|
390
|
+
function hasMultiByteUnit(text: string): boolean {
|
|
391
|
+
for (let index = 0; index < text.length; index++) {
|
|
392
|
+
if (text.charCodeAt(index) > 0xff) return true;
|
|
393
|
+
}
|
|
394
|
+
return false;
|
|
395
|
+
}
|
|
396
|
+
|
|
351
397
|
export function escapeBytes(bytes: readonly number[]): string {
|
|
352
398
|
return bytes
|
|
353
399
|
.map((byte) => {
|
|
@@ -0,0 +1,78 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Source text is a byte sequence.
|
|
3
|
+
*
|
|
4
|
+
* A compiled string is a sequence of UTF-8 bytes (`"e".length` is 1 for an
|
|
5
|
+
* accent and 4 for an emoji - see doc/ai/language-support.md), and the compiler
|
|
6
|
+
* is built by itself: there the self-hosted `readFileSync` hands the scanner a
|
|
7
|
+
* file's own bytes, one per code unit, and the runtime has no UTF-8 decoder to
|
|
8
|
+
* change that. A host whose strings are UTF-16 must be given the text in the
|
|
9
|
+
* same shape, or the same program emits different bytes depending on which
|
|
10
|
+
* generation of the compiler read it - which is how a non-ASCII literal ended
|
|
11
|
+
* up double-encoded and broke the self-hosting fixpoint.
|
|
12
|
+
*
|
|
13
|
+
* `sourceTextBytes` is that shape: a source file's bytes, one code unit each.
|
|
14
|
+
* The driver reads every source file through it, and codegen emits those bytes
|
|
15
|
+
* unchanged.
|
|
16
|
+
*/
|
|
17
|
+
|
|
18
|
+
/**
|
|
19
|
+
* The UTF-8 bytes of `text`, one code unit each. Text that is already a byte
|
|
20
|
+
* sequence is kept as it is, so this is idempotent and safe to apply to a read
|
|
21
|
+
* on either host.
|
|
22
|
+
*/
|
|
23
|
+
export function sourceTextBytes(text: string): string {
|
|
24
|
+
if (!hasCharacter(text)) return startsWithBom(text) ? text.slice(3) : text;
|
|
25
|
+
const encoded = encodeUtf8(text);
|
|
26
|
+
return startsWithBom(encoded) ? encoded.slice(3) : encoded;
|
|
27
|
+
}
|
|
28
|
+
|
|
29
|
+
/** Encode text as UTF-8, spelled one code unit per byte. */
|
|
30
|
+
function encodeUtf8(text: string): string {
|
|
31
|
+
const chunks: string[] = [];
|
|
32
|
+
const units: number[] = [];
|
|
33
|
+
for (let index = 0; index < text.length; index++) {
|
|
34
|
+
let code = text.charCodeAt(index);
|
|
35
|
+
if (code >= 0xd800 && code <= 0xdbff && index + 1 < text.length) {
|
|
36
|
+
const low = text.charCodeAt(index + 1);
|
|
37
|
+
if (low >= 0xdc00 && low <= 0xdfff) {
|
|
38
|
+
code = 0x10000 + ((code - 0xd800) << 10) + (low - 0xdc00);
|
|
39
|
+
index++;
|
|
40
|
+
}
|
|
41
|
+
}
|
|
42
|
+
if (code < 0x80) units.push(code);
|
|
43
|
+
else if (code < 0x800) units.push(0xc0 | (code >> 6), 0x80 | (code & 0x3f));
|
|
44
|
+
else if (code < 0x10000) {
|
|
45
|
+
units.push(0xe0 | (code >> 12), 0x80 | ((code >> 6) & 0x3f), 0x80 | (code & 0x3f));
|
|
46
|
+
} else {
|
|
47
|
+
units.push(
|
|
48
|
+
0xf0 | (code >> 18),
|
|
49
|
+
0x80 | ((code >> 12) & 0x3f),
|
|
50
|
+
0x80 | ((code >> 6) & 0x3f),
|
|
51
|
+
0x80 | (code & 0x3f),
|
|
52
|
+
);
|
|
53
|
+
}
|
|
54
|
+
if (units.length >= 512) flushUnits(units, chunks);
|
|
55
|
+
}
|
|
56
|
+
flushUnits(units, chunks);
|
|
57
|
+
return chunks.join("");
|
|
58
|
+
}
|
|
59
|
+
|
|
60
|
+
/** Whether any code unit is above one byte, which means text, not bytes. */
|
|
61
|
+
function hasCharacter(text: string): boolean {
|
|
62
|
+
for (let index = 0; index < text.length; index++) {
|
|
63
|
+
if (text.charCodeAt(index) > 0xff) return true;
|
|
64
|
+
}
|
|
65
|
+
return false;
|
|
66
|
+
}
|
|
67
|
+
|
|
68
|
+
/** Whether a decoded read started with a UTF-8 byte order mark. */
|
|
69
|
+
function startsWithBom(text: string): boolean {
|
|
70
|
+
return text.charCodeAt(0) === 0xef && text.charCodeAt(1) === 0xbb && text.charCodeAt(2) === 0xbf;
|
|
71
|
+
}
|
|
72
|
+
|
|
73
|
+
/** Append the pending code units as one chunk of the byte string. */
|
|
74
|
+
function flushUnits(units: number[], chunks: string[]): void {
|
|
75
|
+
if (units.length === 0) return;
|
|
76
|
+
chunks.push(String.fromCharCode(...units));
|
|
77
|
+
units.length = 0;
|
|
78
|
+
}
|
|
@@ -9,6 +9,7 @@ import { SyntaxKind, type ExportDeclaration, type ImportDeclaration, type Statem
|
|
|
9
9
|
import { bind } from "../../binder/binder.js";
|
|
10
10
|
import { DiagnosticBag, DiagnosticCode } from "../../diagnostics/diagnostic.js";
|
|
11
11
|
import { SourceFile } from "../../diagnostics/source.js";
|
|
12
|
+
import { sourceTextBytes } from "../../diagnostics/source-text.js";
|
|
12
13
|
import { Parser } from "../../parser/parser.js";
|
|
13
14
|
import type { AssetLoader } from "../../extensions/registry.js";
|
|
14
15
|
import { classifyDependency } from "./resolve.js";
|
|
@@ -53,7 +54,7 @@ export function loadGraph(
|
|
|
53
54
|
return undefined;
|
|
54
55
|
}
|
|
55
56
|
visiting.add(path);
|
|
56
|
-
const raw = readFileSync(path, "utf8");
|
|
57
|
+
const raw = sourceTextBytes(readFileSync(path, "utf8"));
|
|
57
58
|
/* An extension may rewrite an imported asset (e.g. `.html`) into a TS
|
|
58
59
|
* module before it is parsed. The core only sees the transformed text. */
|
|
59
60
|
const loader = assetLoaders[extname(path).toLowerCase()];
|
package/src/driver/compiler.ts
CHANGED
|
@@ -11,6 +11,7 @@ import { existsSync, mkdirSync, readFileSync, readdirSync, writeFileSync } from
|
|
|
11
11
|
import { basename, extname, join, resolve } from "node:path";
|
|
12
12
|
import { DiagnosticBag, DiagnosticCode, type Diagnostic } from "../diagnostics/diagnostic.js";
|
|
13
13
|
import { SourceFile } from "../diagnostics/source.js";
|
|
14
|
+
import { sourceTextBytes } from "../diagnostics/source-text.js";
|
|
14
15
|
import { Parser } from "../parser/parser.js";
|
|
15
16
|
import { generate } from "../codegen/llvm.js";
|
|
16
17
|
import { SyntaxKind } from "../ast/nodes.js";
|
|
@@ -106,7 +107,7 @@ function externalModuleSpecifiers(registry: ExtensionRegistry): Set<string> {
|
|
|
106
107
|
*/
|
|
107
108
|
export function compileEntry(entryPath: string, extensions?: ExtensionRegistry): CompileStringResult {
|
|
108
109
|
const absoluteEntry = resolve(entryPath);
|
|
109
|
-
const sourceText = readFileSync(absoluteEntry, "utf8");
|
|
110
|
+
const sourceText = sourceTextBytes(readFileSync(absoluteEntry, "utf8"));
|
|
110
111
|
const file = new SourceFile(absoluteEntry, sourceText);
|
|
111
112
|
const diagnostics = new DiagnosticBag();
|
|
112
113
|
const parser = new Parser(file, diagnostics);
|
|
@@ -170,7 +171,7 @@ export function build(entryPath: string, options: BuildOptions = {}): BuildResul
|
|
|
170
171
|
const optimize = options.optimize ?? "2";
|
|
171
172
|
|
|
172
173
|
const absoluteEntry = resolve(entryPath);
|
|
173
|
-
const sourceText = readFileSync(absoluteEntry, "utf8");
|
|
174
|
+
const sourceText = sourceTextBytes(readFileSync(absoluteEntry, "utf8"));
|
|
174
175
|
|
|
175
176
|
const baseName = basename(absoluteEntry, extname(absoluteEntry));
|
|
176
177
|
const outputPath =
|
|
@@ -6,8 +6,21 @@
|
|
|
6
6
|
import { DiagnosticCode } from "../../diagnostics/diagnostic.js";
|
|
7
7
|
import { Token, TokenKind } from "../token.js";
|
|
8
8
|
import { Char, isDigit, isLineBreak } from "../char.js";
|
|
9
|
+
import { codePointBytes } from "../../codegen/generator/tables.js";
|
|
9
10
|
import type { Scanner } from "../scanner.js";
|
|
10
11
|
|
|
12
|
+
/**
|
|
13
|
+
* The bytes one escape sequence stands for. An escape names a code point, and a
|
|
14
|
+
* literal holds the UTF-8 bytes of that code point - the same bytes
|
|
15
|
+
* `String.fromCodePoint` produces at run time - spelled as one code unit per
|
|
16
|
+
* byte, so both hosts agree on what a literal's text is.
|
|
17
|
+
*/
|
|
18
|
+
function escapeText(code: number): string {
|
|
19
|
+
let text = "";
|
|
20
|
+
for (const byte of codePointBytes(code)) text += String.fromCharCode(byte);
|
|
21
|
+
return text;
|
|
22
|
+
}
|
|
23
|
+
|
|
11
24
|
export interface StringMethods {
|
|
12
25
|
scanString(this: Scanner, start: number, precededByLineBreak: boolean, quote: number): Token;
|
|
13
26
|
scanEscapeSequence(this: Scanner): string;
|
|
@@ -65,16 +78,16 @@ export const stringMethods: StringMethods = {
|
|
|
65
78
|
const close = this.text.indexOf("}", this.pos);
|
|
66
79
|
const hex = this.text.slice(this.pos + 1, close);
|
|
67
80
|
this.pos = close + 1;
|
|
68
|
-
return
|
|
81
|
+
return escapeText(parseInt(hex, 16) || 0);
|
|
69
82
|
} else {
|
|
70
83
|
const hex = this.text.slice(this.pos, this.pos + 4);
|
|
71
84
|
this.pos += 4;
|
|
72
|
-
return
|
|
85
|
+
return escapeText(parseInt(hex, 16) || 0);
|
|
73
86
|
}
|
|
74
87
|
case Char.LowerX: {
|
|
75
88
|
const hex = this.text.slice(this.pos + 1, this.pos + 3);
|
|
76
89
|
this.pos += 3;
|
|
77
|
-
return
|
|
90
|
+
return escapeText(parseInt(hex, 16) || 0);
|
|
78
91
|
}
|
|
79
92
|
case Char.Zero:
|
|
80
93
|
// \0 is only a null escape when not followed by a digit.
|