xbintsc 0.3.46 → 0.3.49

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (91) hide show
  1. package/AGENTS.md +95 -0
  2. package/README.md +25 -0
  3. package/README.zh-CN.md +23 -0
  4. package/dist/src/cli/hints.d.ts +54 -0
  5. package/dist/src/cli/hints.js +165 -0
  6. package/dist/src/cli/hints.js.map +1 -0
  7. package/dist/src/cli/main.js +73 -9
  8. package/dist/src/cli/main.js.map +1 -1
  9. package/dist/src/codegen/generator/tables.d.ts +26 -0
  10. package/dist/src/codegen/generator/tables.js +64 -12
  11. package/dist/src/codegen/generator/tables.js.map +1 -1
  12. package/dist/src/diagnostics/source-text.d.ts +22 -0
  13. package/dist/src/diagnostics/source-text.js +76 -0
  14. package/dist/src/diagnostics/source-text.js.map +1 -0
  15. package/dist/src/driver/bundler/graph.js +2 -1
  16. package/dist/src/driver/bundler/graph.js.map +1 -1
  17. package/dist/src/driver/compiler.js +3 -2
  18. package/dist/src/driver/compiler.js.map +1 -1
  19. package/dist/src/lexer/scanner/strings.js +16 -3
  20. package/dist/src/lexer/scanner/strings.js.map +1 -1
  21. package/dist/tests/cli/hints.test.d.ts +9 -0
  22. package/dist/tests/cli/hints.test.js +143 -0
  23. package/dist/tests/cli/hints.test.js.map +1 -0
  24. package/dist/tests/cli/main.test.js +6 -4
  25. package/dist/tests/cli/main.test.js.map +1 -1
  26. package/dist/tests/codegen/llvm.test.js +17 -2
  27. package/dist/tests/codegen/llvm.test.js.map +1 -1
  28. package/dist/tests/helpers.js +3 -2
  29. package/dist/tests/helpers.js.map +1 -1
  30. package/dist/tests/lexer/strings.test.js +14 -2
  31. package/dist/tests/lexer/strings.test.js.map +1 -1
  32. package/doc/DESIGN.md +117 -0
  33. package/doc/ai/README.md +63 -0
  34. package/doc/ai/build-recipe.md +137 -0
  35. package/doc/ai/cli.md +142 -0
  36. package/doc/ai/contributing.md +196 -0
  37. package/doc/ai/extensions.md +148 -0
  38. package/doc/ai/language-support.md +152 -0
  39. package/doc/ai/troubleshooting.md +163 -0
  40. package/doc/ai/zh-CN/README.md +56 -0
  41. package/doc/ai/zh-CN/build-recipe.md +132 -0
  42. package/doc/ai/zh-CN/cli.md +127 -0
  43. package/doc/ai/zh-CN/contributing.md +173 -0
  44. package/doc/ai/zh-CN/extensions.md +139 -0
  45. package/doc/ai/zh-CN/language-support.md +147 -0
  46. package/doc/ai/zh-CN/troubleshooting.md +150 -0
  47. package/doc/gui-scripts.md +350 -0
  48. package/doc/gui.md +646 -0
  49. package/doc/icon.md +265 -0
  50. package/doc/implemented.md +373 -0
  51. package/doc/node-implemented.md +588 -0
  52. package/doc/node-unimplemented.md +167 -0
  53. package/doc/post/announce.md +43 -0
  54. package/doc/requirements.md +145 -0
  55. package/doc/unimplemented.md +286 -0
  56. package/doc/xbintsc.config.schema.json +67 -0
  57. package/doc/zh-CN/DESIGN.md +104 -0
  58. package/doc/zh-CN/gui-scripts.md +329 -0
  59. package/doc/zh-CN/gui.md +588 -0
  60. package/doc/zh-CN/icon.md +241 -0
  61. package/doc/zh-CN/implemented.md +365 -0
  62. package/doc/zh-CN/node-implemented.md +533 -0
  63. package/doc/zh-CN/node-unimplemented.md +141 -0
  64. package/doc/zh-CN/plan-require-node-modules.md +284 -0
  65. package/doc/zh-CN/post/announce.md +47 -0
  66. package/doc/zh-CN/requirements.md +134 -0
  67. package/doc/zh-CN/unimplemented.md +247 -0
  68. package/llms.txt +45 -0
  69. package/package.json +4 -1
  70. package/runtime/ext_gui/gui.cpp +3 -1
  71. package/runtime/ext_gui/renderer.cpp +13 -11
  72. package/runtime/ext_gui/renderer_image.cpp +12 -8
  73. package/runtime/ext_gui/renderer_shaders.h +131 -4
  74. package/runtime/ext_gui/renderer_shaders_data.h +1809 -0
  75. package/runtime/ext_gui/renderer_text.cpp +12 -8
  76. package/runtime/ext_gui/shaders.hlsl +98 -0
  77. package/runtime/ext_gui/spirv/fill.frag +19 -0
  78. package/runtime/ext_gui/spirv/fill.vert +42 -0
  79. package/runtime/ext_gui/spirv/image.frag +16 -0
  80. package/runtime/ext_gui/spirv/quad.vert +30 -0
  81. package/runtime/ext_gui/spirv/text.frag +16 -0
  82. package/scripts/build-gui-shaders.mjs +204 -0
  83. package/scripts/build-gui.ts +35 -0
  84. package/scripts/check-file-length.ts +5 -1
  85. package/src/cli/hints.ts +194 -0
  86. package/src/cli/main.ts +82 -9
  87. package/src/codegen/generator/tables.ts +60 -14
  88. package/src/diagnostics/source-text.ts +78 -0
  89. package/src/driver/bundler/graph.ts +2 -1
  90. package/src/driver/compiler.ts +3 -2
  91. package/src/lexer/scanner/strings.ts +16 -3
@@ -0,0 +1,194 @@
1
+ /**
2
+ * Documentation pointers for diagnostics.
3
+ *
4
+ * Every failure printed by the CLI ends with a `hint:` line naming the document
5
+ * that explains what to do next. The messages already say what is wrong; these
6
+ * lines say where to read more, which is what keeps an agent (or a person) from
7
+ * re-deriving the language subset one failed build at a time.
8
+ *
9
+ * The paths are relative to the repository root, so they stay clickable in
10
+ * GitHub's rendered output and on a terminal that links paths.
11
+ */
12
+
13
+ import { DiagnosticCode, type Diagnostic } from "../diagnostics/diagnostic.js";
14
+
15
+ /** Where a reader should go next. */
16
+ export interface DocumentationPointer {
17
+ /** Message printed after `hint:`. */
18
+ readonly text: string;
19
+ /** Repository-relative path to the document. */
20
+ readonly path: string;
21
+ }
22
+
23
+ export const DOC_LANGUAGE_SUPPORT = "doc/ai/language-support.md";
24
+ export const DOC_EXTENSIONS = "doc/ai/extensions.md";
25
+ export const DOC_TROUBLESHOOTING = "doc/ai/troubleshooting.md";
26
+ export const DOC_CLI = "doc/ai/cli.md";
27
+ export const DOC_REQUIREMENTS = "doc/requirements.md";
28
+
29
+ const CHECKER_HINT: DocumentationPointer = {
30
+ text: "the compiler supports a subset of TypeScript; check what is implemented before rewriting",
31
+ path: DOC_LANGUAGE_SUPPORT,
32
+ };
33
+
34
+ const REQUIREMENT_HINT: DocumentationPointer = {
35
+ text: "xbintsc needs clang 16 or newer; run `xbintsc doctor` to see the resolved toolchain",
36
+ path: DOC_REQUIREMENTS,
37
+ };
38
+
39
+ /** Hint for a clang failure: a missing or too-old clang has a specific answer. */
40
+ export function toolchainPointer(message: string): DocumentationPointer {
41
+ const usesTypedPointers = /defined with type '\[.*i8\]\*' but expected 'i8\*'/.test(message);
42
+ if (usesTypedPointers || /No C compiler found/i.test(message)) return REQUIREMENT_HINT;
43
+ return {
44
+ text: "read the clang output above, then the toolchain section of the troubleshooting guide",
45
+ path: DOC_TROUBLESHOOTING,
46
+ };
47
+ }
48
+
49
+ /** A thrown failure the CLI can name, in table order. */
50
+ export interface ThrownFailure {
51
+ /** Tested against the thrown message. */
52
+ readonly match: RegExp;
53
+ readonly code: DiagnosticCode;
54
+ readonly pointer: DocumentationPointer;
55
+ }
56
+
57
+ /**
58
+ * Failures the CLI reports as a message instead of a diagnostic. Matching on the
59
+ * text is what lets each one reach the document that actually explains it —
60
+ * a missing manifest and an unknown extension are both "cannot read something",
61
+ * but only one of them is answered by the CLI reference.
62
+ */
63
+ const THROWN_FAILURES: readonly ThrownFailure[] = [
64
+ {
65
+ match: /native extension manifest/i,
66
+ code: DiagnosticCode.IOError,
67
+ pointer: {
68
+ text: "check the manifest path and its JSON; objects are relative to the manifest file",
69
+ path: DOC_EXTENSIONS,
70
+ },
71
+ },
72
+ {
73
+ match: /Unknown extension/i,
74
+ code: DiagnosticCode.IOError,
75
+ pointer: {
76
+ text: "run `xbintsc help` for the flag, and see the built-in extensions this build ships",
77
+ path: DOC_EXTENSIONS,
78
+ },
79
+ },
80
+ {
81
+ match: /config|JSON/i,
82
+ code: DiagnosticCode.IOError,
83
+ pointer: {
84
+ text: "check the project config path and its fields",
85
+ path: DOC_CLI,
86
+ },
87
+ },
88
+ ];
89
+
90
+ /** The failure a thrown error's message describes, if any. */
91
+ export function thrownFailure(message: string): ThrownFailure | undefined {
92
+ return THROWN_FAILURES.find((failure) => failure.match.test(message));
93
+ }
94
+
95
+ /**
96
+ * The completion for an unclassified thrown error: the toolchain's own wording
97
+ * is the only thing worth directing a reader to.
98
+ */
99
+ export const UNEXPECTED_FAILURE_HINT: DocumentationPointer = {
100
+ text: "this failure has no dedicated guide; check the resolved toolchain and the source paths",
101
+ path: DOC_TROUBLESHOOTING,
102
+ };
103
+
104
+ /**
105
+ * The document that explains a diagnostic, or `undefined` for internal errors
106
+ * that no reader can act on.
107
+ */
108
+ export function documentationFor(diagnostic: Diagnostic): DocumentationPointer | undefined {
109
+ switch (diagnostic.code) {
110
+ // Lexer, parser and binder report a precise location; the guidance is the
111
+ // same subset reference.
112
+ case DiagnosticCode.UnterminatedString:
113
+ case DiagnosticCode.UnterminatedTemplate:
114
+ case DiagnosticCode.UnterminatedComment:
115
+ case DiagnosticCode.InvalidCharacter:
116
+ case DiagnosticCode.InvalidNumber:
117
+ case DiagnosticCode.InvalidEscape:
118
+ case DiagnosticCode.UnexpectedToken:
119
+ case DiagnosticCode.ExpectedToken:
120
+ case DiagnosticCode.ExpectedIdentifier:
121
+ case DiagnosticCode.UnexpectedEof:
122
+ case DiagnosticCode.InvalidAssignmentTarget:
123
+ case DiagnosticCode.DuplicateDefault:
124
+ case DiagnosticCode.InvalidTypeSyntax:
125
+ case DiagnosticCode.DuplicateIdentifier:
126
+ case DiagnosticCode.CannotFindName:
127
+ case DiagnosticCode.IllegalRedeclaration:
128
+ case DiagnosticCode.CannotAssignToConst:
129
+ case DiagnosticCode.UnsupportedFeature:
130
+ return CHECKER_HINT;
131
+
132
+ // Every codegen and bundler error carries its own message; the subset page
133
+ // is what separates "you wrote unsupported code" from "you wrote a bug".
134
+ case DiagnosticCode.CodegenError:
135
+ return CHECKER_HINT;
136
+
137
+ case DiagnosticCode.ModuleNotFound:
138
+ return {
139
+ text: "Node modules need the `node` extension (`--ext node`); third-party npm packages are not supported",
140
+ path: DOC_EXTENSIONS,
141
+ };
142
+
143
+ case DiagnosticCode.IOError:
144
+ return {
145
+ text: "check the path and the project config; paths in xbintsc.config.json resolve against that file",
146
+ path: DOC_TROUBLESHOOTING,
147
+ };
148
+
149
+ case DiagnosticCode.ToolchainError:
150
+ return toolchainPointer(diagnostic.message);
151
+
152
+ case DiagnosticCode.IncrementalCacheError:
153
+ return {
154
+ text: "the incremental cache could not be trusted; retry with `--force` or remove the cache directory",
155
+ path: DOC_CLI,
156
+ };
157
+
158
+ // The checker codes are defined but never emitted today.
159
+ case DiagnosticCode.TypeMismatch:
160
+ case DiagnosticCode.NotCallable:
161
+ case DiagnosticCode.PropertyNotFound:
162
+ case DiagnosticCode.ArgumentCountMismatch:
163
+ return CHECKER_HINT;
164
+ }
165
+ }
166
+
167
+ /** Render one `hint:` line, or `undefined` when the code has no document. */
168
+ export function formatDocumentationHint(diagnostic: Diagnostic): string | undefined {
169
+ const pointer = documentationFor(diagnostic);
170
+ if (!pointer) return undefined;
171
+ return `hint: ${pointer.text} — see ${pointer.path}`;
172
+ }
173
+
174
+ /**
175
+ * One hint line per failing diagnostic, in first-seen order and without
176
+ * repeating a document the reader has already been sent to.
177
+ */
178
+ export function documentationHints(diagnostics: readonly Diagnostic[]): string[] {
179
+ const lines: string[] = [];
180
+ const seen = new Set<string>();
181
+ for (const diagnostic of diagnostics) {
182
+ if (diagnostic.category !== "error") continue;
183
+ const line = formatDocumentationHint(diagnostic);
184
+ if (line === undefined || seen.has(line)) continue;
185
+ seen.add(line);
186
+ lines.push(line);
187
+ }
188
+ return lines;
189
+ }
190
+
191
+ /** Render one hint line for an already-classified thrown failure. */
192
+ export function formatThrownHint(pointer: DocumentationPointer): string {
193
+ return `hint: ${pointer.text} — see ${pointer.path}`;
194
+ }
package/src/cli/main.ts CHANGED
@@ -22,7 +22,16 @@ import { build, compileEntry, COMPILER_VERSION, type EmitKind } from "../driver/
22
22
  import { findRuntimeDir, platformSlug } from "../driver/paths.js";
23
23
  import { runtimeLibDir } from "../driver/runtime-lib.js";
24
24
  import { resolveToolchain } from "../driver/toolchain-provider.js";
25
- import { realRunner } from "../driver/toolchain.js";
25
+ import { realRunner, ToolchainError } from "../driver/toolchain.js";
26
+ import {
27
+ DOC_CLI,
28
+ DOC_TROUBLESHOOTING,
29
+ documentationHints,
30
+ thrownFailure,
31
+ toolchainPointer,
32
+ UNEXPECTED_FAILURE_HINT,
33
+ type DocumentationPointer,
34
+ } from "./hints.js";
26
35
  import { createDefaultRegistry, type ExtensionRegistry } from "../extensions/registry.js";
27
36
  import { bundledExtensions } from "../extensions/catalog.js";
28
37
  import { nativeExtensionFromManifest } from "../extensions/native.js";
@@ -46,6 +55,13 @@ const defaultIo: CliIo = {
46
55
  stderr: (text) => process.stderr.write(text),
47
56
  };
48
57
 
58
+ /**
59
+ * `--no-hints` state for the current {@link run} call. A module-level flag (and
60
+ * not a parameter) keeps the hint layer out of the signatures of every helper
61
+ * that can fail.
62
+ */
63
+ let hintsDisabled = false;
64
+
49
65
  interface ParsedArgs {
50
66
  readonly command?: string;
51
67
  readonly positionals: string[];
@@ -152,6 +168,46 @@ function printDiagnostics(diagnostics: readonly Diagnostic[], fileName: string,
152
168
  }
153
169
  io.stderr(formatDiagnostic(diagnostic, source) + "\n");
154
170
  }
171
+ for (const hint of documentationHints(diagnostics)) emitHint(io, hint);
172
+ }
173
+
174
+ /** Print one `hint:` line unless the caller asked for plain output. */
175
+ function emitHint(io: CliIo, line: string): void {
176
+ if (hintsDisabled) return;
177
+ io.stderr(`${line}\n`);
178
+ }
179
+
180
+ /**
181
+ * Report a failure that threw instead of becoming a diagnostic, ending with the
182
+ * documentation pointer that explains it. Without this a toolchain failure is a
183
+ * raw stack trace, which tells a reader nothing about which document answers
184
+ * the question.
185
+ */
186
+ function reportThrown(error: unknown, io: CliIo): number {
187
+ const message =
188
+ error instanceof ToolchainError
189
+ ? `xbintsc: ${error.message}`
190
+ : `xbintsc: ${error instanceof Error ? error.message : String(error)}`;
191
+ io.stderr(`${message.trimEnd()}\n`);
192
+ const text = error instanceof ToolchainError ? error.message : message;
193
+ const pointer = error instanceof ToolchainError ? toolchainPointer(text) : thrownPointer(text);
194
+ emitHint(io, `hint: ${pointer.text} — see ${pointer.path}`);
195
+ return 1;
196
+ }
197
+
198
+ /** The document that explains a failure the CLI reports as a plain message. */
199
+ function thrownPointer(message: string): DocumentationPointer {
200
+ return thrownFailure(message)?.pointer ?? UNEXPECTED_FAILURE_HINT;
201
+ }
202
+
203
+ /** Report a project-config failure, which never becomes a diagnostic bag. */
204
+ function reportConfigError(error: string, io: CliIo): number {
205
+ io.stderr(`xbintsc: ${error}\n`);
206
+ const pointer = /JSON/i.test(error)
207
+ ? `hint: fix the JSON; an unreadable or malformed config fails the build — see ${DOC_TROUBLESHOOTING}`
208
+ : `hint: check the config path and its fields — see ${DOC_CLI}`;
209
+ emitHint(io, pointer);
210
+ return 1;
155
211
  }
156
212
 
157
213
  const HELP = `xbintsc ${COMPILER_VERSION} - TypeScript binary compiler
@@ -181,6 +237,14 @@ Options:
181
237
  --app-id <id> macOS bundle identifier (e.g. com.example.demo)
182
238
  --force Ignore the incremental cache
183
239
  --verbose Print progress information
240
+ --no-hints Do not append the "hint: … see <doc>" line to failures
241
+
242
+ xbintsc compiles a subset of TypeScript to a native binary. Types are erased
243
+ (never checked), Node modules need --ext node, third-party npm imports are not
244
+ supported, and clang 16+ is required to link.
245
+
246
+ Docs: doc/ai/README.md (task guide) - doc/ai/language-support.md (subset) -
247
+ doc/ai/troubleshooting.md (failures) - llms.txt (index of every document)
184
248
  `;
185
249
 
186
250
  function firstLine(text: string): string {
@@ -281,6 +345,19 @@ function listFlag(flags: Map<string, string | boolean>, name: string): string[]
281
345
  }
282
346
 
283
347
  export function run(argv: readonly string[], io: CliIo = defaultIo): number {
348
+ const disableHints = parseArgs(argv).flags.has("no-hints");
349
+ const previous = hintsDisabled;
350
+ hintsDisabled = disableHints;
351
+ try {
352
+ return dispatch(argv, io);
353
+ } catch (error) {
354
+ return reportThrown(error, io);
355
+ } finally {
356
+ hintsDisabled = previous;
357
+ }
358
+ }
359
+
360
+ function dispatch(argv: readonly string[], io: CliIo): number {
284
361
  const args = parseArgs(argv);
285
362
  const command = args.command;
286
363
 
@@ -301,13 +378,11 @@ export function run(argv: readonly string[], io: CliIo = defaultIo): number {
301
378
  if (command === "emit") {
302
379
  const cliEntry = args.positionals[0];
303
380
  const resolved = resolveProjectConfig(args.flags, cliEntry);
304
- if (!resolved.ok) {
305
- io.stderr(`xbintsc: ${resolved.error}\n`);
306
- return 1;
307
- }
381
+ if (!resolved.ok) return reportConfigError(resolved.error, io);
308
382
  const entry = cliEntry ?? resolved.config.entry;
309
383
  if (!entry) {
310
384
  io.stderr(`xbintsc: emit requires a source file (or 'entry' in ${CONFIG_FILE_NAME})\n`);
385
+ emitHint(io, `hint: pass the entry file, or set 'entry' in ${CONFIG_FILE_NAME} — see ${DOC_CLI}`);
311
386
  return 1;
312
387
  }
313
388
  const extNames = listFlag(args.flags, "ext") ?? resolved.config.extensions ?? [];
@@ -325,14 +400,12 @@ export function run(argv: readonly string[], io: CliIo = defaultIo): number {
325
400
  if (command === "build" || command === "run") {
326
401
  const cliEntry = args.positionals[0];
327
402
  const resolved = resolveProjectConfig(args.flags, cliEntry);
328
- if (!resolved.ok) {
329
- io.stderr(`xbintsc: ${resolved.error}\n`);
330
- return 1;
331
- }
403
+ if (!resolved.ok) return reportConfigError(resolved.error, io);
332
404
  const config = resolved.config;
333
405
  const entry = cliEntry ?? config.entry;
334
406
  if (!entry) {
335
407
  io.stderr(`xbintsc: ${command} requires a source file (or 'entry' in ${CONFIG_FILE_NAME})\n`);
408
+ emitHint(io, `hint: pass the entry file, or set 'entry' in ${CONFIG_FILE_NAME} — see ${DOC_CLI}`);
336
409
  return 1;
337
410
  }
338
411
  const emit = (args.flags.get("emit") as EmitKind | undefined) ?? "exe";
@@ -327,27 +327,73 @@ export const KIND_NAMES = new Map<number, string>([
327
327
  [SyntaxKind.SwitchStatement, "switch statement"],
328
328
  ]);
329
329
 
330
+ /**
331
+ * The UTF-8 bytes of one code point, which must be an ASCII byte or a Unicode
332
+ * scalar (`0..0x10ffff`, not a surrogate). Encoded by hand rather than with
333
+ * `String.fromCharCode`/`String.fromCodePoint`, which a self-hosted build would
334
+ * run through the runtime and encode a second time.
335
+ */
336
+ export function codePointBytes(code: number): number[] {
337
+ if (code < 0x80) return [code];
338
+ if (code < 0x800) return [0xc0 | (code >> 6), 0x80 | (code & 0x3f)];
339
+ if (code < 0x10000) return [0xe0 | (code >> 12), 0x80 | ((code >> 6) & 0x3f), 0x80 | (code & 0x3f)];
340
+ return [
341
+ 0xf0 | (code >> 18),
342
+ 0x80 | ((code >> 12) & 0x3f),
343
+ 0x80 | ((code >> 6) & 0x3f),
344
+ 0x80 | (code & 0x3f),
345
+ ];
346
+ }
347
+
348
+ /**
349
+ * The bytes a string literal stands for.
350
+ *
351
+ * A compiled string is a sequence of UTF-8 bytes (`"é".length` is 1 and
352
+ * `"😀".length` is 4 - see doc/ai/language-support.md). The text reaching this
353
+ * point is whatever a host's strings are, and both shapes have to produce the
354
+ * same bytes:
355
+ *
356
+ * - a UTF-16 host decodes a file into characters, so `é` is one code unit
357
+ * (0xe9) and has to be encoded here;
358
+ * - the self-hosted runtime has no UTF-8 decoder, so `readFileSync` hands over
359
+ * a file's bytes, and `é` is already `c3 a9`. Encoding that again is what
360
+ * turned the three bytes of `—` into six and broke the self-hosting fixpoint.
361
+ *
362
+ * A code unit above 0xff can only be text, so its presence settles which shape
363
+ * this is: text is encoded (once), and text that is already a byte sequence is
364
+ * emitted as it is. A source file read on either host therefore lands on the
365
+ * same bytes, which is what the fixpoint needs.
366
+ */
330
367
  export function utf8Bytes(text: string): number[] {
368
+ if (!hasMultiByteUnit(text)) {
369
+ const bytes: number[] = [];
370
+ for (let index = 0; index < text.length; index++) bytes.push(text.charCodeAt(index) & 0xff);
371
+ return bytes;
372
+ }
331
373
  const bytes: number[] = [];
332
- for (const character of text) {
333
- const code = character.codePointAt(0)!;
334
- if (code < 0x80) bytes.push(code);
335
- else if (code < 0x800) {
336
- bytes.push(0xc0 | (code >> 6), 0x80 | (code & 0x3f));
337
- } else if (code < 0x10000) {
338
- bytes.push(0xe0 | (code >> 12), 0x80 | ((code >> 6) & 0x3f), 0x80 | (code & 0x3f));
339
- } else {
340
- bytes.push(
341
- 0xf0 | (code >> 18),
342
- 0x80 | ((code >> 12) & 0x3f),
343
- 0x80 | ((code >> 6) & 0x3f),
344
- 0x80 | (code & 0x3f),
345
- );
374
+ const length = text.length;
375
+ for (let index = 0; index < length; index++) {
376
+ let code = text.charCodeAt(index);
377
+ if (code >= 0xd800 && code <= 0xdbff && index + 1 < length) {
378
+ const low = text.charCodeAt(index + 1);
379
+ if (low >= 0xdc00 && low <= 0xdfff) {
380
+ code = 0x10000 + ((code - 0xd800) << 10) + (low - 0xdc00);
381
+ index++;
382
+ }
346
383
  }
384
+ for (const byte of codePointBytes(code)) bytes.push(byte);
347
385
  }
348
386
  return bytes;
349
387
  }
350
388
 
389
+ /** Whether any code unit is above one byte, which means text, not bytes. */
390
+ function hasMultiByteUnit(text: string): boolean {
391
+ for (let index = 0; index < text.length; index++) {
392
+ if (text.charCodeAt(index) > 0xff) return true;
393
+ }
394
+ return false;
395
+ }
396
+
351
397
  export function escapeBytes(bytes: readonly number[]): string {
352
398
  return bytes
353
399
  .map((byte) => {
@@ -0,0 +1,78 @@
1
+ /**
2
+ * Source text is a byte sequence.
3
+ *
4
+ * A compiled string is a sequence of UTF-8 bytes (`"e".length` is 1 for an
5
+ * accent and 4 for an emoji - see doc/ai/language-support.md), and the compiler
6
+ * is built by itself: there the self-hosted `readFileSync` hands the scanner a
7
+ * file's own bytes, one per code unit, and the runtime has no UTF-8 decoder to
8
+ * change that. A host whose strings are UTF-16 must be given the text in the
9
+ * same shape, or the same program emits different bytes depending on which
10
+ * generation of the compiler read it - which is how a non-ASCII literal ended
11
+ * up double-encoded and broke the self-hosting fixpoint.
12
+ *
13
+ * `sourceTextBytes` is that shape: a source file's bytes, one code unit each.
14
+ * The driver reads every source file through it, and codegen emits those bytes
15
+ * unchanged.
16
+ */
17
+
18
+ /**
19
+ * The UTF-8 bytes of `text`, one code unit each. Text that is already a byte
20
+ * sequence is kept as it is, so this is idempotent and safe to apply to a read
21
+ * on either host.
22
+ */
23
+ export function sourceTextBytes(text: string): string {
24
+ if (!hasCharacter(text)) return startsWithBom(text) ? text.slice(3) : text;
25
+ const encoded = encodeUtf8(text);
26
+ return startsWithBom(encoded) ? encoded.slice(3) : encoded;
27
+ }
28
+
29
+ /** Encode text as UTF-8, spelled one code unit per byte. */
30
+ function encodeUtf8(text: string): string {
31
+ const chunks: string[] = [];
32
+ const units: number[] = [];
33
+ for (let index = 0; index < text.length; index++) {
34
+ let code = text.charCodeAt(index);
35
+ if (code >= 0xd800 && code <= 0xdbff && index + 1 < text.length) {
36
+ const low = text.charCodeAt(index + 1);
37
+ if (low >= 0xdc00 && low <= 0xdfff) {
38
+ code = 0x10000 + ((code - 0xd800) << 10) + (low - 0xdc00);
39
+ index++;
40
+ }
41
+ }
42
+ if (code < 0x80) units.push(code);
43
+ else if (code < 0x800) units.push(0xc0 | (code >> 6), 0x80 | (code & 0x3f));
44
+ else if (code < 0x10000) {
45
+ units.push(0xe0 | (code >> 12), 0x80 | ((code >> 6) & 0x3f), 0x80 | (code & 0x3f));
46
+ } else {
47
+ units.push(
48
+ 0xf0 | (code >> 18),
49
+ 0x80 | ((code >> 12) & 0x3f),
50
+ 0x80 | ((code >> 6) & 0x3f),
51
+ 0x80 | (code & 0x3f),
52
+ );
53
+ }
54
+ if (units.length >= 512) flushUnits(units, chunks);
55
+ }
56
+ flushUnits(units, chunks);
57
+ return chunks.join("");
58
+ }
59
+
60
+ /** Whether any code unit is above one byte, which means text, not bytes. */
61
+ function hasCharacter(text: string): boolean {
62
+ for (let index = 0; index < text.length; index++) {
63
+ if (text.charCodeAt(index) > 0xff) return true;
64
+ }
65
+ return false;
66
+ }
67
+
68
+ /** Whether a decoded read started with a UTF-8 byte order mark. */
69
+ function startsWithBom(text: string): boolean {
70
+ return text.charCodeAt(0) === 0xef && text.charCodeAt(1) === 0xbb && text.charCodeAt(2) === 0xbf;
71
+ }
72
+
73
+ /** Append the pending code units as one chunk of the byte string. */
74
+ function flushUnits(units: number[], chunks: string[]): void {
75
+ if (units.length === 0) return;
76
+ chunks.push(String.fromCharCode(...units));
77
+ units.length = 0;
78
+ }
@@ -9,6 +9,7 @@ import { SyntaxKind, type ExportDeclaration, type ImportDeclaration, type Statem
9
9
  import { bind } from "../../binder/binder.js";
10
10
  import { DiagnosticBag, DiagnosticCode } from "../../diagnostics/diagnostic.js";
11
11
  import { SourceFile } from "../../diagnostics/source.js";
12
+ import { sourceTextBytes } from "../../diagnostics/source-text.js";
12
13
  import { Parser } from "../../parser/parser.js";
13
14
  import type { AssetLoader } from "../../extensions/registry.js";
14
15
  import { classifyDependency } from "./resolve.js";
@@ -53,7 +54,7 @@ export function loadGraph(
53
54
  return undefined;
54
55
  }
55
56
  visiting.add(path);
56
- const raw = readFileSync(path, "utf8");
57
+ const raw = sourceTextBytes(readFileSync(path, "utf8"));
57
58
  /* An extension may rewrite an imported asset (e.g. `.html`) into a TS
58
59
  * module before it is parsed. The core only sees the transformed text. */
59
60
  const loader = assetLoaders[extname(path).toLowerCase()];
@@ -11,6 +11,7 @@ import { existsSync, mkdirSync, readFileSync, readdirSync, writeFileSync } from
11
11
  import { basename, extname, join, resolve } from "node:path";
12
12
  import { DiagnosticBag, DiagnosticCode, type Diagnostic } from "../diagnostics/diagnostic.js";
13
13
  import { SourceFile } from "../diagnostics/source.js";
14
+ import { sourceTextBytes } from "../diagnostics/source-text.js";
14
15
  import { Parser } from "../parser/parser.js";
15
16
  import { generate } from "../codegen/llvm.js";
16
17
  import { SyntaxKind } from "../ast/nodes.js";
@@ -106,7 +107,7 @@ function externalModuleSpecifiers(registry: ExtensionRegistry): Set<string> {
106
107
  */
107
108
  export function compileEntry(entryPath: string, extensions?: ExtensionRegistry): CompileStringResult {
108
109
  const absoluteEntry = resolve(entryPath);
109
- const sourceText = readFileSync(absoluteEntry, "utf8");
110
+ const sourceText = sourceTextBytes(readFileSync(absoluteEntry, "utf8"));
110
111
  const file = new SourceFile(absoluteEntry, sourceText);
111
112
  const diagnostics = new DiagnosticBag();
112
113
  const parser = new Parser(file, diagnostics);
@@ -170,7 +171,7 @@ export function build(entryPath: string, options: BuildOptions = {}): BuildResul
170
171
  const optimize = options.optimize ?? "2";
171
172
 
172
173
  const absoluteEntry = resolve(entryPath);
173
- const sourceText = readFileSync(absoluteEntry, "utf8");
174
+ const sourceText = sourceTextBytes(readFileSync(absoluteEntry, "utf8"));
174
175
 
175
176
  const baseName = basename(absoluteEntry, extname(absoluteEntry));
176
177
  const outputPath =
@@ -6,8 +6,21 @@
6
6
  import { DiagnosticCode } from "../../diagnostics/diagnostic.js";
7
7
  import { Token, TokenKind } from "../token.js";
8
8
  import { Char, isDigit, isLineBreak } from "../char.js";
9
+ import { codePointBytes } from "../../codegen/generator/tables.js";
9
10
  import type { Scanner } from "../scanner.js";
10
11
 
12
+ /**
13
+ * The bytes one escape sequence stands for. An escape names a code point, and a
14
+ * literal holds the UTF-8 bytes of that code point - the same bytes
15
+ * `String.fromCodePoint` produces at run time - spelled as one code unit per
16
+ * byte, so both hosts agree on what a literal's text is.
17
+ */
18
+ function escapeText(code: number): string {
19
+ let text = "";
20
+ for (const byte of codePointBytes(code)) text += String.fromCharCode(byte);
21
+ return text;
22
+ }
23
+
11
24
  export interface StringMethods {
12
25
  scanString(this: Scanner, start: number, precededByLineBreak: boolean, quote: number): Token;
13
26
  scanEscapeSequence(this: Scanner): string;
@@ -65,16 +78,16 @@ export const stringMethods: StringMethods = {
65
78
  const close = this.text.indexOf("}", this.pos);
66
79
  const hex = this.text.slice(this.pos + 1, close);
67
80
  this.pos = close + 1;
68
- return String.fromCodePoint(parseInt(hex, 16) || 0);
81
+ return escapeText(parseInt(hex, 16) || 0);
69
82
  } else {
70
83
  const hex = this.text.slice(this.pos, this.pos + 4);
71
84
  this.pos += 4;
72
- return String.fromCharCode(parseInt(hex, 16) || 0);
85
+ return escapeText(parseInt(hex, 16) || 0);
73
86
  }
74
87
  case Char.LowerX: {
75
88
  const hex = this.text.slice(this.pos + 1, this.pos + 3);
76
89
  this.pos += 3;
77
- return String.fromCharCode(parseInt(hex, 16) || 0);
90
+ return escapeText(parseInt(hex, 16) || 0);
78
91
  }
79
92
  case Char.Zero:
80
93
  // \0 is only a null escape when not followed by a digit.