openalgo-script 0.1.0-alpha.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (266) hide show
  1. package/LICENSE +201 -0
  2. package/NOTICE +12 -0
  3. package/README.md +255 -0
  4. package/dist/core/ast/annotations.d.ts +56 -0
  5. package/dist/core/ast/annotations.d.ts.map +1 -0
  6. package/dist/core/ast/annotations.js +27 -0
  7. package/dist/core/ast/annotations.js.map +1 -0
  8. package/dist/core/ast/build.d.ts +25 -0
  9. package/dist/core/ast/build.d.ts.map +1 -0
  10. package/dist/core/ast/build.js +26 -0
  11. package/dist/core/ast/build.js.map +1 -0
  12. package/dist/core/ast/children.d.ts +15 -0
  13. package/dist/core/ast/children.d.ts.map +1 -0
  14. package/dist/core/ast/children.js +103 -0
  15. package/dist/core/ast/children.js.map +1 -0
  16. package/dist/core/ast/expressions.d.ts +178 -0
  17. package/dist/core/ast/expressions.d.ts.map +1 -0
  18. package/dist/core/ast/expressions.js +50 -0
  19. package/dist/core/ast/expressions.js.map +1 -0
  20. package/dist/core/ast/header.d.ts +50 -0
  21. package/dist/core/ast/header.d.ts.map +1 -0
  22. package/dist/core/ast/header.js +2 -0
  23. package/dist/core/ast/header.js.map +1 -0
  24. package/dist/core/ast/index.d.ts +30 -0
  25. package/dist/core/ast/index.d.ts.map +1 -0
  26. package/dist/core/ast/index.js +21 -0
  27. package/dist/core/ast/index.js.map +1 -0
  28. package/dist/core/ast/name.d.ts +24 -0
  29. package/dist/core/ast/name.d.ts.map +1 -0
  30. package/dist/core/ast/name.js +2 -0
  31. package/dist/core/ast/name.js.map +1 -0
  32. package/dist/core/ast/node.d.ts +34 -0
  33. package/dist/core/ast/node.d.ts.map +1 -0
  34. package/dist/core/ast/node.js +70 -0
  35. package/dist/core/ast/node.js.map +1 -0
  36. package/dist/core/ast/script.d.ts +57 -0
  37. package/dist/core/ast/script.d.ts.map +1 -0
  38. package/dist/core/ast/script.js +2 -0
  39. package/dist/core/ast/script.js.map +1 -0
  40. package/dist/core/ast/statements.d.ts +161 -0
  41. package/dist/core/ast/statements.d.ts.map +1 -0
  42. package/dist/core/ast/statements.js +10 -0
  43. package/dist/core/ast/statements.js.map +1 -0
  44. package/dist/core/ast/walk.d.ts +39 -0
  45. package/dist/core/ast/walk.d.ts.map +1 -0
  46. package/dist/core/ast/walk.js +54 -0
  47. package/dist/core/ast/walk.js.map +1 -0
  48. package/dist/core/catalogue/catalogue.generated.d.ts +1590 -0
  49. package/dist/core/catalogue/catalogue.generated.d.ts.map +1 -0
  50. package/dist/core/catalogue/catalogue.generated.js +162 -0
  51. package/dist/core/catalogue/catalogue.generated.js.map +1 -0
  52. package/dist/core/catalogue/index.d.ts +15 -0
  53. package/dist/core/catalogue/index.d.ts.map +1 -0
  54. package/dist/core/catalogue/index.js +4 -0
  55. package/dist/core/catalogue/index.js.map +1 -0
  56. package/dist/core/catalogue/lookup.d.ts +14 -0
  57. package/dist/core/catalogue/lookup.d.ts.map +1 -0
  58. package/dist/core/catalogue/lookup.js +19 -0
  59. package/dist/core/catalogue/lookup.js.map +1 -0
  60. package/dist/core/catalogue/template.d.ts +15 -0
  61. package/dist/core/catalogue/template.d.ts.map +1 -0
  62. package/dist/core/catalogue/template.js +33 -0
  63. package/dist/core/catalogue/template.js.map +1 -0
  64. package/dist/core/catalogue/types.d.ts +34 -0
  65. package/dist/core/catalogue/types.d.ts.map +1 -0
  66. package/dist/core/catalogue/types.js +6 -0
  67. package/dist/core/catalogue/types.js.map +1 -0
  68. package/dist/core/catalogue/values.generated.d.ts +565 -0
  69. package/dist/core/catalogue/values.generated.d.ts.map +1 -0
  70. package/dist/core/catalogue/values.generated.js +6 -0
  71. package/dist/core/catalogue/values.generated.js.map +1 -0
  72. package/dist/core/diagnostics/collector.d.ts +46 -0
  73. package/dist/core/diagnostics/collector.d.ts.map +1 -0
  74. package/dist/core/diagnostics/collector.js +55 -0
  75. package/dist/core/diagnostics/collector.js.map +1 -0
  76. package/dist/core/diagnostics/diagnostic.d.ts +40 -0
  77. package/dist/core/diagnostics/diagnostic.d.ts.map +1 -0
  78. package/dist/core/diagnostics/diagnostic.js +29 -0
  79. package/dist/core/diagnostics/diagnostic.js.map +1 -0
  80. package/dist/core/diagnostics/index.d.ts +12 -0
  81. package/dist/core/diagnostics/index.d.ts.map +1 -0
  82. package/dist/core/diagnostics/index.js +3 -0
  83. package/dist/core/diagnostics/index.js.map +1 -0
  84. package/dist/core/index.d.ts +28 -0
  85. package/dist/core/index.d.ts.map +1 -0
  86. package/dist/core/index.js +22 -0
  87. package/dist/core/index.js.map +1 -0
  88. package/dist/core/lex/blocks.d.ts +51 -0
  89. package/dist/core/lex/blocks.d.ts.map +1 -0
  90. package/dist/core/lex/blocks.js +56 -0
  91. package/dist/core/lex/blocks.js.map +1 -0
  92. package/dist/core/lex/characters.d.ts +38 -0
  93. package/dist/core/lex/characters.d.ts.map +1 -0
  94. package/dist/core/lex/characters.js +50 -0
  95. package/dist/core/lex/characters.js.map +1 -0
  96. package/dist/core/lex/comments.d.ts +50 -0
  97. package/dist/core/lex/comments.d.ts.map +1 -0
  98. package/dist/core/lex/comments.js +64 -0
  99. package/dist/core/lex/comments.js.map +1 -0
  100. package/dist/core/lex/index.d.ts +12 -0
  101. package/dist/core/lex/index.d.ts.map +1 -0
  102. package/dist/core/lex/index.js +12 -0
  103. package/dist/core/lex/index.js.map +1 -0
  104. package/dist/core/lex/layout.d.ts +40 -0
  105. package/dist/core/lex/layout.d.ts.map +1 -0
  106. package/dist/core/lex/layout.js +143 -0
  107. package/dist/core/lex/layout.js.map +1 -0
  108. package/dist/core/lex/lexer.d.ts +13 -0
  109. package/dist/core/lex/lexer.d.ts.map +1 -0
  110. package/dist/core/lex/lexer.js +419 -0
  111. package/dist/core/lex/lexer.js.map +1 -0
  112. package/dist/core/lex/numbers.d.ts +22 -0
  113. package/dist/core/lex/numbers.d.ts.map +1 -0
  114. package/dist/core/lex/numbers.js +58 -0
  115. package/dist/core/lex/numbers.js.map +1 -0
  116. package/dist/core/lex/statements.d.ts +24 -0
  117. package/dist/core/lex/statements.d.ts.map +1 -0
  118. package/dist/core/lex/statements.js +90 -0
  119. package/dist/core/lex/statements.js.map +1 -0
  120. package/dist/core/lex/strings.d.ts +27 -0
  121. package/dist/core/lex/strings.d.ts.map +1 -0
  122. package/dist/core/lex/strings.js +92 -0
  123. package/dist/core/lex/strings.js.map +1 -0
  124. package/dist/core/lex/unexpected.d.ts +31 -0
  125. package/dist/core/lex/unexpected.d.ts.map +1 -0
  126. package/dist/core/lex/unexpected.js +114 -0
  127. package/dist/core/lex/unexpected.js.map +1 -0
  128. package/dist/core/parse/annotations.d.ts +4 -0
  129. package/dist/core/parse/annotations.d.ts.map +1 -0
  130. package/dist/core/parse/annotations.js +49 -0
  131. package/dist/core/parse/annotations.js.map +1 -0
  132. package/dist/core/parse/blocks.d.ts +18 -0
  133. package/dist/core/parse/blocks.d.ts.map +1 -0
  134. package/dist/core/parse/blocks.js +47 -0
  135. package/dist/core/parse/blocks.js.map +1 -0
  136. package/dist/core/parse/branches.d.ts +17 -0
  137. package/dist/core/parse/branches.d.ts.map +1 -0
  138. package/dist/core/parse/branches.js +66 -0
  139. package/dist/core/parse/branches.js.map +1 -0
  140. package/dist/core/parse/cursor.d.ts +119 -0
  141. package/dist/core/parse/cursor.d.ts.map +1 -0
  142. package/dist/core/parse/cursor.js +305 -0
  143. package/dist/core/parse/cursor.js.map +1 -0
  144. package/dist/core/parse/expressions.d.ts +57 -0
  145. package/dist/core/parse/expressions.d.ts.map +1 -0
  146. package/dist/core/parse/expressions.js +373 -0
  147. package/dist/core/parse/expressions.js.map +1 -0
  148. package/dist/core/parse/functions.d.ts +16 -0
  149. package/dist/core/parse/functions.d.ts.map +1 -0
  150. package/dist/core/parse/functions.js +92 -0
  151. package/dist/core/parse/functions.js.map +1 -0
  152. package/dist/core/parse/index.d.ts +13 -0
  153. package/dist/core/parse/index.d.ts.map +1 -0
  154. package/dist/core/parse/index.js +13 -0
  155. package/dist/core/parse/index.js.map +1 -0
  156. package/dist/core/parse/loops.d.ts +13 -0
  157. package/dist/core/parse/loops.d.ts.map +1 -0
  158. package/dist/core/parse/loops.js +78 -0
  159. package/dist/core/parse/loops.js.map +1 -0
  160. package/dist/core/parse/names.d.ts +34 -0
  161. package/dist/core/parse/names.d.ts.map +1 -0
  162. package/dist/core/parse/names.js +122 -0
  163. package/dist/core/parse/names.js.map +1 -0
  164. package/dist/core/parse/script.d.ts +23 -0
  165. package/dist/core/parse/script.d.ts.map +1 -0
  166. package/dist/core/parse/script.js +55 -0
  167. package/dist/core/parse/script.js.map +1 -0
  168. package/dist/core/parse/statements.d.ts +14 -0
  169. package/dist/core/parse/statements.d.ts.map +1 -0
  170. package/dist/core/parse/statements.js +292 -0
  171. package/dist/core/parse/statements.js.map +1 -0
  172. package/dist/core/parse/switches.d.ts +16 -0
  173. package/dist/core/parse/switches.d.ts.map +1 -0
  174. package/dist/core/parse/switches.js +98 -0
  175. package/dist/core/parse/switches.js.map +1 -0
  176. package/dist/core/render/index.d.ts +10 -0
  177. package/dist/core/render/index.d.ts.map +1 -0
  178. package/dist/core/render/index.js +10 -0
  179. package/dist/core/render/index.js.map +1 -0
  180. package/dist/core/render/terminal.d.ts +11 -0
  181. package/dist/core/render/terminal.d.ts.map +1 -0
  182. package/dist/core/render/terminal.js +51 -0
  183. package/dist/core/render/terminal.js.map +1 -0
  184. package/dist/core/source/index.d.ts +7 -0
  185. package/dist/core/source/index.d.ts.map +1 -0
  186. package/dist/core/source/index.js +2 -0
  187. package/dist/core/source/index.js.map +1 -0
  188. package/dist/core/source/source.d.ts +41 -0
  189. package/dist/core/source/source.d.ts.map +1 -0
  190. package/dist/core/source/source.js +67 -0
  191. package/dist/core/source/source.js.map +1 -0
  192. package/dist/core/span/index.d.ts +3 -0
  193. package/dist/core/span/index.d.ts.map +1 -0
  194. package/dist/core/span/index.js +2 -0
  195. package/dist/core/span/index.js.map +1 -0
  196. package/dist/core/span/span.d.ts +67 -0
  197. package/dist/core/span/span.d.ts.map +1 -0
  198. package/dist/core/span/span.js +27 -0
  199. package/dist/core/span/span.js.map +1 -0
  200. package/dist/core/tokens/index.d.ts +8 -0
  201. package/dist/core/tokens/index.d.ts.map +1 -0
  202. package/dist/core/tokens/index.js +2 -0
  203. package/dist/core/tokens/index.js.map +1 -0
  204. package/dist/core/tokens/kind.d.ts +53 -0
  205. package/dist/core/tokens/kind.d.ts.map +1 -0
  206. package/dist/core/tokens/kind.js +94 -0
  207. package/dist/core/tokens/kind.js.map +1 -0
  208. package/dist/core/tokens/token.d.ts +36 -0
  209. package/dist/core/tokens/token.d.ts.map +1 -0
  210. package/dist/core/tokens/token.js +2 -0
  211. package/dist/core/tokens/token.js.map +1 -0
  212. package/package.json +52 -0
  213. package/spec/README.md +48 -0
  214. package/spec/errors.json +3204 -0
  215. package/src/core/ast/annotations.ts +68 -0
  216. package/src/core/ast/build.ts +35 -0
  217. package/src/core/ast/children.ts +110 -0
  218. package/src/core/ast/expressions.ts +254 -0
  219. package/src/core/ast/header.ts +53 -0
  220. package/src/core/ast/index.ts +102 -0
  221. package/src/core/ast/name.ts +24 -0
  222. package/src/core/ast/node.ts +120 -0
  223. package/src/core/ast/script.ts +60 -0
  224. package/src/core/ast/statements.ts +193 -0
  225. package/src/core/ast/walk.ts +67 -0
  226. package/src/core/catalogue/catalogue.generated.ts +171 -0
  227. package/src/core/catalogue/index.ts +20 -0
  228. package/src/core/catalogue/lookup.ts +23 -0
  229. package/src/core/catalogue/template.ts +38 -0
  230. package/src/core/catalogue/types.ts +37 -0
  231. package/src/core/catalogue/values.generated.ts +160 -0
  232. package/src/core/diagnostics/collector.ts +87 -0
  233. package/src/core/diagnostics/diagnostic.ts +72 -0
  234. package/src/core/diagnostics/index.ts +12 -0
  235. package/src/core/index.ts +146 -0
  236. package/src/core/lex/blocks.ts +92 -0
  237. package/src/core/lex/characters.ts +56 -0
  238. package/src/core/lex/comments.ts +75 -0
  239. package/src/core/lex/index.ts +11 -0
  240. package/src/core/lex/layout.ts +188 -0
  241. package/src/core/lex/lexer.ts +486 -0
  242. package/src/core/lex/numbers.ts +77 -0
  243. package/src/core/lex/statements.ts +94 -0
  244. package/src/core/lex/strings.ts +119 -0
  245. package/src/core/lex/unexpected.ts +125 -0
  246. package/src/core/parse/annotations.ts +60 -0
  247. package/src/core/parse/blocks.ts +52 -0
  248. package/src/core/parse/branches.ts +77 -0
  249. package/src/core/parse/cursor.ts +348 -0
  250. package/src/core/parse/expressions.ts +441 -0
  251. package/src/core/parse/functions.ts +109 -0
  252. package/src/core/parse/index.ts +12 -0
  253. package/src/core/parse/loops.ts +97 -0
  254. package/src/core/parse/names.ts +135 -0
  255. package/src/core/parse/script.ts +64 -0
  256. package/src/core/parse/statements.ts +329 -0
  257. package/src/core/parse/switches.ts +105 -0
  258. package/src/core/render/index.ts +9 -0
  259. package/src/core/render/terminal.ts +62 -0
  260. package/src/core/source/index.ts +6 -0
  261. package/src/core/source/source.ts +102 -0
  262. package/src/core/span/index.ts +2 -0
  263. package/src/core/span/span.ts +84 -0
  264. package/src/core/tokens/index.ts +15 -0
  265. package/src/core/tokens/kind.ts +123 -0
  266. package/src/core/tokens/token.ts +39 -0
@@ -0,0 +1,62 @@
1
+ import type { Diagnostic } from '../diagnostics/index.js';
2
+ import type { SourceFile } from '../source/index.js';
3
+
4
+ /**
5
+ * The renderer for a terminal, and the shape is the one errors.md section 2
6
+ * fixes as the contract:
7
+ *
8
+ * 14 | len = 9
9
+ * | ^^^
10
+ * OS2002: len is already declared at line 1, so a second one cannot be declared here.
11
+ * Fix: rename this one, or drop the inner declaration and let the assignment update the len at line 1.
12
+ *
13
+ * The line, then the caret, then what happened, then what to do. The fix is on
14
+ * its own line because it is the line the reader acts on, and a reader who has
15
+ * understood the message from the caret alone should be able to skip to it.
16
+ */
17
+
18
+ /** A tab is one column wide here, so the caret below the line still lines up. */
19
+ const TAB_WIDTH_FOR_DISPLAY = ' ';
20
+
21
+ function countCharacters(text: string): number {
22
+ // Code points, not code units: a terminal draws one character per code point,
23
+ // and padding by code units would push the caret right by one for every
24
+ // astral character earlier on the line. See the span module on columns.
25
+ return [...text].length;
26
+ }
27
+
28
+ export function renderDiagnostic(file: SourceFile, diagnostic: Diagnostic): string {
29
+ const { line, column, length } = diagnostic.span;
30
+ const shown = file.lineText(line).replace(/\t/g, TAB_WIDTH_FOR_DISPLAY);
31
+
32
+ const lineNumber = String(line);
33
+ const gutter = ' '.repeat(lineNumber.length);
34
+
35
+ const beforeCaret = shown.slice(0, Math.max(column - 1, 0));
36
+ // A span that runs past the end of its line, which an unterminated bracket
37
+ // produces, is drawn to the end of the line rather than off it.
38
+ const underCaret = shown.slice(beforeCaret.length, beforeCaret.length + length);
39
+ const caretWidth = Math.max(countCharacters(underCaret), 1);
40
+
41
+ return [
42
+ `${lineNumber} | ${shown}`,
43
+ `${gutter} | ${' '.repeat(countCharacters(beforeCaret))}${'^'.repeat(caretWidth)}`,
44
+ `${diagnostic.code}: ${diagnostic.message}`,
45
+ `Fix: ${diagnostic.fix}`,
46
+ ].join('\n');
47
+ }
48
+
49
+ /**
50
+ * Every diagnostic, in the order a reader walks the file, separated by a blank
51
+ * line. The file's name is printed once above them rather than on each one,
52
+ * because a compile reports on one file and repeating its name on twenty
53
+ * diagnostics buries the twenty.
54
+ */
55
+ export function renderDiagnostics(
56
+ file: SourceFile,
57
+ diagnostics: readonly Diagnostic[],
58
+ ): string {
59
+ if (diagnostics.length === 0) return '';
60
+ const body = diagnostics.map((diagnostic) => renderDiagnostic(file, diagnostic));
61
+ return [`${file.name}`, ...body].join('\n\n');
62
+ }
@@ -0,0 +1,6 @@
1
+ /**
2
+ * The source text and the index over it. The one place that normalises a file,
3
+ * so every offset in the compiler means the same thing.
4
+ */
5
+ export type { SourceFile, SourcePosition } from './source.js';
6
+ export { normaliseSource, sourceFile } from './source.js';
@@ -0,0 +1,102 @@
1
+ import { makeSpan } from '../span/index.js';
2
+ import type { Span } from '../span/index.js';
3
+
4
+ /** A place in the source, one-based on both axes. See the span module on columns. */
5
+ export interface SourcePosition {
6
+ readonly line: number;
7
+ readonly column: number;
8
+ }
9
+
10
+ /**
11
+ * A file the compiler is working on: its normalised text, and the index that
12
+ * turns an offset into a line and a column.
13
+ *
14
+ * Every stage needs the same two answers, and computing them by scanning from
15
+ * the top of the file each time is how a compiler becomes quadratic on the one
16
+ * input that has many diagnostics. The line starts are built once.
17
+ */
18
+ export interface SourceFile {
19
+ /** What the file is called, for the first line of a rendered diagnostic. */
20
+ readonly name: string;
21
+ /** The text after normalisation. Offsets and spans index this, not the bytes on disk. */
22
+ readonly text: string;
23
+ readonly lineCount: number;
24
+ /** The line's text, without its newline. An out of range line is empty. */
25
+ lineText(line: number): string;
26
+ /** The offset the line starts at. */
27
+ lineStart(line: number): number;
28
+ positionAt(offset: number): SourcePosition;
29
+ offsetAt(position: SourcePosition): number;
30
+ /** The span covering length code units from offset, with its line and column worked out. */
31
+ spanAt(offset: number, length: number): Span;
32
+ }
33
+
34
+ const BYTE_ORDER_MARK = '';
35
+
36
+ /**
37
+ * Drops a leading byte order mark and turns CRLF into LF, which language.md 3.1
38
+ * requires to happen before anything else reads the file, so that a script
39
+ * written on one operating system compiles identically on another.
40
+ *
41
+ * A lone carriage return is deliberately left alone. It is not a line ending
42
+ * the language accepts, so it belongs to the lexer as OS1001 on the character
43
+ * rather than being silently repaired here.
44
+ */
45
+ export function normaliseSource(raw: string): string {
46
+ const withoutMark = raw.startsWith(BYTE_ORDER_MARK) ? raw.slice(BYTE_ORDER_MARK.length) : raw;
47
+ return withoutMark.includes('\r\n') ? withoutMark.replace(/\r\n/g, '\n') : withoutMark;
48
+ }
49
+
50
+ function lineStartsOf(text: string): readonly number[] {
51
+ const starts = [0];
52
+ for (let i = 0; i < text.length; i++) {
53
+ if (text.charCodeAt(i) === 10) starts.push(i + 1);
54
+ }
55
+ return starts;
56
+ }
57
+
58
+ export function sourceFile(name: string, raw: string): SourceFile {
59
+ const text = normaliseSource(raw);
60
+ const starts = lineStartsOf(text);
61
+
62
+ const clampLine = (line: number): number => Math.min(Math.max(Math.trunc(line), 1), starts.length);
63
+
64
+ const lineStart = (line: number): number => starts[clampLine(line) - 1] ?? 0;
65
+
66
+ const lineEnd = (line: number): number => {
67
+ const next = starts[clampLine(line)];
68
+ if (next === undefined) return text.length;
69
+ // The next line starts one past the newline, which this line does not own.
70
+ return next - 1;
71
+ };
72
+
73
+ const positionAt = (offset: number): SourcePosition => {
74
+ const at = Math.min(Math.max(Math.trunc(offset), 0), text.length);
75
+ let low = 0;
76
+ let high = starts.length - 1;
77
+ while (low < high) {
78
+ const middle = (low + high + 1) >> 1;
79
+ if ((starts[middle] ?? 0) <= at) low = middle;
80
+ else high = middle - 1;
81
+ }
82
+ return { line: low + 1, column: at - (starts[low] ?? 0) + 1 };
83
+ };
84
+
85
+ return {
86
+ name,
87
+ text,
88
+ lineCount: starts.length,
89
+ lineText: (line) => text.slice(lineStart(line), lineEnd(line)),
90
+ lineStart,
91
+ positionAt,
92
+ offsetAt: (position) => {
93
+ const start = lineStart(position.line);
94
+ const end = lineEnd(position.line);
95
+ return Math.min(start + Math.max(Math.trunc(position.column), 1) - 1, end);
96
+ },
97
+ spanAt: (offset, length) => {
98
+ const { line, column } = positionAt(offset);
99
+ return makeSpan(offset, Math.max(length, 0), line, column);
100
+ },
101
+ };
102
+ }
@@ -0,0 +1,2 @@
1
+ export type { Positioned, Span } from './span.js';
2
+ export { containsOffset, endOffset, makeSpan, spanning } from './span.js';
@@ -0,0 +1,84 @@
1
+ /**
2
+ * Where something is in the source text.
3
+ *
4
+ * Every token and every node carries one of these, and every diagnostic is
5
+ * drawn under one, so the definition of a column here decides where a caret
6
+ * lands for every reader of the language. It is stated rather than implied.
7
+ *
8
+ * ## A column counts UTF-16 code units
9
+ *
10
+ * An editor and a terminal disagree about what a column is, and there is no
11
+ * answer that is right for both, so the question is which one gets the stored
12
+ * number and which one converts.
13
+ *
14
+ * The stored number is UTF-16 code units, one-based, because that is what the
15
+ * things that consume a column ask for: a browser text component indexes the
16
+ * text it holds in UTF-16, and the language server protocol an editor speaks
17
+ * counts positions in UTF-16 by default. Storing anything else means converting
18
+ * at every token on the hot path of a compile that runs while somebody waits,
19
+ * and converting is exactly where an off-by-one caret comes from.
20
+ *
21
+ * A terminal counts characters, not code units, so the renderer does not use
22
+ * the column at all. It takes the source line and the span's offsets and counts
23
+ * code points itself when it pads the caret, which is why a caret sitting after
24
+ * a string literal holding an astral character still lands under the right
25
+ * character in a terminal.
26
+ *
27
+ * The practical reach of the difference is small and worth knowing: outside a
28
+ * string literal the language accepts ASCII only (language.md 3.1), so a column
29
+ * can only diverge from a character count on a line that holds a string literal
30
+ * with non-ASCII text in it.
31
+ *
32
+ * ## Offsets index the normalised text
33
+ *
34
+ * A byte order mark is dropped and CRLF becomes LF before anything else reads
35
+ * the file (language.md 3.1), so an offset counts into the text after those two
36
+ * changes. A host that has to point back into the file on disk adds the mark
37
+ * and the carriage returns back itself; see the source module, which is the one
38
+ * place that does the normalising.
39
+ */
40
+ export interface Span {
41
+ /** Zero-based UTF-16 code unit index into the normalised source text. */
42
+ readonly offset: number;
43
+ /** Length in UTF-16 code units. Zero where a diagnostic points between two characters. */
44
+ readonly length: number;
45
+ /** One-based, because that is the first line an editor shows. */
46
+ readonly line: number;
47
+ /** One-based UTF-16 code unit count from the start of the line. */
48
+ readonly column: number;
49
+ }
50
+
51
+ /** Anything the compiler can point a caret at. */
52
+ export interface Positioned {
53
+ readonly span: Span;
54
+ }
55
+
56
+ export function makeSpan(offset: number, length: number, line: number, column: number): Span {
57
+ return { offset, length, line, column };
58
+ }
59
+
60
+ /** One past the last code unit the span covers. */
61
+ export function endOffset(span: Span): number {
62
+ return span.offset + span.length;
63
+ }
64
+
65
+ /**
66
+ * The span from the start of the first to the end of the last.
67
+ *
68
+ * This is how a node gets a span: a call expression runs from its callee to its
69
+ * closing bracket, and the caret under it covers the whole call rather than the
70
+ * one token the parser happened to be holding.
71
+ */
72
+ export function spanning(first: Span, last: Span): Span {
73
+ return {
74
+ offset: first.offset,
75
+ length: Math.max(endOffset(last) - first.offset, 0),
76
+ line: first.line,
77
+ column: first.column,
78
+ };
79
+ }
80
+
81
+ /** Whether an offset falls inside the span. A zero length span contains nothing. */
82
+ export function containsOffset(span: Span, offset: number): boolean {
83
+ return offset >= span.offset && offset < endOffset(span);
84
+ }
@@ -0,0 +1,15 @@
1
+ /**
2
+ * The token vocabulary of language.md section 3. Types and the two tables the
3
+ * types are derived from; the lexer that produces them is the next stage.
4
+ */
5
+ export type {
6
+ KeywordKind,
7
+ LayoutKind,
8
+ LiteralKind,
9
+ NameKind,
10
+ PunctuationKind,
11
+ TokenKind,
12
+ } from './kind.js';
13
+ export { PUNCTUATORS, RESERVED_WORDS } from './kind.js';
14
+
15
+ export type { NumberToken, SimpleToken, StringToken, Token } from './token.js';
@@ -0,0 +1,123 @@
1
+ /**
2
+ * Every kind of token language.md section 3 describes. Types and the two tables
3
+ * they are derived from, and no scanning: the lexer is the next stage.
4
+ *
5
+ * A keyword's kind is the word, and a punctuation token's kind is the symbol,
6
+ * so a parser reads as `token.kind === 'if'` and `token.kind === '('` rather
7
+ * than through a second vocabulary that has to be learned and kept in step.
8
+ */
9
+
10
+ /**
11
+ * The reserved words of language.md 3.4, which cannot be used as names.
12
+ *
13
+ * `import`, `map`, `matrix`, `type` and `as` are reserved and unused in version
14
+ * 1. They are here so that adding them later cannot break a script that used
15
+ * one as a variable name.
16
+ *
17
+ * Two words the grammar treats as keywords are deliberately absent, because 3.4
18
+ * does not reserve them: `version` and `limits` are recognised by position in
19
+ * the parser and are ordinary identifiers to the lexer.
20
+ */
21
+ export const RESERVED_WORDS = [
22
+ 'and',
23
+ 'array',
24
+ 'as',
25
+ 'bool',
26
+ 'break',
27
+ 'case',
28
+ 'color',
29
+ 'continue',
30
+ 'default',
31
+ 'else',
32
+ 'false',
33
+ 'fn',
34
+ 'for',
35
+ 'if',
36
+ 'import',
37
+ 'in',
38
+ 'is',
39
+ 'live',
40
+ 'map',
41
+ 'matrix',
42
+ 'none',
43
+ 'not',
44
+ 'number',
45
+ 'or',
46
+ 'return',
47
+ 'series',
48
+ 'step',
49
+ 'string',
50
+ 'strategy',
51
+ 'study',
52
+ 'switch',
53
+ 'to',
54
+ 'true',
55
+ 'type',
56
+ 'var',
57
+ 'while',
58
+ ] as const;
59
+
60
+ export type KeywordKind = (typeof RESERVED_WORDS)[number];
61
+
62
+ /**
63
+ * The operator and punctuation tokens of language.md 3.12, longest first so a
64
+ * lexer taking them in order takes the longest match.
65
+ *
66
+ * `!` is not in the list and is not an operator on its own: it exists only as
67
+ * the first half of `!=`, and a lone one is OS1001 with the fix to write `not`.
68
+ * There are no bitwise, increment or exponent operators for the same reason
69
+ * they are missing from the specification, so there is nothing here to add.
70
+ */
71
+ export const PUNCTUATORS = [
72
+ '==',
73
+ '!=',
74
+ '<=',
75
+ '>=',
76
+ '+=',
77
+ '-=',
78
+ '*=',
79
+ '/=',
80
+ '%=',
81
+ '+',
82
+ '-',
83
+ '*',
84
+ '/',
85
+ '%',
86
+ '<',
87
+ '>',
88
+ '=',
89
+ '(',
90
+ ')',
91
+ '[',
92
+ ']',
93
+ ',',
94
+ '.',
95
+ '?',
96
+ ':',
97
+ ] as const;
98
+
99
+ export type PunctuationKind = (typeof PUNCTUATORS)[number];
100
+
101
+ /**
102
+ * A name, and the literals that are not words.
103
+ *
104
+ * A named colour such as `aqua` is an identifier rather than a literal: 3.8
105
+ * makes the nineteen names ordinary globals so the standard library can add
106
+ * more without a grammar change. `hexColor` is only the `#rrggbb` and
107
+ * `#rrggbbaa` forms. `true`, `false` and `none` are keywords, being words.
108
+ */
109
+ export type LiteralKind = 'numberLiteral' | 'stringLiteral' | 'hexColor';
110
+
111
+ export type NameKind = 'identifier';
112
+
113
+ /**
114
+ * The layout tokens of 3.10.
115
+ *
116
+ * A comment and a line carrying no token produce nothing at all, which is the
117
+ * whole of the rule that lets a commented-out statement sit at column zero
118
+ * inside a block without closing it. So there is no comment kind and no blank
119
+ * line kind here, and nothing downstream has to skip them.
120
+ */
121
+ export type LayoutKind = 'newline' | 'indent' | 'dedent' | 'endOfFile';
122
+
123
+ export type TokenKind = KeywordKind | PunctuationKind | LiteralKind | NameKind | LayoutKind;
@@ -0,0 +1,39 @@
1
+ import type { Span } from '../span/index.js';
2
+ import type { TokenKind } from './kind.js';
3
+
4
+ interface TokenFields {
5
+ readonly kind: TokenKind;
6
+ readonly span: Span;
7
+ /**
8
+ * The source text the token was read from, exactly as written.
9
+ *
10
+ * An indent token's text is the leading spaces of the line it opens, which is
11
+ * what OS1003 counts when it says how deeply a line was indented. A dedent, a
12
+ * newline and the end of the file carry no text.
13
+ */
14
+ readonly text: string;
15
+ }
16
+
17
+ /**
18
+ * A number literal and the value it denotes.
19
+ *
20
+ * The value is carried rather than re-read later because the literal forms of
21
+ * 3.5 include digit group underscores and a hexadecimal form, and a second
22
+ * reader of the text would be a second place that has to agree about them.
23
+ */
24
+ export interface NumberToken extends TokenFields {
25
+ readonly kind: 'numberLiteral';
26
+ readonly value: number;
27
+ }
28
+
29
+ /** A string literal with its escape sequences already resolved. */
30
+ export interface StringToken extends TokenFields {
31
+ readonly kind: 'stringLiteral';
32
+ readonly value: string;
33
+ }
34
+
35
+ export interface SimpleToken extends TokenFields {
36
+ readonly kind: Exclude<TokenKind, 'numberLiteral' | 'stringLiteral'>;
37
+ }
38
+
39
+ export type Token = NumberToken | StringToken | SimpleToken;