openalgo-script 0.1.0-alpha.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (266) hide show
  1. package/LICENSE +201 -0
  2. package/NOTICE +12 -0
  3. package/README.md +255 -0
  4. package/dist/core/ast/annotations.d.ts +56 -0
  5. package/dist/core/ast/annotations.d.ts.map +1 -0
  6. package/dist/core/ast/annotations.js +27 -0
  7. package/dist/core/ast/annotations.js.map +1 -0
  8. package/dist/core/ast/build.d.ts +25 -0
  9. package/dist/core/ast/build.d.ts.map +1 -0
  10. package/dist/core/ast/build.js +26 -0
  11. package/dist/core/ast/build.js.map +1 -0
  12. package/dist/core/ast/children.d.ts +15 -0
  13. package/dist/core/ast/children.d.ts.map +1 -0
  14. package/dist/core/ast/children.js +103 -0
  15. package/dist/core/ast/children.js.map +1 -0
  16. package/dist/core/ast/expressions.d.ts +178 -0
  17. package/dist/core/ast/expressions.d.ts.map +1 -0
  18. package/dist/core/ast/expressions.js +50 -0
  19. package/dist/core/ast/expressions.js.map +1 -0
  20. package/dist/core/ast/header.d.ts +50 -0
  21. package/dist/core/ast/header.d.ts.map +1 -0
  22. package/dist/core/ast/header.js +2 -0
  23. package/dist/core/ast/header.js.map +1 -0
  24. package/dist/core/ast/index.d.ts +30 -0
  25. package/dist/core/ast/index.d.ts.map +1 -0
  26. package/dist/core/ast/index.js +21 -0
  27. package/dist/core/ast/index.js.map +1 -0
  28. package/dist/core/ast/name.d.ts +24 -0
  29. package/dist/core/ast/name.d.ts.map +1 -0
  30. package/dist/core/ast/name.js +2 -0
  31. package/dist/core/ast/name.js.map +1 -0
  32. package/dist/core/ast/node.d.ts +34 -0
  33. package/dist/core/ast/node.d.ts.map +1 -0
  34. package/dist/core/ast/node.js +70 -0
  35. package/dist/core/ast/node.js.map +1 -0
  36. package/dist/core/ast/script.d.ts +57 -0
  37. package/dist/core/ast/script.d.ts.map +1 -0
  38. package/dist/core/ast/script.js +2 -0
  39. package/dist/core/ast/script.js.map +1 -0
  40. package/dist/core/ast/statements.d.ts +161 -0
  41. package/dist/core/ast/statements.d.ts.map +1 -0
  42. package/dist/core/ast/statements.js +10 -0
  43. package/dist/core/ast/statements.js.map +1 -0
  44. package/dist/core/ast/walk.d.ts +39 -0
  45. package/dist/core/ast/walk.d.ts.map +1 -0
  46. package/dist/core/ast/walk.js +54 -0
  47. package/dist/core/ast/walk.js.map +1 -0
  48. package/dist/core/catalogue/catalogue.generated.d.ts +1590 -0
  49. package/dist/core/catalogue/catalogue.generated.d.ts.map +1 -0
  50. package/dist/core/catalogue/catalogue.generated.js +162 -0
  51. package/dist/core/catalogue/catalogue.generated.js.map +1 -0
  52. package/dist/core/catalogue/index.d.ts +15 -0
  53. package/dist/core/catalogue/index.d.ts.map +1 -0
  54. package/dist/core/catalogue/index.js +4 -0
  55. package/dist/core/catalogue/index.js.map +1 -0
  56. package/dist/core/catalogue/lookup.d.ts +14 -0
  57. package/dist/core/catalogue/lookup.d.ts.map +1 -0
  58. package/dist/core/catalogue/lookup.js +19 -0
  59. package/dist/core/catalogue/lookup.js.map +1 -0
  60. package/dist/core/catalogue/template.d.ts +15 -0
  61. package/dist/core/catalogue/template.d.ts.map +1 -0
  62. package/dist/core/catalogue/template.js +33 -0
  63. package/dist/core/catalogue/template.js.map +1 -0
  64. package/dist/core/catalogue/types.d.ts +34 -0
  65. package/dist/core/catalogue/types.d.ts.map +1 -0
  66. package/dist/core/catalogue/types.js +6 -0
  67. package/dist/core/catalogue/types.js.map +1 -0
  68. package/dist/core/catalogue/values.generated.d.ts +565 -0
  69. package/dist/core/catalogue/values.generated.d.ts.map +1 -0
  70. package/dist/core/catalogue/values.generated.js +6 -0
  71. package/dist/core/catalogue/values.generated.js.map +1 -0
  72. package/dist/core/diagnostics/collector.d.ts +46 -0
  73. package/dist/core/diagnostics/collector.d.ts.map +1 -0
  74. package/dist/core/diagnostics/collector.js +55 -0
  75. package/dist/core/diagnostics/collector.js.map +1 -0
  76. package/dist/core/diagnostics/diagnostic.d.ts +40 -0
  77. package/dist/core/diagnostics/diagnostic.d.ts.map +1 -0
  78. package/dist/core/diagnostics/diagnostic.js +29 -0
  79. package/dist/core/diagnostics/diagnostic.js.map +1 -0
  80. package/dist/core/diagnostics/index.d.ts +12 -0
  81. package/dist/core/diagnostics/index.d.ts.map +1 -0
  82. package/dist/core/diagnostics/index.js +3 -0
  83. package/dist/core/diagnostics/index.js.map +1 -0
  84. package/dist/core/index.d.ts +28 -0
  85. package/dist/core/index.d.ts.map +1 -0
  86. package/dist/core/index.js +22 -0
  87. package/dist/core/index.js.map +1 -0
  88. package/dist/core/lex/blocks.d.ts +51 -0
  89. package/dist/core/lex/blocks.d.ts.map +1 -0
  90. package/dist/core/lex/blocks.js +56 -0
  91. package/dist/core/lex/blocks.js.map +1 -0
  92. package/dist/core/lex/characters.d.ts +38 -0
  93. package/dist/core/lex/characters.d.ts.map +1 -0
  94. package/dist/core/lex/characters.js +50 -0
  95. package/dist/core/lex/characters.js.map +1 -0
  96. package/dist/core/lex/comments.d.ts +50 -0
  97. package/dist/core/lex/comments.d.ts.map +1 -0
  98. package/dist/core/lex/comments.js +64 -0
  99. package/dist/core/lex/comments.js.map +1 -0
  100. package/dist/core/lex/index.d.ts +12 -0
  101. package/dist/core/lex/index.d.ts.map +1 -0
  102. package/dist/core/lex/index.js +12 -0
  103. package/dist/core/lex/index.js.map +1 -0
  104. package/dist/core/lex/layout.d.ts +40 -0
  105. package/dist/core/lex/layout.d.ts.map +1 -0
  106. package/dist/core/lex/layout.js +143 -0
  107. package/dist/core/lex/layout.js.map +1 -0
  108. package/dist/core/lex/lexer.d.ts +13 -0
  109. package/dist/core/lex/lexer.d.ts.map +1 -0
  110. package/dist/core/lex/lexer.js +419 -0
  111. package/dist/core/lex/lexer.js.map +1 -0
  112. package/dist/core/lex/numbers.d.ts +22 -0
  113. package/dist/core/lex/numbers.d.ts.map +1 -0
  114. package/dist/core/lex/numbers.js +58 -0
  115. package/dist/core/lex/numbers.js.map +1 -0
  116. package/dist/core/lex/statements.d.ts +24 -0
  117. package/dist/core/lex/statements.d.ts.map +1 -0
  118. package/dist/core/lex/statements.js +90 -0
  119. package/dist/core/lex/statements.js.map +1 -0
  120. package/dist/core/lex/strings.d.ts +27 -0
  121. package/dist/core/lex/strings.d.ts.map +1 -0
  122. package/dist/core/lex/strings.js +92 -0
  123. package/dist/core/lex/strings.js.map +1 -0
  124. package/dist/core/lex/unexpected.d.ts +31 -0
  125. package/dist/core/lex/unexpected.d.ts.map +1 -0
  126. package/dist/core/lex/unexpected.js +114 -0
  127. package/dist/core/lex/unexpected.js.map +1 -0
  128. package/dist/core/parse/annotations.d.ts +4 -0
  129. package/dist/core/parse/annotations.d.ts.map +1 -0
  130. package/dist/core/parse/annotations.js +49 -0
  131. package/dist/core/parse/annotations.js.map +1 -0
  132. package/dist/core/parse/blocks.d.ts +18 -0
  133. package/dist/core/parse/blocks.d.ts.map +1 -0
  134. package/dist/core/parse/blocks.js +47 -0
  135. package/dist/core/parse/blocks.js.map +1 -0
  136. package/dist/core/parse/branches.d.ts +17 -0
  137. package/dist/core/parse/branches.d.ts.map +1 -0
  138. package/dist/core/parse/branches.js +66 -0
  139. package/dist/core/parse/branches.js.map +1 -0
  140. package/dist/core/parse/cursor.d.ts +119 -0
  141. package/dist/core/parse/cursor.d.ts.map +1 -0
  142. package/dist/core/parse/cursor.js +305 -0
  143. package/dist/core/parse/cursor.js.map +1 -0
  144. package/dist/core/parse/expressions.d.ts +57 -0
  145. package/dist/core/parse/expressions.d.ts.map +1 -0
  146. package/dist/core/parse/expressions.js +373 -0
  147. package/dist/core/parse/expressions.js.map +1 -0
  148. package/dist/core/parse/functions.d.ts +16 -0
  149. package/dist/core/parse/functions.d.ts.map +1 -0
  150. package/dist/core/parse/functions.js +92 -0
  151. package/dist/core/parse/functions.js.map +1 -0
  152. package/dist/core/parse/index.d.ts +13 -0
  153. package/dist/core/parse/index.d.ts.map +1 -0
  154. package/dist/core/parse/index.js +13 -0
  155. package/dist/core/parse/index.js.map +1 -0
  156. package/dist/core/parse/loops.d.ts +13 -0
  157. package/dist/core/parse/loops.d.ts.map +1 -0
  158. package/dist/core/parse/loops.js +78 -0
  159. package/dist/core/parse/loops.js.map +1 -0
  160. package/dist/core/parse/names.d.ts +34 -0
  161. package/dist/core/parse/names.d.ts.map +1 -0
  162. package/dist/core/parse/names.js +122 -0
  163. package/dist/core/parse/names.js.map +1 -0
  164. package/dist/core/parse/script.d.ts +23 -0
  165. package/dist/core/parse/script.d.ts.map +1 -0
  166. package/dist/core/parse/script.js +55 -0
  167. package/dist/core/parse/script.js.map +1 -0
  168. package/dist/core/parse/statements.d.ts +14 -0
  169. package/dist/core/parse/statements.d.ts.map +1 -0
  170. package/dist/core/parse/statements.js +292 -0
  171. package/dist/core/parse/statements.js.map +1 -0
  172. package/dist/core/parse/switches.d.ts +16 -0
  173. package/dist/core/parse/switches.d.ts.map +1 -0
  174. package/dist/core/parse/switches.js +98 -0
  175. package/dist/core/parse/switches.js.map +1 -0
  176. package/dist/core/render/index.d.ts +10 -0
  177. package/dist/core/render/index.d.ts.map +1 -0
  178. package/dist/core/render/index.js +10 -0
  179. package/dist/core/render/index.js.map +1 -0
  180. package/dist/core/render/terminal.d.ts +11 -0
  181. package/dist/core/render/terminal.d.ts.map +1 -0
  182. package/dist/core/render/terminal.js +51 -0
  183. package/dist/core/render/terminal.js.map +1 -0
  184. package/dist/core/source/index.d.ts +7 -0
  185. package/dist/core/source/index.d.ts.map +1 -0
  186. package/dist/core/source/index.js +2 -0
  187. package/dist/core/source/index.js.map +1 -0
  188. package/dist/core/source/source.d.ts +41 -0
  189. package/dist/core/source/source.d.ts.map +1 -0
  190. package/dist/core/source/source.js +67 -0
  191. package/dist/core/source/source.js.map +1 -0
  192. package/dist/core/span/index.d.ts +3 -0
  193. package/dist/core/span/index.d.ts.map +1 -0
  194. package/dist/core/span/index.js +2 -0
  195. package/dist/core/span/index.js.map +1 -0
  196. package/dist/core/span/span.d.ts +67 -0
  197. package/dist/core/span/span.d.ts.map +1 -0
  198. package/dist/core/span/span.js +27 -0
  199. package/dist/core/span/span.js.map +1 -0
  200. package/dist/core/tokens/index.d.ts +8 -0
  201. package/dist/core/tokens/index.d.ts.map +1 -0
  202. package/dist/core/tokens/index.js +2 -0
  203. package/dist/core/tokens/index.js.map +1 -0
  204. package/dist/core/tokens/kind.d.ts +53 -0
  205. package/dist/core/tokens/kind.d.ts.map +1 -0
  206. package/dist/core/tokens/kind.js +94 -0
  207. package/dist/core/tokens/kind.js.map +1 -0
  208. package/dist/core/tokens/token.d.ts +36 -0
  209. package/dist/core/tokens/token.d.ts.map +1 -0
  210. package/dist/core/tokens/token.js +2 -0
  211. package/dist/core/tokens/token.js.map +1 -0
  212. package/package.json +52 -0
  213. package/spec/README.md +48 -0
  214. package/spec/errors.json +3204 -0
  215. package/src/core/ast/annotations.ts +68 -0
  216. package/src/core/ast/build.ts +35 -0
  217. package/src/core/ast/children.ts +110 -0
  218. package/src/core/ast/expressions.ts +254 -0
  219. package/src/core/ast/header.ts +53 -0
  220. package/src/core/ast/index.ts +102 -0
  221. package/src/core/ast/name.ts +24 -0
  222. package/src/core/ast/node.ts +120 -0
  223. package/src/core/ast/script.ts +60 -0
  224. package/src/core/ast/statements.ts +193 -0
  225. package/src/core/ast/walk.ts +67 -0
  226. package/src/core/catalogue/catalogue.generated.ts +171 -0
  227. package/src/core/catalogue/index.ts +20 -0
  228. package/src/core/catalogue/lookup.ts +23 -0
  229. package/src/core/catalogue/template.ts +38 -0
  230. package/src/core/catalogue/types.ts +37 -0
  231. package/src/core/catalogue/values.generated.ts +160 -0
  232. package/src/core/diagnostics/collector.ts +87 -0
  233. package/src/core/diagnostics/diagnostic.ts +72 -0
  234. package/src/core/diagnostics/index.ts +12 -0
  235. package/src/core/index.ts +146 -0
  236. package/src/core/lex/blocks.ts +92 -0
  237. package/src/core/lex/characters.ts +56 -0
  238. package/src/core/lex/comments.ts +75 -0
  239. package/src/core/lex/index.ts +11 -0
  240. package/src/core/lex/layout.ts +188 -0
  241. package/src/core/lex/lexer.ts +486 -0
  242. package/src/core/lex/numbers.ts +77 -0
  243. package/src/core/lex/statements.ts +94 -0
  244. package/src/core/lex/strings.ts +119 -0
  245. package/src/core/lex/unexpected.ts +125 -0
  246. package/src/core/parse/annotations.ts +60 -0
  247. package/src/core/parse/blocks.ts +52 -0
  248. package/src/core/parse/branches.ts +77 -0
  249. package/src/core/parse/cursor.ts +348 -0
  250. package/src/core/parse/expressions.ts +441 -0
  251. package/src/core/parse/functions.ts +109 -0
  252. package/src/core/parse/index.ts +12 -0
  253. package/src/core/parse/loops.ts +97 -0
  254. package/src/core/parse/names.ts +135 -0
  255. package/src/core/parse/script.ts +64 -0
  256. package/src/core/parse/statements.ts +329 -0
  257. package/src/core/parse/switches.ts +105 -0
  258. package/src/core/render/index.ts +9 -0
  259. package/src/core/render/terminal.ts +62 -0
  260. package/src/core/source/index.ts +6 -0
  261. package/src/core/source/source.ts +102 -0
  262. package/src/core/span/index.ts +2 -0
  263. package/src/core/span/span.ts +84 -0
  264. package/src/core/tokens/index.ts +15 -0
  265. package/src/core/tokens/kind.ts +123 -0
  266. package/src/core/tokens/token.ts +39 -0
@@ -0,0 +1,188 @@
1
+ import type { DiagnosticSink } from '../diagnostics/index.js';
2
+ import type { SourceFile } from '../source/index.js';
3
+ import type { Span } from '../span/index.js';
4
+ import type { Token, TokenKind } from '../tokens/index.js';
5
+ import { BlockStack } from './blocks.js';
6
+ import { endsWithArrow, opensBlock } from './statements.js';
7
+
8
+ /** The leading whitespace of a line, held until the line turns out to carry a token. */
9
+ interface PendingLine {
10
+ readonly line: number;
11
+ readonly offset: number;
12
+ readonly width: number;
13
+ readonly text: string;
14
+ readonly hasTab: boolean;
15
+ }
16
+
17
+ /** The statement being read, which may have begun several lines above. */
18
+ interface OpenStatement {
19
+ readonly line: number;
20
+ readonly indent: number;
21
+ firstKind: TokenKind | undefined;
22
+ }
23
+
24
+ /** The statement above, which is all it takes to know whether a block may open here. */
25
+ interface ClosedStatement {
26
+ readonly line: number;
27
+ readonly firstKind: TokenKind;
28
+ readonly arrowAtEnd: boolean;
29
+ }
30
+
31
+ /**
32
+ * What the shape of a file means: where a statement begins, where a block opens
33
+ * and closes, and what a line's leading whitespace is worth.
34
+ *
35
+ * It is a separate thing from the scanner because it works in lines and
36
+ * statements where the scanner works in characters, and because it is where the
37
+ * two rules that a naive lexer gets wrong live: a line that carries no token is
38
+ * never asked about its indentation, and a line indented past its siblings is
39
+ * one error rather than a new block.
40
+ *
41
+ * It writes the layout tokens straight into the list the scanner is filling, so
42
+ * an indent lands in front of the token that opened the block rather than being
43
+ * threaded back in afterwards.
44
+ */
45
+ export class Layout {
46
+ readonly #file: SourceFile;
47
+ readonly #sink: DiagnosticSink;
48
+ readonly #tokens: Token[];
49
+ readonly #blocks = new BlockStack();
50
+
51
+ #pending: PendingLine | undefined;
52
+ #statement: OpenStatement | undefined;
53
+ #previous: ClosedStatement | undefined;
54
+
55
+ constructor(file: SourceFile, sink: DiagnosticSink, tokens: Token[]) {
56
+ this.#file = file;
57
+ this.#sink = sink;
58
+ this.#tokens = tokens;
59
+ }
60
+
61
+ /** Holds a line's leading whitespace, which may turn out to mean nothing at all. */
62
+ beginLine(line: number, offset: number, width: number, text: string, hasTab: boolean): void {
63
+ this.#pending = { line, offset, width, text, hasTab };
64
+ }
65
+
66
+ /** The line is over. Whatever it was holding was never wanted. */
67
+ endLine(): void {
68
+ this.#pending = undefined;
69
+ }
70
+
71
+ /**
72
+ * Turns the leading whitespace into block structure, at the moment the line
73
+ * turns out to carry a token and not before. A line that produces no token
74
+ * never reaches here, which is the whole of the rule that a blank line and a
75
+ * comment-only line carry no indentation at all (3.10).
76
+ */
77
+ admit(continuing: boolean): void {
78
+ const pending = this.#pending;
79
+ if (pending === undefined) return;
80
+ this.#pending = undefined;
81
+
82
+ if (pending.hasTab) {
83
+ this.#sink.report('OS1002', this.#span(pending.offset, pending.width), {});
84
+ }
85
+
86
+ if (continuing) {
87
+ this.#checkContinuation(pending);
88
+ return;
89
+ }
90
+
91
+ this.#enterBlock(pending);
92
+ this.#statement = { line: pending.line, indent: pending.width, firstKind: undefined };
93
+ }
94
+
95
+ /**
96
+ * A continuation line belongs to a statement that began above it, so it opens
97
+ * and closes nothing. It only has to sit further right than the line that
98
+ * began the statement, so that it can never be read as a new one (3.11).
99
+ *
100
+ * That is OS1028 rather than the block rule of OS1003, and the difference is
101
+ * not a formality: there is no block here, so OS1003's message would name an
102
+ * indentation no line carries and its fix would offer to end a block the file
103
+ * never opened.
104
+ */
105
+ #checkContinuation(pending: PendingLine): void {
106
+ const statement = this.#statement;
107
+ if (statement === undefined || pending.width > statement.indent) return;
108
+ this.#sink.report('OS1028', this.#span(pending.offset, pending.width), {
109
+ found: pending.width,
110
+ statement: statement.indent,
111
+ line: statement.line,
112
+ });
113
+ }
114
+
115
+ #enterBlock(pending: PendingLine): void {
116
+ const firstToken = pending.offset + pending.width;
117
+ const event = this.#blocks.enter(
118
+ pending.width,
119
+ this.#afterHeader(),
120
+ this.#previous?.line ?? pending.line,
121
+ );
122
+
123
+ switch (event.kind) {
124
+ case 'same':
125
+ break;
126
+ case 'open':
127
+ this.#tokens.push({
128
+ kind: 'indent',
129
+ span: this.#span(pending.offset, pending.width),
130
+ text: pending.text,
131
+ });
132
+ break;
133
+ case 'close':
134
+ this.#dedent(event.count, firstToken);
135
+ break;
136
+ case 'mismatch':
137
+ this.#dedent(event.count, firstToken);
138
+ this.#sink.report('OS1003', this.#span(pending.offset, pending.width), {
139
+ found: pending.width,
140
+ expected: event.expected,
141
+ line: event.openerLine,
142
+ });
143
+ break;
144
+ }
145
+ }
146
+
147
+ /** The first token of a statement decides whether a block may open under it. */
148
+ observe(kind: TokenKind): void {
149
+ const statement = this.#statement;
150
+ if (statement !== undefined && statement.firstKind === undefined) statement.firstKind = kind;
151
+ }
152
+
153
+ endStatement(): void {
154
+ const statement = this.#statement;
155
+ this.#statement = undefined;
156
+ if (statement === undefined || statement.firstKind === undefined) return;
157
+ this.#previous = {
158
+ line: statement.line,
159
+ firstKind: statement.firstKind,
160
+ arrowAtEnd: endsWithArrow(this.#tokens),
161
+ };
162
+ }
163
+
164
+ /** A second statement beginning on a line that already held one, after a `;`. */
165
+ beginStatement(line: number, indent: number): void {
166
+ this.#statement = { line, indent, firstKind: undefined };
167
+ }
168
+
169
+ /** The end of the file closes every block still open. */
170
+ closeBlocks(at: number): void {
171
+ this.#dedent(this.#blocks.closeAll(), at);
172
+ }
173
+
174
+ #dedent(count: number, at: number): void {
175
+ for (let i = 0; i < count; i++) {
176
+ this.#tokens.push({ kind: 'dedent', span: this.#span(at, 0), text: '' });
177
+ }
178
+ }
179
+
180
+ #afterHeader(): boolean {
181
+ const previous = this.#previous;
182
+ return previous !== undefined && opensBlock(previous.firstKind, previous.arrowAtEnd);
183
+ }
184
+
185
+ #span(offset: number, length: number): Span {
186
+ return this.#file.spanAt(offset, length);
187
+ }
188
+ }
@@ -0,0 +1,486 @@
1
+ import type { DiagnosticSink } from '../diagnostics/index.js';
2
+ import type { SourceFile } from '../source/index.js';
3
+ import type { Span } from '../span/index.js';
4
+ import type { KeywordKind, SimpleToken, Token } from '../tokens/index.js';
5
+ import { PUNCTUATORS, RESERVED_WORDS } from '../tokens/index.js';
6
+ import {
7
+ APOSTROPHE,
8
+ BACKSLASH,
9
+ DOT,
10
+ HASH,
11
+ isBlank,
12
+ isDigit,
13
+ isNamePart,
14
+ isNameStart,
15
+ LINE_FEED,
16
+ QUOTE,
17
+ SEMICOLON,
18
+ SLASH,
19
+ SPACE,
20
+ TAB,
21
+ } from './characters.js';
22
+ import { closesComment, commentRegion, hasCloser, opensComment } from './comments.js';
23
+ import { Layout } from './layout.js';
24
+ import { scanNumber } from './numbers.js';
25
+ import { continuesLine } from './statements.js';
26
+ import { scanString } from './strings.js';
27
+ import { describeUnexpected, isNonAsciiLetter } from './unexpected.js';
28
+
29
+ const RESERVED: ReadonlySet<string> = new Set<string>(RESERVED_WORDS);
30
+ const ONE_CHARACTER: ReadonlySet<string> = new Set(PUNCTUATORS.filter((mark) => mark.length === 1));
31
+ const TWO_CHARACTER: ReadonlySet<string> = new Set(PUNCTUATORS.filter((mark) => mark.length === 2));
32
+
33
+ /**
34
+ * The two character operators of other languages, which this one does not have
35
+ * (3.12). They are matched as a pair so that the message names the operator the
36
+ * reader wrote and offers the one to write instead, rather than reporting one
37
+ * ampersand and leaving them to work out which half was wrong.
38
+ */
39
+ const REJECTED_PAIRS: ReadonlySet<string> = new Set(['&&', '||', '**', '++']);
40
+
41
+ const OPENING_BRACKETS: ReadonlySet<string> = new Set(['(', '[']);
42
+ const CLOSING_BRACKETS: ReadonlySet<string> = new Set([')', ']']);
43
+
44
+ const HEX_ONLY = /^[0-9a-fA-F]+$/;
45
+
46
+ /** A run of name characters, and the first one in it the language does not allow. */
47
+ interface Word {
48
+ readonly end: number;
49
+ readonly foreign: { readonly offset: number; readonly written: string } | undefined;
50
+ }
51
+
52
+ class Lexer {
53
+ readonly #file: SourceFile;
54
+ readonly #text: string;
55
+ readonly #sink: DiagnosticSink;
56
+ readonly #tokens: Token[] = [];
57
+ readonly #layout: Layout;
58
+
59
+ #offset = 0;
60
+ #line = 1;
61
+ #indent = 0;
62
+ #brackets = 0;
63
+ #continuing = false;
64
+ #lineHasToken = false;
65
+ #droppedAfterToken = false;
66
+ /** Whether a block comment marker opened a region that no closer has ended yet. */
67
+ #inComment = false;
68
+
69
+ constructor(file: SourceFile, sink: DiagnosticSink) {
70
+ this.#file = file;
71
+ this.#text = file.text;
72
+ this.#sink = sink;
73
+ this.#layout = new Layout(file, sink, this.#tokens);
74
+ }
75
+
76
+ run(): readonly Token[] {
77
+ while (this.#offset < this.#text.length) this.#readLine();
78
+ this.#finish();
79
+ return this.#tokens;
80
+ }
81
+
82
+ // -------------------------------------------------------------------------
83
+ // Lines
84
+ // -------------------------------------------------------------------------
85
+
86
+ #readLine(): void {
87
+ const start = this.#offset;
88
+ let i = start;
89
+ // A tab is one character of indentation here, and OS1002 as soon as the line
90
+ // turns out to carry a token. It is counted rather than expanded because the
91
+ // report is what matters and a width would be a guess at an editor setting.
92
+ let hasTab = false;
93
+ while (isBlank(this.#text.charCodeAt(i))) {
94
+ if (this.#text.charCodeAt(i) === TAB) hasTab = true;
95
+ i++;
96
+ }
97
+
98
+ this.#indent = i - start;
99
+ this.#layout.beginLine(this.#line, start, this.#indent, this.#text.slice(start, i), hasTab);
100
+ this.#offset = i;
101
+ this.#lineHasToken = false;
102
+ this.#droppedAfterToken = false;
103
+
104
+ if (this.#inComment) this.#skipComment(this.#offset);
105
+
106
+ let backslash = false;
107
+ while (this.#offset < this.#text.length) {
108
+ const code = this.#text.charCodeAt(this.#offset);
109
+ if (code === LINE_FEED) break;
110
+ if (code === SPACE) {
111
+ this.#offset++;
112
+ continue;
113
+ }
114
+ if (code === SLASH && this.#text.charCodeAt(this.#offset + 1) === SLASH) {
115
+ // A comment produces no token, which is what lets a commented out line
116
+ // sit at column zero inside a block without closing it (3.10).
117
+ this.#offset = this.#endOfLine(this.#offset);
118
+ break;
119
+ }
120
+ if (opensComment(this.#text, this.#offset)) {
121
+ this.#openComment();
122
+ continue;
123
+ }
124
+ if (closesComment(this.#text, this.#offset)) {
125
+ // A closer with nothing open. The region it was meant to end was never
126
+ // read as one, so the marker is reported where it sits and read as
127
+ // nothing at all.
128
+ this.#sink.report('OS1026', this.#span(this.#offset, 2), { marker: '*/' });
129
+ this.#offset += 2;
130
+ continue;
131
+ }
132
+ if (code === BACKSLASH && this.#isBlankTo(this.#offset + 1)) {
133
+ backslash = true;
134
+ this.#offset = this.#endOfLine(this.#offset);
135
+ break;
136
+ }
137
+ this.#readToken();
138
+ }
139
+
140
+ this.#endLine(backslash);
141
+ }
142
+
143
+ #endLine(backslash: boolean): void {
144
+ // A line carrying no token says nothing about the statement around it, so a
145
+ // blank line and a comment-only line inside a continuation leave it open
146
+ // (3.11) by never being asked.
147
+ if (this.#lineHasToken || backslash) {
148
+ // A trailing operator whose operand was rejected promised nothing: the
149
+ // operand was written, it was just not written in this language. Letting
150
+ // it continue would hand the next line to a broken statement, and one bad
151
+ // character would take a line that has nothing wrong with it.
152
+ const promised = !this.#droppedAfterToken && continuesLine(this.#tokens);
153
+ this.#continuing = this.#brackets > 0 || backslash || promised;
154
+ }
155
+
156
+ if (this.#lineHasToken && !this.#continuing) {
157
+ this.#layout.endStatement();
158
+ const atEnd = this.#offset >= this.#text.length;
159
+ this.#emitNewline(this.#span(this.#offset, atEnd ? 0 : 1));
160
+ }
161
+
162
+ if (this.#offset < this.#text.length) {
163
+ this.#offset++;
164
+ this.#line++;
165
+ }
166
+ this.#layout.endLine();
167
+ }
168
+
169
+ #finish(): void {
170
+ const last = this.#tokens[this.#tokens.length - 1];
171
+ if (last !== undefined && last.kind !== 'newline') {
172
+ // A bracket that is never closed reads the rest of the file as one
173
+ // statement (3.11), and that statement still has to end somewhere for the
174
+ // parser to report it against.
175
+ this.#layout.endStatement();
176
+ this.#emitNewline(this.#span(this.#text.length, 0));
177
+ }
178
+
179
+ this.#layout.closeBlocks(this.#text.length);
180
+ this.#tokens.push({ kind: 'endOfFile', span: this.#span(this.#text.length, 0), text: '' });
181
+ }
182
+
183
+ /**
184
+ * A region the writer meant as a comment, from its opener to its closer.
185
+ *
186
+ * It is reported once, at the opener, and then skipped. Reading the prose
187
+ * inside it as tokens would report the operators it happens to contain and
188
+ * the statements it appears to hold, which is several diagnostics about a
189
+ * program the writer did not write, and none of them would name the comment
190
+ * (3.2).
191
+ */
192
+ #openComment(): void {
193
+ this.#sink.report('OS1026', this.#span(this.#offset, 2), { marker: '/*' });
194
+ this.#skipComment(this.#offset + 2);
195
+ // A closer written nowhere makes no region: the marker costs this line and
196
+ // the file is read on from the next one, rather than being swallowed whole
197
+ // by a form the language does not have (3.2).
198
+ this.#inComment = this.#inComment && hasCloser(this.#text, this.#offset);
199
+ }
200
+
201
+ /**
202
+ * The region, to its closer or to the end of the line, said once and then
203
+ * skipped: one region is one mistake however many lines it runs to, so a
204
+ * line below the opener is read as prose rather than reported again.
205
+ */
206
+ #skipComment(from: number): void {
207
+ const region = commentRegion(this.#text, from);
208
+ this.#inComment = region.open;
209
+ this.#offset = region.end;
210
+ }
211
+
212
+ #endOfLine(from: number): number {
213
+ const at = this.#text.indexOf('\n', from);
214
+ return at === -1 ? this.#text.length : at;
215
+ }
216
+
217
+ #isBlankTo(from: number): boolean {
218
+ for (let i = from; i < this.#text.length; i++) {
219
+ const code = this.#text.charCodeAt(i);
220
+ if (code === LINE_FEED) return true;
221
+ if (!isBlank(code)) return false;
222
+ }
223
+ return true;
224
+ }
225
+
226
+ // -------------------------------------------------------------------------
227
+ // Tokens
228
+ // -------------------------------------------------------------------------
229
+
230
+ #readToken(): void {
231
+ const start = this.#offset;
232
+ const code = this.#text.charCodeAt(start);
233
+
234
+ if (isNameStart(code)) {
235
+ this.#readName(start);
236
+ return;
237
+ }
238
+ if (isDigit(code) || (code === DOT && isDigit(this.#text.charCodeAt(start + 1)))) {
239
+ this.#readNumber(start);
240
+ return;
241
+ }
242
+ if (code === QUOTE || code === APOSTROPHE) {
243
+ this.#readString(start);
244
+ return;
245
+ }
246
+ if (code === HASH) {
247
+ this.#readColour(start);
248
+ return;
249
+ }
250
+ if (code === SEMICOLON) {
251
+ this.#readSemicolon(start);
252
+ return;
253
+ }
254
+ this.#readPunctuation(start);
255
+ }
256
+
257
+ /**
258
+ * A name, and with it every reserved word, which is spelled like one and told
259
+ * apart by the table rather than by the scanner.
260
+ *
261
+ * A reserved word keeps its own kind here even where the parser will accept it
262
+ * as a named argument label (3.4). The lexer cannot make that call: the same
263
+ * word before an `=` is a legal label inside a call and is OS1019 inside a
264
+ * parameter list, and only the parser knows which of the two it is reading.
265
+ */
266
+ #readName(start: number): void {
267
+ const word = this.#word(start);
268
+ if (word.foreign !== undefined) {
269
+ // Identifiers are ASCII (3.3). The whole word is taken as one name so that
270
+ // the statement around it still parses and one bad character reports once.
271
+ this.#sink.report(
272
+ 'OS1001',
273
+ this.#span(word.foreign.offset, word.foreign.written.length),
274
+ describeUnexpected(word.foreign.written),
275
+ );
276
+ }
277
+ const text = this.#text.slice(start, word.end);
278
+ const kind: SimpleToken['kind'] = RESERVED.has(text) ? (text as KeywordKind) : 'identifier';
279
+ this.#emit({ kind, span: this.#span(start, word.end - start), text });
280
+ this.#offset = word.end;
281
+ }
282
+
283
+ /**
284
+ * A number literal, and the run that is not one: a base prefix the language
285
+ * does not have, a unit written against a quantity, a literal that ends in an
286
+ * underscore (3.5).
287
+ *
288
+ * Such a run is reported whole, as OS1029. Every character in it is one the
289
+ * language accepts, so naming the leading digit would name something legal,
290
+ * and deleting it, which is what OS1001 tells a reader to do with the
291
+ * character it names, leaves a valid name behind and a program that compiles
292
+ * and means something else.
293
+ */
294
+ #readNumber(start: number): void {
295
+ const literal = scanNumber(this.#text, start);
296
+ const after = this.#text.codePointAt(literal.end);
297
+
298
+ if (after !== undefined && (isNamePart(after) || isNonAsciiLetter(after))) {
299
+ // A name cannot begin with a digit (3.3), and a letter written against a
300
+ // number is that mistake rather than two tokens that happen to touch.
301
+ const word = this.#word(literal.end);
302
+ const written = this.#text.slice(start, word.end);
303
+ this.#sink.report('OS1029', this.#span(start, word.end - start), {
304
+ written,
305
+ number: this.#text.slice(start, literal.end),
306
+ });
307
+ this.#emit({
308
+ kind: 'identifier',
309
+ span: this.#span(start, word.end - start),
310
+ text: written,
311
+ });
312
+ this.#offset = word.end;
313
+ return;
314
+ }
315
+
316
+ this.#emit({
317
+ kind: 'numberLiteral',
318
+ span: this.#span(start, literal.end - start),
319
+ text: this.#text.slice(start, literal.end),
320
+ value: literal.value,
321
+ });
322
+ this.#offset = literal.end;
323
+ }
324
+
325
+ #readString(start: number): void {
326
+ const literal = scanString(this.#text, start);
327
+
328
+ for (const bad of literal.badEscapes) {
329
+ this.#sink.report('OS1005', this.#span(bad.offset, bad.length), { sequence: bad.text });
330
+ }
331
+ if (literal.unterminated) {
332
+ this.#sink.report('OS1004', this.#span(start, 1), { quote: this.#text.charAt(start) });
333
+ }
334
+
335
+ // The token is emitted either way. The reader is told once that the quote is
336
+ // missing; they do not also need to be told that the statement around it
337
+ // makes no sense without it.
338
+ this.#emit({
339
+ kind: 'stringLiteral',
340
+ span: this.#span(start, literal.end - start),
341
+ text: this.#text.slice(start, literal.end),
342
+ value: literal.value,
343
+ });
344
+ this.#offset = literal.end;
345
+ }
346
+
347
+ /**
348
+ * `#rrggbb` and `#rrggbbaa` (3.8).
349
+ *
350
+ * A run of the wrong length, or one carrying a character that is not a
351
+ * hexadecimal digit, is a colour the writer had nearly right, and it is
352
+ * OS1027 over the whole of what they wrote. The OS1001 sentence about `#`
353
+ * says to delete the character or to put the text in a string, which is the
354
+ * answer for a `#` standing on its own and would throw the colour away here.
355
+ */
356
+ #readColour(start: number): void {
357
+ const word = this.#word(start + 1);
358
+ const digits = this.#text.slice(start + 1, word.end);
359
+ const isColour =
360
+ word.foreign === undefined &&
361
+ (digits.length === 6 || digits.length === 8) &&
362
+ HEX_ONLY.test(digits);
363
+
364
+ if (isColour) {
365
+ this.#emit({
366
+ kind: 'hexColor',
367
+ span: this.#span(start, word.end - start),
368
+ text: this.#text.slice(start, word.end),
369
+ });
370
+ this.#offset = word.end;
371
+ return;
372
+ }
373
+
374
+ if (digits.length === 0) {
375
+ // Nothing was written against it, so there is no colour here to have got
376
+ // wrong: this is the character on its own (3.1).
377
+ this.#sink.report('OS1001', this.#span(start, 1), describeUnexpected('#'));
378
+ this.#droppedAfterToken = true;
379
+ this.#offset = start + 1;
380
+ return;
381
+ }
382
+
383
+ this.#sink.report('OS1027', this.#span(start, word.end - start), {
384
+ written: this.#text.slice(start, word.end),
385
+ });
386
+ this.#droppedAfterToken = true;
387
+ this.#offset = word.end;
388
+ }
389
+
390
+ #readSemicolon(start: number): void {
391
+ this.#sink.report('OS1007', this.#span(start, 1), {});
392
+ this.#droppedAfterToken = true;
393
+ this.#offset = start + 1;
394
+
395
+ // The fix is to put the second statement on its own line, so the token
396
+ // stream is given exactly that and the rest of the line is read as the
397
+ // statement the writer meant. Inside brackets there is no statement to end,
398
+ // so there the character is only dropped.
399
+ if (this.#brackets > 0 || !this.#lineHasToken) return;
400
+ this.#layout.endStatement();
401
+ this.#emitNewline(this.#span(start, 1));
402
+ this.#layout.beginStatement(this.#line, this.#indent);
403
+ }
404
+
405
+ #readPunctuation(start: number): void {
406
+ const pair = this.#text.slice(start, start + 2);
407
+
408
+ if (REJECTED_PAIRS.has(pair)) {
409
+ this.#sink.report('OS1001', this.#span(start, 2), describeUnexpected(pair));
410
+ this.#droppedAfterToken = true;
411
+ this.#offset = start + 2;
412
+ return;
413
+ }
414
+
415
+ if (TWO_CHARACTER.has(pair)) {
416
+ this.#emit({ kind: pair as SimpleToken['kind'], span: this.#span(start, 2), text: pair });
417
+ this.#offset = start + 2;
418
+ return;
419
+ }
420
+
421
+ const one = this.#text.charAt(start);
422
+ if (ONE_CHARACTER.has(one)) {
423
+ if (OPENING_BRACKETS.has(one)) this.#brackets++;
424
+ else if (CLOSING_BRACKETS.has(one) && this.#brackets > 0) this.#brackets--;
425
+ this.#emit({ kind: one as SimpleToken['kind'], span: this.#span(start, 1), text: one });
426
+ this.#offset = start + 1;
427
+ return;
428
+ }
429
+
430
+ // Everything the language does not have, reported where it sits rather than
431
+ // three tokens later (3.1). An astral character is two code units and is
432
+ // taken whole, so a message never quotes half of one.
433
+ const written = String.fromCodePoint(this.#text.codePointAt(start) ?? 0);
434
+ this.#sink.report('OS1001', this.#span(start, written.length), describeUnexpected(written));
435
+ this.#droppedAfterToken = true;
436
+ this.#offset = start + written.length;
437
+ }
438
+
439
+ #word(from: number): Word {
440
+ let i = from;
441
+ let foreign: Word['foreign'];
442
+ for (;;) {
443
+ while (isNamePart(this.#text.charCodeAt(i))) i++;
444
+ const point = this.#text.codePointAt(i);
445
+ if (point === undefined || !isNonAsciiLetter(point)) return { end: i, foreign };
446
+ const written = String.fromCodePoint(point);
447
+ foreign ??= { offset: i, written };
448
+ i += written.length;
449
+ }
450
+ }
451
+
452
+ // -------------------------------------------------------------------------
453
+ // Emitting
454
+ // -------------------------------------------------------------------------
455
+
456
+ #emit(token: Token): void {
457
+ this.#layout.admit(this.#continuing);
458
+ this.#layout.observe(token.kind);
459
+ this.#lineHasToken = true;
460
+ this.#droppedAfterToken = false;
461
+ this.#tokens.push(token);
462
+ }
463
+
464
+ /** One newline per statement, never two, and never one before a statement has begun. */
465
+ #emitNewline(span: Span): void {
466
+ const last = this.#tokens[this.#tokens.length - 1];
467
+ if (last === undefined || last.kind === 'newline') return;
468
+ this.#tokens.push({ kind: 'newline', span, text: '' });
469
+ }
470
+
471
+ #span(offset: number, length: number): Span {
472
+ return this.#file.spanAt(offset, length);
473
+ }
474
+ }
475
+
476
+ /**
477
+ * Reads a file into tokens, reporting what it cannot read and carrying on.
478
+ *
479
+ * Nothing here throws. One character the language does not have should cost the
480
+ * reader that character and not the rest of the file, so every lexical error is
481
+ * reported against its own span and the scan resumes at the next thing it can
482
+ * recognise.
483
+ */
484
+ export function lex(file: SourceFile, diagnostics: DiagnosticSink): readonly Token[] {
485
+ return new Lexer(file, diagnostics).run();
486
+ }