openalgo-script 0.1.0-alpha.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (266) hide show
  1. package/LICENSE +201 -0
  2. package/NOTICE +12 -0
  3. package/README.md +255 -0
  4. package/dist/core/ast/annotations.d.ts +56 -0
  5. package/dist/core/ast/annotations.d.ts.map +1 -0
  6. package/dist/core/ast/annotations.js +27 -0
  7. package/dist/core/ast/annotations.js.map +1 -0
  8. package/dist/core/ast/build.d.ts +25 -0
  9. package/dist/core/ast/build.d.ts.map +1 -0
  10. package/dist/core/ast/build.js +26 -0
  11. package/dist/core/ast/build.js.map +1 -0
  12. package/dist/core/ast/children.d.ts +15 -0
  13. package/dist/core/ast/children.d.ts.map +1 -0
  14. package/dist/core/ast/children.js +103 -0
  15. package/dist/core/ast/children.js.map +1 -0
  16. package/dist/core/ast/expressions.d.ts +178 -0
  17. package/dist/core/ast/expressions.d.ts.map +1 -0
  18. package/dist/core/ast/expressions.js +50 -0
  19. package/dist/core/ast/expressions.js.map +1 -0
  20. package/dist/core/ast/header.d.ts +50 -0
  21. package/dist/core/ast/header.d.ts.map +1 -0
  22. package/dist/core/ast/header.js +2 -0
  23. package/dist/core/ast/header.js.map +1 -0
  24. package/dist/core/ast/index.d.ts +30 -0
  25. package/dist/core/ast/index.d.ts.map +1 -0
  26. package/dist/core/ast/index.js +21 -0
  27. package/dist/core/ast/index.js.map +1 -0
  28. package/dist/core/ast/name.d.ts +24 -0
  29. package/dist/core/ast/name.d.ts.map +1 -0
  30. package/dist/core/ast/name.js +2 -0
  31. package/dist/core/ast/name.js.map +1 -0
  32. package/dist/core/ast/node.d.ts +34 -0
  33. package/dist/core/ast/node.d.ts.map +1 -0
  34. package/dist/core/ast/node.js +70 -0
  35. package/dist/core/ast/node.js.map +1 -0
  36. package/dist/core/ast/script.d.ts +57 -0
  37. package/dist/core/ast/script.d.ts.map +1 -0
  38. package/dist/core/ast/script.js +2 -0
  39. package/dist/core/ast/script.js.map +1 -0
  40. package/dist/core/ast/statements.d.ts +161 -0
  41. package/dist/core/ast/statements.d.ts.map +1 -0
  42. package/dist/core/ast/statements.js +10 -0
  43. package/dist/core/ast/statements.js.map +1 -0
  44. package/dist/core/ast/walk.d.ts +39 -0
  45. package/dist/core/ast/walk.d.ts.map +1 -0
  46. package/dist/core/ast/walk.js +54 -0
  47. package/dist/core/ast/walk.js.map +1 -0
  48. package/dist/core/catalogue/catalogue.generated.d.ts +1590 -0
  49. package/dist/core/catalogue/catalogue.generated.d.ts.map +1 -0
  50. package/dist/core/catalogue/catalogue.generated.js +162 -0
  51. package/dist/core/catalogue/catalogue.generated.js.map +1 -0
  52. package/dist/core/catalogue/index.d.ts +15 -0
  53. package/dist/core/catalogue/index.d.ts.map +1 -0
  54. package/dist/core/catalogue/index.js +4 -0
  55. package/dist/core/catalogue/index.js.map +1 -0
  56. package/dist/core/catalogue/lookup.d.ts +14 -0
  57. package/dist/core/catalogue/lookup.d.ts.map +1 -0
  58. package/dist/core/catalogue/lookup.js +19 -0
  59. package/dist/core/catalogue/lookup.js.map +1 -0
  60. package/dist/core/catalogue/template.d.ts +15 -0
  61. package/dist/core/catalogue/template.d.ts.map +1 -0
  62. package/dist/core/catalogue/template.js +33 -0
  63. package/dist/core/catalogue/template.js.map +1 -0
  64. package/dist/core/catalogue/types.d.ts +34 -0
  65. package/dist/core/catalogue/types.d.ts.map +1 -0
  66. package/dist/core/catalogue/types.js +6 -0
  67. package/dist/core/catalogue/types.js.map +1 -0
  68. package/dist/core/catalogue/values.generated.d.ts +565 -0
  69. package/dist/core/catalogue/values.generated.d.ts.map +1 -0
  70. package/dist/core/catalogue/values.generated.js +6 -0
  71. package/dist/core/catalogue/values.generated.js.map +1 -0
  72. package/dist/core/diagnostics/collector.d.ts +46 -0
  73. package/dist/core/diagnostics/collector.d.ts.map +1 -0
  74. package/dist/core/diagnostics/collector.js +55 -0
  75. package/dist/core/diagnostics/collector.js.map +1 -0
  76. package/dist/core/diagnostics/diagnostic.d.ts +40 -0
  77. package/dist/core/diagnostics/diagnostic.d.ts.map +1 -0
  78. package/dist/core/diagnostics/diagnostic.js +29 -0
  79. package/dist/core/diagnostics/diagnostic.js.map +1 -0
  80. package/dist/core/diagnostics/index.d.ts +12 -0
  81. package/dist/core/diagnostics/index.d.ts.map +1 -0
  82. package/dist/core/diagnostics/index.js +3 -0
  83. package/dist/core/diagnostics/index.js.map +1 -0
  84. package/dist/core/index.d.ts +28 -0
  85. package/dist/core/index.d.ts.map +1 -0
  86. package/dist/core/index.js +22 -0
  87. package/dist/core/index.js.map +1 -0
  88. package/dist/core/lex/blocks.d.ts +51 -0
  89. package/dist/core/lex/blocks.d.ts.map +1 -0
  90. package/dist/core/lex/blocks.js +56 -0
  91. package/dist/core/lex/blocks.js.map +1 -0
  92. package/dist/core/lex/characters.d.ts +38 -0
  93. package/dist/core/lex/characters.d.ts.map +1 -0
  94. package/dist/core/lex/characters.js +50 -0
  95. package/dist/core/lex/characters.js.map +1 -0
  96. package/dist/core/lex/comments.d.ts +50 -0
  97. package/dist/core/lex/comments.d.ts.map +1 -0
  98. package/dist/core/lex/comments.js +64 -0
  99. package/dist/core/lex/comments.js.map +1 -0
  100. package/dist/core/lex/index.d.ts +12 -0
  101. package/dist/core/lex/index.d.ts.map +1 -0
  102. package/dist/core/lex/index.js +12 -0
  103. package/dist/core/lex/index.js.map +1 -0
  104. package/dist/core/lex/layout.d.ts +40 -0
  105. package/dist/core/lex/layout.d.ts.map +1 -0
  106. package/dist/core/lex/layout.js +143 -0
  107. package/dist/core/lex/layout.js.map +1 -0
  108. package/dist/core/lex/lexer.d.ts +13 -0
  109. package/dist/core/lex/lexer.d.ts.map +1 -0
  110. package/dist/core/lex/lexer.js +419 -0
  111. package/dist/core/lex/lexer.js.map +1 -0
  112. package/dist/core/lex/numbers.d.ts +22 -0
  113. package/dist/core/lex/numbers.d.ts.map +1 -0
  114. package/dist/core/lex/numbers.js +58 -0
  115. package/dist/core/lex/numbers.js.map +1 -0
  116. package/dist/core/lex/statements.d.ts +24 -0
  117. package/dist/core/lex/statements.d.ts.map +1 -0
  118. package/dist/core/lex/statements.js +90 -0
  119. package/dist/core/lex/statements.js.map +1 -0
  120. package/dist/core/lex/strings.d.ts +27 -0
  121. package/dist/core/lex/strings.d.ts.map +1 -0
  122. package/dist/core/lex/strings.js +92 -0
  123. package/dist/core/lex/strings.js.map +1 -0
  124. package/dist/core/lex/unexpected.d.ts +31 -0
  125. package/dist/core/lex/unexpected.d.ts.map +1 -0
  126. package/dist/core/lex/unexpected.js +114 -0
  127. package/dist/core/lex/unexpected.js.map +1 -0
  128. package/dist/core/parse/annotations.d.ts +4 -0
  129. package/dist/core/parse/annotations.d.ts.map +1 -0
  130. package/dist/core/parse/annotations.js +49 -0
  131. package/dist/core/parse/annotations.js.map +1 -0
  132. package/dist/core/parse/blocks.d.ts +18 -0
  133. package/dist/core/parse/blocks.d.ts.map +1 -0
  134. package/dist/core/parse/blocks.js +47 -0
  135. package/dist/core/parse/blocks.js.map +1 -0
  136. package/dist/core/parse/branches.d.ts +17 -0
  137. package/dist/core/parse/branches.d.ts.map +1 -0
  138. package/dist/core/parse/branches.js +66 -0
  139. package/dist/core/parse/branches.js.map +1 -0
  140. package/dist/core/parse/cursor.d.ts +119 -0
  141. package/dist/core/parse/cursor.d.ts.map +1 -0
  142. package/dist/core/parse/cursor.js +305 -0
  143. package/dist/core/parse/cursor.js.map +1 -0
  144. package/dist/core/parse/expressions.d.ts +57 -0
  145. package/dist/core/parse/expressions.d.ts.map +1 -0
  146. package/dist/core/parse/expressions.js +373 -0
  147. package/dist/core/parse/expressions.js.map +1 -0
  148. package/dist/core/parse/functions.d.ts +16 -0
  149. package/dist/core/parse/functions.d.ts.map +1 -0
  150. package/dist/core/parse/functions.js +92 -0
  151. package/dist/core/parse/functions.js.map +1 -0
  152. package/dist/core/parse/index.d.ts +13 -0
  153. package/dist/core/parse/index.d.ts.map +1 -0
  154. package/dist/core/parse/index.js +13 -0
  155. package/dist/core/parse/index.js.map +1 -0
  156. package/dist/core/parse/loops.d.ts +13 -0
  157. package/dist/core/parse/loops.d.ts.map +1 -0
  158. package/dist/core/parse/loops.js +78 -0
  159. package/dist/core/parse/loops.js.map +1 -0
  160. package/dist/core/parse/names.d.ts +34 -0
  161. package/dist/core/parse/names.d.ts.map +1 -0
  162. package/dist/core/parse/names.js +122 -0
  163. package/dist/core/parse/names.js.map +1 -0
  164. package/dist/core/parse/script.d.ts +23 -0
  165. package/dist/core/parse/script.d.ts.map +1 -0
  166. package/dist/core/parse/script.js +55 -0
  167. package/dist/core/parse/script.js.map +1 -0
  168. package/dist/core/parse/statements.d.ts +14 -0
  169. package/dist/core/parse/statements.d.ts.map +1 -0
  170. package/dist/core/parse/statements.js +292 -0
  171. package/dist/core/parse/statements.js.map +1 -0
  172. package/dist/core/parse/switches.d.ts +16 -0
  173. package/dist/core/parse/switches.d.ts.map +1 -0
  174. package/dist/core/parse/switches.js +98 -0
  175. package/dist/core/parse/switches.js.map +1 -0
  176. package/dist/core/render/index.d.ts +10 -0
  177. package/dist/core/render/index.d.ts.map +1 -0
  178. package/dist/core/render/index.js +10 -0
  179. package/dist/core/render/index.js.map +1 -0
  180. package/dist/core/render/terminal.d.ts +11 -0
  181. package/dist/core/render/terminal.d.ts.map +1 -0
  182. package/dist/core/render/terminal.js +51 -0
  183. package/dist/core/render/terminal.js.map +1 -0
  184. package/dist/core/source/index.d.ts +7 -0
  185. package/dist/core/source/index.d.ts.map +1 -0
  186. package/dist/core/source/index.js +2 -0
  187. package/dist/core/source/index.js.map +1 -0
  188. package/dist/core/source/source.d.ts +41 -0
  189. package/dist/core/source/source.d.ts.map +1 -0
  190. package/dist/core/source/source.js +67 -0
  191. package/dist/core/source/source.js.map +1 -0
  192. package/dist/core/span/index.d.ts +3 -0
  193. package/dist/core/span/index.d.ts.map +1 -0
  194. package/dist/core/span/index.js +2 -0
  195. package/dist/core/span/index.js.map +1 -0
  196. package/dist/core/span/span.d.ts +67 -0
  197. package/dist/core/span/span.d.ts.map +1 -0
  198. package/dist/core/span/span.js +27 -0
  199. package/dist/core/span/span.js.map +1 -0
  200. package/dist/core/tokens/index.d.ts +8 -0
  201. package/dist/core/tokens/index.d.ts.map +1 -0
  202. package/dist/core/tokens/index.js +2 -0
  203. package/dist/core/tokens/index.js.map +1 -0
  204. package/dist/core/tokens/kind.d.ts +53 -0
  205. package/dist/core/tokens/kind.d.ts.map +1 -0
  206. package/dist/core/tokens/kind.js +94 -0
  207. package/dist/core/tokens/kind.js.map +1 -0
  208. package/dist/core/tokens/token.d.ts +36 -0
  209. package/dist/core/tokens/token.d.ts.map +1 -0
  210. package/dist/core/tokens/token.js +2 -0
  211. package/dist/core/tokens/token.js.map +1 -0
  212. package/package.json +52 -0
  213. package/spec/README.md +48 -0
  214. package/spec/errors.json +3204 -0
  215. package/src/core/ast/annotations.ts +68 -0
  216. package/src/core/ast/build.ts +35 -0
  217. package/src/core/ast/children.ts +110 -0
  218. package/src/core/ast/expressions.ts +254 -0
  219. package/src/core/ast/header.ts +53 -0
  220. package/src/core/ast/index.ts +102 -0
  221. package/src/core/ast/name.ts +24 -0
  222. package/src/core/ast/node.ts +120 -0
  223. package/src/core/ast/script.ts +60 -0
  224. package/src/core/ast/statements.ts +193 -0
  225. package/src/core/ast/walk.ts +67 -0
  226. package/src/core/catalogue/catalogue.generated.ts +171 -0
  227. package/src/core/catalogue/index.ts +20 -0
  228. package/src/core/catalogue/lookup.ts +23 -0
  229. package/src/core/catalogue/template.ts +38 -0
  230. package/src/core/catalogue/types.ts +37 -0
  231. package/src/core/catalogue/values.generated.ts +160 -0
  232. package/src/core/diagnostics/collector.ts +87 -0
  233. package/src/core/diagnostics/diagnostic.ts +72 -0
  234. package/src/core/diagnostics/index.ts +12 -0
  235. package/src/core/index.ts +146 -0
  236. package/src/core/lex/blocks.ts +92 -0
  237. package/src/core/lex/characters.ts +56 -0
  238. package/src/core/lex/comments.ts +75 -0
  239. package/src/core/lex/index.ts +11 -0
  240. package/src/core/lex/layout.ts +188 -0
  241. package/src/core/lex/lexer.ts +486 -0
  242. package/src/core/lex/numbers.ts +77 -0
  243. package/src/core/lex/statements.ts +94 -0
  244. package/src/core/lex/strings.ts +119 -0
  245. package/src/core/lex/unexpected.ts +125 -0
  246. package/src/core/parse/annotations.ts +60 -0
  247. package/src/core/parse/blocks.ts +52 -0
  248. package/src/core/parse/branches.ts +77 -0
  249. package/src/core/parse/cursor.ts +348 -0
  250. package/src/core/parse/expressions.ts +441 -0
  251. package/src/core/parse/functions.ts +109 -0
  252. package/src/core/parse/index.ts +12 -0
  253. package/src/core/parse/loops.ts +97 -0
  254. package/src/core/parse/names.ts +135 -0
  255. package/src/core/parse/script.ts +64 -0
  256. package/src/core/parse/statements.ts +329 -0
  257. package/src/core/parse/switches.ts +105 -0
  258. package/src/core/render/index.ts +9 -0
  259. package/src/core/render/terminal.ts +62 -0
  260. package/src/core/source/index.ts +6 -0
  261. package/src/core/source/source.ts +102 -0
  262. package/src/core/span/index.ts +2 -0
  263. package/src/core/span/span.ts +84 -0
  264. package/src/core/tokens/index.ts +15 -0
  265. package/src/core/tokens/kind.ts +123 -0
  266. package/src/core/tokens/token.ts +39 -0
@@ -0,0 +1,77 @@
1
+ import { DOT, isDigit, isHexDigit, UNDERSCORE } from './characters.js';
2
+
3
+ /**
4
+ * A number literal, in every form of language.md 3.5: decimal with or without a
5
+ * fractional part, a fractional part with no leading digit, an exponent, digit
6
+ * group underscores and the hexadecimal form. There is no octal form, so `010`
7
+ * is ten, which falls out of reading the digits rather than being a special
8
+ * case.
9
+ */
10
+ export interface NumberLiteral {
11
+ /** One past the last character of the literal. */
12
+ readonly end: number;
13
+ readonly value: number;
14
+ }
15
+
16
+ /**
17
+ * An underscore separates digit groups, so it only belongs to the literal when
18
+ * a digit follows it. `1_` and `1__0` therefore end the literal at the first
19
+ * underscore, and the caller reports what follows as a name that starts with a
20
+ * digit, which is what they are.
21
+ */
22
+ function runOfDigits(text: string, from: number, digit: (code: number) => boolean): number {
23
+ let i = from;
24
+ for (;;) {
25
+ const code = text.charCodeAt(i);
26
+ if (digit(code)) {
27
+ i++;
28
+ continue;
29
+ }
30
+ if (code === UNDERSCORE && digit(text.charCodeAt(i + 1))) {
31
+ i += 2;
32
+ continue;
33
+ }
34
+ return i;
35
+ }
36
+ }
37
+
38
+ /**
39
+ * Reads the literal starting at `start`, which the caller has established
40
+ * begins a number: a digit, or a dot with a digit after it.
41
+ *
42
+ * The value is computed here and carried on the token so that no later stage
43
+ * re-reads the text and has to agree a second time about underscores and about
44
+ * the hexadecimal form.
45
+ */
46
+ export function scanNumber(text: string, start: number): NumberLiteral {
47
+ const second = text.charCodeAt(start + 1);
48
+ const isHex =
49
+ text.charCodeAt(start) === 0x30 &&
50
+ (second === 0x78 || second === 0x58) &&
51
+ isHexDigit(text.charCodeAt(start + 2));
52
+
53
+ if (isHex) {
54
+ const end = runOfDigits(text, start + 2, isHexDigit);
55
+ return { end, value: Number(text.slice(start, end).replace(/_/g, '')) };
56
+ }
57
+
58
+ let i = runOfDigits(text, start, isDigit);
59
+
60
+ // A dot only belongs to the literal when a digit follows it, so `x.y` reads as
61
+ // an element access and `1.` reads as a number and a dot.
62
+ if (text.charCodeAt(i) === DOT && isDigit(text.charCodeAt(i + 1))) {
63
+ i = runOfDigits(text, i + 1, isDigit);
64
+ }
65
+
66
+ const exponent = text.charCodeAt(i);
67
+ if (exponent === 0x65 || exponent === 0x45) {
68
+ const sign = text.charCodeAt(i + 1);
69
+ const digits = sign === 0x2b || sign === 0x2d ? i + 2 : i + 1;
70
+ // Without a digit the `e` is not an exponent, and the caller reports the
71
+ // letter sitting against a number rather than this reading a literal that
72
+ // is not there.
73
+ if (isDigit(text.charCodeAt(digits))) i = runOfDigits(text, digits, isDigit);
74
+ }
75
+
76
+ return { end: i, value: Number(text.slice(start, i).replace(/_/g, '')) };
77
+ }
@@ -0,0 +1,94 @@
1
+ import type { Token, TokenKind } from '../tokens/index.js';
2
+
3
+ /**
4
+ * Where a statement ends, and whether it opens a block. Both questions are
5
+ * asked of the tokens of a line, which is the whole of what language.md 3.10
6
+ * and 3.11 need to know about a statement before it is parsed.
7
+ */
8
+
9
+ /**
10
+ * The tokens that promise a right-hand side, from 3.11: a binary operator, a
11
+ * comma, a `?`, a `:` or an `=`. A line ending in one of them continues onto
12
+ * the next line.
13
+ *
14
+ * The compound assignments are here on the same terms as `=`: each one promises
15
+ * a value just as plainly. `.` is deliberately absent, because 3.11 does not
16
+ * list it and a line that ends in a dot is a mistake rather than a wrap.
17
+ */
18
+ const CONTINUING: ReadonlySet<TokenKind> = new Set<TokenKind>([
19
+ '+',
20
+ '-',
21
+ '*',
22
+ '/',
23
+ '%',
24
+ '==',
25
+ '!=',
26
+ '<',
27
+ '<=',
28
+ '>',
29
+ '>=',
30
+ '=',
31
+ '+=',
32
+ '-=',
33
+ '*=',
34
+ '/=',
35
+ '%=',
36
+ ',',
37
+ '?',
38
+ ':',
39
+ 'and',
40
+ 'or',
41
+ ]);
42
+
43
+ /** The words that introduce a block, per 3.10 and the grammar of section 19. */
44
+ const HEADERS: ReadonlySet<TokenKind> = new Set<TokenKind>([
45
+ 'if',
46
+ 'else',
47
+ 'for',
48
+ 'while',
49
+ 'switch',
50
+ 'case',
51
+ 'default',
52
+ ]);
53
+
54
+ /**
55
+ * Whether the last two tokens are the `=>` of a function declaration.
56
+ *
57
+ * Section 3.12 does not list `=>` among the punctuation, so it arrives as an
58
+ * `=` and a `>` that touch. Both readings need saying apart here: a line whose
59
+ * last token is `>` continues onto the next line, and a line that ends in `=>`
60
+ * is a function header whose body is the indented block below it, so it must
61
+ * not.
62
+ */
63
+ export function endsWithArrow(tokens: readonly Token[]): boolean {
64
+ const last = tokens[tokens.length - 1];
65
+ const before = tokens[tokens.length - 2];
66
+ if (last === undefined || before === undefined) return false;
67
+ return (
68
+ last.kind === '>' &&
69
+ before.kind === '=' &&
70
+ before.span.offset + before.span.length === last.span.offset
71
+ );
72
+ }
73
+
74
+ /** Whether the statement so far promises more, so the next line belongs to it. */
75
+ export function continuesLine(tokens: readonly Token[]): boolean {
76
+ const last = tokens[tokens.length - 1];
77
+ if (last === undefined) return false;
78
+ if (endsWithArrow(tokens)) return false;
79
+ return CONTINUING.has(last.kind);
80
+ }
81
+
82
+ /**
83
+ * Whether a statement that begins with this token and ends this way opens a
84
+ * block.
85
+ *
86
+ * `fn` is the one header with a form that fits on one line: with its body after
87
+ * the `=>` it opens nothing, and with nothing after the `=>` the block below it
88
+ * is the body (3.10, 11.1). Every other header opens a block always, because
89
+ * there is no statement separator to end a one-line body with.
90
+ */
91
+ export function opensBlock(firstKind: TokenKind, arrowAtEnd: boolean): boolean {
92
+ if (firstKind === 'fn') return arrowAtEnd;
93
+ return HEADERS.has(firstKind);
94
+ }
@@ -0,0 +1,119 @@
1
+ import { BACKSLASH, isHexDigit, LINE_FEED } from './characters.js';
2
+
3
+ /** A backslash sequence the language does not define, for OS1005. */
4
+ export interface BadEscape {
5
+ readonly offset: number;
6
+ readonly length: number;
7
+ /** The backslash and what followed it, which is the {sequence} of OS1005. */
8
+ readonly text: string;
9
+ }
10
+
11
+ export interface StringLiteral {
12
+ /** One past the closing quote, or the end of the line when there was none. */
13
+ readonly end: number;
14
+ /** The text the literal denotes, with its escapes already resolved. */
15
+ readonly value: string;
16
+ readonly unterminated: boolean;
17
+ readonly badEscapes: readonly BadEscape[];
18
+ }
19
+
20
+ /** The escapes of language.md 3.6, apart from \uXXXX which carries its own digits. */
21
+ const SIMPLE = new Map<number, string>([
22
+ [0x5c, '\\'],
23
+ [0x22, '"'],
24
+ [0x27, "'"],
25
+ [0x6e, '\n'],
26
+ [0x74, '\t'],
27
+ [0x72, '\r'],
28
+ [0x30, '\u0000'],
29
+ ]);
30
+
31
+ const UNICODE_ESCAPE = 0x75;
32
+
33
+ /** How many hexadecimal digits follow, up to the four a \uXXXX escape needs. */
34
+ function hexDigitsAfter(text: string, from: number): number {
35
+ let count = 0;
36
+ while (count < 4 && isHexDigit(text.charCodeAt(from + count))) count++;
37
+ return count;
38
+ }
39
+
40
+ /**
41
+ * Reads the literal opening at `start`, which is the quote character, in either
42
+ * delimiter of language.md 3.6.
43
+ *
44
+ * A literal may not span a line, so the scan stops at the newline and says so
45
+ * rather than swallowing the rest of the file. The caller reports OS1004 at the
46
+ * opening quote, which is where the mistake is, and keeps the text that was
47
+ * read so that the statement around it still parses and the reader is not shown
48
+ * a second error for the same missing character.
49
+ */
50
+ export function scanString(text: string, start: number): StringLiteral {
51
+ const quote = text.charCodeAt(start);
52
+ const badEscapes: BadEscape[] = [];
53
+ let value = '';
54
+ let plainFrom = start + 1;
55
+ let i = start + 1;
56
+
57
+ const takePlain = (upTo: number): void => {
58
+ if (upTo > plainFrom) value += text.slice(plainFrom, upTo);
59
+ };
60
+
61
+ while (i < text.length) {
62
+ const code = text.charCodeAt(i);
63
+
64
+ if (code === LINE_FEED) break;
65
+
66
+ if (code === quote) {
67
+ takePlain(i);
68
+ return { end: i + 1, value, unterminated: false, badEscapes };
69
+ }
70
+
71
+ if (code !== BACKSLASH) {
72
+ i++;
73
+ continue;
74
+ }
75
+
76
+ const next = text.charCodeAt(i + 1);
77
+ // A backslash at the end of the line escapes nothing, because the line is
78
+ // where the literal has to end. That is one missing quote, not two errors.
79
+ if (Number.isNaN(next) || next === LINE_FEED) break;
80
+
81
+ takePlain(i);
82
+
83
+ const simple = SIMPLE.get(next);
84
+ if (simple !== undefined) {
85
+ value += simple;
86
+ i += 2;
87
+ plainFrom = i;
88
+ continue;
89
+ }
90
+
91
+ if (next === UNICODE_ESCAPE) {
92
+ const digits = hexDigitsAfter(text, i + 2);
93
+ if (digits === 4) {
94
+ value += String.fromCharCode(Number.parseInt(text.slice(i + 2, i + 6), 16));
95
+ i += 6;
96
+ plainFrom = i;
97
+ continue;
98
+ }
99
+ const length = 2 + digits;
100
+ badEscapes.push({ offset: i, length, text: text.slice(i, i + length) });
101
+ value += text.slice(i + 1, i + length);
102
+ i += length;
103
+ plainFrom = i;
104
+ continue;
105
+ }
106
+
107
+ // The written character, whole: an astral character is two code units, and
108
+ // reporting half of one puts a broken surrogate in the message.
109
+ const written = String.fromCodePoint(text.codePointAt(i + 1) ?? next);
110
+ const length = 1 + written.length;
111
+ badEscapes.push({ offset: i, length, text: text.slice(i, i + length) });
112
+ value += written;
113
+ i += length;
114
+ plainFrom = i;
115
+ }
116
+
117
+ takePlain(i);
118
+ return { end: i, value, unterminated: true, badEscapes };
119
+ }
@@ -0,0 +1,125 @@
1
+ /**
2
+ * OS1001: naming the character that was found, and the plain spelling to write
3
+ * instead.
4
+ *
5
+ * The replacement sentences are the substitution table of errors.md section 7,
6
+ * which is the one part of an OS1001 message that errors.json does not carry:
7
+ * the catalogue holds the message and the fix, and section 7 holds the third
8
+ * column that fills {suggestion}. The table is transcribed here rather than
9
+ * paraphrased, and the moment it moves into errors.json this file is generated
10
+ * away instead of being edited.
11
+ *
12
+ * A character the table does not name takes the generic sentence, which is what
13
+ * section 7 says to do with one.
14
+ */
15
+
16
+ export interface Unexpected {
17
+ /** The {char} slot: what was found, quoted, and named when it cannot be seen. */
18
+ readonly char: string;
19
+ /** The {suggestion} slot: one sentence naming the plain spelling to use. */
20
+ readonly suggestion: string;
21
+ }
22
+
23
+ const PLAIN_SPACE = 'Use a plain space instead.';
24
+ const STRAIGHT_QUOTE = 'Use a straight quote instead.';
25
+ const ASCII_NAME = 'Use the ASCII spelling of the name instead.';
26
+ const INDENTATION = 'Use indentation instead, which is how a block is written.';
27
+ const PUT_IN_A_STRING = 'Delete it, or put the text in a string.';
28
+ const GENERIC = 'Delete it, or move it inside a string literal.';
29
+
30
+ /** The rows of section 7 that name an exact spelling. */
31
+ const SPELLINGS = new Map<string, string>([
32
+ ['!', 'Write not instead.'],
33
+ ['&&', 'Write and instead.'],
34
+ ['||', 'Write or instead.'],
35
+ ['^', 'Write pow(a, b) instead.'],
36
+ ['**', 'Write pow(a, b) instead.'],
37
+ ['++', 'Write a += 1 instead.'],
38
+ ['{', INDENTATION],
39
+ ['}', INDENTATION],
40
+ ['#', PUT_IN_A_STRING],
41
+ ['$', PUT_IN_A_STRING],
42
+ ['@', PUT_IN_A_STRING],
43
+ ]);
44
+
45
+ /**
46
+ * Names for the characters a reader cannot see on their screen.
47
+ *
48
+ * A message that quotes an invisible character quotes nothing, and the whole
49
+ * point of OS1001 is that an invisible character produces a baffling error
50
+ * three tokens later. The name and the code point are what makes the message
51
+ * act on: a reader can search their file for the one and read the other aloud.
52
+ */
53
+ const NAMES = new Map<number, string>([
54
+ [0x0009, 'a tab'],
55
+ [0x000d, 'a carriage return'],
56
+ [0x00a0, 'a no-break space'],
57
+ [0x2002, 'an en space'],
58
+ [0x2003, 'an em space'],
59
+ [0x2007, 'a figure space'],
60
+ [0x2009, 'a thin space'],
61
+ [0x200a, 'a hair space'],
62
+ [0x202f, 'a narrow no-break space'],
63
+ [0x3000, 'an ideographic space'],
64
+ [0x200b, 'a zero width space'],
65
+ [0x200c, 'a zero width non-joiner'],
66
+ [0x200d, 'a zero width joiner'],
67
+ [0xfeff, 'a byte order mark'],
68
+ [0x2018, 'a left single quotation mark'],
69
+ [0x2019, 'a right single quotation mark'],
70
+ [0x201c, 'a left double quotation mark'],
71
+ [0x201d, 'a right double quotation mark'],
72
+ [0x2013, 'an en dash'],
73
+ [0x2014, 'an em dash'],
74
+ [0x2026, 'a horizontal ellipsis'],
75
+ ]);
76
+
77
+ const TYPOGRAPHIC_QUOTES = '\u2018\u2019\u201c\u201d\u201a\u201b\u201e\u201f';
78
+
79
+ const SPACE_SEPARATOR = /\p{Zs}/u;
80
+ const LETTER = /\p{L}/u;
81
+ const ASCII_GRAPHIC = /^[\x21-\x7e]+$/;
82
+
83
+ function codePointLabel(text: string): string {
84
+ const point = text.codePointAt(0) ?? 0;
85
+ return `U+${point.toString(16).toUpperCase().padStart(4, '0')}`;
86
+ }
87
+
88
+ /**
89
+ * The {char} slot. An ASCII graphic speaks for itself; everything else is
90
+ * quoted and then identified, because the quotes may well hold nothing a reader
91
+ * can see.
92
+ */
93
+ function describe(written: string): string {
94
+ if (ASCII_GRAPHIC.test(written)) return `"${written}"`;
95
+ const name = NAMES.get(written.codePointAt(0) ?? -1);
96
+ const label = codePointLabel(written);
97
+ return name === undefined ? `"${written}" (${label})` : `"${written}" (${name}, ${label})`;
98
+ }
99
+
100
+ function suggestionFor(written: string): string {
101
+ const spelling = SPELLINGS.get(written);
102
+ if (spelling !== undefined) return spelling;
103
+ // A tab is a space that happens to be a control character, and the table's
104
+ // answer to every other invisible space is the same plain space.
105
+ if (written === '\t' || SPACE_SEPARATOR.test(written)) return PLAIN_SPACE;
106
+ if (TYPOGRAPHIC_QUOTES.includes(written)) return STRAIGHT_QUOTE;
107
+ if (LETTER.test(written)) return ASCII_NAME;
108
+ return GENERIC;
109
+ }
110
+
111
+ /** What OS1001 says about one rejected character, or one rejected operator. */
112
+ export function describeUnexpected(written: string): Unexpected {
113
+ return { char: describe(written), suggestion: suggestionFor(written) };
114
+ }
115
+
116
+ /**
117
+ * Whether a code point is a letter the language does not allow in a name.
118
+ *
119
+ * It is asked once per word, at the character that ended the word, so that
120
+ * `l\u00e4ngd` is one name with one error on the one character 3.3 refuses,
121
+ * rather than two names with a stray character between them.
122
+ */
123
+ export function isNonAsciiLetter(point: number): boolean {
124
+ return point > 0x7f && LETTER.test(String.fromCodePoint(point));
125
+ }
@@ -0,0 +1,60 @@
1
+ import type { TypeAnnotation } from '../ast/index.js';
2
+ import { makeNode } from '../ast/index.js';
3
+ import { spanning } from '../span/index.js';
4
+ import type { TokenKind } from '../tokens/index.js';
5
+ import { RESERVED_WORDS } from '../tokens/index.js';
6
+ import type { Cursor } from './cursor.js';
7
+
8
+ /**
9
+ * A type as a script writes it, on a parameter or a `var`.
10
+ *
11
+ * The grammar of section 19 admits only the real type names. The prose wins
12
+ * here, because OS2016 and OS2019 are check stage: this reads any word into a
13
+ * `namedType` and lets the checker be the one to say that `whole` is not a
14
+ * type, which is the difference between a caret with no sentence and a sentence
15
+ * naming the types that do exist.
16
+ */
17
+
18
+ const RESERVED: ReadonlySet<TokenKind> = new Set<TokenKind>(RESERVED_WORDS);
19
+
20
+ /**
21
+ * What may stand as a type name.
22
+ *
23
+ * A literal is taken as well as a word, so `len: 5` reaches OS2016 with `5` in
24
+ * its message instead of failing here with a caret and nothing to act on.
25
+ */
26
+ function isTypeWord(kind: TokenKind): boolean {
27
+ return (
28
+ kind === 'identifier' ||
29
+ kind === 'numberLiteral' ||
30
+ kind === 'stringLiteral' ||
31
+ RESERVED.has(kind)
32
+ );
33
+ }
34
+
35
+ export function parseTypeAnnotation(cursor: Cursor): TypeAnnotation {
36
+ const token = cursor.token;
37
+
38
+ if (token.kind === 'series') {
39
+ cursor.advance();
40
+ const element = parseTypeAnnotation(cursor);
41
+ return makeNode('seriesType', spanning(token.span, element.span), { element });
42
+ }
43
+
44
+ if (token.kind === 'array' && cursor.peek().kind === '<') {
45
+ cursor.advance();
46
+ cursor.advance();
47
+ const element = parseTypeAnnotation(cursor);
48
+ const close = cursor.take('>');
49
+ if (close === undefined) cursor.unexpected(cursor.token);
50
+ return makeNode('arrayType', spanning(token.span, close?.span ?? element.span), { element });
51
+ }
52
+
53
+ if (isTypeWord(token.kind)) {
54
+ cursor.advance();
55
+ return makeNode('namedType', token.span, { name: token.text });
56
+ }
57
+
58
+ cursor.unexpected(token);
59
+ return makeNode('namedType', cursor.holeSpan(), { name: '' });
60
+ }
@@ -0,0 +1,52 @@
1
+ import type { Block, FunctionDeclaration, Statement } from '../ast/index.js';
2
+ import { makeNode } from '../ast/index.js';
3
+ import type { Span } from '../span/index.js';
4
+ import { spanning } from '../span/index.js';
5
+ import type { Cursor } from './cursor.js';
6
+ import { parseStatement } from './statements.js';
7
+
8
+ /**
9
+ * The indented lines under a header, language.md 3.10.
10
+ *
11
+ * The lexer has already decided where a block begins and ends, so this reads an
12
+ * `indent`, statements, and a `dedent`, and the one rule it owns is what to do
13
+ * when the `indent` is not there: OS1010, reported against the header rather
14
+ * than against the line that failed to be a body, because the header is what
15
+ * the reader has to change.
16
+ *
17
+ * The statement loop insists on progress. A rule that reports without consuming
18
+ * anything would otherwise spin on the token it could not read, and a compiler
19
+ * that hangs on a malformed file is worse than one that gives up on it.
20
+ */
21
+ export function parseBlock(cursor: Cursor, header: string, at: Span): Block {
22
+ if (!cursor.at('indent')) {
23
+ // A header that was already reported on has nothing to add by saying its
24
+ // body is missing too: the body is missing because of what it was told.
25
+ if (!cursor.reportedHere) cursor.report('OS1010', at, { header });
26
+ const hole = cursor.holeSpan();
27
+ return makeNode('block', hole, { statements: [] });
28
+ }
29
+
30
+ if (!cursor.openBlock(cursor.token.span)) {
31
+ const span = cursor.token.span;
32
+ cursor.skipBlockBody();
33
+ return makeNode('block', span, { statements: [] });
34
+ }
35
+
36
+ const indent = cursor.advance();
37
+ const statements: (Statement | FunctionDeclaration)[] = [];
38
+
39
+ while (!cursor.at('dedent') && !cursor.at('endOfFile')) {
40
+ const before = cursor.position;
41
+ const statement = parseStatement(cursor);
42
+ if (statement !== undefined) statements.push(statement);
43
+ if (cursor.position === before) cursor.skipStatement();
44
+ }
45
+
46
+ cursor.closeBlock();
47
+ cursor.take('dedent');
48
+
49
+ const last = statements[statements.length - 1];
50
+ const span = last === undefined ? indent.span : spanning(indent.span, last.span);
51
+ return makeNode('block', span, { statements });
52
+ }
@@ -0,0 +1,77 @@
1
+ import type { ElseBranch, IfBranch, IfStatement } from '../ast/index.js';
2
+ import { makeNode } from '../ast/index.js';
3
+ import type { Span } from '../span/index.js';
4
+ import { spanning } from '../span/index.js';
5
+ import type { Token } from '../tokens/index.js';
6
+ import type { Cursor } from './cursor.js';
7
+ import { parseBlock } from './blocks.js';
8
+ import { parseCondition } from './expressions.js';
9
+ import { finishStatement } from './statements.js';
10
+
11
+ /**
12
+ * `if`, `else if` and `else`, language.md 10.2.
13
+ *
14
+ * The branches are kept beside each other rather than nested, because `else if`
15
+ * is two words on one line and does not increase indentation, so a reader sees
16
+ * one statement with several arms and every pass over the tree should see the
17
+ * same thing.
18
+ *
19
+ * An `else` pairs with the `if` at its own indentation. The lexer has already
20
+ * closed the blocks between them, so the `else` that arrives here is the one
21
+ * that belongs to this `if`; what is left to check is that the two line up, and
22
+ * OS1016 when they do not.
23
+ */
24
+ export function parseIfStatement(cursor: Cursor): IfStatement {
25
+ const keyword = cursor.token;
26
+ cursor.noteIf(keyword.span.column);
27
+
28
+ const branches: IfBranch[] = [parseBranch(cursor, 'if', keyword.span)];
29
+ let elseBranch: ElseBranch | undefined;
30
+ let end = branches[0]?.span ?? keyword.span;
31
+
32
+ while (cursor.at('else')) {
33
+ const word = cursor.token;
34
+ pairWithIf(cursor, word, keyword);
35
+ cursor.advance();
36
+
37
+ if (cursor.at('if')) {
38
+ const branch = parseBranch(cursor, 'else if', word.span);
39
+ branches.push(branch);
40
+ end = branch.span;
41
+ continue;
42
+ }
43
+
44
+ elseBranch = parseElse(cursor, word);
45
+ end = elseBranch.span;
46
+ break;
47
+ }
48
+
49
+ return makeNode('ifStatement', spanning(keyword.span, end), { branches, elseBranch });
50
+ }
51
+
52
+ /** One `if` or `else if` header and the block under it. */
53
+ function parseBranch(cursor: Cursor, header: string, start: Span): IfBranch {
54
+ cursor.beginStatement();
55
+ const keyword = cursor.advance();
56
+ const condition = parseCondition(cursor);
57
+ finishStatement(cursor);
58
+ const body = parseBlock(cursor, header, spanning(start, keyword.span));
59
+ return makeNode('ifBranch', spanning(start, body.span), { condition, body });
60
+ }
61
+
62
+ function parseElse(cursor: Cursor, word: Token): ElseBranch {
63
+ cursor.beginStatement();
64
+ // Nothing may follow `else` on its line: there is no statement separator to
65
+ // end a one line body with, so a body is written indented underneath.
66
+ finishStatement(cursor);
67
+ const body = parseBlock(cursor, 'else', word.span);
68
+ return makeNode('elseBranch', spanning(word.span, body.span), { body });
69
+ }
70
+
71
+ function pairWithIf(cursor: Cursor, word: Token, keyword: Token): void {
72
+ if (word.span.column === keyword.span.column) return;
73
+ cursor.report('OS1016', word.span, {
74
+ found: word.span.column - 1,
75
+ expected: keyword.span.column - 1,
76
+ });
77
+ }