rapydscript-ng 0.8.0 → 0.8.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +9 -0
- package/README.md +8 -0
- package/bin/build.ts +117 -0
- package/build_wheels.py +133 -0
- package/editor-plugins/README.md +80 -0
- package/editor-plugins/nvim.lua +321 -0
- package/package.json +1 -1
- package/publish.py +27 -17
- package/release/compiler.js +8125 -8093
- package/release/signatures.json +27 -27
- package/release/stdlib_modules.json +1 -0
- package/src/ast.pyj +379 -279
- package/src/baselib-builtins.pyj +47 -26
- package/src/baselib-containers.pyj +105 -92
- package/src/baselib-errors.pyj +8 -1
- package/src/baselib-internal.pyj +35 -20
- package/src/baselib-itertools.pyj +13 -7
- package/src/baselib-str.pyj +55 -80
- package/src/compiler.pyj +4 -4
- package/src/errors.pyj +3 -4
- package/src/lib/aes.pyj +46 -32
- package/src/lib/elementmaker.pyj +11 -9
- package/src/lib/encodings.pyj +15 -5
- package/src/lib/gettext.pyj +9 -4
- package/src/lib/math.pyj +106 -45
- package/src/lib/operator.pyj +10 -10
- package/src/lib/pythonize.pyj +2 -1
- package/src/lib/random.pyj +28 -21
- package/src/lib/re.pyj +490 -17
- package/src/lib/traceback.pyj +7 -2
- package/src/lib/uuid.pyj +2 -2
- package/src/output/classes.pyj +28 -27
- package/src/output/codegen.pyj +80 -83
- package/src/output/comments.pyj +8 -8
- package/src/output/exceptions.pyj +20 -21
- package/src/output/functions.pyj +37 -26
- package/src/output/literals.pyj +14 -10
- package/src/output/loops.pyj +63 -59
- package/src/output/modules.pyj +52 -23
- package/src/output/operators.pyj +40 -29
- package/src/output/statements.pyj +20 -14
- package/src/output/stream.pyj +33 -34
- package/src/output/utils.pyj +12 -8
- package/src/parse.pyj +234 -233
- package/src/string_interpolation.pyj +5 -3
- package/src/tokenizer.pyj +176 -148
- package/src/utils.pyj +31 -14
- package/test/fmt.pyj +94 -4
- package/test/generic.pyj +9 -0
- package/test/lsp.pyj +35 -0
- package/test/str.pyj +8 -0
- package/tools/cli.mjs +25 -4
- package/tools/compile.mjs +7 -3
- package/tools/compiler.mjs +34 -2
- package/tools/fmt.mjs +262 -22
- package/tools/ini.mjs +112 -1
- package/tools/lsp.mjs +56 -6
- package/tools/repl.mjs +5 -2
- package/tools/self.mjs +15 -0
- package/tools/test.mjs +100 -0
- package/tools/web_repl_export.mjs +4 -0
- package/tree-sitter/package.json +1 -1
- package/tree-sitter/tree-sitter.json +1 -1
- package/editor-plugins/nvim/rapydscript/README.md +0 -184
- package/editor-plugins/nvim/rapydscript/bin/rapydscript.so +0 -0
- package/editor-plugins/nvim/rapydscript/ftdetect/rapydscript.lua +0 -10
- package/editor-plugins/nvim/rapydscript/lua/rapydscript/health.lua +0 -88
- package/editor-plugins/nvim/rapydscript/lua/rapydscript/init.lua +0 -279
- /package/tree-sitter/queries/{rapydscript/highlights.scm → highlights.scm} +0 -0
- /package/tree-sitter/queries/{rapydscript/indents.scm → indents.scm} +0 -0
- /package/tree-sitter/queries/{rapydscript/injections.scm → injections.scm} +0 -0
- /package/tree-sitter/queries/{rapydscript/locals.scm → locals.scm} +0 -0
package/src/tokenizer.pyj
CHANGED
|
@@ -2,97 +2,97 @@
|
|
|
2
2
|
# License: BSD Copyright: 2016, Kovid Goyal <kovid at kovidgoyal.net>
|
|
3
3
|
from __python__ import hash_literals
|
|
4
4
|
|
|
5
|
-
from unicode_aliases import ALIAS_MAP
|
|
6
|
-
from utils import make_predicate, characters
|
|
7
5
|
from ast import AST_Token
|
|
8
6
|
from errors import SyntaxError
|
|
9
7
|
from string_interpolation import interpolate, quoted_string
|
|
8
|
+
from unicode_aliases import ALIAS_MAP
|
|
9
|
+
from utils import make_predicate, characters
|
|
10
10
|
|
|
11
11
|
RE_HEX_NUMBER = /^0x[0-9a-f]+$/i
|
|
12
12
|
RE_OCT_NUMBER = /^0[0-7]+$/
|
|
13
13
|
RE_DEC_NUMBER = /^\d*\.?\d*(?:e[+-]?\d*(?:\d\.?|\.?\d)\d*)?$/i
|
|
14
14
|
|
|
15
|
-
OPERATOR_CHARS = make_predicate(characters(
|
|
15
|
+
OPERATOR_CHARS = make_predicate(characters('+-*&%=<>!?|~^@'))
|
|
16
16
|
|
|
17
|
-
ASCII_CONTROL_CHARS = {'a':7, 'b':8, 'f': 12, 'n': 10, 'r': 13, 't': 9, 'v': 11}
|
|
17
|
+
ASCII_CONTROL_CHARS = {'a': 7, 'b': 8, 'f': 12, 'n': 10, 'r': 13, 't': 9, 'v': 11}
|
|
18
18
|
HEX_PAT = /[a-fA-F0-9]/
|
|
19
19
|
NAME_PAT = /[a-zA-Z ]/
|
|
20
20
|
|
|
21
21
|
OPERATORS = make_predicate([
|
|
22
|
-
|
|
23
|
-
|
|
24
|
-
|
|
25
|
-
|
|
26
|
-
|
|
27
|
-
|
|
28
|
-
|
|
29
|
-
|
|
30
|
-
|
|
31
|
-
|
|
32
|
-
|
|
33
|
-
|
|
34
|
-
|
|
35
|
-
|
|
36
|
-
|
|
37
|
-
|
|
38
|
-
|
|
39
|
-
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
|
|
43
|
-
|
|
44
|
-
|
|
45
|
-
|
|
46
|
-
|
|
47
|
-
|
|
48
|
-
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
|
|
52
|
-
|
|
53
|
-
|
|
54
|
-
|
|
55
|
-
|
|
56
|
-
|
|
57
|
-
|
|
58
|
-
|
|
59
|
-
|
|
60
|
-
|
|
61
|
-
|
|
62
|
-
|
|
63
|
-
|
|
64
|
-
|
|
65
|
-
|
|
66
|
-
|
|
22
|
+
'in',
|
|
23
|
+
'instanceof',
|
|
24
|
+
'typeof',
|
|
25
|
+
'new',
|
|
26
|
+
'void',
|
|
27
|
+
'del',
|
|
28
|
+
'+',
|
|
29
|
+
'-',
|
|
30
|
+
'not',
|
|
31
|
+
'~',
|
|
32
|
+
'&',
|
|
33
|
+
'|',
|
|
34
|
+
'^',
|
|
35
|
+
'**',
|
|
36
|
+
'*',
|
|
37
|
+
'//',
|
|
38
|
+
'/',
|
|
39
|
+
'%',
|
|
40
|
+
'>>',
|
|
41
|
+
'<<',
|
|
42
|
+
'>>>',
|
|
43
|
+
'<',
|
|
44
|
+
'>',
|
|
45
|
+
'<=',
|
|
46
|
+
'>=',
|
|
47
|
+
'==',
|
|
48
|
+
'is',
|
|
49
|
+
'!=',
|
|
50
|
+
'=',
|
|
51
|
+
'+=',
|
|
52
|
+
'-=',
|
|
53
|
+
'//=',
|
|
54
|
+
'/=',
|
|
55
|
+
'*=',
|
|
56
|
+
'%=',
|
|
57
|
+
'>>=',
|
|
58
|
+
'<<=',
|
|
59
|
+
'>>>=',
|
|
60
|
+
'|=',
|
|
61
|
+
'^=',
|
|
62
|
+
'&=',
|
|
63
|
+
'and',
|
|
64
|
+
'or',
|
|
65
|
+
'@',
|
|
66
|
+
'->'
|
|
67
67
|
])
|
|
68
68
|
|
|
69
69
|
OP_MAP = {
|
|
70
|
-
'or':
|
|
71
|
-
'and':
|
|
72
|
-
'not':
|
|
73
|
-
'del':
|
|
74
|
-
'None':
|
|
75
|
-
'is':
|
|
70
|
+
'or': '||',
|
|
71
|
+
'and': '&&',
|
|
72
|
+
'not': '!',
|
|
73
|
+
'del': 'delete',
|
|
74
|
+
'None': 'null',
|
|
75
|
+
'is': '===',
|
|
76
76
|
}
|
|
77
77
|
|
|
78
|
-
WHITESPACE_CHARS = make_predicate(characters(
|
|
78
|
+
WHITESPACE_CHARS = make_predicate(characters(' \u00a0\n\r\t\f\u000b\u200b\u180e\u2000\u2001\u2002\u2003\u2004\u2005\u2006\u2007\u2008\u2009\u200a\u202f\u205f\u3000'))
|
|
79
79
|
|
|
80
|
-
PUNC_BEFORE_EXPRESSION = make_predicate(characters(
|
|
80
|
+
PUNC_BEFORE_EXPRESSION = make_predicate(characters('[{(,.;:'))
|
|
81
81
|
|
|
82
|
-
PUNC_CHARS = make_predicate(characters(
|
|
82
|
+
PUNC_CHARS = make_predicate(characters('[]{}(),;:?'))
|
|
83
83
|
|
|
84
|
-
KEYWORDS =
|
|
84
|
+
KEYWORDS = 'as assert async await break class continue def del do elif else except finally for from global if import in is new nonlocal pass raise return yield try while with or and not'
|
|
85
85
|
|
|
86
|
-
KEYWORDS_ATOM =
|
|
86
|
+
KEYWORDS_ATOM = 'False None True'
|
|
87
87
|
|
|
88
88
|
# see https://developer.mozilla.org/en-US/docs/Web/JavaScript/Reference/Lexical_grammar
|
|
89
|
-
RESERVED_WORDS = (
|
|
90
|
-
|
|
91
|
-
|
|
89
|
+
RESERVED_WORDS = ('break case class catch const continue debugger default delete do else export extends'
|
|
90
|
+
' finally for function if import in instanceof new return super switch this throw try typeof var void'
|
|
91
|
+
' while with yield enum implements static private package let public protected interface null true false')
|
|
92
92
|
|
|
93
|
-
KEYWORDS_BEFORE_EXPRESSION =
|
|
93
|
+
KEYWORDS_BEFORE_EXPRESSION = 'return yield new del raise elif else if await'
|
|
94
94
|
|
|
95
|
-
ALL_KEYWORDS = KEYWORDS +
|
|
95
|
+
ALL_KEYWORDS = KEYWORDS + ' ' + KEYWORDS_ATOM
|
|
96
96
|
|
|
97
97
|
KEYWORDS = make_predicate(KEYWORDS)
|
|
98
98
|
RESERVED_WORDS = make_predicate(RESERVED_WORDS)
|
|
@@ -100,37 +100,47 @@ KEYWORDS_BEFORE_EXPRESSION = make_predicate(KEYWORDS_BEFORE_EXPRESSION)
|
|
|
100
100
|
KEYWORDS_ATOM = make_predicate(KEYWORDS_ATOM)
|
|
101
101
|
IDENTIFIER_PAT = /^[a-z_$][_a-z0-9$]*$/i
|
|
102
102
|
|
|
103
|
+
|
|
103
104
|
def is_string_modifier(val):
|
|
104
105
|
for ch in val:
|
|
105
106
|
if 'vrufVRUF'.indexOf(ch) is -1:
|
|
106
107
|
return False
|
|
107
108
|
return True
|
|
108
109
|
|
|
110
|
+
|
|
109
111
|
def is_letter(code):
|
|
110
112
|
return code >= 97 and code <= 122 or code >= 65 and code <= 90 or code >= 170 and UNICODE.letter.test(String.fromCharCode(code))
|
|
111
113
|
|
|
114
|
+
|
|
112
115
|
def is_digit(code):
|
|
113
116
|
return code >= 48 and code <= 57
|
|
114
117
|
|
|
118
|
+
|
|
115
119
|
def is_alphanumeric_char(code):
|
|
116
120
|
return is_digit(code) or is_letter(code)
|
|
117
121
|
|
|
122
|
+
|
|
118
123
|
def is_unicode_combining_mark(ch):
|
|
119
124
|
return UNICODE.non_spacing_mark.test(ch) or UNICODE.space_combining_mark.test(ch)
|
|
120
125
|
|
|
126
|
+
|
|
121
127
|
def is_unicode_connector_punctuation(ch):
|
|
122
128
|
return UNICODE.connector_punctuation.test(ch)
|
|
123
129
|
|
|
130
|
+
|
|
124
131
|
def is_identifier(name):
|
|
125
132
|
return not RESERVED_WORDS[name] and not KEYWORDS[name] and not KEYWORDS_ATOM[name] and IDENTIFIER_PAT.test(name)
|
|
126
133
|
|
|
134
|
+
|
|
127
135
|
def is_identifier_start(code):
|
|
128
136
|
return code is 36 or code is 95 or is_letter(code)
|
|
129
137
|
|
|
138
|
+
|
|
130
139
|
def is_identifier_char(ch):
|
|
131
140
|
code = ch.charCodeAt(0)
|
|
132
141
|
return is_identifier_start(code) or is_digit(code) or code is 8204 or code is 8205 or is_unicode_combining_mark(ch) or is_unicode_connector_punctuation(ch)
|
|
133
142
|
|
|
143
|
+
|
|
134
144
|
def parse_js_number(num):
|
|
135
145
|
if RE_HEX_NUMBER.test(num):
|
|
136
146
|
return parseInt(num.substr(2), 16)
|
|
@@ -141,10 +151,10 @@ def parse_js_number(num):
|
|
|
141
151
|
|
|
142
152
|
# regexps adapted from http://xregexp.com/plugins/#unicode
|
|
143
153
|
UNICODE = { # {{{
|
|
144
|
-
'letter': RegExp(
|
|
145
|
-
'non_spacing_mark': RegExp(
|
|
146
|
-
'space_combining_mark': RegExp(
|
|
147
|
-
'connector_punctuation': RegExp(
|
|
154
|
+
'letter': RegExp('[\\u0041-\\u005A\\u0061-\\u007A\\u00AA\\u00B5\\u00BA\\u00C0-\\u00D6\\u00D8-\\u00F6\\u00F8-\\u02C1\\u02C6-\\u02D1\\u02E0-\\u02E4\\u02EC\\u02EE\\u0370-\\u0374\\u0376\\u0377\\u037A-\\u037D\\u0386\\u0388-\\u038A\\u038C\\u038E-\\u03A1\\u03A3-\\u03F5\\u03F7-\\u0481\\u048A-\\u0523\\u0531-\\u0556\\u0559\\u0561-\\u0587\\u05D0-\\u05EA\\u05F0-\\u05F2\\u0621-\\u064A\\u066E\\u066F\\u0671-\\u06D3\\u06D5\\u06E5\\u06E6\\u06EE\\u06EF\\u06FA-\\u06FC\\u06FF\\u0710\\u0712-\\u072F\\u074D-\\u07A5\\u07B1\\u07CA-\\u07EA\\u07F4\\u07F5\\u07FA\\u0904-\\u0939\\u093D\\u0950\\u0958-\\u0961\\u0971\\u0972\\u097B-\\u097F\\u0985-\\u098C\\u098F\\u0990\\u0993-\\u09A8\\u09AA-\\u09B0\\u09B2\\u09B6-\\u09B9\\u09BD\\u09CE\\u09DC\\u09DD\\u09DF-\\u09E1\\u09F0\\u09F1\\u0A05-\\u0A0A\\u0A0F\\u0A10\\u0A13-\\u0A28\\u0A2A-\\u0A30\\u0A32\\u0A33\\u0A35\\u0A36\\u0A38\\u0A39\\u0A59-\\u0A5C\\u0A5E\\u0A72-\\u0A74\\u0A85-\\u0A8D\\u0A8F-\\u0A91\\u0A93-\\u0AA8\\u0AAA-\\u0AB0\\u0AB2\\u0AB3\\u0AB5-\\u0AB9\\u0ABD\\u0AD0\\u0AE0\\u0AE1\\u0B05-\\u0B0C\\u0B0F\\u0B10\\u0B13-\\u0B28\\u0B2A-\\u0B30\\u0B32\\u0B33\\u0B35-\\u0B39\\u0B3D\\u0B5C\\u0B5D\\u0B5F-\\u0B61\\u0B71\\u0B83\\u0B85-\\u0B8A\\u0B8E-\\u0B90\\u0B92-\\u0B95\\u0B99\\u0B9A\\u0B9C\\u0B9E\\u0B9F\\u0BA3\\u0BA4\\u0BA8-\\u0BAA\\u0BAE-\\u0BB9\\u0BD0\\u0C05-\\u0C0C\\u0C0E-\\u0C10\\u0C12-\\u0C28\\u0C2A-\\u0C33\\u0C35-\\u0C39\\u0C3D\\u0C58\\u0C59\\u0C60\\u0C61\\u0C85-\\u0C8C\\u0C8E-\\u0C90\\u0C92-\\u0CA8\\u0CAA-\\u0CB3\\u0CB5-\\u0CB9\\u0CBD\\u0CDE\\u0CE0\\u0CE1\\u0D05-\\u0D0C\\u0D0E-\\u0D10\\u0D12-\\u0D28\\u0D2A-\\u0D39\\u0D3D\\u0D60\\u0D61\\u0D7A-\\u0D7F\\u0D85-\\u0D96\\u0D9A-\\u0DB1\\u0DB3-\\u0DBB\\u0DBD\\u0DC0-\\u0DC6\\u0E01-\\u0E30\\u0E32\\u0E33\\u0E40-\\u0E46\\u0E81\\u0E82\\u0E84\\u0E87\\u0E88\\u0E8A\\u0E8D\\u0E94-\\u0E97\\u0E99-\\u0E9F\\u0EA1-\\u0EA3\\u0EA5\\u0EA7\\u0EAA\\u0EAB\\u0EAD-\\u0EB0\\u0EB2\\u0EB3\\u0EBD\\u0EC0-\\u0EC4\\u0EC6\\u0EDC\\u0EDD\\u0F00\\u0F40-\\u0F47\\u0F49-\\u0F6C\\u0F88-\\u0F8B\\u1000-\\u102A\\u103F\\u1050-\\u1055\\u105A-\\u105D\\u1061\\u1065\\u1066\\u106E-\\u1070\\u1075-\\u1081\\u108E\\u10A0-\\u10C5\\u10D0-\\u10FA\\u10FC\\u1100-\\u1159\\u115F-\\u11A2\\u11A8-\\u11F9\\u1200-\\u1248\\u124A-\\u124D\\u1250-\\u1256\\u1258\\u125A-\\u125D\\u1260-\\u1288\\u128A-\\u128D\\u1290-\\u12B0\\u12B2-\\u12B5\\u12B8-\\u12BE\\u12C0\\u12C2-\\u12C5\\u12C8-\\u12D6\\u12D8-\\u1310\\u1312-\\u1315\\u1318-\\u135A\\u1380-\\u138F\\u13A0-\\u13F4\\u1401-\\u166C\\u166F-\\u1676\\u1681-\\u169A\\u16A0-\\u16EA\\u1700-\\u170C\\u170E-\\u1711\\u1720-\\u1731\\u1740-\\u1751\\u1760-\\u176C\\u176E-\\u1770\\u1780-\\u17B3\\u17D7\\u17DC\\u1820-\\u1877\\u1880-\\u18A8\\u18AA\\u1900-\\u191C\\u1950-\\u196D\\u1970-\\u1974\\u1980-\\u19A9\\u19C1-\\u19C7\\u1A00-\\u1A16\\u1B05-\\u1B33\\u1B45-\\u1B4B\\u1B83-\\u1BA0\\u1BAE\\u1BAF\\u1C00-\\u1C23\\u1C4D-\\u1C4F\\u1C5A-\\u1C7D\\u1D00-\\u1DBF\\u1E00-\\u1F15\\u1F18-\\u1F1D\\u1F20-\\u1F45\\u1F48-\\u1F4D\\u1F50-\\u1F57\\u1F59\\u1F5B\\u1F5D\\u1F5F-\\u1F7D\\u1F80-\\u1FB4\\u1FB6-\\u1FBC\\u1FBE\\u1FC2-\\u1FC4\\u1FC6-\\u1FCC\\u1FD0-\\u1FD3\\u1FD6-\\u1FDB\\u1FE0-\\u1FEC\\u1FF2-\\u1FF4\\u1FF6-\\u1FFC\\u2071\\u207F\\u2090-\\u2094\\u2102\\u2107\\u210A-\\u2113\\u2115\\u2119-\\u211D\\u2124\\u2126\\u2128\\u212A-\\u212D\\u212F-\\u2139\\u213C-\\u213F\\u2145-\\u2149\\u214E\\u2183\\u2184\\u2C00-\\u2C2E\\u2C30-\\u2C5E\\u2C60-\\u2C6F\\u2C71-\\u2C7D\\u2C80-\\u2CE4\\u2D00-\\u2D25\\u2D30-\\u2D65\\u2D6F\\u2D80-\\u2D96\\u2DA0-\\u2DA6\\u2DA8-\\u2DAE\\u2DB0-\\u2DB6\\u2DB8-\\u2DBE\\u2DC0-\\u2DC6\\u2DC8-\\u2DCE\\u2DD0-\\u2DD6\\u2DD8-\\u2DDE\\u2E2F\\u3005\\u3006\\u3031-\\u3035\\u303B\\u303C\\u3041-\\u3096\\u309D-\\u309F\\u30A1-\\u30FA\\u30FC-\\u30FF\\u3105-\\u312D\\u3131-\\u318E\\u31A0-\\u31B7\\u31F0-\\u31FF\\u3400\\u4DB5\\u4E00\\u9FC3\\uA000-\\uA48C\\uA500-\\uA60C\\uA610-\\uA61F\\uA62A\\uA62B\\uA640-\\uA65F\\uA662-\\uA66E\\uA67F-\\uA697\\uA717-\\uA71F\\uA722-\\uA788\\uA78B\\uA78C\\uA7FB-\\uA801\\uA803-\\uA805\\uA807-\\uA80A\\uA80C-\\uA822\\uA840-\\uA873\\uA882-\\uA8B3\\uA90A-\\uA925\\uA930-\\uA946\\uAA00-\\uAA28\\uAA40-\\uAA42\\uAA44-\\uAA4B\\uAC00\\uD7A3\\uF900-\\uFA2D\\uFA30-\\uFA6A\\uFA70-\\uFAD9\\uFB00-\\uFB06\\uFB13-\\uFB17\\uFB1D\\uFB1F-\\uFB28\\uFB2A-\\uFB36\\uFB38-\\uFB3C\\uFB3E\\uFB40\\uFB41\\uFB43\\uFB44\\uFB46-\\uFBB1\\uFBD3-\\uFD3D\\uFD50-\\uFD8F\\uFD92-\\uFDC7\\uFDF0-\\uFDFB\\uFE70-\\uFE74\\uFE76-\\uFEFC\\uFF21-\\uFF3A\\uFF41-\\uFF5A\\uFF66-\\uFFBE\\uFFC2-\\uFFC7\\uFFCA-\\uFFCF\\uFFD2-\\uFFD7\\uFFDA-\\uFFDC]'),
|
|
155
|
+
'non_spacing_mark': RegExp('[\\u0300-\\u036F\\u0483-\\u0487\\u0591-\\u05BD\\u05BF\\u05C1\\u05C2\\u05C4\\u05C5\\u05C7\\u0610-\\u061A\\u064B-\\u065E\\u0670\\u06D6-\\u06DC\\u06DF-\\u06E4\\u06E7\\u06E8\\u06EA-\\u06ED\\u0711\\u0730-\\u074A\\u07A6-\\u07B0\\u07EB-\\u07F3\\u0816-\\u0819\\u081B-\\u0823\\u0825-\\u0827\\u0829-\\u082D\\u0900-\\u0902\\u093C\\u0941-\\u0948\\u094D\\u0951-\\u0955\\u0962\\u0963\\u0981\\u09BC\\u09C1-\\u09C4\\u09CD\\u09E2\\u09E3\\u0A01\\u0A02\\u0A3C\\u0A41\\u0A42\\u0A47\\u0A48\\u0A4B-\\u0A4D\\u0A51\\u0A70\\u0A71\\u0A75\\u0A81\\u0A82\\u0ABC\\u0AC1-\\u0AC5\\u0AC7\\u0AC8\\u0ACD\\u0AE2\\u0AE3\\u0B01\\u0B3C\\u0B3F\\u0B41-\\u0B44\\u0B4D\\u0B56\\u0B62\\u0B63\\u0B82\\u0BC0\\u0BCD\\u0C3E-\\u0C40\\u0C46-\\u0C48\\u0C4A-\\u0C4D\\u0C55\\u0C56\\u0C62\\u0C63\\u0CBC\\u0CBF\\u0CC6\\u0CCC\\u0CCD\\u0CE2\\u0CE3\\u0D41-\\u0D44\\u0D4D\\u0D62\\u0D63\\u0DCA\\u0DD2-\\u0DD4\\u0DD6\\u0E31\\u0E34-\\u0E3A\\u0E47-\\u0E4E\\u0EB1\\u0EB4-\\u0EB9\\u0EBB\\u0EBC\\u0EC8-\\u0ECD\\u0F18\\u0F19\\u0F35\\u0F37\\u0F39\\u0F71-\\u0F7E\\u0F80-\\u0F84\\u0F86\\u0F87\\u0F90-\\u0F97\\u0F99-\\u0FBC\\u0FC6\\u102D-\\u1030\\u1032-\\u1037\\u1039\\u103A\\u103D\\u103E\\u1058\\u1059\\u105E-\\u1060\\u1071-\\u1074\\u1082\\u1085\\u1086\\u108D\\u109D\\u135F\\u1712-\\u1714\\u1732-\\u1734\\u1752\\u1753\\u1772\\u1773\\u17B7-\\u17BD\\u17C6\\u17C9-\\u17D3\\u17DD\\u180B-\\u180D\\u18A9\\u1920-\\u1922\\u1927\\u1928\\u1932\\u1939-\\u193B\\u1A17\\u1A18\\u1A56\\u1A58-\\u1A5E\\u1A60\\u1A62\\u1A65-\\u1A6C\\u1A73-\\u1A7C\\u1A7F\\u1B00-\\u1B03\\u1B34\\u1B36-\\u1B3A\\u1B3C\\u1B42\\u1B6B-\\u1B73\\u1B80\\u1B81\\u1BA2-\\u1BA5\\u1BA8\\u1BA9\\u1C2C-\\u1C33\\u1C36\\u1C37\\u1CD0-\\u1CD2\\u1CD4-\\u1CE0\\u1CE2-\\u1CE8\\u1CED\\u1DC0-\\u1DE6\\u1DFD-\\u1DFF\\u20D0-\\u20DC\\u20E1\\u20E5-\\u20F0\\u2CEF-\\u2CF1\\u2DE0-\\u2DFF\\u302A-\\u302F\\u3099\\u309A\\uA66F\\uA67C\\uA67D\\uA6F0\\uA6F1\\uA802\\uA806\\uA80B\\uA825\\uA826\\uA8C4\\uA8E0-\\uA8F1\\uA926-\\uA92D\\uA947-\\uA951\\uA980-\\uA982\\uA9B3\\uA9B6-\\uA9B9\\uA9BC\\uAA29-\\uAA2E\\uAA31\\uAA32\\uAA35\\uAA36\\uAA43\\uAA4C\\uAAB0\\uAAB2-\\uAAB4\\uAAB7\\uAAB8\\uAABE\\uAABF\\uAAC1\\uABE5\\uABE8\\uABED\\uFB1E\\uFE00-\\uFE0F\\uFE20-\\uFE26]'),
|
|
156
|
+
'space_combining_mark': RegExp('[\\u0903\\u093E-\\u0940\\u0949-\\u094C\\u094E\\u0982\\u0983\\u09BE-\\u09C0\\u09C7\\u09C8\\u09CB\\u09CC\\u09D7\\u0A03\\u0A3E-\\u0A40\\u0A83\\u0ABE-\\u0AC0\\u0AC9\\u0ACB\\u0ACC\\u0B02\\u0B03\\u0B3E\\u0B40\\u0B47\\u0B48\\u0B4B\\u0B4C\\u0B57\\u0BBE\\u0BBF\\u0BC1\\u0BC2\\u0BC6-\\u0BC8\\u0BCA-\\u0BCC\\u0BD7\\u0C01-\\u0C03\\u0C41-\\u0C44\\u0C82\\u0C83\\u0CBE\\u0CC0-\\u0CC4\\u0CC7\\u0CC8\\u0CCA\\u0CCB\\u0CD5\\u0CD6\\u0D02\\u0D03\\u0D3E-\\u0D40\\u0D46-\\u0D48\\u0D4A-\\u0D4C\\u0D57\\u0D82\\u0D83\\u0DCF-\\u0DD1\\u0DD8-\\u0DDF\\u0DF2\\u0DF3\\u0F3E\\u0F3F\\u0F7F\\u102B\\u102C\\u1031\\u1038\\u103B\\u103C\\u1056\\u1057\\u1062-\\u1064\\u1067-\\u106D\\u1083\\u1084\\u1087-\\u108C\\u108F\\u109A-\\u109C\\u17B6\\u17BE-\\u17C5\\u17C7\\u17C8\\u1923-\\u1926\\u1929-\\u192B\\u1930\\u1931\\u1933-\\u1938\\u19B0-\\u19C0\\u19C8\\u19C9\\u1A19-\\u1A1B\\u1A55\\u1A57\\u1A61\\u1A63\\u1A64\\u1A6D-\\u1A72\\u1B04\\u1B35\\u1B3B\\u1B3D-\\u1B41\\u1B43\\u1B44\\u1B82\\u1BA1\\u1BA6\\u1BA7\\u1BAA\\u1C24-\\u1C2B\\u1C34\\u1C35\\u1CE1\\u1CF2\\uA823\\uA824\\uA827\\uA880\\uA881\\uA8B4-\\uA8C3\\uA952\\uA953\\uA983\\uA9B4\\uA9B5\\uA9BA\\uA9BB\\uA9BD-\\uA9C0\\uAA2F\\uAA30\\uAA33\\uAA34\\uAA4D\\uAA7B\\uABE3\\uABE4\\uABE6\\uABE7\\uABE9\\uABEA\\uABEC]'),
|
|
157
|
+
'connector_punctuation': RegExp('[\\u005F\\u203F\\u2040\\u2054\\uFE33\\uFE34\\uFE4D-\\uFE4F\\uFF3F]')
|
|
148
158
|
} # }}}
|
|
149
159
|
|
|
150
160
|
|
|
@@ -153,9 +163,10 @@ def is_token(token, type, val):
|
|
|
153
163
|
|
|
154
164
|
EX_EOF = {}
|
|
155
165
|
|
|
166
|
+
|
|
156
167
|
def tokenizer(raw_text, filename, recover_errors):
|
|
157
168
|
S = {
|
|
158
|
-
'text': raw_text.replace(/\r\n?|[\n\u2028\u2029]/g,
|
|
169
|
+
'text': raw_text.replace(/\r\n?|[\n\u2028\u2029]/g, '\n').replace(/\uFEFF/g, ''),
|
|
159
170
|
'filename': filename,
|
|
160
171
|
'recover_errors': not not recover_errors, # LSP error-recovery mode (see parse())
|
|
161
172
|
'recovered_errors': v'[]', # collected errors when recover_errors is on
|
|
@@ -172,11 +183,12 @@ def tokenizer(raw_text, filename, recover_errors):
|
|
|
172
183
|
'newblock': False,
|
|
173
184
|
'endblock': False,
|
|
174
185
|
'indentation_matters': v'[ true ]',
|
|
175
|
-
'cached_whitespace':
|
|
186
|
+
'cached_whitespace': '',
|
|
176
187
|
'prev': undefined,
|
|
177
188
|
'index_or_slice': v'[ false ]',
|
|
178
189
|
'expecting_object_literal_key': False, # This is set by the parser when it is expecting an object literal key
|
|
179
190
|
}
|
|
191
|
+
|
|
180
192
|
def peek():
|
|
181
193
|
return S.text.charAt(S.pos)
|
|
182
194
|
|
|
@@ -189,7 +201,7 @@ def tokenizer(raw_text, filename, recover_errors):
|
|
|
189
201
|
if signal_eof and not ch:
|
|
190
202
|
raise EX_EOF
|
|
191
203
|
|
|
192
|
-
if ch is
|
|
204
|
+
if ch is '\n':
|
|
193
205
|
S.newline_before = S.newline_before or not in_string
|
|
194
206
|
S.line += 1
|
|
195
207
|
S.col = 0
|
|
@@ -209,15 +221,15 @@ def tokenizer(raw_text, filename, recover_errors):
|
|
|
209
221
|
S.tokpos = S.pos
|
|
210
222
|
|
|
211
223
|
def token(type, value, is_comment, keep_newline):
|
|
212
|
-
S.regex_allowed = (type is
|
|
213
|
-
or type is
|
|
214
|
-
or type is
|
|
224
|
+
S.regex_allowed = (type is 'operator'
|
|
225
|
+
or type is 'keyword' and KEYWORDS_BEFORE_EXPRESSION[value]
|
|
226
|
+
or type is 'punc' and PUNC_BEFORE_EXPRESSION[value])
|
|
215
227
|
|
|
216
|
-
if type is
|
|
228
|
+
if type is 'operator' and value is 'is' and S.text.substr(S.pos).trimLeft().substr(0, 4).trimRight() is 'not':
|
|
217
229
|
next_token()
|
|
218
|
-
value =
|
|
230
|
+
value = '!=='
|
|
219
231
|
|
|
220
|
-
if type is
|
|
232
|
+
if type is 'operator' and OP_MAP[value]:
|
|
221
233
|
value = OP_MAP[value]
|
|
222
234
|
|
|
223
235
|
ret = {
|
|
@@ -241,29 +253,29 @@ def tokenizer(raw_text, filename, recover_errors):
|
|
|
241
253
|
if not keep_newline:
|
|
242
254
|
S.newline_before = False
|
|
243
255
|
|
|
244
|
-
if type is
|
|
256
|
+
if type is 'punc':
|
|
245
257
|
# if (value is ":" && peek() is "\n") {
|
|
246
|
-
if value is
|
|
258
|
+
if value is ':' and not S.index_or_slice[-1]
|
|
247
259
|
and not S.expecting_object_literal_key
|
|
248
|
-
and (not S.text.substring(S.pos + 1, find(
|
|
260
|
+
and (not S.text.substring(S.pos + 1, find('\n')).trim() or not S.text.substring(S.pos + 1, find('#')).trim()):
|
|
249
261
|
S.newblock = True
|
|
250
262
|
S.indentation_matters.push(True)
|
|
251
263
|
|
|
252
|
-
if value is
|
|
264
|
+
if value is '[':
|
|
253
265
|
if S.prev and (
|
|
254
|
-
S.prev.type is
|
|
266
|
+
S.prev.type is 'name' or
|
|
255
267
|
(S.prev.type is 'punc' and ')]'.indexOf(S.prev.value) is not -1)
|
|
256
268
|
):
|
|
257
269
|
S.index_or_slice.push(True)
|
|
258
270
|
else:
|
|
259
271
|
S.index_or_slice.push(False)
|
|
260
272
|
S.indentation_matters.push(False)
|
|
261
|
-
elif value is
|
|
273
|
+
elif value is '{' or value is '(':
|
|
262
274
|
S.indentation_matters.push(False)
|
|
263
|
-
elif value is
|
|
275
|
+
elif value is ']':
|
|
264
276
|
S.index_or_slice.pop()
|
|
265
277
|
S.indentation_matters.pop()
|
|
266
|
-
elif value is
|
|
278
|
+
elif value is '}' or value is ')':
|
|
267
279
|
S.indentation_matters.pop()
|
|
268
280
|
S.prev = AST_Token(ret)
|
|
269
281
|
return S.prev
|
|
@@ -271,16 +283,16 @@ def tokenizer(raw_text, filename, recover_errors):
|
|
|
271
283
|
# this will transform leading whitespace to block tokens unless
|
|
272
284
|
# part of array/hash, and skip non-leading whitespace
|
|
273
285
|
def parse_whitespace():
|
|
274
|
-
leading_whitespace =
|
|
286
|
+
leading_whitespace = ''
|
|
275
287
|
whitespace_exists = False
|
|
276
288
|
while WHITESPACE_CHARS[peek()]:
|
|
277
289
|
whitespace_exists = True
|
|
278
290
|
ch = next()
|
|
279
|
-
if ch is
|
|
280
|
-
leading_whitespace =
|
|
291
|
+
if ch is '\n':
|
|
292
|
+
leading_whitespace = ''
|
|
281
293
|
else:
|
|
282
294
|
leading_whitespace += ch
|
|
283
|
-
if peek() is not
|
|
295
|
+
if peek() is not '#':
|
|
284
296
|
if not whitespace_exists:
|
|
285
297
|
leading_whitespace = S.cached_whitespace
|
|
286
298
|
else:
|
|
@@ -289,7 +301,7 @@ def tokenizer(raw_text, filename, recover_errors):
|
|
|
289
301
|
return test_indent_token(leading_whitespace)
|
|
290
302
|
|
|
291
303
|
def test_indent_token(leading_whitespace):
|
|
292
|
-
most_recent = S.whitespace_before[-1] or
|
|
304
|
+
most_recent = S.whitespace_before[-1] or ''
|
|
293
305
|
S.endblock = False
|
|
294
306
|
if S.indentation_matters[-1] and leading_whitespace is not most_recent:
|
|
295
307
|
if S.newblock and leading_whitespace and leading_whitespace.indexOf(most_recent) is 0:
|
|
@@ -304,11 +316,11 @@ def tokenizer(raw_text, filename, recover_errors):
|
|
|
304
316
|
return -1
|
|
305
317
|
else:
|
|
306
318
|
# indent mismatch, inconsistent indentation
|
|
307
|
-
parse_error(
|
|
319
|
+
parse_error('Inconsistent indentation')
|
|
308
320
|
return 0
|
|
309
321
|
|
|
310
322
|
def read_while(pred):
|
|
311
|
-
ret =
|
|
323
|
+
ret = ''
|
|
312
324
|
i = 0
|
|
313
325
|
ch = ''
|
|
314
326
|
while (ch = peek()) and pred(ch, i):
|
|
@@ -322,16 +334,16 @@ def tokenizer(raw_text, filename, recover_errors):
|
|
|
322
334
|
def read_num(prefix):
|
|
323
335
|
has_e = False
|
|
324
336
|
has_x = False
|
|
325
|
-
has_dot = prefix is
|
|
337
|
+
has_dot = prefix is '.'
|
|
326
338
|
if not prefix and peek() is '0' and S.text.charAt(S.pos + 1) is 'b':
|
|
327
339
|
next(), next()
|
|
328
|
-
num = read_while(def(ch): return ch is '0' or ch is '1';)
|
|
340
|
+
num = read_while(def (ch): return ch is '0' or ch is '1';)
|
|
329
341
|
valid = parseInt(num, 2)
|
|
330
342
|
if isNaN(valid):
|
|
331
343
|
parse_error('Invalid syntax for a binary number')
|
|
332
344
|
return token('num', valid)
|
|
333
345
|
seen = v'[]'
|
|
334
|
-
num = read_while(def(ch, i):
|
|
346
|
+
num = read_while(def (ch, i):
|
|
335
347
|
nonlocal has_dot, has_e, has_x
|
|
336
348
|
seen.push(ch)
|
|
337
349
|
if ch is 'x' or ch is 'X':
|
|
@@ -349,11 +361,11 @@ def tokenizer(raw_text, filename, recover_errors):
|
|
|
349
361
|
elif ch is '-':
|
|
350
362
|
if i is 0 and not prefix:
|
|
351
363
|
return True
|
|
352
|
-
if has_e and seen[i-1].toLowerCase() is 'e':
|
|
364
|
+
if has_e and seen[i - 1].toLowerCase() is 'e':
|
|
353
365
|
return True
|
|
354
366
|
return False
|
|
355
367
|
elif ch is '+':
|
|
356
|
-
if has_e and seen[i-1].toLowerCase() is 'e':
|
|
368
|
+
if has_e and seen[i - 1].toLowerCase() is 'e':
|
|
357
369
|
return True
|
|
358
370
|
return False
|
|
359
371
|
elif ch is '.':
|
|
@@ -365,9 +377,9 @@ def tokenizer(raw_text, filename, recover_errors):
|
|
|
365
377
|
|
|
366
378
|
valid = parse_js_number(num)
|
|
367
379
|
if not isNaN(valid):
|
|
368
|
-
return token(
|
|
380
|
+
return token('num', valid)
|
|
369
381
|
else:
|
|
370
|
-
parse_error(
|
|
382
|
+
parse_error('Invalid syntax: ' + num)
|
|
371
383
|
|
|
372
384
|
def read_hex_digits(count):
|
|
373
385
|
ans = ''
|
|
@@ -417,7 +429,7 @@ def tokenizer(raw_text, filename, recover_errors):
|
|
|
417
429
|
if code <= 0xFFFF:
|
|
418
430
|
return String.fromCharCode(code)
|
|
419
431
|
code -= 0x10000
|
|
420
|
-
return String.fromCharCode(0xD800+(code>>10), 0xDC00+(code&0x3FF))
|
|
432
|
+
return String.fromCharCode(0xD800 + (code >> 10), 0xDC00 + (code & 0x3FF))
|
|
421
433
|
return '\\U' + code
|
|
422
434
|
if q is 'N' and peek() is '{':
|
|
423
435
|
next()
|
|
@@ -432,11 +444,11 @@ def tokenizer(raw_text, filename, recover_errors):
|
|
|
432
444
|
if code <= 0xFFFF:
|
|
433
445
|
return String.fromCharCode(code)
|
|
434
446
|
code -= 0x10000
|
|
435
|
-
return String.fromCharCode(0xD800+(code>>10), 0xDC00+(code&0x3FF))
|
|
447
|
+
return String.fromCharCode(0xD800 + (code >> 10), 0xDC00 + (code & 0x3FF))
|
|
436
448
|
return '\\' + q
|
|
437
449
|
|
|
438
450
|
def with_eof_error(eof_error, cont):
|
|
439
|
-
return def():
|
|
451
|
+
return def ():
|
|
440
452
|
try:
|
|
441
453
|
return cont.apply(None, arguments)
|
|
442
454
|
except as ex:
|
|
@@ -445,10 +457,10 @@ def tokenizer(raw_text, filename, recover_errors):
|
|
|
445
457
|
else:
|
|
446
458
|
raise
|
|
447
459
|
|
|
448
|
-
read_string = with_eof_error(
|
|
460
|
+
read_string = with_eof_error('Unterminated string constant', def (is_raw_literal, is_js_literal):
|
|
449
461
|
quote = next()
|
|
450
462
|
tok_type = 'js' if is_js_literal else 'string'
|
|
451
|
-
ret =
|
|
463
|
+
ret = ''
|
|
452
464
|
is_multiline = False
|
|
453
465
|
if peek() is quote:
|
|
454
466
|
# two quotes in a row
|
|
@@ -459,18 +471,15 @@ def tokenizer(raw_text, filename, recover_errors):
|
|
|
459
471
|
is_multiline = True
|
|
460
472
|
else:
|
|
461
473
|
return token(tok_type, '')
|
|
462
|
-
|
|
463
474
|
while True:
|
|
464
475
|
ch = next(True, True)
|
|
465
476
|
if not ch:
|
|
466
477
|
break
|
|
467
|
-
if ch is
|
|
468
|
-
parse_error(
|
|
469
|
-
|
|
470
|
-
if ch is "\\":
|
|
478
|
+
if ch is '\n' and not is_multiline:
|
|
479
|
+
parse_error('End of line while scanning string literal')
|
|
480
|
+
if ch is '\\':
|
|
471
481
|
ret += ('\\' + next(True)) if is_raw_literal else read_escape_sequence()
|
|
472
482
|
continue
|
|
473
|
-
|
|
474
483
|
if ch is quote:
|
|
475
484
|
if not is_multiline:
|
|
476
485
|
break
|
|
@@ -485,10 +494,13 @@ def tokenizer(raw_text, filename, recover_errors):
|
|
|
485
494
|
return token(tok_type, ret)
|
|
486
495
|
)
|
|
487
496
|
|
|
488
|
-
def handle_interpolated_string(string, start_tok):
|
|
497
|
+
def handle_interpolated_string(string, start_tok, is_fstring=True):
|
|
489
498
|
def raise_error(err):
|
|
490
499
|
raise new SyntaxError(err, filename, start_tok.line, start_tok.col, start_tok.pos, False)
|
|
491
|
-
|
|
500
|
+
if is_fstring:
|
|
501
|
+
parts = v'[interpolate(string, raise_error)]'
|
|
502
|
+
else:
|
|
503
|
+
parts = v'[quoted_string(string)]'
|
|
492
504
|
# Look ahead for consecutive string literals to concatenate (e.g. f'a'f'b' or f'a''b')
|
|
493
505
|
while True:
|
|
494
506
|
# Skip horizontal whitespace (spaces and tabs, not newlines)
|
|
@@ -532,7 +544,7 @@ def tokenizer(raw_text, filename, recover_errors):
|
|
|
532
544
|
def read_line_comment(shebang):
|
|
533
545
|
if not shebang:
|
|
534
546
|
next()
|
|
535
|
-
i = find(
|
|
547
|
+
i = find('\n')
|
|
536
548
|
|
|
537
549
|
if i is -1:
|
|
538
550
|
ret = S.text.substr(S.pos)
|
|
@@ -541,13 +553,13 @@ def tokenizer(raw_text, filename, recover_errors):
|
|
|
541
553
|
ret = S.text.substring(S.pos, i)
|
|
542
554
|
S.pos = i
|
|
543
555
|
|
|
544
|
-
return token(
|
|
556
|
+
return token('shebang' if shebang else 'comment1', ret, True)
|
|
545
557
|
|
|
546
558
|
def read_name():
|
|
547
|
-
name = ch =
|
|
559
|
+
name = ch = ''
|
|
548
560
|
while (ch = peek()) is not None:
|
|
549
|
-
if ch is
|
|
550
|
-
if S.text.charAt(S.pos + 1) is
|
|
561
|
+
if ch is '\\':
|
|
562
|
+
if S.text.charAt(S.pos + 1) is '\n':
|
|
551
563
|
S.pos += 2
|
|
552
564
|
continue
|
|
553
565
|
break
|
|
@@ -557,21 +569,20 @@ def tokenizer(raw_text, filename, recover_errors):
|
|
|
557
569
|
break
|
|
558
570
|
return name
|
|
559
571
|
|
|
560
|
-
read_regexp = with_eof_error(
|
|
572
|
+
read_regexp = with_eof_error('Unterminated regular expression', def ():
|
|
561
573
|
prev_backslash = False
|
|
562
574
|
regexp = ch = ''
|
|
563
575
|
in_class = False
|
|
564
576
|
verbose_regexp = False
|
|
565
577
|
in_comment = False
|
|
566
|
-
|
|
567
578
|
if peek() is '/':
|
|
568
579
|
next(True)
|
|
569
580
|
if peek() is '/':
|
|
570
|
-
verbose_regexp
|
|
581
|
+
verbose_regexp = True
|
|
571
582
|
next(True)
|
|
572
|
-
else:
|
|
583
|
+
else: # empty regexp (//)
|
|
573
584
|
mods = read_name()
|
|
574
|
-
return token(
|
|
585
|
+
return token('regexp', RegExp(regexp, mods))
|
|
575
586
|
while True:
|
|
576
587
|
ch = next(True)
|
|
577
588
|
if not ch:
|
|
@@ -581,15 +592,15 @@ def tokenizer(raw_text, filename, recover_errors):
|
|
|
581
592
|
in_comment = False
|
|
582
593
|
continue
|
|
583
594
|
if prev_backslash:
|
|
584
|
-
regexp +=
|
|
595
|
+
regexp += '\\' + ch
|
|
585
596
|
prev_backslash = False
|
|
586
|
-
elif ch is
|
|
597
|
+
elif ch is '[':
|
|
587
598
|
in_class = True
|
|
588
599
|
regexp += ch
|
|
589
|
-
elif ch is
|
|
600
|
+
elif ch is ']' and in_class:
|
|
590
601
|
in_class = False
|
|
591
602
|
regexp += ch
|
|
592
|
-
elif ch is
|
|
603
|
+
elif ch is '/' and not in_class:
|
|
593
604
|
if verbose_regexp:
|
|
594
605
|
if peek() is not '/':
|
|
595
606
|
regexp += '\\/'
|
|
@@ -600,7 +611,7 @@ def tokenizer(raw_text, filename, recover_errors):
|
|
|
600
611
|
continue
|
|
601
612
|
next(True)
|
|
602
613
|
break
|
|
603
|
-
elif ch is
|
|
614
|
+
elif ch is '\\':
|
|
604
615
|
prev_backslash = True
|
|
605
616
|
elif verbose_regexp and not in_class and ' \n\r\t'.indexOf(ch) is not -1:
|
|
606
617
|
pass
|
|
@@ -608,9 +619,8 @@ def tokenizer(raw_text, filename, recover_errors):
|
|
|
608
619
|
in_comment = True
|
|
609
620
|
else:
|
|
610
621
|
regexp += ch
|
|
611
|
-
|
|
612
622
|
mods = read_name()
|
|
613
|
-
return token(
|
|
623
|
+
return token('regexp', RegExp(regexp, mods))
|
|
614
624
|
)
|
|
615
625
|
|
|
616
626
|
def read_operator(prefix):
|
|
@@ -629,60 +639,78 @@ def tokenizer(raw_text, filename, recover_errors):
|
|
|
629
639
|
# pretend that this is an operator as the tokenizer only allows
|
|
630
640
|
# one character punctuation.
|
|
631
641
|
return token('punc', op)
|
|
632
|
-
return token(
|
|
642
|
+
return token('operator', op)
|
|
633
643
|
|
|
634
644
|
def handle_slash():
|
|
635
645
|
next()
|
|
636
|
-
return read_regexp(
|
|
646
|
+
return read_regexp('') if S.regex_allowed else read_operator('/')
|
|
637
647
|
|
|
638
648
|
def handle_dot():
|
|
639
649
|
next()
|
|
640
|
-
return read_num(
|
|
650
|
+
return read_num('.') if is_digit(peek().charCodeAt(0)) else token('punc', '.')
|
|
641
651
|
|
|
642
652
|
def read_word():
|
|
643
653
|
word = read_name()
|
|
644
|
-
return token(
|
|
654
|
+
return token(
|
|
655
|
+
'atom',
|
|
656
|
+
word
|
|
657
|
+
) if KEYWORDS_ATOM[word] else (token('name', word) if not KEYWORDS[word] else (token('operator', word) if OPERATORS[word] and prevChar() is not '.' else token('keyword', word)))
|
|
645
658
|
|
|
646
659
|
def next_token():
|
|
647
|
-
|
|
648
660
|
indent = parse_whitespace()
|
|
649
661
|
# if indent is 1:
|
|
650
662
|
# return token("punc", "{")
|
|
651
663
|
if indent is -1:
|
|
652
|
-
return token(
|
|
664
|
+
return token('punc', '}', False, True)
|
|
653
665
|
|
|
654
666
|
start_token()
|
|
655
667
|
ch = peek()
|
|
656
668
|
if not ch:
|
|
657
|
-
return token(
|
|
669
|
+
return token('eof')
|
|
658
670
|
|
|
659
671
|
code = ch.charCodeAt(0)
|
|
660
672
|
tmp_ = code
|
|
661
|
-
if tmp_ is 34 or tmp_ is 39:
|
|
662
|
-
|
|
663
|
-
|
|
673
|
+
if tmp_ is 34 or tmp_ is 39: # double-quote (") or single quote (')
|
|
674
|
+
stok = read_string(False)
|
|
675
|
+
# Look ahead to detect a following f-string (e.g. "a" f"b"), which would
|
|
676
|
+
# otherwise be parsed as a call expression "a"(...) after the tokenizer
|
|
677
|
+
# injects the f-string result as a parenthesised expression.
|
|
678
|
+
j = S.pos
|
|
679
|
+
while j < S.text.length and (S.text.charAt(j) is ' ' or S.text.charAt(j) is '\t'):
|
|
680
|
+
j += 1
|
|
681
|
+
if j < S.text.length and is_identifier_start(S.text.charAt(j).charCodeAt(0)):
|
|
682
|
+
k = j
|
|
683
|
+
while k < S.text.length and is_identifier_char(S.text.charAt(k)):
|
|
684
|
+
k += 1
|
|
685
|
+
potential_mod = S.text.substring(j, k)
|
|
686
|
+
if is_string_modifier(potential_mod) and k < S.text.length and '\'"'.indexOf(S.text.charAt(k)) is not -1:
|
|
687
|
+
mods_check = potential_mod.toLowerCase()
|
|
688
|
+
if mods_check.indexOf('f') is not -1 and mods_check.indexOf('v') is -1:
|
|
689
|
+
return handle_interpolated_string(stok.value, stok, False)
|
|
690
|
+
return stok
|
|
691
|
+
elif tmp_ is 35: # pound-sign (#)
|
|
664
692
|
if S.pos is 0 and S.text.charAt(1) is '!':
|
|
665
|
-
#shebang
|
|
693
|
+
# shebang
|
|
666
694
|
return read_line_comment(True)
|
|
667
695
|
regex_allowed = S.regex_allowed
|
|
668
696
|
S.comments_before.push(read_line_comment())
|
|
669
697
|
S.regex_allowed = regex_allowed
|
|
670
698
|
return next_token()
|
|
671
|
-
elif tmp_ is 46:
|
|
699
|
+
elif tmp_ is 46: # dot (.)
|
|
672
700
|
return handle_dot()
|
|
673
|
-
elif tmp_ is 47:
|
|
701
|
+
elif tmp_ is 47: # slash (/)
|
|
674
702
|
return handle_slash()
|
|
675
703
|
|
|
676
704
|
if is_digit(code):
|
|
677
705
|
return read_num()
|
|
678
706
|
|
|
679
707
|
if PUNC_CHARS[ch]:
|
|
680
|
-
return token(
|
|
708
|
+
return token('punc', next())
|
|
681
709
|
|
|
682
710
|
if OPERATOR_CHARS[ch]:
|
|
683
711
|
return read_operator()
|
|
684
712
|
|
|
685
|
-
if code is 92 and S.text.charAt(S.pos + 1) is
|
|
713
|
+
if code is 92 and S.text.charAt(S.pos + 1) is '\n':
|
|
686
714
|
# backslash will consume the newline character that follows
|
|
687
715
|
next()
|
|
688
716
|
# backslash
|
|
@@ -711,13 +739,13 @@ def tokenizer(raw_text, filename, recover_errors):
|
|
|
711
739
|
# loop forever (parse_error is raised before the character is
|
|
712
740
|
# consumed). Record it and skip the single offending character so the
|
|
713
741
|
# tokenizer always makes forward progress.
|
|
714
|
-
S.recovered_errors.push({'message':
|
|
742
|
+
S.recovered_errors.push({'message': 'Unexpected character «' + ch + '»',
|
|
715
743
|
'line': S.tokline, 'col': S.tokcol, 'pos': S.tokpos, 'is_eof': False})
|
|
716
744
|
next()
|
|
717
745
|
return next_token()
|
|
718
|
-
parse_error(
|
|
746
|
+
parse_error('Unexpected character «' + ch + '»')
|
|
719
747
|
|
|
720
|
-
next_token.context = def(nc):
|
|
748
|
+
next_token.context = def (nc):
|
|
721
749
|
nonlocal S
|
|
722
750
|
if nc:
|
|
723
751
|
S = nc
|