rapydscript-ng 0.7.24 → 0.8.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (132) hide show
  1. package/CHANGELOG.md +28 -0
  2. package/README.md +80 -26
  3. package/bin/build.ts +117 -0
  4. package/bin/package.json +1 -0
  5. package/bin/rapydscript +65 -62
  6. package/build_wheels.py +133 -0
  7. package/editor-plugins/README.md +80 -0
  8. package/editor-plugins/nvim.lua +321 -0
  9. package/local-agent.md +28 -0
  10. package/package.json +4 -5
  11. package/publish.py +27 -17
  12. package/release/baselib-plain-pretty.js +448 -326
  13. package/release/baselib-plain-ugly.js +3 -3
  14. package/release/compiler.js +9027 -8139
  15. package/release/signatures.json +27 -27
  16. package/release/stdlib_modules.json +1 -0
  17. package/src/ast.pyj +428 -280
  18. package/src/baselib-builtins.pyj +49 -28
  19. package/src/baselib-containers.pyj +241 -226
  20. package/src/baselib-errors.pyj +8 -1
  21. package/src/baselib-internal.pyj +38 -23
  22. package/src/baselib-itertools.pyj +13 -7
  23. package/src/baselib-str.pyj +59 -84
  24. package/src/compiler.pyj +4 -4
  25. package/src/errors.pyj +3 -4
  26. package/src/lib/aes.pyj +46 -32
  27. package/src/lib/elementmaker.pyj +11 -9
  28. package/src/lib/encodings.pyj +15 -5
  29. package/src/lib/gettext.pyj +9 -4
  30. package/src/lib/math.pyj +106 -45
  31. package/src/lib/operator.pyj +10 -10
  32. package/src/lib/pythonize.pyj +2 -1
  33. package/src/lib/random.pyj +28 -21
  34. package/src/lib/re.pyj +492 -19
  35. package/src/lib/traceback.pyj +171 -29
  36. package/src/lib/uuid.pyj +2 -2
  37. package/src/output/classes.pyj +28 -27
  38. package/src/output/codegen.pyj +87 -83
  39. package/src/output/comments.pyj +8 -8
  40. package/src/output/exceptions.pyj +20 -21
  41. package/src/output/functions.pyj +66 -64
  42. package/src/output/literals.pyj +24 -23
  43. package/src/output/loops.pyj +84 -142
  44. package/src/output/modules.pyj +55 -68
  45. package/src/output/operators.pyj +40 -29
  46. package/src/output/statements.pyj +21 -15
  47. package/src/output/stream.pyj +82 -56
  48. package/src/output/utils.pyj +12 -8
  49. package/src/parse.pyj +971 -620
  50. package/src/string_interpolation.pyj +5 -3
  51. package/src/tokenizer.pyj +188 -148
  52. package/src/utils.pyj +32 -15
  53. package/test/_treeshake_mod.pyj +30 -0
  54. package/test/_treeshake_side_effect.pyj +9 -0
  55. package/test/ast_serialization.pyj +392 -0
  56. package/test/async.pyj +95 -0
  57. package/test/baselib.pyj +1 -2
  58. package/test/embedded_compiler.pyj +57 -0
  59. package/test/fmt.pyj +291 -0
  60. package/test/generic.pyj +16 -3
  61. package/test/imports.pyj +8 -6
  62. package/test/internationalization.pyj +38 -37
  63. package/test/lint.pyj +120 -117
  64. package/test/lsp.pyj +257 -0
  65. package/test/repl.pyj +70 -65
  66. package/test/sourcemaps.pyj +315 -0
  67. package/test/starargs.pyj +46 -39
  68. package/test/str.pyj +8 -0
  69. package/test/traceback.pyj +263 -0
  70. package/test/treeshaking.pyj +181 -0
  71. package/test/web_repl_export.pyj +88 -0
  72. package/tools/ast_serialize.mjs +557 -0
  73. package/tools/cli.mjs +531 -0
  74. package/tools/compile.mjs +205 -0
  75. package/tools/compiler.mjs +159 -0
  76. package/tools/{completer.js → completer.mjs} +6 -6
  77. package/tools/{embedded_compiler.js → embedded_compiler.mjs} +17 -13
  78. package/tools/export.js +109 -56
  79. package/tools/fmt.mjs +975 -0
  80. package/tools/{gettext.js → gettext.mjs} +46 -53
  81. package/tools/ini.mjs +175 -0
  82. package/tools/{lint.js → lint.mjs} +78 -55
  83. package/tools/lsp.mjs +964 -0
  84. package/tools/lsp_protocol.mjs +259 -0
  85. package/tools/lsp_symbols.mjs +418 -0
  86. package/tools/{msgfmt.js → msgfmt.mjs} +18 -18
  87. package/tools/{repl.js → repl.mjs} +56 -42
  88. package/tools/{self.js → self.mjs} +71 -46
  89. package/tools/sourcemap.mjs +123 -0
  90. package/tools/test.mjs +252 -0
  91. package/tools/treeshake.mjs +400 -0
  92. package/tools/{utils.js → utils.mjs} +26 -23
  93. package/tools/{web_repl.js → web_repl.mjs} +15 -13
  94. package/tools/web_repl_export.mjs +97 -0
  95. package/tree-sitter/README.md +101 -0
  96. package/tree-sitter/grammar.js +992 -0
  97. package/tree-sitter/package.json +43 -0
  98. package/tree-sitter/queries/highlights.scm +232 -0
  99. package/tree-sitter/queries/indents.scm +64 -0
  100. package/tree-sitter/queries/injections.scm +14 -0
  101. package/tree-sitter/queries/locals.scm +108 -0
  102. package/tree-sitter/src/grammar.json +4976 -0
  103. package/tree-sitter/src/node-types.json +2901 -0
  104. package/tree-sitter/src/parser.c +196465 -0
  105. package/tree-sitter/src/scanner.c +294 -0
  106. package/tree-sitter/src/tree_sitter/alloc.h +54 -0
  107. package/tree-sitter/src/tree_sitter/array.h +330 -0
  108. package/tree-sitter/src/tree_sitter/parser.h +286 -0
  109. package/tree-sitter/test/corpus/chaining.txt +99 -0
  110. package/tree-sitter/test/corpus/classes.txt +147 -0
  111. package/tree-sitter/test/corpus/comprehensions.txt +155 -0
  112. package/tree-sitter/test/corpus/containers.txt +242 -0
  113. package/tree-sitter/test/corpus/control_flow.txt +361 -0
  114. package/tree-sitter/test/corpus/decorators.txt +64 -0
  115. package/tree-sitter/test/corpus/existential.txt +102 -0
  116. package/tree-sitter/test/corpus/functions.txt +293 -0
  117. package/tree-sitter/test/corpus/imports.txt +117 -0
  118. package/tree-sitter/test/corpus/literals.txt +135 -0
  119. package/tree-sitter/test/corpus/operators.txt +296 -0
  120. package/tree-sitter/test/corpus/statements.txt +189 -0
  121. package/tree-sitter/test/corpus/strings.txt +90 -0
  122. package/tree-sitter/test/corpus/subscripts.txt +227 -0
  123. package/tree-sitter/tree-sitter.json +36 -0
  124. package/web-repl/env.js +35 -23
  125. package/web-repl/main.js +8 -8
  126. package/bin/export +0 -75
  127. package/bin/web-repl-export +0 -102
  128. package/tools/cli.js +0 -523
  129. package/tools/compile.js +0 -184
  130. package/tools/compiler.js +0 -84
  131. package/tools/ini.js +0 -65
  132. package/tools/test.js +0 -110
package/src/tokenizer.pyj CHANGED
@@ -2,97 +2,97 @@
2
2
  # License: BSD Copyright: 2016, Kovid Goyal <kovid at kovidgoyal.net>
3
3
  from __python__ import hash_literals
4
4
 
5
- from unicode_aliases import ALIAS_MAP
6
- from utils import make_predicate, characters
7
5
  from ast import AST_Token
8
6
  from errors import SyntaxError
9
7
  from string_interpolation import interpolate, quoted_string
8
+ from unicode_aliases import ALIAS_MAP
9
+ from utils import make_predicate, characters
10
10
 
11
11
  RE_HEX_NUMBER = /^0x[0-9a-f]+$/i
12
12
  RE_OCT_NUMBER = /^0[0-7]+$/
13
13
  RE_DEC_NUMBER = /^\d*\.?\d*(?:e[+-]?\d*(?:\d\.?|\.?\d)\d*)?$/i
14
14
 
15
- OPERATOR_CHARS = make_predicate(characters("+-*&%=<>!?|~^@"))
15
+ OPERATOR_CHARS = make_predicate(characters('+-*&%=<>!?|~^@'))
16
16
 
17
- ASCII_CONTROL_CHARS = {'a':7, 'b':8, 'f': 12, 'n': 10, 'r': 13, 't': 9, 'v': 11}
17
+ ASCII_CONTROL_CHARS = {'a': 7, 'b': 8, 'f': 12, 'n': 10, 'r': 13, 't': 9, 'v': 11}
18
18
  HEX_PAT = /[a-fA-F0-9]/
19
19
  NAME_PAT = /[a-zA-Z ]/
20
20
 
21
21
  OPERATORS = make_predicate([
22
- "in",
23
- "instanceof",
24
- "typeof",
25
- "new",
26
- "void",
27
- "del",
28
- "+",
29
- "-",
30
- "not",
31
- "~",
32
- "&",
33
- "|",
34
- "^",
35
- "**",
36
- "*",
37
- "//",
38
- "/",
39
- "%",
40
- ">>",
41
- "<<",
42
- ">>>",
43
- "<",
44
- ">",
45
- "<=",
46
- ">=",
47
- "==",
48
- "is",
49
- "!=",
50
- "=",
51
- "+=",
52
- "-=",
53
- "//=",
54
- "/=",
55
- "*=",
56
- "%=",
57
- ">>=",
58
- "<<=",
59
- ">>>=",
60
- "|=",
61
- "^=",
62
- "&=",
63
- "and",
64
- "or",
65
- "@",
66
- "->"
22
+ 'in',
23
+ 'instanceof',
24
+ 'typeof',
25
+ 'new',
26
+ 'void',
27
+ 'del',
28
+ '+',
29
+ '-',
30
+ 'not',
31
+ '~',
32
+ '&',
33
+ '|',
34
+ '^',
35
+ '**',
36
+ '*',
37
+ '//',
38
+ '/',
39
+ '%',
40
+ '>>',
41
+ '<<',
42
+ '>>>',
43
+ '<',
44
+ '>',
45
+ '<=',
46
+ '>=',
47
+ '==',
48
+ 'is',
49
+ '!=',
50
+ '=',
51
+ '+=',
52
+ '-=',
53
+ '//=',
54
+ '/=',
55
+ '*=',
56
+ '%=',
57
+ '>>=',
58
+ '<<=',
59
+ '>>>=',
60
+ '|=',
61
+ '^=',
62
+ '&=',
63
+ 'and',
64
+ 'or',
65
+ '@',
66
+ '->'
67
67
  ])
68
68
 
69
69
  OP_MAP = {
70
- 'or': "||",
71
- 'and': "&&",
72
- 'not': "!",
73
- 'del': "delete",
74
- 'None': "null",
75
- 'is': "===",
70
+ 'or': '||',
71
+ 'and': '&&',
72
+ 'not': '!',
73
+ 'del': 'delete',
74
+ 'None': 'null',
75
+ 'is': '===',
76
76
  }
77
77
 
78
- WHITESPACE_CHARS = make_predicate(characters(" \u00a0\n\r\t\f\u000b\u200b\u180e\u2000\u2001\u2002\u2003\u2004\u2005\u2006\u2007\u2008\u2009\u200a\u202f\u205f\u3000"))
78
+ WHITESPACE_CHARS = make_predicate(characters(' \u00a0\n\r\t\f\u000b\u200b\u180e\u2000\u2001\u2002\u2003\u2004\u2005\u2006\u2007\u2008\u2009\u200a\u202f\u205f\u3000'))
79
79
 
80
- PUNC_BEFORE_EXPRESSION = make_predicate(characters("[{(,.;:"))
80
+ PUNC_BEFORE_EXPRESSION = make_predicate(characters('[{(,.;:'))
81
81
 
82
- PUNC_CHARS = make_predicate(characters("[]{}(),;:?"))
82
+ PUNC_CHARS = make_predicate(characters('[]{}(),;:?'))
83
83
 
84
- KEYWORDS = "as assert break class continue def del do elif else except finally for from global if import in is new nonlocal pass raise return yield try while with or and not"
84
+ KEYWORDS = 'as assert async await break class continue def del do elif else except finally for from global if import in is new nonlocal pass raise return yield try while with or and not'
85
85
 
86
- KEYWORDS_ATOM = "False None True"
86
+ KEYWORDS_ATOM = 'False None True'
87
87
 
88
88
  # see https://developer.mozilla.org/en-US/docs/Web/JavaScript/Reference/Lexical_grammar
89
- RESERVED_WORDS = ("break case class catch const continue debugger default delete do else export extends"
90
- " finally for function if import in instanceof new return super switch this throw try typeof var void"
91
- " while with yield enum implements static private package let public protected interface await null true false" )
89
+ RESERVED_WORDS = ('break case class catch const continue debugger default delete do else export extends'
90
+ ' finally for function if import in instanceof new return super switch this throw try typeof var void'
91
+ ' while with yield enum implements static private package let public protected interface null true false')
92
92
 
93
- KEYWORDS_BEFORE_EXPRESSION = "return yield new del raise elif else if"
93
+ KEYWORDS_BEFORE_EXPRESSION = 'return yield new del raise elif else if await'
94
94
 
95
- ALL_KEYWORDS = KEYWORDS + " " + KEYWORDS_ATOM
95
+ ALL_KEYWORDS = KEYWORDS + ' ' + KEYWORDS_ATOM
96
96
 
97
97
  KEYWORDS = make_predicate(KEYWORDS)
98
98
  RESERVED_WORDS = make_predicate(RESERVED_WORDS)
@@ -100,37 +100,47 @@ KEYWORDS_BEFORE_EXPRESSION = make_predicate(KEYWORDS_BEFORE_EXPRESSION)
100
100
  KEYWORDS_ATOM = make_predicate(KEYWORDS_ATOM)
101
101
  IDENTIFIER_PAT = /^[a-z_$][_a-z0-9$]*$/i
102
102
 
103
+
103
104
  def is_string_modifier(val):
104
105
  for ch in val:
105
106
  if 'vrufVRUF'.indexOf(ch) is -1:
106
107
  return False
107
108
  return True
108
109
 
110
+
109
111
  def is_letter(code):
110
112
  return code >= 97 and code <= 122 or code >= 65 and code <= 90 or code >= 170 and UNICODE.letter.test(String.fromCharCode(code))
111
113
 
114
+
112
115
  def is_digit(code):
113
116
  return code >= 48 and code <= 57
114
117
 
118
+
115
119
  def is_alphanumeric_char(code):
116
120
  return is_digit(code) or is_letter(code)
117
121
 
122
+
118
123
  def is_unicode_combining_mark(ch):
119
124
  return UNICODE.non_spacing_mark.test(ch) or UNICODE.space_combining_mark.test(ch)
120
125
 
126
+
121
127
  def is_unicode_connector_punctuation(ch):
122
128
  return UNICODE.connector_punctuation.test(ch)
123
129
 
130
+
124
131
  def is_identifier(name):
125
132
  return not RESERVED_WORDS[name] and not KEYWORDS[name] and not KEYWORDS_ATOM[name] and IDENTIFIER_PAT.test(name)
126
133
 
134
+
127
135
  def is_identifier_start(code):
128
136
  return code is 36 or code is 95 or is_letter(code)
129
137
 
138
+
130
139
  def is_identifier_char(ch):
131
140
  code = ch.charCodeAt(0)
132
141
  return is_identifier_start(code) or is_digit(code) or code is 8204 or code is 8205 or is_unicode_combining_mark(ch) or is_unicode_connector_punctuation(ch)
133
142
 
143
+
134
144
  def parse_js_number(num):
135
145
  if RE_HEX_NUMBER.test(num):
136
146
  return parseInt(num.substr(2), 16)
@@ -141,10 +151,10 @@ def parse_js_number(num):
141
151
 
142
152
  # regexps adapted from http://xregexp.com/plugins/#unicode
143
153
  UNICODE = { # {{{
144
- 'letter': RegExp("[\\u0041-\\u005A\\u0061-\\u007A\\u00AA\\u00B5\\u00BA\\u00C0-\\u00D6\\u00D8-\\u00F6\\u00F8-\\u02C1\\u02C6-\\u02D1\\u02E0-\\u02E4\\u02EC\\u02EE\\u0370-\\u0374\\u0376\\u0377\\u037A-\\u037D\\u0386\\u0388-\\u038A\\u038C\\u038E-\\u03A1\\u03A3-\\u03F5\\u03F7-\\u0481\\u048A-\\u0523\\u0531-\\u0556\\u0559\\u0561-\\u0587\\u05D0-\\u05EA\\u05F0-\\u05F2\\u0621-\\u064A\\u066E\\u066F\\u0671-\\u06D3\\u06D5\\u06E5\\u06E6\\u06EE\\u06EF\\u06FA-\\u06FC\\u06FF\\u0710\\u0712-\\u072F\\u074D-\\u07A5\\u07B1\\u07CA-\\u07EA\\u07F4\\u07F5\\u07FA\\u0904-\\u0939\\u093D\\u0950\\u0958-\\u0961\\u0971\\u0972\\u097B-\\u097F\\u0985-\\u098C\\u098F\\u0990\\u0993-\\u09A8\\u09AA-\\u09B0\\u09B2\\u09B6-\\u09B9\\u09BD\\u09CE\\u09DC\\u09DD\\u09DF-\\u09E1\\u09F0\\u09F1\\u0A05-\\u0A0A\\u0A0F\\u0A10\\u0A13-\\u0A28\\u0A2A-\\u0A30\\u0A32\\u0A33\\u0A35\\u0A36\\u0A38\\u0A39\\u0A59-\\u0A5C\\u0A5E\\u0A72-\\u0A74\\u0A85-\\u0A8D\\u0A8F-\\u0A91\\u0A93-\\u0AA8\\u0AAA-\\u0AB0\\u0AB2\\u0AB3\\u0AB5-\\u0AB9\\u0ABD\\u0AD0\\u0AE0\\u0AE1\\u0B05-\\u0B0C\\u0B0F\\u0B10\\u0B13-\\u0B28\\u0B2A-\\u0B30\\u0B32\\u0B33\\u0B35-\\u0B39\\u0B3D\\u0B5C\\u0B5D\\u0B5F-\\u0B61\\u0B71\\u0B83\\u0B85-\\u0B8A\\u0B8E-\\u0B90\\u0B92-\\u0B95\\u0B99\\u0B9A\\u0B9C\\u0B9E\\u0B9F\\u0BA3\\u0BA4\\u0BA8-\\u0BAA\\u0BAE-\\u0BB9\\u0BD0\\u0C05-\\u0C0C\\u0C0E-\\u0C10\\u0C12-\\u0C28\\u0C2A-\\u0C33\\u0C35-\\u0C39\\u0C3D\\u0C58\\u0C59\\u0C60\\u0C61\\u0C85-\\u0C8C\\u0C8E-\\u0C90\\u0C92-\\u0CA8\\u0CAA-\\u0CB3\\u0CB5-\\u0CB9\\u0CBD\\u0CDE\\u0CE0\\u0CE1\\u0D05-\\u0D0C\\u0D0E-\\u0D10\\u0D12-\\u0D28\\u0D2A-\\u0D39\\u0D3D\\u0D60\\u0D61\\u0D7A-\\u0D7F\\u0D85-\\u0D96\\u0D9A-\\u0DB1\\u0DB3-\\u0DBB\\u0DBD\\u0DC0-\\u0DC6\\u0E01-\\u0E30\\u0E32\\u0E33\\u0E40-\\u0E46\\u0E81\\u0E82\\u0E84\\u0E87\\u0E88\\u0E8A\\u0E8D\\u0E94-\\u0E97\\u0E99-\\u0E9F\\u0EA1-\\u0EA3\\u0EA5\\u0EA7\\u0EAA\\u0EAB\\u0EAD-\\u0EB0\\u0EB2\\u0EB3\\u0EBD\\u0EC0-\\u0EC4\\u0EC6\\u0EDC\\u0EDD\\u0F00\\u0F40-\\u0F47\\u0F49-\\u0F6C\\u0F88-\\u0F8B\\u1000-\\u102A\\u103F\\u1050-\\u1055\\u105A-\\u105D\\u1061\\u1065\\u1066\\u106E-\\u1070\\u1075-\\u1081\\u108E\\u10A0-\\u10C5\\u10D0-\\u10FA\\u10FC\\u1100-\\u1159\\u115F-\\u11A2\\u11A8-\\u11F9\\u1200-\\u1248\\u124A-\\u124D\\u1250-\\u1256\\u1258\\u125A-\\u125D\\u1260-\\u1288\\u128A-\\u128D\\u1290-\\u12B0\\u12B2-\\u12B5\\u12B8-\\u12BE\\u12C0\\u12C2-\\u12C5\\u12C8-\\u12D6\\u12D8-\\u1310\\u1312-\\u1315\\u1318-\\u135A\\u1380-\\u138F\\u13A0-\\u13F4\\u1401-\\u166C\\u166F-\\u1676\\u1681-\\u169A\\u16A0-\\u16EA\\u1700-\\u170C\\u170E-\\u1711\\u1720-\\u1731\\u1740-\\u1751\\u1760-\\u176C\\u176E-\\u1770\\u1780-\\u17B3\\u17D7\\u17DC\\u1820-\\u1877\\u1880-\\u18A8\\u18AA\\u1900-\\u191C\\u1950-\\u196D\\u1970-\\u1974\\u1980-\\u19A9\\u19C1-\\u19C7\\u1A00-\\u1A16\\u1B05-\\u1B33\\u1B45-\\u1B4B\\u1B83-\\u1BA0\\u1BAE\\u1BAF\\u1C00-\\u1C23\\u1C4D-\\u1C4F\\u1C5A-\\u1C7D\\u1D00-\\u1DBF\\u1E00-\\u1F15\\u1F18-\\u1F1D\\u1F20-\\u1F45\\u1F48-\\u1F4D\\u1F50-\\u1F57\\u1F59\\u1F5B\\u1F5D\\u1F5F-\\u1F7D\\u1F80-\\u1FB4\\u1FB6-\\u1FBC\\u1FBE\\u1FC2-\\u1FC4\\u1FC6-\\u1FCC\\u1FD0-\\u1FD3\\u1FD6-\\u1FDB\\u1FE0-\\u1FEC\\u1FF2-\\u1FF4\\u1FF6-\\u1FFC\\u2071\\u207F\\u2090-\\u2094\\u2102\\u2107\\u210A-\\u2113\\u2115\\u2119-\\u211D\\u2124\\u2126\\u2128\\u212A-\\u212D\\u212F-\\u2139\\u213C-\\u213F\\u2145-\\u2149\\u214E\\u2183\\u2184\\u2C00-\\u2C2E\\u2C30-\\u2C5E\\u2C60-\\u2C6F\\u2C71-\\u2C7D\\u2C80-\\u2CE4\\u2D00-\\u2D25\\u2D30-\\u2D65\\u2D6F\\u2D80-\\u2D96\\u2DA0-\\u2DA6\\u2DA8-\\u2DAE\\u2DB0-\\u2DB6\\u2DB8-\\u2DBE\\u2DC0-\\u2DC6\\u2DC8-\\u2DCE\\u2DD0-\\u2DD6\\u2DD8-\\u2DDE\\u2E2F\\u3005\\u3006\\u3031-\\u3035\\u303B\\u303C\\u3041-\\u3096\\u309D-\\u309F\\u30A1-\\u30FA\\u30FC-\\u30FF\\u3105-\\u312D\\u3131-\\u318E\\u31A0-\\u31B7\\u31F0-\\u31FF\\u3400\\u4DB5\\u4E00\\u9FC3\\uA000-\\uA48C\\uA500-\\uA60C\\uA610-\\uA61F\\uA62A\\uA62B\\uA640-\\uA65F\\uA662-\\uA66E\\uA67F-\\uA697\\uA717-\\uA71F\\uA722-\\uA788\\uA78B\\uA78C\\uA7FB-\\uA801\\uA803-\\uA805\\uA807-\\uA80A\\uA80C-\\uA822\\uA840-\\uA873\\uA882-\\uA8B3\\uA90A-\\uA925\\uA930-\\uA946\\uAA00-\\uAA28\\uAA40-\\uAA42\\uAA44-\\uAA4B\\uAC00\\uD7A3\\uF900-\\uFA2D\\uFA30-\\uFA6A\\uFA70-\\uFAD9\\uFB00-\\uFB06\\uFB13-\\uFB17\\uFB1D\\uFB1F-\\uFB28\\uFB2A-\\uFB36\\uFB38-\\uFB3C\\uFB3E\\uFB40\\uFB41\\uFB43\\uFB44\\uFB46-\\uFBB1\\uFBD3-\\uFD3D\\uFD50-\\uFD8F\\uFD92-\\uFDC7\\uFDF0-\\uFDFB\\uFE70-\\uFE74\\uFE76-\\uFEFC\\uFF21-\\uFF3A\\uFF41-\\uFF5A\\uFF66-\\uFFBE\\uFFC2-\\uFFC7\\uFFCA-\\uFFCF\\uFFD2-\\uFFD7\\uFFDA-\\uFFDC]"),
145
- 'non_spacing_mark': RegExp("[\\u0300-\\u036F\\u0483-\\u0487\\u0591-\\u05BD\\u05BF\\u05C1\\u05C2\\u05C4\\u05C5\\u05C7\\u0610-\\u061A\\u064B-\\u065E\\u0670\\u06D6-\\u06DC\\u06DF-\\u06E4\\u06E7\\u06E8\\u06EA-\\u06ED\\u0711\\u0730-\\u074A\\u07A6-\\u07B0\\u07EB-\\u07F3\\u0816-\\u0819\\u081B-\\u0823\\u0825-\\u0827\\u0829-\\u082D\\u0900-\\u0902\\u093C\\u0941-\\u0948\\u094D\\u0951-\\u0955\\u0962\\u0963\\u0981\\u09BC\\u09C1-\\u09C4\\u09CD\\u09E2\\u09E3\\u0A01\\u0A02\\u0A3C\\u0A41\\u0A42\\u0A47\\u0A48\\u0A4B-\\u0A4D\\u0A51\\u0A70\\u0A71\\u0A75\\u0A81\\u0A82\\u0ABC\\u0AC1-\\u0AC5\\u0AC7\\u0AC8\\u0ACD\\u0AE2\\u0AE3\\u0B01\\u0B3C\\u0B3F\\u0B41-\\u0B44\\u0B4D\\u0B56\\u0B62\\u0B63\\u0B82\\u0BC0\\u0BCD\\u0C3E-\\u0C40\\u0C46-\\u0C48\\u0C4A-\\u0C4D\\u0C55\\u0C56\\u0C62\\u0C63\\u0CBC\\u0CBF\\u0CC6\\u0CCC\\u0CCD\\u0CE2\\u0CE3\\u0D41-\\u0D44\\u0D4D\\u0D62\\u0D63\\u0DCA\\u0DD2-\\u0DD4\\u0DD6\\u0E31\\u0E34-\\u0E3A\\u0E47-\\u0E4E\\u0EB1\\u0EB4-\\u0EB9\\u0EBB\\u0EBC\\u0EC8-\\u0ECD\\u0F18\\u0F19\\u0F35\\u0F37\\u0F39\\u0F71-\\u0F7E\\u0F80-\\u0F84\\u0F86\\u0F87\\u0F90-\\u0F97\\u0F99-\\u0FBC\\u0FC6\\u102D-\\u1030\\u1032-\\u1037\\u1039\\u103A\\u103D\\u103E\\u1058\\u1059\\u105E-\\u1060\\u1071-\\u1074\\u1082\\u1085\\u1086\\u108D\\u109D\\u135F\\u1712-\\u1714\\u1732-\\u1734\\u1752\\u1753\\u1772\\u1773\\u17B7-\\u17BD\\u17C6\\u17C9-\\u17D3\\u17DD\\u180B-\\u180D\\u18A9\\u1920-\\u1922\\u1927\\u1928\\u1932\\u1939-\\u193B\\u1A17\\u1A18\\u1A56\\u1A58-\\u1A5E\\u1A60\\u1A62\\u1A65-\\u1A6C\\u1A73-\\u1A7C\\u1A7F\\u1B00-\\u1B03\\u1B34\\u1B36-\\u1B3A\\u1B3C\\u1B42\\u1B6B-\\u1B73\\u1B80\\u1B81\\u1BA2-\\u1BA5\\u1BA8\\u1BA9\\u1C2C-\\u1C33\\u1C36\\u1C37\\u1CD0-\\u1CD2\\u1CD4-\\u1CE0\\u1CE2-\\u1CE8\\u1CED\\u1DC0-\\u1DE6\\u1DFD-\\u1DFF\\u20D0-\\u20DC\\u20E1\\u20E5-\\u20F0\\u2CEF-\\u2CF1\\u2DE0-\\u2DFF\\u302A-\\u302F\\u3099\\u309A\\uA66F\\uA67C\\uA67D\\uA6F0\\uA6F1\\uA802\\uA806\\uA80B\\uA825\\uA826\\uA8C4\\uA8E0-\\uA8F1\\uA926-\\uA92D\\uA947-\\uA951\\uA980-\\uA982\\uA9B3\\uA9B6-\\uA9B9\\uA9BC\\uAA29-\\uAA2E\\uAA31\\uAA32\\uAA35\\uAA36\\uAA43\\uAA4C\\uAAB0\\uAAB2-\\uAAB4\\uAAB7\\uAAB8\\uAABE\\uAABF\\uAAC1\\uABE5\\uABE8\\uABED\\uFB1E\\uFE00-\\uFE0F\\uFE20-\\uFE26]"),
146
- 'space_combining_mark': RegExp("[\\u0903\\u093E-\\u0940\\u0949-\\u094C\\u094E\\u0982\\u0983\\u09BE-\\u09C0\\u09C7\\u09C8\\u09CB\\u09CC\\u09D7\\u0A03\\u0A3E-\\u0A40\\u0A83\\u0ABE-\\u0AC0\\u0AC9\\u0ACB\\u0ACC\\u0B02\\u0B03\\u0B3E\\u0B40\\u0B47\\u0B48\\u0B4B\\u0B4C\\u0B57\\u0BBE\\u0BBF\\u0BC1\\u0BC2\\u0BC6-\\u0BC8\\u0BCA-\\u0BCC\\u0BD7\\u0C01-\\u0C03\\u0C41-\\u0C44\\u0C82\\u0C83\\u0CBE\\u0CC0-\\u0CC4\\u0CC7\\u0CC8\\u0CCA\\u0CCB\\u0CD5\\u0CD6\\u0D02\\u0D03\\u0D3E-\\u0D40\\u0D46-\\u0D48\\u0D4A-\\u0D4C\\u0D57\\u0D82\\u0D83\\u0DCF-\\u0DD1\\u0DD8-\\u0DDF\\u0DF2\\u0DF3\\u0F3E\\u0F3F\\u0F7F\\u102B\\u102C\\u1031\\u1038\\u103B\\u103C\\u1056\\u1057\\u1062-\\u1064\\u1067-\\u106D\\u1083\\u1084\\u1087-\\u108C\\u108F\\u109A-\\u109C\\u17B6\\u17BE-\\u17C5\\u17C7\\u17C8\\u1923-\\u1926\\u1929-\\u192B\\u1930\\u1931\\u1933-\\u1938\\u19B0-\\u19C0\\u19C8\\u19C9\\u1A19-\\u1A1B\\u1A55\\u1A57\\u1A61\\u1A63\\u1A64\\u1A6D-\\u1A72\\u1B04\\u1B35\\u1B3B\\u1B3D-\\u1B41\\u1B43\\u1B44\\u1B82\\u1BA1\\u1BA6\\u1BA7\\u1BAA\\u1C24-\\u1C2B\\u1C34\\u1C35\\u1CE1\\u1CF2\\uA823\\uA824\\uA827\\uA880\\uA881\\uA8B4-\\uA8C3\\uA952\\uA953\\uA983\\uA9B4\\uA9B5\\uA9BA\\uA9BB\\uA9BD-\\uA9C0\\uAA2F\\uAA30\\uAA33\\uAA34\\uAA4D\\uAA7B\\uABE3\\uABE4\\uABE6\\uABE7\\uABE9\\uABEA\\uABEC]"),
147
- 'connector_punctuation': RegExp("[\\u005F\\u203F\\u2040\\u2054\\uFE33\\uFE34\\uFE4D-\\uFE4F\\uFF3F]")
154
+ 'letter': RegExp('[\\u0041-\\u005A\\u0061-\\u007A\\u00AA\\u00B5\\u00BA\\u00C0-\\u00D6\\u00D8-\\u00F6\\u00F8-\\u02C1\\u02C6-\\u02D1\\u02E0-\\u02E4\\u02EC\\u02EE\\u0370-\\u0374\\u0376\\u0377\\u037A-\\u037D\\u0386\\u0388-\\u038A\\u038C\\u038E-\\u03A1\\u03A3-\\u03F5\\u03F7-\\u0481\\u048A-\\u0523\\u0531-\\u0556\\u0559\\u0561-\\u0587\\u05D0-\\u05EA\\u05F0-\\u05F2\\u0621-\\u064A\\u066E\\u066F\\u0671-\\u06D3\\u06D5\\u06E5\\u06E6\\u06EE\\u06EF\\u06FA-\\u06FC\\u06FF\\u0710\\u0712-\\u072F\\u074D-\\u07A5\\u07B1\\u07CA-\\u07EA\\u07F4\\u07F5\\u07FA\\u0904-\\u0939\\u093D\\u0950\\u0958-\\u0961\\u0971\\u0972\\u097B-\\u097F\\u0985-\\u098C\\u098F\\u0990\\u0993-\\u09A8\\u09AA-\\u09B0\\u09B2\\u09B6-\\u09B9\\u09BD\\u09CE\\u09DC\\u09DD\\u09DF-\\u09E1\\u09F0\\u09F1\\u0A05-\\u0A0A\\u0A0F\\u0A10\\u0A13-\\u0A28\\u0A2A-\\u0A30\\u0A32\\u0A33\\u0A35\\u0A36\\u0A38\\u0A39\\u0A59-\\u0A5C\\u0A5E\\u0A72-\\u0A74\\u0A85-\\u0A8D\\u0A8F-\\u0A91\\u0A93-\\u0AA8\\u0AAA-\\u0AB0\\u0AB2\\u0AB3\\u0AB5-\\u0AB9\\u0ABD\\u0AD0\\u0AE0\\u0AE1\\u0B05-\\u0B0C\\u0B0F\\u0B10\\u0B13-\\u0B28\\u0B2A-\\u0B30\\u0B32\\u0B33\\u0B35-\\u0B39\\u0B3D\\u0B5C\\u0B5D\\u0B5F-\\u0B61\\u0B71\\u0B83\\u0B85-\\u0B8A\\u0B8E-\\u0B90\\u0B92-\\u0B95\\u0B99\\u0B9A\\u0B9C\\u0B9E\\u0B9F\\u0BA3\\u0BA4\\u0BA8-\\u0BAA\\u0BAE-\\u0BB9\\u0BD0\\u0C05-\\u0C0C\\u0C0E-\\u0C10\\u0C12-\\u0C28\\u0C2A-\\u0C33\\u0C35-\\u0C39\\u0C3D\\u0C58\\u0C59\\u0C60\\u0C61\\u0C85-\\u0C8C\\u0C8E-\\u0C90\\u0C92-\\u0CA8\\u0CAA-\\u0CB3\\u0CB5-\\u0CB9\\u0CBD\\u0CDE\\u0CE0\\u0CE1\\u0D05-\\u0D0C\\u0D0E-\\u0D10\\u0D12-\\u0D28\\u0D2A-\\u0D39\\u0D3D\\u0D60\\u0D61\\u0D7A-\\u0D7F\\u0D85-\\u0D96\\u0D9A-\\u0DB1\\u0DB3-\\u0DBB\\u0DBD\\u0DC0-\\u0DC6\\u0E01-\\u0E30\\u0E32\\u0E33\\u0E40-\\u0E46\\u0E81\\u0E82\\u0E84\\u0E87\\u0E88\\u0E8A\\u0E8D\\u0E94-\\u0E97\\u0E99-\\u0E9F\\u0EA1-\\u0EA3\\u0EA5\\u0EA7\\u0EAA\\u0EAB\\u0EAD-\\u0EB0\\u0EB2\\u0EB3\\u0EBD\\u0EC0-\\u0EC4\\u0EC6\\u0EDC\\u0EDD\\u0F00\\u0F40-\\u0F47\\u0F49-\\u0F6C\\u0F88-\\u0F8B\\u1000-\\u102A\\u103F\\u1050-\\u1055\\u105A-\\u105D\\u1061\\u1065\\u1066\\u106E-\\u1070\\u1075-\\u1081\\u108E\\u10A0-\\u10C5\\u10D0-\\u10FA\\u10FC\\u1100-\\u1159\\u115F-\\u11A2\\u11A8-\\u11F9\\u1200-\\u1248\\u124A-\\u124D\\u1250-\\u1256\\u1258\\u125A-\\u125D\\u1260-\\u1288\\u128A-\\u128D\\u1290-\\u12B0\\u12B2-\\u12B5\\u12B8-\\u12BE\\u12C0\\u12C2-\\u12C5\\u12C8-\\u12D6\\u12D8-\\u1310\\u1312-\\u1315\\u1318-\\u135A\\u1380-\\u138F\\u13A0-\\u13F4\\u1401-\\u166C\\u166F-\\u1676\\u1681-\\u169A\\u16A0-\\u16EA\\u1700-\\u170C\\u170E-\\u1711\\u1720-\\u1731\\u1740-\\u1751\\u1760-\\u176C\\u176E-\\u1770\\u1780-\\u17B3\\u17D7\\u17DC\\u1820-\\u1877\\u1880-\\u18A8\\u18AA\\u1900-\\u191C\\u1950-\\u196D\\u1970-\\u1974\\u1980-\\u19A9\\u19C1-\\u19C7\\u1A00-\\u1A16\\u1B05-\\u1B33\\u1B45-\\u1B4B\\u1B83-\\u1BA0\\u1BAE\\u1BAF\\u1C00-\\u1C23\\u1C4D-\\u1C4F\\u1C5A-\\u1C7D\\u1D00-\\u1DBF\\u1E00-\\u1F15\\u1F18-\\u1F1D\\u1F20-\\u1F45\\u1F48-\\u1F4D\\u1F50-\\u1F57\\u1F59\\u1F5B\\u1F5D\\u1F5F-\\u1F7D\\u1F80-\\u1FB4\\u1FB6-\\u1FBC\\u1FBE\\u1FC2-\\u1FC4\\u1FC6-\\u1FCC\\u1FD0-\\u1FD3\\u1FD6-\\u1FDB\\u1FE0-\\u1FEC\\u1FF2-\\u1FF4\\u1FF6-\\u1FFC\\u2071\\u207F\\u2090-\\u2094\\u2102\\u2107\\u210A-\\u2113\\u2115\\u2119-\\u211D\\u2124\\u2126\\u2128\\u212A-\\u212D\\u212F-\\u2139\\u213C-\\u213F\\u2145-\\u2149\\u214E\\u2183\\u2184\\u2C00-\\u2C2E\\u2C30-\\u2C5E\\u2C60-\\u2C6F\\u2C71-\\u2C7D\\u2C80-\\u2CE4\\u2D00-\\u2D25\\u2D30-\\u2D65\\u2D6F\\u2D80-\\u2D96\\u2DA0-\\u2DA6\\u2DA8-\\u2DAE\\u2DB0-\\u2DB6\\u2DB8-\\u2DBE\\u2DC0-\\u2DC6\\u2DC8-\\u2DCE\\u2DD0-\\u2DD6\\u2DD8-\\u2DDE\\u2E2F\\u3005\\u3006\\u3031-\\u3035\\u303B\\u303C\\u3041-\\u3096\\u309D-\\u309F\\u30A1-\\u30FA\\u30FC-\\u30FF\\u3105-\\u312D\\u3131-\\u318E\\u31A0-\\u31B7\\u31F0-\\u31FF\\u3400\\u4DB5\\u4E00\\u9FC3\\uA000-\\uA48C\\uA500-\\uA60C\\uA610-\\uA61F\\uA62A\\uA62B\\uA640-\\uA65F\\uA662-\\uA66E\\uA67F-\\uA697\\uA717-\\uA71F\\uA722-\\uA788\\uA78B\\uA78C\\uA7FB-\\uA801\\uA803-\\uA805\\uA807-\\uA80A\\uA80C-\\uA822\\uA840-\\uA873\\uA882-\\uA8B3\\uA90A-\\uA925\\uA930-\\uA946\\uAA00-\\uAA28\\uAA40-\\uAA42\\uAA44-\\uAA4B\\uAC00\\uD7A3\\uF900-\\uFA2D\\uFA30-\\uFA6A\\uFA70-\\uFAD9\\uFB00-\\uFB06\\uFB13-\\uFB17\\uFB1D\\uFB1F-\\uFB28\\uFB2A-\\uFB36\\uFB38-\\uFB3C\\uFB3E\\uFB40\\uFB41\\uFB43\\uFB44\\uFB46-\\uFBB1\\uFBD3-\\uFD3D\\uFD50-\\uFD8F\\uFD92-\\uFDC7\\uFDF0-\\uFDFB\\uFE70-\\uFE74\\uFE76-\\uFEFC\\uFF21-\\uFF3A\\uFF41-\\uFF5A\\uFF66-\\uFFBE\\uFFC2-\\uFFC7\\uFFCA-\\uFFCF\\uFFD2-\\uFFD7\\uFFDA-\\uFFDC]'),
155
+ 'non_spacing_mark': RegExp('[\\u0300-\\u036F\\u0483-\\u0487\\u0591-\\u05BD\\u05BF\\u05C1\\u05C2\\u05C4\\u05C5\\u05C7\\u0610-\\u061A\\u064B-\\u065E\\u0670\\u06D6-\\u06DC\\u06DF-\\u06E4\\u06E7\\u06E8\\u06EA-\\u06ED\\u0711\\u0730-\\u074A\\u07A6-\\u07B0\\u07EB-\\u07F3\\u0816-\\u0819\\u081B-\\u0823\\u0825-\\u0827\\u0829-\\u082D\\u0900-\\u0902\\u093C\\u0941-\\u0948\\u094D\\u0951-\\u0955\\u0962\\u0963\\u0981\\u09BC\\u09C1-\\u09C4\\u09CD\\u09E2\\u09E3\\u0A01\\u0A02\\u0A3C\\u0A41\\u0A42\\u0A47\\u0A48\\u0A4B-\\u0A4D\\u0A51\\u0A70\\u0A71\\u0A75\\u0A81\\u0A82\\u0ABC\\u0AC1-\\u0AC5\\u0AC7\\u0AC8\\u0ACD\\u0AE2\\u0AE3\\u0B01\\u0B3C\\u0B3F\\u0B41-\\u0B44\\u0B4D\\u0B56\\u0B62\\u0B63\\u0B82\\u0BC0\\u0BCD\\u0C3E-\\u0C40\\u0C46-\\u0C48\\u0C4A-\\u0C4D\\u0C55\\u0C56\\u0C62\\u0C63\\u0CBC\\u0CBF\\u0CC6\\u0CCC\\u0CCD\\u0CE2\\u0CE3\\u0D41-\\u0D44\\u0D4D\\u0D62\\u0D63\\u0DCA\\u0DD2-\\u0DD4\\u0DD6\\u0E31\\u0E34-\\u0E3A\\u0E47-\\u0E4E\\u0EB1\\u0EB4-\\u0EB9\\u0EBB\\u0EBC\\u0EC8-\\u0ECD\\u0F18\\u0F19\\u0F35\\u0F37\\u0F39\\u0F71-\\u0F7E\\u0F80-\\u0F84\\u0F86\\u0F87\\u0F90-\\u0F97\\u0F99-\\u0FBC\\u0FC6\\u102D-\\u1030\\u1032-\\u1037\\u1039\\u103A\\u103D\\u103E\\u1058\\u1059\\u105E-\\u1060\\u1071-\\u1074\\u1082\\u1085\\u1086\\u108D\\u109D\\u135F\\u1712-\\u1714\\u1732-\\u1734\\u1752\\u1753\\u1772\\u1773\\u17B7-\\u17BD\\u17C6\\u17C9-\\u17D3\\u17DD\\u180B-\\u180D\\u18A9\\u1920-\\u1922\\u1927\\u1928\\u1932\\u1939-\\u193B\\u1A17\\u1A18\\u1A56\\u1A58-\\u1A5E\\u1A60\\u1A62\\u1A65-\\u1A6C\\u1A73-\\u1A7C\\u1A7F\\u1B00-\\u1B03\\u1B34\\u1B36-\\u1B3A\\u1B3C\\u1B42\\u1B6B-\\u1B73\\u1B80\\u1B81\\u1BA2-\\u1BA5\\u1BA8\\u1BA9\\u1C2C-\\u1C33\\u1C36\\u1C37\\u1CD0-\\u1CD2\\u1CD4-\\u1CE0\\u1CE2-\\u1CE8\\u1CED\\u1DC0-\\u1DE6\\u1DFD-\\u1DFF\\u20D0-\\u20DC\\u20E1\\u20E5-\\u20F0\\u2CEF-\\u2CF1\\u2DE0-\\u2DFF\\u302A-\\u302F\\u3099\\u309A\\uA66F\\uA67C\\uA67D\\uA6F0\\uA6F1\\uA802\\uA806\\uA80B\\uA825\\uA826\\uA8C4\\uA8E0-\\uA8F1\\uA926-\\uA92D\\uA947-\\uA951\\uA980-\\uA982\\uA9B3\\uA9B6-\\uA9B9\\uA9BC\\uAA29-\\uAA2E\\uAA31\\uAA32\\uAA35\\uAA36\\uAA43\\uAA4C\\uAAB0\\uAAB2-\\uAAB4\\uAAB7\\uAAB8\\uAABE\\uAABF\\uAAC1\\uABE5\\uABE8\\uABED\\uFB1E\\uFE00-\\uFE0F\\uFE20-\\uFE26]'),
156
+ 'space_combining_mark': RegExp('[\\u0903\\u093E-\\u0940\\u0949-\\u094C\\u094E\\u0982\\u0983\\u09BE-\\u09C0\\u09C7\\u09C8\\u09CB\\u09CC\\u09D7\\u0A03\\u0A3E-\\u0A40\\u0A83\\u0ABE-\\u0AC0\\u0AC9\\u0ACB\\u0ACC\\u0B02\\u0B03\\u0B3E\\u0B40\\u0B47\\u0B48\\u0B4B\\u0B4C\\u0B57\\u0BBE\\u0BBF\\u0BC1\\u0BC2\\u0BC6-\\u0BC8\\u0BCA-\\u0BCC\\u0BD7\\u0C01-\\u0C03\\u0C41-\\u0C44\\u0C82\\u0C83\\u0CBE\\u0CC0-\\u0CC4\\u0CC7\\u0CC8\\u0CCA\\u0CCB\\u0CD5\\u0CD6\\u0D02\\u0D03\\u0D3E-\\u0D40\\u0D46-\\u0D48\\u0D4A-\\u0D4C\\u0D57\\u0D82\\u0D83\\u0DCF-\\u0DD1\\u0DD8-\\u0DDF\\u0DF2\\u0DF3\\u0F3E\\u0F3F\\u0F7F\\u102B\\u102C\\u1031\\u1038\\u103B\\u103C\\u1056\\u1057\\u1062-\\u1064\\u1067-\\u106D\\u1083\\u1084\\u1087-\\u108C\\u108F\\u109A-\\u109C\\u17B6\\u17BE-\\u17C5\\u17C7\\u17C8\\u1923-\\u1926\\u1929-\\u192B\\u1930\\u1931\\u1933-\\u1938\\u19B0-\\u19C0\\u19C8\\u19C9\\u1A19-\\u1A1B\\u1A55\\u1A57\\u1A61\\u1A63\\u1A64\\u1A6D-\\u1A72\\u1B04\\u1B35\\u1B3B\\u1B3D-\\u1B41\\u1B43\\u1B44\\u1B82\\u1BA1\\u1BA6\\u1BA7\\u1BAA\\u1C24-\\u1C2B\\u1C34\\u1C35\\u1CE1\\u1CF2\\uA823\\uA824\\uA827\\uA880\\uA881\\uA8B4-\\uA8C3\\uA952\\uA953\\uA983\\uA9B4\\uA9B5\\uA9BA\\uA9BB\\uA9BD-\\uA9C0\\uAA2F\\uAA30\\uAA33\\uAA34\\uAA4D\\uAA7B\\uABE3\\uABE4\\uABE6\\uABE7\\uABE9\\uABEA\\uABEC]'),
157
+ 'connector_punctuation': RegExp('[\\u005F\\u203F\\u2040\\u2054\\uFE33\\uFE34\\uFE4D-\\uFE4F\\uFF3F]')
148
158
  } # }}}
149
159
 
150
160
 
@@ -153,10 +163,13 @@ def is_token(token, type, val):
153
163
 
154
164
  EX_EOF = {}
155
165
 
156
- def tokenizer(raw_text, filename):
166
+
167
+ def tokenizer(raw_text, filename, recover_errors):
157
168
  S = {
158
- 'text': raw_text.replace(/\r\n?|[\n\u2028\u2029]/g, "\n").replace(/\uFEFF/g, ""),
169
+ 'text': raw_text.replace(/\r\n?|[\n\u2028\u2029]/g, '\n').replace(/\uFEFF/g, ''),
159
170
  'filename': filename,
171
+ 'recover_errors': not not recover_errors, # LSP error-recovery mode (see parse())
172
+ 'recovered_errors': v'[]', # collected errors when recover_errors is on
160
173
  'pos': 0,
161
174
  'tokpos': 0,
162
175
  'line': 1,
@@ -170,11 +183,12 @@ def tokenizer(raw_text, filename):
170
183
  'newblock': False,
171
184
  'endblock': False,
172
185
  'indentation_matters': v'[ true ]',
173
- 'cached_whitespace': "",
186
+ 'cached_whitespace': '',
174
187
  'prev': undefined,
175
188
  'index_or_slice': v'[ false ]',
176
189
  'expecting_object_literal_key': False, # This is set by the parser when it is expecting an object literal key
177
190
  }
191
+
178
192
  def peek():
179
193
  return S.text.charAt(S.pos)
180
194
 
@@ -187,7 +201,7 @@ def tokenizer(raw_text, filename):
187
201
  if signal_eof and not ch:
188
202
  raise EX_EOF
189
203
 
190
- if ch is "\n":
204
+ if ch is '\n':
191
205
  S.newline_before = S.newline_before or not in_string
192
206
  S.line += 1
193
207
  S.col = 0
@@ -207,15 +221,15 @@ def tokenizer(raw_text, filename):
207
221
  S.tokpos = S.pos
208
222
 
209
223
  def token(type, value, is_comment, keep_newline):
210
- S.regex_allowed = (type is "operator"
211
- or type is "keyword" and KEYWORDS_BEFORE_EXPRESSION[value]
212
- or type is "punc" and PUNC_BEFORE_EXPRESSION[value])
224
+ S.regex_allowed = (type is 'operator'
225
+ or type is 'keyword' and KEYWORDS_BEFORE_EXPRESSION[value]
226
+ or type is 'punc' and PUNC_BEFORE_EXPRESSION[value])
213
227
 
214
- if type is "operator" and value is "is" and S.text.substr(S.pos).trimLeft().substr(0, 4).trimRight() is "not":
228
+ if type is 'operator' and value is 'is' and S.text.substr(S.pos).trimLeft().substr(0, 4).trimRight() is 'not':
215
229
  next_token()
216
- value = "!=="
230
+ value = '!=='
217
231
 
218
- if type is "operator" and OP_MAP[value]:
232
+ if type is 'operator' and OP_MAP[value]:
219
233
  value = OP_MAP[value]
220
234
 
221
235
  ret = {
@@ -239,29 +253,29 @@ def tokenizer(raw_text, filename):
239
253
  if not keep_newline:
240
254
  S.newline_before = False
241
255
 
242
- if type is "punc":
256
+ if type is 'punc':
243
257
  # if (value is ":" && peek() is "\n") {
244
- if value is ":" and not S.index_or_slice[-1]
258
+ if value is ':' and not S.index_or_slice[-1]
245
259
  and not S.expecting_object_literal_key
246
- and (not S.text.substring(S.pos + 1, find("\n")).trim() or not S.text.substring(S.pos + 1, find("#")).trim()):
260
+ and (not S.text.substring(S.pos + 1, find('\n')).trim() or not S.text.substring(S.pos + 1, find('#')).trim()):
247
261
  S.newblock = True
248
262
  S.indentation_matters.push(True)
249
263
 
250
- if value is "[":
264
+ if value is '[':
251
265
  if S.prev and (
252
- S.prev.type is "name" or
266
+ S.prev.type is 'name' or
253
267
  (S.prev.type is 'punc' and ')]'.indexOf(S.prev.value) is not -1)
254
268
  ):
255
269
  S.index_or_slice.push(True)
256
270
  else:
257
271
  S.index_or_slice.push(False)
258
272
  S.indentation_matters.push(False)
259
- elif value is "{" or value is "(":
273
+ elif value is '{' or value is '(':
260
274
  S.indentation_matters.push(False)
261
- elif value is "]":
275
+ elif value is ']':
262
276
  S.index_or_slice.pop()
263
277
  S.indentation_matters.pop()
264
- elif value is "}" or value is ")":
278
+ elif value is '}' or value is ')':
265
279
  S.indentation_matters.pop()
266
280
  S.prev = AST_Token(ret)
267
281
  return S.prev
@@ -269,16 +283,16 @@ def tokenizer(raw_text, filename):
269
283
  # this will transform leading whitespace to block tokens unless
270
284
  # part of array/hash, and skip non-leading whitespace
271
285
  def parse_whitespace():
272
- leading_whitespace = ""
286
+ leading_whitespace = ''
273
287
  whitespace_exists = False
274
288
  while WHITESPACE_CHARS[peek()]:
275
289
  whitespace_exists = True
276
290
  ch = next()
277
- if ch is "\n":
278
- leading_whitespace = ""
291
+ if ch is '\n':
292
+ leading_whitespace = ''
279
293
  else:
280
294
  leading_whitespace += ch
281
- if peek() is not "#":
295
+ if peek() is not '#':
282
296
  if not whitespace_exists:
283
297
  leading_whitespace = S.cached_whitespace
284
298
  else:
@@ -287,7 +301,7 @@ def tokenizer(raw_text, filename):
287
301
  return test_indent_token(leading_whitespace)
288
302
 
289
303
  def test_indent_token(leading_whitespace):
290
- most_recent = S.whitespace_before[-1] or ""
304
+ most_recent = S.whitespace_before[-1] or ''
291
305
  S.endblock = False
292
306
  if S.indentation_matters[-1] and leading_whitespace is not most_recent:
293
307
  if S.newblock and leading_whitespace and leading_whitespace.indexOf(most_recent) is 0:
@@ -302,11 +316,11 @@ def tokenizer(raw_text, filename):
302
316
  return -1
303
317
  else:
304
318
  # indent mismatch, inconsistent indentation
305
- parse_error("Inconsistent indentation")
319
+ parse_error('Inconsistent indentation')
306
320
  return 0
307
321
 
308
322
  def read_while(pred):
309
- ret = ""
323
+ ret = ''
310
324
  i = 0
311
325
  ch = ''
312
326
  while (ch = peek()) and pred(ch, i):
@@ -320,16 +334,16 @@ def tokenizer(raw_text, filename):
320
334
  def read_num(prefix):
321
335
  has_e = False
322
336
  has_x = False
323
- has_dot = prefix is "."
337
+ has_dot = prefix is '.'
324
338
  if not prefix and peek() is '0' and S.text.charAt(S.pos + 1) is 'b':
325
339
  next(), next()
326
- num = read_while(def(ch): return ch is '0' or ch is '1';)
340
+ num = read_while(def (ch): return ch is '0' or ch is '1';)
327
341
  valid = parseInt(num, 2)
328
342
  if isNaN(valid):
329
343
  parse_error('Invalid syntax for a binary number')
330
344
  return token('num', valid)
331
345
  seen = v'[]'
332
- num = read_while(def(ch, i):
346
+ num = read_while(def (ch, i):
333
347
  nonlocal has_dot, has_e, has_x
334
348
  seen.push(ch)
335
349
  if ch is 'x' or ch is 'X':
@@ -347,11 +361,11 @@ def tokenizer(raw_text, filename):
347
361
  elif ch is '-':
348
362
  if i is 0 and not prefix:
349
363
  return True
350
- if has_e and seen[i-1].toLowerCase() is 'e':
364
+ if has_e and seen[i - 1].toLowerCase() is 'e':
351
365
  return True
352
366
  return False
353
367
  elif ch is '+':
354
- if has_e and seen[i-1].toLowerCase() is 'e':
368
+ if has_e and seen[i - 1].toLowerCase() is 'e':
355
369
  return True
356
370
  return False
357
371
  elif ch is '.':
@@ -363,9 +377,9 @@ def tokenizer(raw_text, filename):
363
377
 
364
378
  valid = parse_js_number(num)
365
379
  if not isNaN(valid):
366
- return token("num", valid)
380
+ return token('num', valid)
367
381
  else:
368
- parse_error("Invalid syntax: " + num)
382
+ parse_error('Invalid syntax: ' + num)
369
383
 
370
384
  def read_hex_digits(count):
371
385
  ans = ''
@@ -415,7 +429,7 @@ def tokenizer(raw_text, filename):
415
429
  if code <= 0xFFFF:
416
430
  return String.fromCharCode(code)
417
431
  code -= 0x10000
418
- return String.fromCharCode(0xD800+(code>>10), 0xDC00+(code&0x3FF))
432
+ return String.fromCharCode(0xD800 + (code >> 10), 0xDC00 + (code & 0x3FF))
419
433
  return '\\U' + code
420
434
  if q is 'N' and peek() is '{':
421
435
  next()
@@ -430,11 +444,11 @@ def tokenizer(raw_text, filename):
430
444
  if code <= 0xFFFF:
431
445
  return String.fromCharCode(code)
432
446
  code -= 0x10000
433
- return String.fromCharCode(0xD800+(code>>10), 0xDC00+(code&0x3FF))
447
+ return String.fromCharCode(0xD800 + (code >> 10), 0xDC00 + (code & 0x3FF))
434
448
  return '\\' + q
435
449
 
436
450
  def with_eof_error(eof_error, cont):
437
- return def():
451
+ return def ():
438
452
  try:
439
453
  return cont.apply(None, arguments)
440
454
  except as ex:
@@ -443,10 +457,10 @@ def tokenizer(raw_text, filename):
443
457
  else:
444
458
  raise
445
459
 
446
- read_string = with_eof_error("Unterminated string constant", def(is_raw_literal, is_js_literal):
460
+ read_string = with_eof_error('Unterminated string constant', def (is_raw_literal, is_js_literal):
447
461
  quote = next()
448
462
  tok_type = 'js' if is_js_literal else 'string'
449
- ret = ""
463
+ ret = ''
450
464
  is_multiline = False
451
465
  if peek() is quote:
452
466
  # two quotes in a row
@@ -457,18 +471,15 @@ def tokenizer(raw_text, filename):
457
471
  is_multiline = True
458
472
  else:
459
473
  return token(tok_type, '')
460
-
461
474
  while True:
462
475
  ch = next(True, True)
463
476
  if not ch:
464
477
  break
465
- if ch is "\n" and not is_multiline:
466
- parse_error("End of line while scanning string literal")
467
-
468
- if ch is "\\":
478
+ if ch is '\n' and not is_multiline:
479
+ parse_error('End of line while scanning string literal')
480
+ if ch is '\\':
469
481
  ret += ('\\' + next(True)) if is_raw_literal else read_escape_sequence()
470
482
  continue
471
-
472
483
  if ch is quote:
473
484
  if not is_multiline:
474
485
  break
@@ -483,10 +494,13 @@ def tokenizer(raw_text, filename):
483
494
  return token(tok_type, ret)
484
495
  )
485
496
 
486
- def handle_interpolated_string(string, start_tok):
497
+ def handle_interpolated_string(string, start_tok, is_fstring=True):
487
498
  def raise_error(err):
488
499
  raise new SyntaxError(err, filename, start_tok.line, start_tok.col, start_tok.pos, False)
489
- parts = v'[interpolate(string, raise_error)]'
500
+ if is_fstring:
501
+ parts = v'[interpolate(string, raise_error)]'
502
+ else:
503
+ parts = v'[quoted_string(string)]'
490
504
  # Look ahead for consecutive string literals to concatenate (e.g. f'a'f'b' or f'a''b')
491
505
  while True:
492
506
  # Skip horizontal whitespace (spaces and tabs, not newlines)
@@ -530,7 +544,7 @@ def tokenizer(raw_text, filename):
530
544
  def read_line_comment(shebang):
531
545
  if not shebang:
532
546
  next()
533
- i = find("\n")
547
+ i = find('\n')
534
548
 
535
549
  if i is -1:
536
550
  ret = S.text.substr(S.pos)
@@ -539,13 +553,13 @@ def tokenizer(raw_text, filename):
539
553
  ret = S.text.substring(S.pos, i)
540
554
  S.pos = i
541
555
 
542
- return token("shebang" if shebang else "comment1", ret, True)
556
+ return token('shebang' if shebang else 'comment1', ret, True)
543
557
 
544
558
  def read_name():
545
- name = ch = ""
559
+ name = ch = ''
546
560
  while (ch = peek()) is not None:
547
- if ch is "\\":
548
- if S.text.charAt(S.pos + 1) is "\n":
561
+ if ch is '\\':
562
+ if S.text.charAt(S.pos + 1) is '\n':
549
563
  S.pos += 2
550
564
  continue
551
565
  break
@@ -555,21 +569,20 @@ def tokenizer(raw_text, filename):
555
569
  break
556
570
  return name
557
571
 
558
- read_regexp = with_eof_error("Unterminated regular expression", def():
572
+ read_regexp = with_eof_error('Unterminated regular expression', def ():
559
573
  prev_backslash = False
560
574
  regexp = ch = ''
561
575
  in_class = False
562
576
  verbose_regexp = False
563
577
  in_comment = False
564
-
565
578
  if peek() is '/':
566
579
  next(True)
567
580
  if peek() is '/':
568
- verbose_regexp = True
581
+ verbose_regexp = True
569
582
  next(True)
570
- else: # empty regexp (//)
583
+ else: # empty regexp (//)
571
584
  mods = read_name()
572
- return token("regexp", RegExp(regexp, mods))
585
+ return token('regexp', RegExp(regexp, mods))
573
586
  while True:
574
587
  ch = next(True)
575
588
  if not ch:
@@ -579,15 +592,15 @@ def tokenizer(raw_text, filename):
579
592
  in_comment = False
580
593
  continue
581
594
  if prev_backslash:
582
- regexp += "\\" + ch
595
+ regexp += '\\' + ch
583
596
  prev_backslash = False
584
- elif ch is "[":
597
+ elif ch is '[':
585
598
  in_class = True
586
599
  regexp += ch
587
- elif ch is "]" and in_class:
600
+ elif ch is ']' and in_class:
588
601
  in_class = False
589
602
  regexp += ch
590
- elif ch is "/" and not in_class:
603
+ elif ch is '/' and not in_class:
591
604
  if verbose_regexp:
592
605
  if peek() is not '/':
593
606
  regexp += '\\/'
@@ -598,7 +611,7 @@ def tokenizer(raw_text, filename):
598
611
  continue
599
612
  next(True)
600
613
  break
601
- elif ch is "\\":
614
+ elif ch is '\\':
602
615
  prev_backslash = True
603
616
  elif verbose_regexp and not in_class and ' \n\r\t'.indexOf(ch) is not -1:
604
617
  pass
@@ -606,9 +619,8 @@ def tokenizer(raw_text, filename):
606
619
  in_comment = True
607
620
  else:
608
621
  regexp += ch
609
-
610
622
  mods = read_name()
611
- return token("regexp", RegExp(regexp, mods))
623
+ return token('regexp', RegExp(regexp, mods))
612
624
  )
613
625
 
614
626
  def read_operator(prefix):
@@ -627,60 +639,78 @@ def tokenizer(raw_text, filename):
627
639
  # pretend that this is an operator as the tokenizer only allows
628
640
  # one character punctuation.
629
641
  return token('punc', op)
630
- return token("operator", op)
642
+ return token('operator', op)
631
643
 
632
644
  def handle_slash():
633
645
  next()
634
- return read_regexp("") if S.regex_allowed else read_operator("/")
646
+ return read_regexp('') if S.regex_allowed else read_operator('/')
635
647
 
636
648
  def handle_dot():
637
649
  next()
638
- return read_num(".") if is_digit(peek().charCodeAt(0)) else token("punc", ".")
650
+ return read_num('.') if is_digit(peek().charCodeAt(0)) else token('punc', '.')
639
651
 
640
652
  def read_word():
641
653
  word = read_name()
642
- return token("atom", word) if KEYWORDS_ATOM[word] else (token("name", word) if not KEYWORDS[word] else (token("operator", word) if OPERATORS[word] and prevChar() is not "." else token("keyword", word)))
654
+ return token(
655
+ 'atom',
656
+ word
657
+ ) if KEYWORDS_ATOM[word] else (token('name', word) if not KEYWORDS[word] else (token('operator', word) if OPERATORS[word] and prevChar() is not '.' else token('keyword', word)))
643
658
 
644
659
  def next_token():
645
-
646
660
  indent = parse_whitespace()
647
661
  # if indent is 1:
648
662
  # return token("punc", "{")
649
663
  if indent is -1:
650
- return token("punc", "}", False, True)
664
+ return token('punc', '}', False, True)
651
665
 
652
666
  start_token()
653
667
  ch = peek()
654
668
  if not ch:
655
- return token("eof")
669
+ return token('eof')
656
670
 
657
671
  code = ch.charCodeAt(0)
658
672
  tmp_ = code
659
- if tmp_ is 34 or tmp_ is 39: # double-quote (") or single quote (')
660
- return read_string(False)
661
- elif tmp_ is 35: # pound-sign (#)
673
+ if tmp_ is 34 or tmp_ is 39: # double-quote (") or single quote (')
674
+ stok = read_string(False)
675
+ # Look ahead to detect a following f-string (e.g. "a" f"b"), which would
676
+ # otherwise be parsed as a call expression "a"(...) after the tokenizer
677
+ # injects the f-string result as a parenthesised expression.
678
+ j = S.pos
679
+ while j < S.text.length and (S.text.charAt(j) is ' ' or S.text.charAt(j) is '\t'):
680
+ j += 1
681
+ if j < S.text.length and is_identifier_start(S.text.charAt(j).charCodeAt(0)):
682
+ k = j
683
+ while k < S.text.length and is_identifier_char(S.text.charAt(k)):
684
+ k += 1
685
+ potential_mod = S.text.substring(j, k)
686
+ if is_string_modifier(potential_mod) and k < S.text.length and '\'"'.indexOf(S.text.charAt(k)) is not -1:
687
+ mods_check = potential_mod.toLowerCase()
688
+ if mods_check.indexOf('f') is not -1 and mods_check.indexOf('v') is -1:
689
+ return handle_interpolated_string(stok.value, stok, False)
690
+ return stok
691
+ elif tmp_ is 35: # pound-sign (#)
662
692
  if S.pos is 0 and S.text.charAt(1) is '!':
663
- #shebang
693
+ # shebang
664
694
  return read_line_comment(True)
665
695
  regex_allowed = S.regex_allowed
666
696
  S.comments_before.push(read_line_comment())
667
697
  S.regex_allowed = regex_allowed
668
698
  return next_token()
669
- elif tmp_ is 46: # dot (.)
699
+ elif tmp_ is 46: # dot (.)
670
700
  return handle_dot()
671
- elif tmp_ is 47: # slash (/)
701
+ elif tmp_ is 47: # slash (/)
672
702
  return handle_slash()
673
703
 
674
704
  if is_digit(code):
675
705
  return read_num()
676
706
 
677
707
  if PUNC_CHARS[ch]:
678
- return token("punc", next())
708
+ return token('punc', next())
679
709
 
680
710
  if OPERATOR_CHARS[ch]:
681
711
  return read_operator()
682
712
 
683
- if code is 92 and S.text.charAt(S.pos + 1) is "\n":
713
+ if code is 92 and S.text.charAt(S.pos + 1) is '\n':
684
714
  # backslash will consume the newline character that follows
685
715
  next()
686
716
  # backslash
@@ -703,9 +733,19 @@ def tokenizer(raw_text, filename):
703
733
  tok.type = stok.type
704
734
  return tok
705
735
 
706
- parse_error("Unexpected character «" + ch + "»")
736
+ if S.recover_errors:
737
+ # In error-recovery mode (used by the LSP) an unrecognized character
738
+ # must not raise, because re-tokenizing from the same position would
739
+ # loop forever (parse_error is raised before the character is
740
+ # consumed). Record it and skip the single offending character so the
741
+ # tokenizer always makes forward progress.
742
+ S.recovered_errors.push({'message': 'Unexpected character «' + ch + '»',
743
+ 'line': S.tokline, 'col': S.tokcol, 'pos': S.tokpos, 'is_eof': False})
744
+ next()
745
+ return next_token()
746
+ parse_error('Unexpected character «' + ch + '»')
707
747
 
708
- next_token.context = def(nc):
748
+ next_token.context = def (nc):
709
749
  nonlocal S
710
750
  if nc:
711
751
  S = nc