opencode-pine2pyne 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
pine2pyne/lexer.py ADDED
@@ -0,0 +1,546 @@
1
+ """
2
+ Lexer (tokenizer) for Pine Script v6.
3
+
4
+ Handles indentation-significant syntax, multi-line continuations,
5
+ and all Pine Script v6 token types.
6
+ """
7
+ import re
8
+ from typing import List, Optional
9
+ from .tokens import Token, TokenType, KEYWORDS, TYPE_KEYWORDS
10
+
11
+
12
+ class LexerError(Exception):
13
+ """Raised when lexer encounters an invalid token."""
14
+ def __init__(self, message: str, line: int, column: int):
15
+ super().__init__(f"Lexer error at {line}:{column}: {message}")
16
+ self.line = line
17
+ self.column = column
18
+
19
+
20
+ class Lexer:
21
+ """Tokenizes Pine Script v6 source code."""
22
+
23
+ def __init__(self, source: str):
24
+ self.source = source
25
+ self.pos = 0
26
+ self.line = 1
27
+ self.column = 1
28
+ self.tokens: List[Token] = []
29
+ self.indent_stack: List[int] = [0] # Track indentation levels
30
+ self.paren_depth = 0 # Track depth inside (), [], {} for indentation-free zones
31
+
32
+ def current_char(self) -> Optional[str]:
33
+ """Get current character without advancing."""
34
+ if self.pos >= len(self.source):
35
+ return None
36
+ return self.source[self.pos]
37
+
38
+ def peek_char(self, offset: int = 1) -> Optional[str]:
39
+ """Look ahead at character at pos + offset."""
40
+ pos = self.pos + offset
41
+ if pos >= len(self.source):
42
+ return None
43
+ return self.source[pos]
44
+
45
+ def advance(self) -> Optional[str]:
46
+ """Consume and return current character, updating position."""
47
+ if self.pos >= len(self.source):
48
+ return None
49
+ char = self.source[self.pos]
50
+ self.pos += 1
51
+ if char == '\n':
52
+ self.line += 1
53
+ self.column = 1
54
+ else:
55
+ self.column += 1
56
+ return char
57
+
58
+ def skip_whitespace(self, skip_newlines: bool = False) -> None:
59
+ """Skip spaces and tabs (and optionally newlines)."""
60
+ while self.current_char() in (' ', '\t') or (skip_newlines and self.current_char() == '\n'):
61
+ self.advance()
62
+
63
+ def read_line_comment(self) -> Token:
64
+ """Read // style comment until end of line."""
65
+ start_line = self.line
66
+ start_col = self.column
67
+
68
+ # Check for compiler annotations
69
+ if self.peek_char() == '@':
70
+ return self.read_annotation()
71
+
72
+ # Regular comment
73
+ comment = ''
74
+ while self.current_char() and self.current_char() != '\n':
75
+ comment += self.advance()
76
+
77
+ return Token(TokenType.COMMENT, comment.strip(), start_line, start_col, comment)
78
+
79
+ def read_annotation(self) -> Token:
80
+ """Read //@annotation style compiler directive."""
81
+ start_line = self.line
82
+ start_col = self.column
83
+
84
+ # Skip //
85
+ self.advance()
86
+ self.advance()
87
+ # Skip @
88
+ self.advance()
89
+
90
+ # Read annotation name
91
+ annotation = ''
92
+ while self.current_char() and self.current_char() not in (' ', '\t', '\n', '='):
93
+ annotation += self.advance()
94
+
95
+ # Check for version annotation
96
+ if annotation == 'version':
97
+ self.skip_whitespace()
98
+ if self.current_char() == '=':
99
+ self.advance()
100
+ self.skip_whitespace()
101
+ version = ''
102
+ while self.current_char() and self.current_char() != '\n':
103
+ version += self.advance()
104
+ return Token(TokenType.VERSION_ANNOTATION, version.strip(), start_line, start_col)
105
+
106
+ # Read rest of annotation
107
+ value = ''
108
+ while self.current_char() and self.current_char() != '\n':
109
+ value += self.advance()
110
+
111
+ return Token(TokenType.COMPILER_ANNOTATION, f"{annotation}{value}".strip(), start_line, start_col)
112
+
113
+ def read_string(self, quote: str) -> Token:
114
+ """Read string literal (single or double quoted)."""
115
+ start_line = self.line
116
+ start_col = self.column
117
+
118
+ # Skip opening quote
119
+ self.advance()
120
+
121
+ string = ''
122
+ while self.current_char() and self.current_char() != quote:
123
+ char = self.current_char()
124
+ if char == '\\':
125
+ # Handle escape sequences — unescape to actual characters
126
+ self.advance()
127
+ next_char = self.current_char()
128
+ if next_char == 'n':
129
+ string += '\n'
130
+ self.advance()
131
+ elif next_char == 't':
132
+ string += '\t'
133
+ self.advance()
134
+ elif next_char == 'r':
135
+ string += '\r'
136
+ self.advance()
137
+ elif next_char in ('"', "'", '\\'):
138
+ string += next_char
139
+ self.advance()
140
+ else:
141
+ string += char
142
+ elif char == '\n':
143
+ # Pine Script allows multiline strings with regular quotes
144
+ string += self.advance()
145
+ else:
146
+ string += self.advance()
147
+
148
+ if not self.current_char():
149
+ raise LexerError(f"Unterminated string literal", start_line, start_col)
150
+
151
+ # Skip closing quote
152
+ self.advance()
153
+
154
+ # Mark if double quoted (for preserving quote style in output)
155
+ is_double_quoted = (quote == '"')
156
+ return Token(TokenType.STRING_LITERAL, string, start_line, start_col,
157
+ f"{quote}{string}{quote}", is_double_quoted=is_double_quoted)
158
+
159
+ def read_number(self) -> Token:
160
+ """Read integer or float literal."""
161
+ start_line = self.line
162
+ start_col = self.column
163
+
164
+ number = ''
165
+ has_dot = False
166
+ has_e = False
167
+
168
+ while self.current_char():
169
+ char = self.current_char()
170
+
171
+ if char.isdigit():
172
+ number += self.advance()
173
+ elif char == '.' and not has_dot and not has_e:
174
+ next_ch = self.peek_char()
175
+ if next_ch and next_ch.isdigit():
176
+ # 100.5 — normal decimal
177
+ has_dot = True
178
+ number += self.advance()
179
+ elif next_ch is None or not (next_ch.isalpha() or next_ch == '_' or next_ch == '.'):
180
+ # 100. — trailing dot float (not member access like 100.toString)
181
+ has_dot = True
182
+ number += self.advance()
183
+ else:
184
+ break
185
+ elif char in ('e', 'E') and not has_e:
186
+ has_e = True
187
+ number += self.advance()
188
+ # Handle optional +/- after e
189
+ if self.current_char() in ('+', '-'):
190
+ number += self.advance()
191
+ else:
192
+ break
193
+
194
+ if has_dot or has_e:
195
+ return Token(TokenType.FLOAT_LITERAL, float(number), start_line, start_col, number)
196
+ else:
197
+ return Token(TokenType.INT_LITERAL, int(number), start_line, start_col, number)
198
+
199
+ def read_color_literal(self) -> Token:
200
+ """Read #RRGGBB or #RRGGBBAA color literal."""
201
+ start_line = self.line
202
+ start_col = self.column
203
+
204
+ color = self.advance() # #
205
+
206
+ # Read hex digits
207
+ while self.current_char() and self.current_char() in '0123456789abcdefABCDEF':
208
+ color += self.advance()
209
+
210
+ # Validate length (6 or 8 hex digits after #)
211
+ if len(color) not in (7, 9):
212
+ raise LexerError(f"Invalid color literal: {color}", start_line, start_col)
213
+
214
+ return Token(TokenType.COLOR_LITERAL, color, start_line, start_col, color)
215
+
216
+ def read_identifier(self) -> Token:
217
+ """Read identifier or keyword."""
218
+ start_line = self.line
219
+ start_col = self.column
220
+
221
+ identifier = ''
222
+ while self.current_char() and (self.current_char().isalnum() or self.current_char() in ('_', '.')):
223
+ identifier += self.advance()
224
+
225
+ # Check if it's a keyword
226
+ if identifier in KEYWORDS:
227
+ token_type = KEYWORDS[identifier]
228
+ return Token(token_type, identifier, start_line, start_col, identifier)
229
+
230
+ # Check if it's a type keyword
231
+ if identifier in TYPE_KEYWORDS:
232
+ return Token(TokenType.TYPE_IDENTIFIER, identifier, start_line, start_col, identifier)
233
+
234
+ return Token(TokenType.IDENTIFIER, identifier, start_line, start_col, identifier)
235
+
236
+ def _is_continuation_line(self) -> bool:
237
+ """Check if this line is a continuation of the previous expression.
238
+
239
+ Pine Script treats indented lines as continuations when:
240
+ 1. The line starts with a binary operator keyword (or/and)
241
+ 2. The previous line ended with an operator that needs a right operand (?, :, and, or, +, -, etc.)
242
+ """
243
+ # Check if line starts with a continuation keyword
244
+ if self.current_char() and self.current_char().isalpha():
245
+ word = ''
246
+ temp_pos = self.pos
247
+ while temp_pos < len(self.source) and self.source[temp_pos].isalpha():
248
+ word += self.source[temp_pos]
249
+ temp_pos += 1
250
+ if word in ('or', 'and'):
251
+ return True
252
+
253
+ # Check if previous line ended with an operator that needs a right operand
254
+ # These operators always indicate the expression continues on the next line
255
+ CONTINUATION_END_TOKENS = {
256
+ TokenType.TERNARY, # ?
257
+ TokenType.COLON, # : (ternary false branch)
258
+ TokenType.AND, # and
259
+ TokenType.OR, # or
260
+ TokenType.PLUS, # +
261
+ TokenType.MINUS, # -
262
+ TokenType.MULT, # *
263
+ TokenType.DIV, # /
264
+ TokenType.MOD, # %
265
+ TokenType.COMMA, # ,
266
+ # Assignment/comparison operators can never end a complete expression,
267
+ # so a line ending with one always continues onto the next indented
268
+ # line, e.g. `x =` \n ` "a" + "b"` (line-wrapped RHS after `=`).
269
+ TokenType.ASSIGN, # =
270
+ TokenType.REASSIGN, # :=
271
+ TokenType.PLUS_ASSIGN, # +=
272
+ TokenType.MINUS_ASSIGN, # -=
273
+ TokenType.MULT_ASSIGN, # *=
274
+ TokenType.DIV_ASSIGN, # /=
275
+ TokenType.MOD_ASSIGN, # %=
276
+ TokenType.GT, # >
277
+ TokenType.LT, # <
278
+ TokenType.GTE, # >=
279
+ TokenType.LTE, # <=
280
+ TokenType.EQ, # ==
281
+ TokenType.NEQ, # !=
282
+ }
283
+ if self.tokens:
284
+ last_token = self.tokens[-1]
285
+ if last_token.type in CONTINUATION_END_TOKENS:
286
+ return True
287
+
288
+ return False
289
+
290
+ def handle_indentation(self, indent_level: int) -> List[Token]:
291
+ """
292
+ Generate INDENT/DEDENT tokens based on indentation change.
293
+
294
+ Uses lenient matching: if exact level not in stack, finds closest level.
295
+ This handles real-world Pine Script files with varied indentation styles.
296
+ """
297
+ tokens = []
298
+ current_level = self.indent_stack[-1]
299
+
300
+ if indent_level > current_level:
301
+ # Increase indentation
302
+ self.indent_stack.append(indent_level)
303
+ tokens.append(Token(TokenType.INDENT, None, self.line, self.column))
304
+ elif indent_level < current_level:
305
+ # Decrease indentation (may generate multiple DEDENT tokens)
306
+ while self.indent_stack and self.indent_stack[-1] > indent_level:
307
+ self.indent_stack.pop()
308
+ tokens.append(Token(TokenType.DEDENT, None, self.line, self.column))
309
+
310
+ # Lenient matching: if exact level not found, find closest match
311
+ if not self.indent_stack:
312
+ # Should never happen, but reset to base
313
+ self.indent_stack.append(0)
314
+ elif self.indent_stack[-1] != indent_level:
315
+ # Find closest level at or below target
316
+ # This handles cases where continuation lines use different indentation
317
+ closest_level = max(level for level in self.indent_stack if level <= indent_level)
318
+
319
+ # If we found a closer match, pop until we reach it
320
+ while self.indent_stack and self.indent_stack[-1] > closest_level:
321
+ self.indent_stack.pop()
322
+ tokens.append(Token(TokenType.DEDENT, None, self.line, self.column))
323
+
324
+ # If we need to add the new level
325
+ if closest_level < indent_level:
326
+ self.indent_stack.append(indent_level)
327
+ tokens.append(Token(TokenType.INDENT, None, self.line, self.column))
328
+
329
+ return tokens
330
+
331
+ def tokenize(self) -> List[Token]:
332
+ """Tokenize the entire source code."""
333
+ self.tokens = []
334
+ at_line_start = True
335
+
336
+ while self.pos < len(self.source):
337
+ char = self.current_char()
338
+
339
+ # Handle newlines and indentation
340
+ if char == '\n':
341
+ # Skip empty lines and lines with only comments
342
+ saved_pos = self.pos
343
+ saved_line = self.line
344
+ saved_col = self.column
345
+
346
+ self.advance() # Skip \n
347
+
348
+ # Count indentation on next line
349
+ indent_level = 0
350
+ while self.current_char() in (' ', '\t'):
351
+ if self.current_char() == ' ':
352
+ indent_level += 1
353
+ else: # tab
354
+ indent_level += 4
355
+ self.advance()
356
+
357
+ # Check if line is empty or comment-only
358
+ if self.current_char() in ('\n', None) or (self.current_char() == '/' and self.peek_char() == '/'):
359
+ # Empty line or comment line - skip indentation tracking
360
+ if self.current_char() is None:
361
+ break
362
+ continue
363
+
364
+ # Check for line continuation: increased indent + starts with binary operator
365
+ # In Pine Script, indented lines starting with or/and continue the previous expression
366
+ if (self.paren_depth == 0
367
+ and indent_level > self.indent_stack[-1]
368
+ and self._is_continuation_line()):
369
+ # Suppress NEWLINE and INDENT — treat as seamless continuation
370
+ at_line_start = False
371
+ continue
372
+
373
+ # Inside parentheses/brackets/braces: suppress NEWLINE and indentation
374
+ # Multi-line expressions inside delimiters are seamless
375
+ if self.paren_depth > 0:
376
+ at_line_start = False
377
+ continue
378
+
379
+ # Emit NEWLINE token for previous line
380
+ self.tokens.append(Token(TokenType.NEWLINE, '\n', saved_line, saved_col))
381
+
382
+ # Handle indentation changes
383
+ if self.paren_depth == 0:
384
+ indent_tokens = self.handle_indentation(indent_level)
385
+ self.tokens.extend(indent_tokens)
386
+
387
+ at_line_start = False
388
+ continue
389
+
390
+ # Skip inline whitespace
391
+ if char in (' ', '\t'):
392
+ self.skip_whitespace()
393
+ continue
394
+
395
+ # Comments and annotations
396
+ if char == '/' and self.peek_char() == '/':
397
+ token = self.read_line_comment()
398
+ # Don't add comments to token stream (or add them with a flag)
399
+ # For now, skip comments except annotations
400
+ if token.type in (TokenType.VERSION_ANNOTATION, TokenType.COMPILER_ANNOTATION):
401
+ self.tokens.append(token)
402
+ continue
403
+
404
+ # String literals
405
+ if char in ('"', "'"):
406
+ self.tokens.append(self.read_string(char))
407
+ continue
408
+
409
+ # Color literals
410
+ if char == '#':
411
+ self.tokens.append(self.read_color_literal())
412
+ continue
413
+
414
+ # Numbers
415
+ if char.isdigit():
416
+ self.tokens.append(self.read_number())
417
+ continue
418
+
419
+ # Identifiers and keywords
420
+ if char.isalpha() or char == '_':
421
+ self.tokens.append(self.read_identifier())
422
+ continue
423
+
424
+ # Operators and delimiters
425
+ start_line = self.line
426
+ start_col = self.column
427
+
428
+ # Two-character operators
429
+ if char == ':' and self.peek_char() == '=':
430
+ self.advance()
431
+ self.advance()
432
+ self.tokens.append(Token(TokenType.REASSIGN, ':=', start_line, start_col, ':='))
433
+ continue
434
+
435
+ # Compound assignment operators
436
+ if char == '+' and self.peek_char() == '=':
437
+ self.advance()
438
+ self.advance()
439
+ self.tokens.append(Token(TokenType.PLUS_ASSIGN, '+=', start_line, start_col, '+='))
440
+ continue
441
+
442
+ if char == '-' and self.peek_char() == '=':
443
+ self.advance()
444
+ self.advance()
445
+ self.tokens.append(Token(TokenType.MINUS_ASSIGN, '-=', start_line, start_col, '-='))
446
+ continue
447
+
448
+ if char == '*' and self.peek_char() == '=':
449
+ self.advance()
450
+ self.advance()
451
+ self.tokens.append(Token(TokenType.MULT_ASSIGN, '*=', start_line, start_col, '*='))
452
+ continue
453
+
454
+ if char == '/' and self.peek_char() == '=':
455
+ self.advance()
456
+ self.advance()
457
+ self.tokens.append(Token(TokenType.DIV_ASSIGN, '/=', start_line, start_col, '/='))
458
+ continue
459
+
460
+ if char == '%' and self.peek_char() == '=':
461
+ self.advance()
462
+ self.advance()
463
+ self.tokens.append(Token(TokenType.MOD_ASSIGN, '%=', start_line, start_col, '%='))
464
+ continue
465
+
466
+ if char == '=' and self.peek_char() == '>':
467
+ self.advance()
468
+ self.advance()
469
+ self.tokens.append(Token(TokenType.ARROW, '=>', start_line, start_col, '=>'))
470
+ continue
471
+
472
+ if char == '=' and self.peek_char() == '=':
473
+ self.advance()
474
+ self.advance()
475
+ self.tokens.append(Token(TokenType.EQ, '==', start_line, start_col, '=='))
476
+ continue
477
+
478
+ if char == '!' and self.peek_char() == '=':
479
+ self.advance()
480
+ self.advance()
481
+ self.tokens.append(Token(TokenType.NEQ, '!=', start_line, start_col, '!='))
482
+ continue
483
+
484
+ if char == '>' and self.peek_char() == '=':
485
+ self.advance()
486
+ self.advance()
487
+ self.tokens.append(Token(TokenType.GTE, '>=', start_line, start_col, '>='))
488
+ continue
489
+
490
+ if char == '<' and self.peek_char() == '=':
491
+ self.advance()
492
+ self.advance()
493
+ self.tokens.append(Token(TokenType.LTE, '<=', start_line, start_col, '<='))
494
+ continue
495
+
496
+ # Single-character operators and delimiters
497
+ single_char_tokens = {
498
+ '=': TokenType.ASSIGN,
499
+ '+': TokenType.PLUS,
500
+ '-': TokenType.MINUS,
501
+ '*': TokenType.MULT,
502
+ '/': TokenType.DIV,
503
+ '%': TokenType.MOD,
504
+ '>': TokenType.GT,
505
+ '<': TokenType.LT,
506
+ '?': TokenType.TERNARY,
507
+ ':': TokenType.COLON,
508
+ '.': TokenType.DOT,
509
+ '(': TokenType.LPAREN,
510
+ ')': TokenType.RPAREN,
511
+ '[': TokenType.LBRACKET,
512
+ ']': TokenType.RBRACKET,
513
+ ',': TokenType.COMMA,
514
+ }
515
+
516
+ if char in single_char_tokens:
517
+ token_type = single_char_tokens[char]
518
+
519
+ # Track delimiter depth for indentation-free zones
520
+ if char in ('(', '[', '{'):
521
+ self.paren_depth += 1
522
+ elif char in (')', ']', '}'):
523
+ self.paren_depth = max(0, self.paren_depth - 1)
524
+
525
+ self.advance()
526
+ self.tokens.append(Token(token_type, char, start_line, start_col, char))
527
+ continue
528
+
529
+ # Unknown character
530
+ raise LexerError(f"Unexpected character: {char!r}", self.line, self.column)
531
+
532
+ # Emit final DEDENT tokens to return to base indentation
533
+ while len(self.indent_stack) > 1:
534
+ self.indent_stack.pop()
535
+ self.tokens.append(Token(TokenType.DEDENT, None, self.line, self.column))
536
+
537
+ # Add EOF token
538
+ self.tokens.append(Token(TokenType.EOF, None, self.line, self.column))
539
+
540
+ return self.tokens
541
+
542
+
543
+ def tokenize(source: str) -> List[Token]:
544
+ """Convenience function to tokenize Pine Script source."""
545
+ lexer = Lexer(source)
546
+ return lexer.tokenize()