pycli-dsl 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
pycli/lexer.py ADDED
@@ -0,0 +1,321 @@
1
+ """Lexer / Scanner for pycli (.spy) source code."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from dataclasses import dataclass
6
+ from enum import Enum, auto
7
+ from typing import List
8
+
9
+
10
+ class TokenType(Enum):
11
+ PYTHON_CODE = auto()
12
+ COMMAND_EXPR = auto()
13
+
14
+
15
+ @dataclass
16
+ class Token:
17
+ type: TokenType
18
+ value: str
19
+ line: int
20
+ column: int
21
+ strict: bool = False # True if followed by '!'
22
+ safe: bool = False # True if followed by '?'
23
+ background: bool = False # True if followed by '&'
24
+
25
+
26
+
27
+ class LexerError(Exception):
28
+ """Raised when lexical scanning fails."""
29
+
30
+ def __init__(self, message: str, line: int, column: int) -> None:
31
+ self.line = line
32
+ self.column = column
33
+ super().__init__(f"Line {line}, Column {column}: {message}")
34
+
35
+
36
+ import os
37
+
38
+ MAX_SOURCE_SIZE_BYTES = 10 * 1024 * 1024
39
+ STRING_PREFIXES = frozenset(["r", "b", "f", "u", "rb", "br", "fr", "rf"])
40
+
41
+
42
+ def get_max_source_size() -> int:
43
+ """Return maximum allowed source size in bytes from env or default."""
44
+ return int(os.environ.get("PYCLI_MAX_SOURCE_SIZE", MAX_SOURCE_SIZE_BYTES))
45
+
46
+
47
+ class Lexer:
48
+ """Scans .spy source code into alternating Python code and CommandExpression tokens."""
49
+
50
+ def __init__(self, source: str) -> None:
51
+ max_size = get_max_source_size()
52
+ source_size = len(source.encode("utf-8"))
53
+ if source_size > max_size:
54
+ raise ValueError(
55
+ f"Source size ({source_size} bytes) exceeds maximum allowed size "
56
+ f"({max_size} bytes)"
57
+ )
58
+ self.source = source
59
+ self.pos = 0
60
+ self.line = 1
61
+ self.col = 1
62
+ self.length = len(source)
63
+
64
+ def _peek_string_prefix(self) -> str:
65
+ """Return any string prefix at current position (e.g. 'r', 'rb'), or ''."""
66
+ if self.pos > 0 and (self.source[self.pos - 1].isalnum() or self.source[self.pos - 1] == "_"):
67
+ return ""
68
+ for length in (2, 1):
69
+ if self.pos + length <= self.length:
70
+ candidate = self.source[self.pos : self.pos + length].lower()
71
+ if candidate in STRING_PREFIXES:
72
+ next_ch = self._peek(length)
73
+ if next_ch in ('"', "'"):
74
+ return candidate
75
+ return ""
76
+
77
+ def _peek(self, offset: int = 0) -> str:
78
+ idx = self.pos + offset
79
+ if idx < self.length:
80
+ return self.source[idx]
81
+ return ""
82
+
83
+ def _advance(self) -> str:
84
+ ch = self.source[self.pos]
85
+ self.pos += 1
86
+ if ch == "\n":
87
+ self.line += 1
88
+ self.col = 1
89
+ else:
90
+ self.col += 1
91
+ return ch
92
+
93
+ def tokenize(self) -> List[Token]:
94
+ tokens: List[Token] = []
95
+ py_buf: list[str] = []
96
+ py_start_line = self.line
97
+ py_start_col = self.col
98
+
99
+ def flush_py():
100
+ if py_buf:
101
+ tokens.append(
102
+ Token(
103
+ type=TokenType.PYTHON_CODE,
104
+ value="".join(py_buf),
105
+ line=py_start_line,
106
+ column=py_start_col,
107
+ )
108
+ )
109
+ py_buf.clear()
110
+
111
+ while self.pos < self.length:
112
+ ch = self._peek()
113
+
114
+ # 1. Check for comments
115
+ if ch == "#":
116
+ py_buf.append(self._advance())
117
+ while self.pos < self.length and self._peek() != "\n":
118
+ py_buf.append(self._advance())
119
+ continue
120
+
121
+ # 2. Check for string prefixes (r, b, f, u, rb, br, fr, rf) before quotes
122
+ prefix = self._peek_string_prefix()
123
+ if prefix:
124
+ for _ in range(len(prefix)):
125
+ py_buf.append(self._advance())
126
+ ch = self._peek()
127
+
128
+ # 3. Check for string literals in Python
129
+ if ch in ("'", '"'):
130
+ # Check for triple-quote
131
+ quote_char = ch
132
+ is_triple = self.source[self.pos : self.pos + 3] == quote_char * 3
133
+ if is_triple:
134
+ py_buf.append(self._advance())
135
+ py_buf.append(self._advance())
136
+ py_buf.append(self._advance())
137
+ delim = quote_char * 3
138
+ while self.pos < self.length:
139
+ if self.source[self.pos : self.pos + 3] == delim:
140
+ py_buf.append(self._advance())
141
+ py_buf.append(self._advance())
142
+ py_buf.append(self._advance())
143
+ break
144
+ if self._peek() == "\\":
145
+ py_buf.append(self._advance())
146
+ if self.pos < self.length:
147
+ py_buf.append(self._advance())
148
+ else:
149
+ py_buf.append(self._advance())
150
+ else:
151
+ # Single quoted string
152
+ py_buf.append(self._advance())
153
+ while self.pos < self.length:
154
+ c = self._peek()
155
+ if c == quote_char:
156
+ py_buf.append(self._advance())
157
+ break
158
+ if c == "\\":
159
+ py_buf.append(self._advance())
160
+ if self.pos < self.length:
161
+ py_buf.append(self._advance())
162
+ elif c == "\n":
163
+ # Unterminated string on this line in Python
164
+ py_buf.append(self._advance())
165
+ break
166
+ else:
167
+ py_buf.append(self._advance())
168
+ continue
169
+
170
+ # 4. Check for $( command expression (strictly outside strings)
171
+ if ch == "$" and self._peek(1) == "(":
172
+ flush_py()
173
+ cmd_token = self._scan_command_expr()
174
+ tokens.append(cmd_token)
175
+ py_start_line = self.line
176
+ py_start_col = self.col
177
+ continue
178
+
179
+ # Regular Python character
180
+ py_buf.append(self._advance())
181
+
182
+ flush_py()
183
+ return tokens
184
+
185
+ def _scan_command_expr(self) -> Token:
186
+ cmd_line = self.line
187
+ cmd_col = self.col
188
+
189
+ # Consume '$('
190
+ self._advance() # $
191
+ self._advance() # (
192
+
193
+ content_buf: list[str] = []
194
+ depth = 1 # Parentheses depth inside $( ... )
195
+
196
+ while self.pos < self.length and depth > 0:
197
+ c = self._peek()
198
+
199
+ # Handle quotes inside command
200
+ if c in ("'", '"'):
201
+ quote = c
202
+ content_buf.append(self._advance())
203
+ while self.pos < self.length:
204
+ qc = self._peek()
205
+ if qc == quote:
206
+ content_buf.append(self._advance())
207
+ break
208
+ if qc == "\\":
209
+ content_buf.append(self._advance())
210
+ if self.pos < self.length:
211
+ content_buf.append(self._advance())
212
+ else:
213
+ content_buf.append(self._advance())
214
+ continue
215
+
216
+ # Handle braces { ... } (interpolation / splat)
217
+ if c == "{":
218
+ brace_depth = 1
219
+ content_buf.append(self._advance())
220
+ while self.pos < self.length and brace_depth > 0:
221
+ bc = self._peek()
222
+ if bc in ("'", '"'):
223
+ b_quote = bc
224
+ content_buf.append(self._advance())
225
+ while self.pos < self.length:
226
+ b_qc = self._peek()
227
+ if b_qc == b_quote:
228
+ content_buf.append(self._advance())
229
+ break
230
+ if b_qc == "\\":
231
+ content_buf.append(self._advance())
232
+ if self.pos < self.length:
233
+ content_buf.append(self._advance())
234
+ else:
235
+ content_buf.append(self._advance())
236
+ continue
237
+
238
+ if bc == "{":
239
+ brace_depth += 1
240
+ content_buf.append(self._advance())
241
+ elif bc == "}":
242
+ brace_depth -= 1
243
+ content_buf.append(self._advance())
244
+ else:
245
+ content_buf.append(self._advance())
246
+ continue
247
+
248
+ # Nested $(...) subcommands or regular ( ... )
249
+ if c == "(":
250
+ depth += 1
251
+ content_buf.append(self._advance())
252
+ elif c == ")":
253
+ depth -= 1
254
+ if depth == 0:
255
+ # Matched outer closing ')'
256
+ self._advance()
257
+ break
258
+ else:
259
+ content_buf.append(self._advance())
260
+ else:
261
+ content_buf.append(self._advance())
262
+
263
+ if depth != 0:
264
+ raise LexerError("Unclosed command expression '$('", cmd_line, cmd_col)
265
+
266
+ # Check for modifier suffix after ')':
267
+ # '!' -> strict mode
268
+ # '?' -> safe mode (suppress errors)
269
+ # '&' -> background non-blocking execution
270
+ strict = False
271
+ safe = False
272
+ background = False
273
+
274
+ idx = self.pos
275
+ while idx < self.length and self.source[idx] in (" ", "\t"):
276
+ idx += 1
277
+
278
+ if idx < self.length:
279
+ nxt = self.source[idx]
280
+ is_adjacent = (idx == self.pos)
281
+ if nxt == "!":
282
+ while self.pos < idx:
283
+ self._advance()
284
+ self._advance()
285
+ strict = True
286
+ elif nxt == "?":
287
+ while self.pos < idx:
288
+ self._advance()
289
+ self._advance()
290
+ safe = True
291
+ elif nxt == "&" and (idx + 1 >= self.length or self.source[idx + 1] != "&"):
292
+ is_bg = is_adjacent
293
+ if not is_bg:
294
+ rem_idx = idx + 1
295
+ while rem_idx < self.length and self.source[rem_idx] in (" ", "\t"):
296
+ rem_idx += 1
297
+ if (
298
+ rem_idx >= self.length
299
+ or self.source[rem_idx] in ("\n", "\r", "#", ";", "]", ")", "}", ",")
300
+ or self.source[rem_idx : rem_idx + 4] == "for "
301
+ or self.source[rem_idx : rem_idx + 3] == "for"
302
+ ):
303
+ is_bg = True
304
+ if is_bg:
305
+ while self.pos < idx:
306
+ self._advance()
307
+ self._advance()
308
+ background = True
309
+
310
+
311
+ raw_cmd = "".join(content_buf)
312
+ return Token(
313
+ type=TokenType.COMMAND_EXPR,
314
+ value=raw_cmd,
315
+ line=cmd_line,
316
+ column=cmd_col,
317
+ strict=strict,
318
+ safe=safe,
319
+ background=background,
320
+ )
321
+