lectes 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- lectes/__init__.py +4 -0
- lectes/config/__init__.py +0 -0
- lectes/config/models.py +32 -0
- lectes/engine/__init__.py +0 -0
- lectes/engine/errors.py +13 -0
- lectes/engine/models.py +97 -0
- lectes/errors.py +4 -0
- lectes/scanner/__init__.py +0 -0
- lectes/scanner/logger.py +68 -0
- lectes/scanner/models.py +19 -0
- lectes/scanner/scanner.py +150 -0
- lectes/tests/__init__.py +0 -0
- lectes/tests/unit/__init__.py +0 -0
- lectes/tests/unit/engine/__init__.py +0 -0
- lectes/tests/unit/engine/test_models.py +150 -0
- lectes/tests/unit/scanner/__init__.py +0 -0
- lectes/tests/unit/scanner/test_scanner.py +285 -0
- lectes-0.1.0.dist-info/METADATA +11 -0
- lectes-0.1.0.dist-info/RECORD +22 -0
- lectes-0.1.0.dist-info/WHEEL +5 -0
- lectes-0.1.0.dist-info/licenses/LICENSE +21 -0
- lectes-0.1.0.dist-info/top_level.txt +1 -0
lectes/__init__.py
ADDED
|
File without changes
|
lectes/config/models.py
ADDED
|
@@ -0,0 +1,32 @@
|
|
|
1
|
+
from dataclasses import dataclass
|
|
2
|
+
|
|
3
|
+
from lectes.engine.models import Regex
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
@dataclass
|
|
7
|
+
class Rule:
|
|
8
|
+
"""
|
|
9
|
+
Represents a scanner configuration rule.
|
|
10
|
+
|
|
11
|
+
Is actually a proxy for a regular expression.
|
|
12
|
+
|
|
13
|
+
## Example
|
|
14
|
+
|
|
15
|
+
```python
|
|
16
|
+
from lectes import Regex
|
|
17
|
+
|
|
18
|
+
Rule(name="INT_LITERAL", regex=Regex("0|([-]?[1-9]+[0-9]*))
|
|
19
|
+
```
|
|
20
|
+
"""
|
|
21
|
+
|
|
22
|
+
name: str
|
|
23
|
+
regex: Regex
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
@dataclass
|
|
27
|
+
class Configuration:
|
|
28
|
+
"""
|
|
29
|
+
Represents the configured rules of the scanner.
|
|
30
|
+
"""
|
|
31
|
+
|
|
32
|
+
rules: list[Rule]
|
|
File without changes
|
lectes/engine/errors.py
ADDED
|
@@ -0,0 +1,13 @@
|
|
|
1
|
+
from lectes.errors import LectesError
|
|
2
|
+
|
|
3
|
+
|
|
4
|
+
class EngineError(LectesError):
|
|
5
|
+
"""
|
|
6
|
+
Base class for all errors occuring in the regex engine.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
class RegexPatternError(EngineError):
|
|
11
|
+
"""
|
|
12
|
+
The regular expression is invalid or not recognized.
|
|
13
|
+
"""
|
lectes/engine/models.py
ADDED
|
@@ -0,0 +1,97 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
from dataclasses import dataclass
|
|
3
|
+
import re
|
|
4
|
+
|
|
5
|
+
from lectes.engine.errors import RegexPatternError
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
@dataclass
|
|
9
|
+
class Match:
|
|
10
|
+
"""
|
|
11
|
+
Represents a regular expression match.
|
|
12
|
+
|
|
13
|
+
Objects hold the matched string, the unmatched part of the string they were
|
|
14
|
+
checked against and the actual Regex object.
|
|
15
|
+
"""
|
|
16
|
+
|
|
17
|
+
unmatched: str | None
|
|
18
|
+
string: str
|
|
19
|
+
re: "Regex"
|
|
20
|
+
|
|
21
|
+
@classmethod
|
|
22
|
+
def from_re(cls, match: re.Match) -> "Match":
|
|
23
|
+
"""
|
|
24
|
+
Create an object from Python's standard library re.Match object.
|
|
25
|
+
"""
|
|
26
|
+
|
|
27
|
+
index = match.span()
|
|
28
|
+
matched = match.string[index[0] : index[1]]
|
|
29
|
+
unmatched = match.string.replace(matched, "")
|
|
30
|
+
unmatched = unmatched if len(unmatched) >= 1 else None
|
|
31
|
+
return Match(unmatched=unmatched, string=matched, re=Regex.from_re(match.re))
|
|
32
|
+
|
|
33
|
+
def __bool__(self) -> bool:
|
|
34
|
+
return True
|
|
35
|
+
|
|
36
|
+
def __len__(self) -> int:
|
|
37
|
+
return len(self.string)
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
class Regex:
|
|
41
|
+
"""
|
|
42
|
+
Represents a regular expression.
|
|
43
|
+
|
|
44
|
+
Can be initialized with any string, but upon invoking a method, the validity
|
|
45
|
+
of the string will be checked and may raise an error.
|
|
46
|
+
"""
|
|
47
|
+
|
|
48
|
+
def __init__(self, pattern: str) -> None:
|
|
49
|
+
self._pattern = pattern
|
|
50
|
+
self._re_pattern = None
|
|
51
|
+
|
|
52
|
+
@classmethod
|
|
53
|
+
def from_re(cls, pattern: re.Pattern) -> "Regex":
|
|
54
|
+
"""
|
|
55
|
+
Create an object from Python's standard library re.Pattern object.
|
|
56
|
+
"""
|
|
57
|
+
return Regex(pattern.pattern)
|
|
58
|
+
|
|
59
|
+
def fullmatch(self, string: str) -> Match | None:
|
|
60
|
+
"""
|
|
61
|
+
If the whole string matches the regular expression, return a Match.
|
|
62
|
+
Return None if the string does not match the regular expression.
|
|
63
|
+
"""
|
|
64
|
+
match = self._compiled_pattern().fullmatch(string)
|
|
65
|
+
|
|
66
|
+
if match is None:
|
|
67
|
+
return None
|
|
68
|
+
|
|
69
|
+
return Match.from_re(match)
|
|
70
|
+
|
|
71
|
+
def search(self, string: str) -> Match | None:
|
|
72
|
+
"""
|
|
73
|
+
Scan through string looking for the first location where the regular
|
|
74
|
+
expression pattern produces a match, and return a Match. Return None if
|
|
75
|
+
no position in the string matches the regular expression.
|
|
76
|
+
"""
|
|
77
|
+
match = self._compiled_pattern().search(string)
|
|
78
|
+
|
|
79
|
+
if match is None:
|
|
80
|
+
return None
|
|
81
|
+
|
|
82
|
+
return Match.from_re(match)
|
|
83
|
+
|
|
84
|
+
def _compiled_pattern(self) -> re.Pattern:
|
|
85
|
+
if self._re_pattern is None:
|
|
86
|
+
self._re_pattern = self._compile_pattern()
|
|
87
|
+
|
|
88
|
+
return self._re_pattern
|
|
89
|
+
|
|
90
|
+
def _compile_pattern(self) -> re.Pattern:
|
|
91
|
+
try:
|
|
92
|
+
return re.compile(self._pattern)
|
|
93
|
+
except re.PatternError as e:
|
|
94
|
+
raise RegexPatternError(str(e)) from None
|
|
95
|
+
|
|
96
|
+
def __repr__(self) -> str:
|
|
97
|
+
return f"<Regex: {self._compiled_pattern().pattern}>"
|
lectes/errors.py
ADDED
|
File without changes
|
lectes/scanner/logger.py
ADDED
|
@@ -0,0 +1,68 @@
|
|
|
1
|
+
import logging
|
|
2
|
+
from enum import Enum
|
|
3
|
+
|
|
4
|
+
|
|
5
|
+
class LogLevel(Enum):
|
|
6
|
+
"""
|
|
7
|
+
Log level options for the scanner's logger.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
DEBUG = "DEBUG"
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
class Logger:
|
|
14
|
+
"""
|
|
15
|
+
Logger class for the scanner.
|
|
16
|
+
"""
|
|
17
|
+
|
|
18
|
+
def __init__(self) -> None:
|
|
19
|
+
self._logger = None
|
|
20
|
+
self._handler = None
|
|
21
|
+
self._formatter = None
|
|
22
|
+
|
|
23
|
+
def set_level(self, level: LogLevel) -> None:
|
|
24
|
+
self.logger().setLevel(self._map_level(level))
|
|
25
|
+
self.handler().setLevel(self._map_level(level))
|
|
26
|
+
|
|
27
|
+
def debug(self, message: str) -> None:
|
|
28
|
+
self.logger().debug(message)
|
|
29
|
+
|
|
30
|
+
def logger(self) -> logging.Logger:
|
|
31
|
+
if self._logger is None:
|
|
32
|
+
self._logger = self._build_logger()
|
|
33
|
+
|
|
34
|
+
return self._logger
|
|
35
|
+
|
|
36
|
+
def handler(self) -> logging.StreamHandler:
|
|
37
|
+
if self._handler is None:
|
|
38
|
+
self._handler = self._build_handler()
|
|
39
|
+
|
|
40
|
+
return self._handler
|
|
41
|
+
|
|
42
|
+
def formatter(self) -> logging.Formatter:
|
|
43
|
+
if self._formatter is None:
|
|
44
|
+
self._formatter = self._build_formatter()
|
|
45
|
+
|
|
46
|
+
return self._formatter
|
|
47
|
+
|
|
48
|
+
def _build_logger(self) -> logging.Logger:
|
|
49
|
+
logger = logging.getLogger(__name__)
|
|
50
|
+
logger.addHandler(self.handler())
|
|
51
|
+
|
|
52
|
+
return logger
|
|
53
|
+
|
|
54
|
+
def _build_handler(self) -> logging.StreamHandler:
|
|
55
|
+
handler = logging.StreamHandler()
|
|
56
|
+
handler.setFormatter(self.formatter())
|
|
57
|
+
|
|
58
|
+
return handler
|
|
59
|
+
|
|
60
|
+
def _build_formatter(self) -> logging.Formatter:
|
|
61
|
+
return logging.Formatter("%(levelname)s: %(message)s")
|
|
62
|
+
|
|
63
|
+
def _map_level(self, level: LogLevel) -> int:
|
|
64
|
+
match level:
|
|
65
|
+
case LogLevel.DEBUG:
|
|
66
|
+
return logging.DEBUG
|
|
67
|
+
case _:
|
|
68
|
+
return logging.INFO
|
lectes/scanner/models.py
ADDED
|
@@ -0,0 +1,19 @@
|
|
|
1
|
+
from dataclasses import dataclass
|
|
2
|
+
|
|
3
|
+
from lectes.config.models import Rule
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
@dataclass
|
|
7
|
+
class Token:
|
|
8
|
+
"""
|
|
9
|
+
Represents a token returned by the scanner.
|
|
10
|
+
|
|
11
|
+
The scanned token is related to a configuration rule and a string literal.
|
|
12
|
+
"""
|
|
13
|
+
|
|
14
|
+
rule: Rule
|
|
15
|
+
literal: str
|
|
16
|
+
|
|
17
|
+
@property
|
|
18
|
+
def name(self) -> str:
|
|
19
|
+
return self.rule.name
|
|
@@ -0,0 +1,150 @@
|
|
|
1
|
+
from typing import Callable, Generator
|
|
2
|
+
|
|
3
|
+
from lectes.config.models import Configuration, Rule
|
|
4
|
+
from lectes.engine.models import Match
|
|
5
|
+
from lectes.scanner.models import Token
|
|
6
|
+
from lectes.scanner.logger import Logger, LogLevel
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
class Scanner:
|
|
10
|
+
"""
|
|
11
|
+
Scans a given text and returns tokens based on the provided configuration.
|
|
12
|
+
|
|
13
|
+
## Example
|
|
14
|
+
|
|
15
|
+
```python
|
|
16
|
+
from lectes import Rule, Configuration, Regex, Scanner
|
|
17
|
+
|
|
18
|
+
config = Configuration(
|
|
19
|
+
[
|
|
20
|
+
Rule(name="FOR", regex=Regex("for")),
|
|
21
|
+
Rule(name="INT", regex=Regex("[1-9]+")),
|
|
22
|
+
Rule(name="ID", regex=Regex("[a-zA-Z][a-zA-Z0-9]*")),
|
|
23
|
+
Rule(name="WHITESPACE", regex=Regex("( )")),
|
|
24
|
+
]
|
|
25
|
+
)
|
|
26
|
+
|
|
27
|
+
scanner = Scanner(config)
|
|
28
|
+
program = "somevar in othervar for 9 let"
|
|
29
|
+
|
|
30
|
+
for token in scanner.scan(program):
|
|
31
|
+
print(token)
|
|
32
|
+
```
|
|
33
|
+
"""
|
|
34
|
+
|
|
35
|
+
def __init__(self, configuration: Configuration, debug: bool = False) -> None:
|
|
36
|
+
self.configuration = configuration
|
|
37
|
+
self.set_text("")
|
|
38
|
+
self._unmatched_handler = self._handle_unmatched
|
|
39
|
+
self._debug = debug
|
|
40
|
+
self._logger = None
|
|
41
|
+
self._match = None
|
|
42
|
+
self._matched_rule = None
|
|
43
|
+
|
|
44
|
+
def scan(self, text: str) -> Generator[Token]:
|
|
45
|
+
"""
|
|
46
|
+
Scan the given text and yield tokens as they are recognized.
|
|
47
|
+
"""
|
|
48
|
+
if len(text) == 0:
|
|
49
|
+
return
|
|
50
|
+
|
|
51
|
+
self.set_text(text)
|
|
52
|
+
|
|
53
|
+
for character in text:
|
|
54
|
+
self.logger().debug(f"character: '{character}'")
|
|
55
|
+
|
|
56
|
+
current_string = self.current_string()
|
|
57
|
+
self.logger().debug(f"current_string: '{current_string}'")
|
|
58
|
+
|
|
59
|
+
lookahead_string = self.lookahead_string()
|
|
60
|
+
self.logger().debug(f"lookahead_string: '{lookahead_string}'")
|
|
61
|
+
|
|
62
|
+
for rule in self.configuration.rules:
|
|
63
|
+
if not self._is_last_char() and rule.regex.fullmatch(lookahead_string):
|
|
64
|
+
self.logger().debug(
|
|
65
|
+
f"rule {rule.name} fullmatched lookahead_string: '{lookahead_string}'"
|
|
66
|
+
)
|
|
67
|
+
break
|
|
68
|
+
|
|
69
|
+
if match := rule.regex.search(current_string):
|
|
70
|
+
self.logger().debug(
|
|
71
|
+
f"rule {rule.name} matched current_string: '{current_string}'"
|
|
72
|
+
)
|
|
73
|
+
self._update_matched_state(rule, match)
|
|
74
|
+
|
|
75
|
+
if self._match is not None:
|
|
76
|
+
if self._match.unmatched is not None:
|
|
77
|
+
self._unmatched_handler(self._match.unmatched)
|
|
78
|
+
|
|
79
|
+
if self._matched_rule is not None:
|
|
80
|
+
yield Token(rule=self._matched_rule, literal=self._match.string)
|
|
81
|
+
|
|
82
|
+
self.last_position = self.current_position
|
|
83
|
+
|
|
84
|
+
self.current_position += 1
|
|
85
|
+
self._reset_matched_state()
|
|
86
|
+
|
|
87
|
+
def set_unmatched_handler(self, handler: Callable[[str], None]) -> None:
|
|
88
|
+
"""
|
|
89
|
+
Set the given function as the handler that executes when a string is not
|
|
90
|
+
matched to a configured rule.
|
|
91
|
+
|
|
92
|
+
The handler receives the string as argument and does returns None.
|
|
93
|
+
"""
|
|
94
|
+
self._unmatched_handler = handler
|
|
95
|
+
|
|
96
|
+
def set_text(self, text: str) -> None:
|
|
97
|
+
"""
|
|
98
|
+
Set the text to scan.
|
|
99
|
+
"""
|
|
100
|
+
self.text = text
|
|
101
|
+
self.current_position = 1
|
|
102
|
+
self.last_position = 0
|
|
103
|
+
|
|
104
|
+
def current_string(self) -> str:
|
|
105
|
+
"""
|
|
106
|
+
Return the string that the scanner is currently reading; that is,
|
|
107
|
+
the characters from the last matched string up to the character that
|
|
108
|
+
the scanner is currently reading.
|
|
109
|
+
"""
|
|
110
|
+
return self.text[self.last_position : self.current_position]
|
|
111
|
+
|
|
112
|
+
def lookahead_string(self) -> str:
|
|
113
|
+
"""
|
|
114
|
+
Return the string that the scanner is currently reading plus one character.
|
|
115
|
+
"""
|
|
116
|
+
return self.text[self.last_position : self.current_position + 1]
|
|
117
|
+
|
|
118
|
+
def logger(self) -> Logger:
|
|
119
|
+
"""
|
|
120
|
+
Return the scanner's logger instance.
|
|
121
|
+
"""
|
|
122
|
+
if self._logger is None:
|
|
123
|
+
self._logger = self._build_logger()
|
|
124
|
+
|
|
125
|
+
return self._logger
|
|
126
|
+
|
|
127
|
+
def _build_logger(self) -> Logger:
|
|
128
|
+
logger = Logger()
|
|
129
|
+
|
|
130
|
+
if self._debug:
|
|
131
|
+
logger.set_level(LogLevel.DEBUG)
|
|
132
|
+
|
|
133
|
+
return logger
|
|
134
|
+
|
|
135
|
+
def _is_last_char(self) -> bool:
|
|
136
|
+
return self.current_position == len(self.text)
|
|
137
|
+
|
|
138
|
+
def _update_matched_state(self, rule: Rule, match: Match) -> None:
|
|
139
|
+
if self._match is None or len(match) > len(self._match):
|
|
140
|
+
self.logger().debug(f"updating match from {self._match} to {match.string}")
|
|
141
|
+
self._matched_rule = rule
|
|
142
|
+
self._match = match
|
|
143
|
+
|
|
144
|
+
def _reset_matched_state(self) -> None:
|
|
145
|
+
self._match = None
|
|
146
|
+
self._matched_rule = None
|
|
147
|
+
|
|
148
|
+
@staticmethod
|
|
149
|
+
def _handle_unmatched(unmatched: str) -> None:
|
|
150
|
+
print(f"unmatched: {unmatched}")
|
lectes/tests/__init__.py
ADDED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
@@ -0,0 +1,150 @@
|
|
|
1
|
+
from re import Pattern
|
|
2
|
+
from unittest_extensions import args, TestCase
|
|
3
|
+
|
|
4
|
+
from lectes.engine.models import Regex
|
|
5
|
+
from lectes.engine.errors import RegexPatternError
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
class TestRegexSearch(TestCase):
|
|
9
|
+
def subject(self, pattern, string=""):
|
|
10
|
+
return Regex(pattern).search(string)
|
|
11
|
+
|
|
12
|
+
def assert_unmatched(self, unmatched):
|
|
13
|
+
self.assertResultTrue()
|
|
14
|
+
self.assertSequenceEqual(self.cachedResult().unmatched, unmatched)
|
|
15
|
+
|
|
16
|
+
def assert_pattern_error(self):
|
|
17
|
+
with self.assertRaises(RegexPatternError):
|
|
18
|
+
self.result()
|
|
19
|
+
|
|
20
|
+
@args(pattern="(")
|
|
21
|
+
def test_unclosed_parentheses(self):
|
|
22
|
+
self.assert_pattern_error()
|
|
23
|
+
|
|
24
|
+
@args(pattern="[a-Z]")
|
|
25
|
+
def test_all_letters_class(self):
|
|
26
|
+
self.assert_pattern_error()
|
|
27
|
+
|
|
28
|
+
@args(pattern="a", string="a b a")
|
|
29
|
+
def test_simple_char_two_occurences_in_string(self):
|
|
30
|
+
self.assert_unmatched(" b ")
|
|
31
|
+
|
|
32
|
+
@args(pattern="a", string="bcd")
|
|
33
|
+
def test_simple_char_absence_in_string(self):
|
|
34
|
+
self.assertResultFalse()
|
|
35
|
+
|
|
36
|
+
@args(pattern="for", string="for i in a:")
|
|
37
|
+
def test_word_occurence(self):
|
|
38
|
+
self.assertResultTrue()
|
|
39
|
+
|
|
40
|
+
@args(pattern="for", string="a: int = 2")
|
|
41
|
+
def test_word_absence(self):
|
|
42
|
+
self.assertResultFalse()
|
|
43
|
+
|
|
44
|
+
@args(pattern="for", string="forum")
|
|
45
|
+
def test_word_occurence_in_substring(self):
|
|
46
|
+
self.assertResultTrue()
|
|
47
|
+
|
|
48
|
+
@args(pattern="ab|cd", string="abd")
|
|
49
|
+
def test_character_alternation(self):
|
|
50
|
+
self.assert_unmatched("d")
|
|
51
|
+
|
|
52
|
+
@args(pattern="ab|cd", string="aca")
|
|
53
|
+
def test_alternation_does_not_match(self):
|
|
54
|
+
self.assertResultFalse()
|
|
55
|
+
|
|
56
|
+
@args(pattern="ab|cd", string="cd")
|
|
57
|
+
def test_character_alternation_presedence(self):
|
|
58
|
+
self.assertResultTrue()
|
|
59
|
+
|
|
60
|
+
@args(pattern="ab|abc|abcd", string="abcde")
|
|
61
|
+
def test_multiple_alternation(self):
|
|
62
|
+
self.assert_unmatched("cde")
|
|
63
|
+
|
|
64
|
+
@args(pattern="this|that", string="this or that")
|
|
65
|
+
def test_alternation_first_match(self):
|
|
66
|
+
self.assert_unmatched(" or that")
|
|
67
|
+
|
|
68
|
+
@args(pattern="a?b", string="bcd")
|
|
69
|
+
def test_zero_or_one_zero(self):
|
|
70
|
+
self.assert_unmatched("cd")
|
|
71
|
+
|
|
72
|
+
@args(pattern="a?b", string="abcd")
|
|
73
|
+
def test_zero_or_one_one(self):
|
|
74
|
+
self.assert_unmatched("cd")
|
|
75
|
+
|
|
76
|
+
@args(pattern="a?b", string="cd")
|
|
77
|
+
def test_zero_or_one_absence(self):
|
|
78
|
+
self.assertResultFalse()
|
|
79
|
+
|
|
80
|
+
@args(pattern="a*b", string="bcd")
|
|
81
|
+
def test_zero_or_more_zero(self):
|
|
82
|
+
self.assert_unmatched("cd")
|
|
83
|
+
|
|
84
|
+
@args(pattern="a*b", string="abcd")
|
|
85
|
+
def test_zero_or_more_one(self):
|
|
86
|
+
self.assert_unmatched("cd")
|
|
87
|
+
|
|
88
|
+
@args(pattern="a*b", string="aabcd")
|
|
89
|
+
def test_zero_or_more_two(self):
|
|
90
|
+
self.assert_unmatched("cd")
|
|
91
|
+
|
|
92
|
+
@args(pattern="a*b", string="cd")
|
|
93
|
+
def test_zero_or_more_absence(self):
|
|
94
|
+
self.assertResultFalse()
|
|
95
|
+
|
|
96
|
+
@args(pattern="a+b", string="bcd")
|
|
97
|
+
def test_one_or_more_zero(self):
|
|
98
|
+
self.assertResultFalse()
|
|
99
|
+
|
|
100
|
+
@args(pattern="a+b", string="abcd")
|
|
101
|
+
def test_one_or_more_one(self):
|
|
102
|
+
self.assert_unmatched("cd")
|
|
103
|
+
|
|
104
|
+
@args(pattern="a+b", string="aabcd")
|
|
105
|
+
def test_one_or_more_two(self):
|
|
106
|
+
self.assert_unmatched("cd")
|
|
107
|
+
|
|
108
|
+
@args(pattern="[a-z]", string="k")
|
|
109
|
+
def test_small_letter_class(self):
|
|
110
|
+
self.assertResultTrue()
|
|
111
|
+
|
|
112
|
+
@args(pattern="[a-z]", string="K")
|
|
113
|
+
def test_small_letter_class_with_capital(self):
|
|
114
|
+
self.assertResultFalse()
|
|
115
|
+
|
|
116
|
+
@args(pattern="[A-Z]", string="U")
|
|
117
|
+
def test_capital_letter_class(self):
|
|
118
|
+
self.assertResultTrue()
|
|
119
|
+
|
|
120
|
+
@args(pattern="[a-d]", string="e")
|
|
121
|
+
def test_letter_class_out_of_range(self):
|
|
122
|
+
self.assertResultFalse()
|
|
123
|
+
|
|
124
|
+
@args(pattern="[a-z]", string=" ")
|
|
125
|
+
def test_letter_class_with_whitespace(self):
|
|
126
|
+
self.assertResultFalse()
|
|
127
|
+
|
|
128
|
+
@args(pattern="[A-Z]", string="y")
|
|
129
|
+
def test_capital_letter_class_with_downcase(self):
|
|
130
|
+
self.assertResultFalse()
|
|
131
|
+
|
|
132
|
+
@args(pattern="[0-4]", string="3")
|
|
133
|
+
def test_numeric_class_in_range(self):
|
|
134
|
+
self.assertResultTrue()
|
|
135
|
+
|
|
136
|
+
@args(pattern="[2-7]", string="1")
|
|
137
|
+
def test_numeric_class_out_of_range(self):
|
|
138
|
+
self.assertResultFalse()
|
|
139
|
+
|
|
140
|
+
@args(pattern="[a-zA-Z]", string="b")
|
|
141
|
+
def test_all_letters_compound_class(self):
|
|
142
|
+
self.assertResultTrue()
|
|
143
|
+
|
|
144
|
+
@args(pattern="[a-zA-Z]", string="G")
|
|
145
|
+
def test_all_letters_compound_class_capital(self):
|
|
146
|
+
self.assertResultTrue()
|
|
147
|
+
|
|
148
|
+
@args(pattern="[a-zA-Z2-6]", string="5")
|
|
149
|
+
def test_letter_and_numeric_class(self):
|
|
150
|
+
self.assertResultTrue()
|
|
File without changes
|
|
@@ -0,0 +1,285 @@
|
|
|
1
|
+
from unittest_extensions import args, TestCase
|
|
2
|
+
|
|
3
|
+
from lectes.config.models import Rule, Configuration
|
|
4
|
+
from lectes.engine.models import Regex
|
|
5
|
+
from lectes.scanner.scanner import Scanner
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
def rule(name, regex):
|
|
9
|
+
return Rule(name=name, regex=Regex(regex))
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
class TestScanner(TestCase):
|
|
13
|
+
def subject(self, text):
|
|
14
|
+
return list(self.scanner().scan(text))
|
|
15
|
+
|
|
16
|
+
def scanner(self):
|
|
17
|
+
return Scanner(self.configuration())
|
|
18
|
+
|
|
19
|
+
def configuration(self):
|
|
20
|
+
return Configuration(self.rules())
|
|
21
|
+
|
|
22
|
+
def assert_tokens(self, *tokens):
|
|
23
|
+
self.assertSequenceEqual(list(map(lambda t: t.name, self.result())), tokens)
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
class TestScannerSimpleGrammar(TestScanner):
|
|
27
|
+
def rules(self):
|
|
28
|
+
return [
|
|
29
|
+
rule("FOR", "for"),
|
|
30
|
+
rule("INT_LITERAL", "[0-9]+"),
|
|
31
|
+
rule("INT", "int"),
|
|
32
|
+
rule("ID", "[a-zA-Z][a-zA-Z0-9]*"),
|
|
33
|
+
rule("WHITESPACE", "( )"),
|
|
34
|
+
]
|
|
35
|
+
|
|
36
|
+
@args("somevar in othervar for 9 let")
|
|
37
|
+
def test_scan(self):
|
|
38
|
+
self.assert_tokens(
|
|
39
|
+
"ID",
|
|
40
|
+
"WHITESPACE",
|
|
41
|
+
"ID",
|
|
42
|
+
"WHITESPACE",
|
|
43
|
+
"ID",
|
|
44
|
+
"WHITESPACE",
|
|
45
|
+
"FOR",
|
|
46
|
+
"WHITESPACE",
|
|
47
|
+
"INT_LITERAL",
|
|
48
|
+
"WHITESPACE",
|
|
49
|
+
"ID",
|
|
50
|
+
)
|
|
51
|
+
|
|
52
|
+
@args("int myint ")
|
|
53
|
+
def test_longer_match(self):
|
|
54
|
+
self.assert_tokens("INT", "WHITESPACE", "ID", "WHITESPACE")
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
class TestScannerClassicGrammar(TestScanner):
|
|
58
|
+
def rules(self):
|
|
59
|
+
return [
|
|
60
|
+
rule("OPER", "oper"),
|
|
61
|
+
rule("EXEMP", "exemp"),
|
|
62
|
+
rule("INT", "int"),
|
|
63
|
+
rule("DUPL", "dupl"),
|
|
64
|
+
rule("STR", "str"),
|
|
65
|
+
rule("ANEF", "anef"),
|
|
66
|
+
rule("EGO", "ego"),
|
|
67
|
+
rule("INITUS", "initus"),
|
|
68
|
+
rule("EXODUS", "exodus"),
|
|
69
|
+
rule("ID", "(_|[a-zA-Z])(_|[a-zA-Z0-9])*"),
|
|
70
|
+
rule("INT_LITERAL", "[-]?[0-9]+"),
|
|
71
|
+
rule("DOUBLE_LITERAL", "[-+]?[0-9]+\.?[0-9]*"),
|
|
72
|
+
rule("PLUS", "\+"),
|
|
73
|
+
rule("MINUS", "-"),
|
|
74
|
+
rule("DIV", "/"),
|
|
75
|
+
rule("MUL", "\*"),
|
|
76
|
+
rule("LPAREN", "\("),
|
|
77
|
+
rule("RPAREN", "\)"),
|
|
78
|
+
rule("LBRACK", "{"),
|
|
79
|
+
rule("RBRACK", "}"),
|
|
80
|
+
rule("COLON", ":"),
|
|
81
|
+
rule("SEMICOLON", ";"),
|
|
82
|
+
rule("DOT", "[.]"),
|
|
83
|
+
rule("COMMA", "[,]"),
|
|
84
|
+
rule("EQUAL", "="),
|
|
85
|
+
rule("WHITESPACE", "( )"),
|
|
86
|
+
rule("NEWLINE", "\\n"),
|
|
87
|
+
rule("TAB", "\\t"),
|
|
88
|
+
rule("BEGIN_COMMENT", "/\*"),
|
|
89
|
+
rule("END_COMMENT", "\*/"),
|
|
90
|
+
]
|
|
91
|
+
|
|
92
|
+
@args(
|
|
93
|
+
"""oper: int simple_function(int myint) {
|
|
94
|
+
exodus myint;
|
|
95
|
+
}
|
|
96
|
+
|
|
97
|
+
oper: int initus() {
|
|
98
|
+
exodus simple_function(myint=0)
|
|
99
|
+
}"""
|
|
100
|
+
)
|
|
101
|
+
def test_scan_simple_program(self):
|
|
102
|
+
self.assert_tokens(
|
|
103
|
+
"OPER",
|
|
104
|
+
"COLON",
|
|
105
|
+
"WHITESPACE",
|
|
106
|
+
"INT",
|
|
107
|
+
"WHITESPACE",
|
|
108
|
+
"ID",
|
|
109
|
+
"LPAREN",
|
|
110
|
+
"INT",
|
|
111
|
+
"WHITESPACE",
|
|
112
|
+
"ID",
|
|
113
|
+
"RPAREN",
|
|
114
|
+
"WHITESPACE",
|
|
115
|
+
"LBRACK",
|
|
116
|
+
"NEWLINE",
|
|
117
|
+
"WHITESPACE",
|
|
118
|
+
"WHITESPACE",
|
|
119
|
+
"WHITESPACE",
|
|
120
|
+
"WHITESPACE",
|
|
121
|
+
"EXODUS",
|
|
122
|
+
"WHITESPACE",
|
|
123
|
+
"ID",
|
|
124
|
+
"SEMICOLON",
|
|
125
|
+
"NEWLINE",
|
|
126
|
+
"RBRACK",
|
|
127
|
+
"NEWLINE",
|
|
128
|
+
"NEWLINE",
|
|
129
|
+
"OPER",
|
|
130
|
+
"COLON",
|
|
131
|
+
"WHITESPACE",
|
|
132
|
+
"INT",
|
|
133
|
+
"WHITESPACE",
|
|
134
|
+
"INITUS",
|
|
135
|
+
"LPAREN",
|
|
136
|
+
"RPAREN",
|
|
137
|
+
"WHITESPACE",
|
|
138
|
+
"LBRACK",
|
|
139
|
+
"NEWLINE",
|
|
140
|
+
"WHITESPACE",
|
|
141
|
+
"WHITESPACE",
|
|
142
|
+
"WHITESPACE",
|
|
143
|
+
"WHITESPACE",
|
|
144
|
+
"EXODUS",
|
|
145
|
+
"WHITESPACE",
|
|
146
|
+
"ID",
|
|
147
|
+
"LPAREN",
|
|
148
|
+
"ID",
|
|
149
|
+
"EQUAL",
|
|
150
|
+
"INT_LITERAL",
|
|
151
|
+
"RPAREN",
|
|
152
|
+
"NEWLINE",
|
|
153
|
+
"RBRACK",
|
|
154
|
+
)
|
|
155
|
+
|
|
156
|
+
@args(
|
|
157
|
+
"""oper: int add(int a, int b) {
|
|
158
|
+
exodus a + b;
|
|
159
|
+
}
|
|
160
|
+
|
|
161
|
+
oper: int mul(int a, int b) {
|
|
162
|
+
exodus a * b;
|
|
163
|
+
}
|
|
164
|
+
|
|
165
|
+
oper: int initus() {
|
|
166
|
+
exodus add(a=2, b=3) + mul(a=4, b=5)
|
|
167
|
+
}"""
|
|
168
|
+
)
|
|
169
|
+
def test_scan_program(self):
|
|
170
|
+
self.assert_tokens(
|
|
171
|
+
"OPER",
|
|
172
|
+
"COLON",
|
|
173
|
+
"WHITESPACE",
|
|
174
|
+
"INT",
|
|
175
|
+
"WHITESPACE",
|
|
176
|
+
"ID",
|
|
177
|
+
"LPAREN",
|
|
178
|
+
"INT",
|
|
179
|
+
"WHITESPACE",
|
|
180
|
+
"ID",
|
|
181
|
+
"COMMA",
|
|
182
|
+
"WHITESPACE",
|
|
183
|
+
"INT",
|
|
184
|
+
"WHITESPACE",
|
|
185
|
+
"ID",
|
|
186
|
+
"RPAREN",
|
|
187
|
+
"WHITESPACE",
|
|
188
|
+
"LBRACK",
|
|
189
|
+
"NEWLINE",
|
|
190
|
+
"WHITESPACE",
|
|
191
|
+
"WHITESPACE",
|
|
192
|
+
"WHITESPACE",
|
|
193
|
+
"WHITESPACE",
|
|
194
|
+
"EXODUS",
|
|
195
|
+
"WHITESPACE",
|
|
196
|
+
"ID",
|
|
197
|
+
"WHITESPACE",
|
|
198
|
+
"PLUS",
|
|
199
|
+
"WHITESPACE",
|
|
200
|
+
"ID",
|
|
201
|
+
"SEMICOLON",
|
|
202
|
+
"NEWLINE",
|
|
203
|
+
"RBRACK",
|
|
204
|
+
"NEWLINE",
|
|
205
|
+
"NEWLINE",
|
|
206
|
+
"OPER",
|
|
207
|
+
"COLON",
|
|
208
|
+
"WHITESPACE",
|
|
209
|
+
"INT",
|
|
210
|
+
"WHITESPACE",
|
|
211
|
+
"ID",
|
|
212
|
+
"LPAREN",
|
|
213
|
+
"INT",
|
|
214
|
+
"WHITESPACE",
|
|
215
|
+
"ID",
|
|
216
|
+
"COMMA",
|
|
217
|
+
"WHITESPACE",
|
|
218
|
+
"INT",
|
|
219
|
+
"WHITESPACE",
|
|
220
|
+
"ID",
|
|
221
|
+
"RPAREN",
|
|
222
|
+
"WHITESPACE",
|
|
223
|
+
"LBRACK",
|
|
224
|
+
"NEWLINE",
|
|
225
|
+
"WHITESPACE",
|
|
226
|
+
"WHITESPACE",
|
|
227
|
+
"WHITESPACE",
|
|
228
|
+
"WHITESPACE",
|
|
229
|
+
"EXODUS",
|
|
230
|
+
"WHITESPACE",
|
|
231
|
+
"ID",
|
|
232
|
+
"WHITESPACE",
|
|
233
|
+
"MUL",
|
|
234
|
+
"WHITESPACE",
|
|
235
|
+
"ID",
|
|
236
|
+
"SEMICOLON",
|
|
237
|
+
"NEWLINE",
|
|
238
|
+
"RBRACK",
|
|
239
|
+
"NEWLINE",
|
|
240
|
+
"NEWLINE",
|
|
241
|
+
"OPER",
|
|
242
|
+
"COLON",
|
|
243
|
+
"WHITESPACE",
|
|
244
|
+
"INT",
|
|
245
|
+
"WHITESPACE",
|
|
246
|
+
"INITUS",
|
|
247
|
+
"LPAREN",
|
|
248
|
+
"RPAREN",
|
|
249
|
+
"WHITESPACE",
|
|
250
|
+
"LBRACK",
|
|
251
|
+
"NEWLINE",
|
|
252
|
+
"WHITESPACE",
|
|
253
|
+
"WHITESPACE",
|
|
254
|
+
"WHITESPACE",
|
|
255
|
+
"WHITESPACE",
|
|
256
|
+
"EXODUS",
|
|
257
|
+
"WHITESPACE",
|
|
258
|
+
"ID",
|
|
259
|
+
"LPAREN",
|
|
260
|
+
"ID",
|
|
261
|
+
"EQUAL",
|
|
262
|
+
"INT_LITERAL",
|
|
263
|
+
"COMMA",
|
|
264
|
+
"WHITESPACE",
|
|
265
|
+
"ID",
|
|
266
|
+
"EQUAL",
|
|
267
|
+
"INT_LITERAL",
|
|
268
|
+
"RPAREN",
|
|
269
|
+
"WHITESPACE",
|
|
270
|
+
"PLUS",
|
|
271
|
+
"WHITESPACE",
|
|
272
|
+
"ID",
|
|
273
|
+
"LPAREN",
|
|
274
|
+
"ID",
|
|
275
|
+
"EQUAL",
|
|
276
|
+
"INT_LITERAL",
|
|
277
|
+
"COMMA",
|
|
278
|
+
"WHITESPACE",
|
|
279
|
+
"ID",
|
|
280
|
+
"EQUAL",
|
|
281
|
+
"INT_LITERAL",
|
|
282
|
+
"RPAREN",
|
|
283
|
+
"NEWLINE",
|
|
284
|
+
"RBRACK",
|
|
285
|
+
)
|
|
@@ -0,0 +1,22 @@
|
|
|
1
|
+
lectes/__init__.py,sha256=OaG0A_KPRZK6-N_JS78Je-czxZLGYF9-TTQ3vI1D4jI,211
|
|
2
|
+
lectes/errors.py,sha256=7DCHpxPzfiKoGo6704tc3RCdv5rKR5pMtfB7JPxtrtM,98
|
|
3
|
+
lectes/config/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
4
|
+
lectes/config/models.py,sha256=h2rtMYxKbhy-OZ3qLcOBQxpSKcCVkEiMffnJNaQYd5w,501
|
|
5
|
+
lectes/engine/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
6
|
+
lectes/engine/errors.py,sha256=1cEuWVNDG5dH0no2EkN30mXfnxifLBk_mu6tHz4BDfA,261
|
|
7
|
+
lectes/engine/models.py,sha256=BH0AsAxyE70dBVG70QhcMAnrSIQ7cVNL_-5Vith-LAE,2780
|
|
8
|
+
lectes/scanner/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
9
|
+
lectes/scanner/logger.py,sha256=-IAi7-rXIkgwVuv2XhHVQTUCesD_gmdb_HIE44lM-UI,1701
|
|
10
|
+
lectes/scanner/models.py,sha256=MF5AdqqAZgrA8dZ6KcryuN2HXOW8Fg9-llhIE-oeJDA,348
|
|
11
|
+
lectes/scanner/scanner.py,sha256=x6WXWheJtRzh9hqGVSPv_0Cvlv-Su_A3bcaKS2BQ-PE,4832
|
|
12
|
+
lectes/tests/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
13
|
+
lectes/tests/unit/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
14
|
+
lectes/tests/unit/engine/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
15
|
+
lectes/tests/unit/engine/test_models.py,sha256=9WS9aWUVX7vgWC5IBnaTVbtlXM-yn2EjaNmIjjD8vOQ,4452
|
|
16
|
+
lectes/tests/unit/scanner/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
17
|
+
lectes/tests/unit/scanner/test_scanner.py,sha256=pBHaesSDr1EH5QN9Xgh_JK7bxXaivoOBEnwJ9gBprLc,6834
|
|
18
|
+
lectes-0.1.0.dist-info/licenses/LICENSE,sha256=7Kt0nEcUePIgPCSj1z3bzBQD7MXF6b2QHspqHCvjWQg,1076
|
|
19
|
+
lectes-0.1.0.dist-info/METADATA,sha256=rI5z6KGnmB36bOHHAEW233_D67eZzG6s1STLvu12DLo,226
|
|
20
|
+
lectes-0.1.0.dist-info/WHEEL,sha256=_zCd3N1l69ArxyTb8rzEoP9TpbYXkqRFSNOD5OuxnTs,91
|
|
21
|
+
lectes-0.1.0.dist-info/top_level.txt,sha256=i3GoBfdsTrXfoaYeaoRHoREwX2KtpFvREBQ6KkYmjUM,7
|
|
22
|
+
lectes-0.1.0.dist-info/RECORD,,
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2025 Maximos Nikiforakis
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
lectes
|