latexwalker 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- latexwalker/__init__.py +72 -0
- latexwalker/_version.py +24 -0
- latexwalker/errors.py +24 -0
- latexwalker/lexer.py +129 -0
- latexwalker/nodes.py +123 -0
- latexwalker/normalize.py +46 -0
- latexwalker/parser.py +288 -0
- latexwalker/resources/rules.yaml +49 -0
- latexwalker/rewrite.py +196 -0
- latexwalker/rules.py +54 -0
- latexwalker/source.py +138 -0
- latexwalker/tokens.py +41 -0
- latexwalker/traverse.py +38 -0
- latexwalker/zones.py +78 -0
- latexwalker-0.1.0.dist-info/METADATA +66 -0
- latexwalker-0.1.0.dist-info/RECORD +18 -0
- latexwalker-0.1.0.dist-info/WHEEL +4 -0
- latexwalker-0.1.0.dist-info/licenses/LICENSE +7 -0
latexwalker/__init__.py
ADDED
|
@@ -0,0 +1,72 @@
|
|
|
1
|
+
"""Public surface of latexwalker.
|
|
2
|
+
|
|
3
|
+
Symbols re-exported here form the supported API; deeper modules remain
|
|
4
|
+
importable for advanced use but may change without notice.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from ._version import __version__
|
|
8
|
+
from .errors import LatexWalkerError, ParseError, RulesError
|
|
9
|
+
from .lexer import tokenize
|
|
10
|
+
from .nodes import (
|
|
11
|
+
BraceGroup,
|
|
12
|
+
BracketGroup,
|
|
13
|
+
Comment,
|
|
14
|
+
Content,
|
|
15
|
+
Document,
|
|
16
|
+
Environment,
|
|
17
|
+
Macro,
|
|
18
|
+
MacroArgument,
|
|
19
|
+
Math,
|
|
20
|
+
MathMode,
|
|
21
|
+
Node,
|
|
22
|
+
ParBreak,
|
|
23
|
+
SpecialChar,
|
|
24
|
+
Text,
|
|
25
|
+
Whitespace,
|
|
26
|
+
)
|
|
27
|
+
from .normalize import normalize
|
|
28
|
+
from .parser import parse_document
|
|
29
|
+
from .rewrite import CompiledRuleSet
|
|
30
|
+
from .rules import Rule, RuleScope, RuleSet, decode_rules, load_rules
|
|
31
|
+
from .source import SourceMap, SourcePoint, SourceSpan, covering_span
|
|
32
|
+
from .tokens import Token, TokenType
|
|
33
|
+
from .traverse import iter_children, walk
|
|
34
|
+
|
|
35
|
+
__all__ = [
|
|
36
|
+
"BraceGroup",
|
|
37
|
+
"BracketGroup",
|
|
38
|
+
"Comment",
|
|
39
|
+
"CompiledRuleSet",
|
|
40
|
+
"Content",
|
|
41
|
+
"Document",
|
|
42
|
+
"Environment",
|
|
43
|
+
"LatexWalkerError",
|
|
44
|
+
"Macro",
|
|
45
|
+
"MacroArgument",
|
|
46
|
+
"Math",
|
|
47
|
+
"MathMode",
|
|
48
|
+
"Node",
|
|
49
|
+
"ParBreak",
|
|
50
|
+
"ParseError",
|
|
51
|
+
"Rule",
|
|
52
|
+
"RuleScope",
|
|
53
|
+
"RuleSet",
|
|
54
|
+
"RulesError",
|
|
55
|
+
"SourceMap",
|
|
56
|
+
"SourcePoint",
|
|
57
|
+
"SourceSpan",
|
|
58
|
+
"SpecialChar",
|
|
59
|
+
"Text",
|
|
60
|
+
"Token",
|
|
61
|
+
"TokenType",
|
|
62
|
+
"Whitespace",
|
|
63
|
+
"__version__",
|
|
64
|
+
"covering_span",
|
|
65
|
+
"decode_rules",
|
|
66
|
+
"iter_children",
|
|
67
|
+
"load_rules",
|
|
68
|
+
"normalize",
|
|
69
|
+
"parse_document",
|
|
70
|
+
"tokenize",
|
|
71
|
+
"walk",
|
|
72
|
+
]
|
latexwalker/_version.py
ADDED
|
@@ -0,0 +1,24 @@
|
|
|
1
|
+
# file generated by vcs-versioning
|
|
2
|
+
# don't change, don't track in version control
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
__all__ = [
|
|
6
|
+
"__version__",
|
|
7
|
+
"__version_tuple__",
|
|
8
|
+
"version",
|
|
9
|
+
"version_tuple",
|
|
10
|
+
"__commit_id__",
|
|
11
|
+
"commit_id",
|
|
12
|
+
]
|
|
13
|
+
|
|
14
|
+
version: str
|
|
15
|
+
__version__: str
|
|
16
|
+
__version_tuple__: tuple[int | str, ...]
|
|
17
|
+
version_tuple: tuple[int | str, ...]
|
|
18
|
+
commit_id: str | None
|
|
19
|
+
__commit_id__: str | None
|
|
20
|
+
|
|
21
|
+
__version__ = version = '0.1.0'
|
|
22
|
+
__version_tuple__ = version_tuple = (0, 1, 0)
|
|
23
|
+
|
|
24
|
+
__commit_id__ = commit_id = None
|
latexwalker/errors.py
ADDED
|
@@ -0,0 +1,24 @@
|
|
|
1
|
+
"""Typed failure surface for lexing and parsing operations."""
|
|
2
|
+
|
|
3
|
+
from .source import SourceSpan
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
class LatexWalkerError(Exception):
|
|
7
|
+
"""Base class for every error raised by latexwalker."""
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
class ParseError(LatexWalkerError):
|
|
11
|
+
"""Raised when source text cannot be tokenized or parsed."""
|
|
12
|
+
|
|
13
|
+
message: str
|
|
14
|
+
span: SourceSpan | None
|
|
15
|
+
|
|
16
|
+
def __init__(self, message: str, *, span: SourceSpan | None = None) -> None:
|
|
17
|
+
rendered = f"{message} ({span.describe()})" if span is not None else message
|
|
18
|
+
super().__init__(rendered)
|
|
19
|
+
self.message = message
|
|
20
|
+
self.span = span
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
class RulesError(LatexWalkerError):
|
|
24
|
+
"""Raised when a normalization rule set cannot be read or compiled."""
|
latexwalker/lexer.py
ADDED
|
@@ -0,0 +1,129 @@
|
|
|
1
|
+
"""Streaming lexer converting raw LaTeX text into a token sequence."""
|
|
2
|
+
|
|
3
|
+
from collections.abc import Iterator
|
|
4
|
+
|
|
5
|
+
from .errors import ParseError
|
|
6
|
+
from .source import SourceMap, SourceSpan
|
|
7
|
+
from .tokens import Token, TokenType
|
|
8
|
+
|
|
9
|
+
_SINGLE_CHAR_TYPES: dict[str, TokenType] = {
|
|
10
|
+
"{": TokenType.GROUP_OPEN,
|
|
11
|
+
"}": TokenType.GROUP_CLOSE,
|
|
12
|
+
"[": TokenType.BRACKET_OPEN,
|
|
13
|
+
"]": TokenType.BRACKET_CLOSE,
|
|
14
|
+
"&": TokenType.ALIGNMENT,
|
|
15
|
+
"#": TokenType.PARAMETER,
|
|
16
|
+
"_": TokenType.SUBSCRIPT,
|
|
17
|
+
"^": TokenType.SUPERSCRIPT,
|
|
18
|
+
"~": TokenType.TIE,
|
|
19
|
+
}
|
|
20
|
+
|
|
21
|
+
_WORD_STOPPERS: frozenset[str] = frozenset(_SINGLE_CHAR_TYPES) | {"\\", "%", "$"}
|
|
22
|
+
_WHITESPACE: frozenset[str] = frozenset(" \t\n\r")
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def tokenize(source: SourceMap) -> Iterator[Token]:
|
|
26
|
+
"""Yield every token in ``source``, ending with a sentinel end-of-input."""
|
|
27
|
+
text = source.text
|
|
28
|
+
length = len(text)
|
|
29
|
+
index = 0
|
|
30
|
+
while index < length:
|
|
31
|
+
char = text[index]
|
|
32
|
+
if char == "\\":
|
|
33
|
+
token, index = _scan_macro(source, index)
|
|
34
|
+
elif char == "%":
|
|
35
|
+
token, index = _scan_comment(source, index)
|
|
36
|
+
elif char == "$":
|
|
37
|
+
token, index = _scan_math_shift(source, index)
|
|
38
|
+
elif char in _WHITESPACE:
|
|
39
|
+
token, index = _scan_whitespace_run(source, index)
|
|
40
|
+
elif char in _SINGLE_CHAR_TYPES:
|
|
41
|
+
token = _make_token(
|
|
42
|
+
source, index, index + 1, _SINGLE_CHAR_TYPES[char], char
|
|
43
|
+
)
|
|
44
|
+
index += 1
|
|
45
|
+
else:
|
|
46
|
+
token, index = _scan_word(source, index)
|
|
47
|
+
yield token
|
|
48
|
+
yield _make_token(source, length, length, TokenType.END_OF_INPUT, "")
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
def _make_token(
|
|
52
|
+
source: SourceMap, start: int, end: int, token_type: TokenType, value: str
|
|
53
|
+
) -> Token:
|
|
54
|
+
return Token(
|
|
55
|
+
type=token_type,
|
|
56
|
+
value=value,
|
|
57
|
+
span=SourceSpan(start=source.point_at(start), end=source.point_at(end)),
|
|
58
|
+
)
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
def _scan_word(source: SourceMap, start: int) -> tuple[Token, int]:
|
|
62
|
+
text = source.text
|
|
63
|
+
end = start + 1
|
|
64
|
+
while (
|
|
65
|
+
end < len(text)
|
|
66
|
+
and text[end] not in _WORD_STOPPERS
|
|
67
|
+
and text[end] not in _WHITESPACE
|
|
68
|
+
):
|
|
69
|
+
end += 1
|
|
70
|
+
return _make_token(source, start, end, TokenType.WORD, text[start:end]), end
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
def _scan_macro(source: SourceMap, start: int) -> tuple[Token, int]:
|
|
74
|
+
"""Scan one control sequence; the stored value excludes the backslash."""
|
|
75
|
+
text = source.text
|
|
76
|
+
name_start = start + 1
|
|
77
|
+
if name_start >= len(text):
|
|
78
|
+
raise ParseError(
|
|
79
|
+
"dangling backslash at end of input",
|
|
80
|
+
span=SourceSpan.empty_at(source.point_at(start)),
|
|
81
|
+
)
|
|
82
|
+
lead = text[name_start]
|
|
83
|
+
if lead == "\n":
|
|
84
|
+
raise ParseError(
|
|
85
|
+
"backslash cannot be continued by a newline",
|
|
86
|
+
span=SourceSpan.empty_at(source.point_at(start)),
|
|
87
|
+
)
|
|
88
|
+
if lead.isalpha():
|
|
89
|
+
name_end = name_start + 1
|
|
90
|
+
while name_end < len(text) and text[name_end].isalpha():
|
|
91
|
+
name_end += 1
|
|
92
|
+
return _make_token(
|
|
93
|
+
source, start, name_end, TokenType.MACRO, text[name_start:name_end]
|
|
94
|
+
), name_end
|
|
95
|
+
return _make_token(
|
|
96
|
+
source, start, name_start + 1, TokenType.MACRO, lead
|
|
97
|
+
), name_start + 1
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
def _scan_comment(source: SourceMap, start: int) -> tuple[Token, int]:
|
|
101
|
+
text = source.text
|
|
102
|
+
newline = text.find("\n", start)
|
|
103
|
+
end = newline if newline != -1 else len(text)
|
|
104
|
+
return _make_token(source, start, end, TokenType.COMMENT, text[start:end]), end
|
|
105
|
+
|
|
106
|
+
|
|
107
|
+
def _scan_math_shift(source: SourceMap, start: int) -> tuple[Token, int]:
|
|
108
|
+
display = source.text.startswith("$$", start)
|
|
109
|
+
if display:
|
|
110
|
+
return (
|
|
111
|
+
_make_token(source, start, start + 2, TokenType.MATH_SHIFT_DISPLAY, "$$"),
|
|
112
|
+
start + 2,
|
|
113
|
+
)
|
|
114
|
+
return (
|
|
115
|
+
_make_token(source, start, start + 1, TokenType.MATH_SHIFT_INLINE, "$"),
|
|
116
|
+
start + 1,
|
|
117
|
+
)
|
|
118
|
+
|
|
119
|
+
|
|
120
|
+
def _scan_whitespace_run(source: SourceMap, start: int) -> tuple[Token, int]:
|
|
121
|
+
text = source.text
|
|
122
|
+
end = start + 1
|
|
123
|
+
while end < len(text) and text[end] in _WHITESPACE:
|
|
124
|
+
end += 1
|
|
125
|
+
raw = text[start:end]
|
|
126
|
+
token_type = (
|
|
127
|
+
TokenType.PARAGRAPH_BREAK if raw.count("\n") >= 2 else TokenType.WHITESPACE
|
|
128
|
+
)
|
|
129
|
+
return _make_token(source, start, end, token_type, raw), end
|
latexwalker/nodes.py
ADDED
|
@@ -0,0 +1,123 @@
|
|
|
1
|
+
"""Immutable syntax tree schemas produced by the parser.
|
|
2
|
+
|
|
3
|
+
Every node kind is a frozen ``msgspec.Struct`` carrying the source extent it
|
|
4
|
+
was built from, so trees stay lossless and cheap to share across stages.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
from enum import StrEnum
|
|
10
|
+
|
|
11
|
+
import msgspec
|
|
12
|
+
|
|
13
|
+
from .source import SourceMap, SourceSpan
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
class MathMode(StrEnum):
|
|
17
|
+
"""Whether a math region is inline or display."""
|
|
18
|
+
|
|
19
|
+
INLINE = "inline"
|
|
20
|
+
DISPLAY = "display"
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
class Node(msgspec.Struct, frozen=True, kw_only=True):
|
|
24
|
+
"""Base schema shared by every syntax node; anchors it in the source."""
|
|
25
|
+
|
|
26
|
+
span: SourceSpan
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
class Text(Node, frozen=True):
|
|
30
|
+
"""A run of ordinary characters, including interior spacing."""
|
|
31
|
+
|
|
32
|
+
value: str
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
class Whitespace(Node, frozen=True):
|
|
36
|
+
"""A run of spaces or newlines that does not separate paragraphs."""
|
|
37
|
+
|
|
38
|
+
value: str
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
class ParBreak(Node, frozen=True):
|
|
42
|
+
"""One or more blank lines separating paragraphs."""
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
class Comment(Node, frozen=True):
|
|
46
|
+
"""A percent-introduced comment up to but excluding its newline."""
|
|
47
|
+
|
|
48
|
+
text: str
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
class SpecialChar(Node, frozen=True):
|
|
52
|
+
"""A category-code character such as ``&``, ``#``, ``_``, or ``^``."""
|
|
53
|
+
|
|
54
|
+
char: str
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
class BraceGroup(Node, frozen=True):
|
|
58
|
+
"""A brace-delimited ``{body}``."""
|
|
59
|
+
|
|
60
|
+
body: tuple[Content, ...]
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
class BracketGroup(Node, frozen=True):
|
|
64
|
+
"""A bracket-delimited ``[body]``."""
|
|
65
|
+
|
|
66
|
+
body: tuple[Content, ...]
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
type MacroArgument = BraceGroup | BracketGroup
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
class Macro(Node, frozen=True):
|
|
73
|
+
"""A control sequence with its immediately attached group arguments."""
|
|
74
|
+
|
|
75
|
+
name: str
|
|
76
|
+
arguments: tuple[MacroArgument, ...]
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
class Environment(Node, frozen=True):
|
|
80
|
+
"""A ``\\begin{name}...\\end{name}`` region with its begin-side arguments."""
|
|
81
|
+
|
|
82
|
+
name: str
|
|
83
|
+
arguments: tuple[MacroArgument, ...]
|
|
84
|
+
body: tuple[Content, ...]
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
class Math(Node, frozen=True):
|
|
88
|
+
"""An inline or display math region delimited by shifts or ``\\[``/``\\]``."""
|
|
89
|
+
|
|
90
|
+
mode: MathMode
|
|
91
|
+
body: tuple[Content, ...]
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
class Document(Node, frozen=True):
|
|
95
|
+
"""Root of a parsed file, owning the offset index over its exact source."""
|
|
96
|
+
|
|
97
|
+
children: tuple[Content, ...]
|
|
98
|
+
source: SourceMap
|
|
99
|
+
|
|
100
|
+
@property
|
|
101
|
+
def text(self) -> str:
|
|
102
|
+
return self.source.text
|
|
103
|
+
|
|
104
|
+
def __str__(self) -> str:
|
|
105
|
+
return self.source.text
|
|
106
|
+
|
|
107
|
+
def source_of(self, node: Node) -> str:
|
|
108
|
+
"""Return the exact original text the given node was built from."""
|
|
109
|
+
return self.source.extract(node.span)
|
|
110
|
+
|
|
111
|
+
|
|
112
|
+
type Content = (
|
|
113
|
+
Text
|
|
114
|
+
| Whitespace
|
|
115
|
+
| ParBreak
|
|
116
|
+
| Comment
|
|
117
|
+
| SpecialChar
|
|
118
|
+
| BraceGroup
|
|
119
|
+
| BracketGroup
|
|
120
|
+
| Macro
|
|
121
|
+
| Environment
|
|
122
|
+
| Math
|
|
123
|
+
)
|
latexwalker/normalize.py
ADDED
|
@@ -0,0 +1,46 @@
|
|
|
1
|
+
"""Public entry point reducing LaTeX source toward its canonical form."""
|
|
2
|
+
|
|
3
|
+
from functools import lru_cache
|
|
4
|
+
from importlib import resources
|
|
5
|
+
|
|
6
|
+
from .errors import RulesError
|
|
7
|
+
from .rewrite import CompiledRuleSet
|
|
8
|
+
from .rules import RuleScope, RuleSet, decode_rules
|
|
9
|
+
from .zones import ZoneKind, classify
|
|
10
|
+
|
|
11
|
+
_ALLOWED_SCOPES: dict[ZoneKind, frozenset[RuleScope]] = {
|
|
12
|
+
ZoneKind.PRESERVE: frozenset(),
|
|
13
|
+
ZoneKind.MATH: frozenset((RuleScope.ANY, RuleScope.MATH)),
|
|
14
|
+
ZoneKind.TEXT: frozenset((RuleScope.ANY, RuleScope.TEXT)),
|
|
15
|
+
}
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
@lru_cache(maxsize=1)
|
|
19
|
+
def default_compiled() -> CompiledRuleSet:
|
|
20
|
+
resource = resources.files(__package__).joinpath("resources", "rules.yaml")
|
|
21
|
+
try:
|
|
22
|
+
raw_yaml = resource.read_text(encoding="utf-8")
|
|
23
|
+
except OSError as exc:
|
|
24
|
+
raise RulesError("packaged default rule set is unreadable") from exc
|
|
25
|
+
return CompiledRuleSet(decode_rules(raw_yaml))
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def normalize(
|
|
29
|
+
source_text: str,
|
|
30
|
+
*,
|
|
31
|
+
rules: RuleSet | CompiledRuleSet | None = None,
|
|
32
|
+
) -> str:
|
|
33
|
+
"""Rewrite ``source_text`` with canonicalization rules over safe regions.
|
|
34
|
+
|
|
35
|
+
``rules`` replaces the packaged defaults entirely; pass a pre-built
|
|
36
|
+
:class:`CompiledRuleSet` to amortize compilation across many snippets.
|
|
37
|
+
"""
|
|
38
|
+
compiled = (
|
|
39
|
+
default_compiled()
|
|
40
|
+
if rules is None
|
|
41
|
+
else (rules if isinstance(rules, CompiledRuleSet) else CompiledRuleSet(rules))
|
|
42
|
+
)
|
|
43
|
+
return "".join(
|
|
44
|
+
compiled.rewrite(zone.text, allowed=_ALLOWED_SCOPES[zone.kind])
|
|
45
|
+
for zone in classify(source_text)
|
|
46
|
+
)
|