latexwalker 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,72 @@
1
+ """Public surface of latexwalker.
2
+
3
+ Symbols re-exported here form the supported API; deeper modules remain
4
+ importable for advanced use but may change without notice.
5
+ """
6
+
7
+ from ._version import __version__
8
+ from .errors import LatexWalkerError, ParseError, RulesError
9
+ from .lexer import tokenize
10
+ from .nodes import (
11
+ BraceGroup,
12
+ BracketGroup,
13
+ Comment,
14
+ Content,
15
+ Document,
16
+ Environment,
17
+ Macro,
18
+ MacroArgument,
19
+ Math,
20
+ MathMode,
21
+ Node,
22
+ ParBreak,
23
+ SpecialChar,
24
+ Text,
25
+ Whitespace,
26
+ )
27
+ from .normalize import normalize
28
+ from .parser import parse_document
29
+ from .rewrite import CompiledRuleSet
30
+ from .rules import Rule, RuleScope, RuleSet, decode_rules, load_rules
31
+ from .source import SourceMap, SourcePoint, SourceSpan, covering_span
32
+ from .tokens import Token, TokenType
33
+ from .traverse import iter_children, walk
34
+
35
+ __all__ = [
36
+ "BraceGroup",
37
+ "BracketGroup",
38
+ "Comment",
39
+ "CompiledRuleSet",
40
+ "Content",
41
+ "Document",
42
+ "Environment",
43
+ "LatexWalkerError",
44
+ "Macro",
45
+ "MacroArgument",
46
+ "Math",
47
+ "MathMode",
48
+ "Node",
49
+ "ParBreak",
50
+ "ParseError",
51
+ "Rule",
52
+ "RuleScope",
53
+ "RuleSet",
54
+ "RulesError",
55
+ "SourceMap",
56
+ "SourcePoint",
57
+ "SourceSpan",
58
+ "SpecialChar",
59
+ "Text",
60
+ "Token",
61
+ "TokenType",
62
+ "Whitespace",
63
+ "__version__",
64
+ "covering_span",
65
+ "decode_rules",
66
+ "iter_children",
67
+ "load_rules",
68
+ "normalize",
69
+ "parse_document",
70
+ "tokenize",
71
+ "walk",
72
+ ]
@@ -0,0 +1,24 @@
1
+ # file generated by vcs-versioning
2
+ # don't change, don't track in version control
3
+ from __future__ import annotations
4
+
5
+ __all__ = [
6
+ "__version__",
7
+ "__version_tuple__",
8
+ "version",
9
+ "version_tuple",
10
+ "__commit_id__",
11
+ "commit_id",
12
+ ]
13
+
14
+ version: str
15
+ __version__: str
16
+ __version_tuple__: tuple[int | str, ...]
17
+ version_tuple: tuple[int | str, ...]
18
+ commit_id: str | None
19
+ __commit_id__: str | None
20
+
21
+ __version__ = version = '0.1.0'
22
+ __version_tuple__ = version_tuple = (0, 1, 0)
23
+
24
+ __commit_id__ = commit_id = None
latexwalker/errors.py ADDED
@@ -0,0 +1,24 @@
1
+ """Typed failure surface for lexing and parsing operations."""
2
+
3
+ from .source import SourceSpan
4
+
5
+
6
+ class LatexWalkerError(Exception):
7
+ """Base class for every error raised by latexwalker."""
8
+
9
+
10
+ class ParseError(LatexWalkerError):
11
+ """Raised when source text cannot be tokenized or parsed."""
12
+
13
+ message: str
14
+ span: SourceSpan | None
15
+
16
+ def __init__(self, message: str, *, span: SourceSpan | None = None) -> None:
17
+ rendered = f"{message} ({span.describe()})" if span is not None else message
18
+ super().__init__(rendered)
19
+ self.message = message
20
+ self.span = span
21
+
22
+
23
+ class RulesError(LatexWalkerError):
24
+ """Raised when a normalization rule set cannot be read or compiled."""
latexwalker/lexer.py ADDED
@@ -0,0 +1,129 @@
1
+ """Streaming lexer converting raw LaTeX text into a token sequence."""
2
+
3
+ from collections.abc import Iterator
4
+
5
+ from .errors import ParseError
6
+ from .source import SourceMap, SourceSpan
7
+ from .tokens import Token, TokenType
8
+
9
+ _SINGLE_CHAR_TYPES: dict[str, TokenType] = {
10
+ "{": TokenType.GROUP_OPEN,
11
+ "}": TokenType.GROUP_CLOSE,
12
+ "[": TokenType.BRACKET_OPEN,
13
+ "]": TokenType.BRACKET_CLOSE,
14
+ "&": TokenType.ALIGNMENT,
15
+ "#": TokenType.PARAMETER,
16
+ "_": TokenType.SUBSCRIPT,
17
+ "^": TokenType.SUPERSCRIPT,
18
+ "~": TokenType.TIE,
19
+ }
20
+
21
+ _WORD_STOPPERS: frozenset[str] = frozenset(_SINGLE_CHAR_TYPES) | {"\\", "%", "$"}
22
+ _WHITESPACE: frozenset[str] = frozenset(" \t\n\r")
23
+
24
+
25
+ def tokenize(source: SourceMap) -> Iterator[Token]:
26
+ """Yield every token in ``source``, ending with a sentinel end-of-input."""
27
+ text = source.text
28
+ length = len(text)
29
+ index = 0
30
+ while index < length:
31
+ char = text[index]
32
+ if char == "\\":
33
+ token, index = _scan_macro(source, index)
34
+ elif char == "%":
35
+ token, index = _scan_comment(source, index)
36
+ elif char == "$":
37
+ token, index = _scan_math_shift(source, index)
38
+ elif char in _WHITESPACE:
39
+ token, index = _scan_whitespace_run(source, index)
40
+ elif char in _SINGLE_CHAR_TYPES:
41
+ token = _make_token(
42
+ source, index, index + 1, _SINGLE_CHAR_TYPES[char], char
43
+ )
44
+ index += 1
45
+ else:
46
+ token, index = _scan_word(source, index)
47
+ yield token
48
+ yield _make_token(source, length, length, TokenType.END_OF_INPUT, "")
49
+
50
+
51
+ def _make_token(
52
+ source: SourceMap, start: int, end: int, token_type: TokenType, value: str
53
+ ) -> Token:
54
+ return Token(
55
+ type=token_type,
56
+ value=value,
57
+ span=SourceSpan(start=source.point_at(start), end=source.point_at(end)),
58
+ )
59
+
60
+
61
+ def _scan_word(source: SourceMap, start: int) -> tuple[Token, int]:
62
+ text = source.text
63
+ end = start + 1
64
+ while (
65
+ end < len(text)
66
+ and text[end] not in _WORD_STOPPERS
67
+ and text[end] not in _WHITESPACE
68
+ ):
69
+ end += 1
70
+ return _make_token(source, start, end, TokenType.WORD, text[start:end]), end
71
+
72
+
73
+ def _scan_macro(source: SourceMap, start: int) -> tuple[Token, int]:
74
+ """Scan one control sequence; the stored value excludes the backslash."""
75
+ text = source.text
76
+ name_start = start + 1
77
+ if name_start >= len(text):
78
+ raise ParseError(
79
+ "dangling backslash at end of input",
80
+ span=SourceSpan.empty_at(source.point_at(start)),
81
+ )
82
+ lead = text[name_start]
83
+ if lead == "\n":
84
+ raise ParseError(
85
+ "backslash cannot be continued by a newline",
86
+ span=SourceSpan.empty_at(source.point_at(start)),
87
+ )
88
+ if lead.isalpha():
89
+ name_end = name_start + 1
90
+ while name_end < len(text) and text[name_end].isalpha():
91
+ name_end += 1
92
+ return _make_token(
93
+ source, start, name_end, TokenType.MACRO, text[name_start:name_end]
94
+ ), name_end
95
+ return _make_token(
96
+ source, start, name_start + 1, TokenType.MACRO, lead
97
+ ), name_start + 1
98
+
99
+
100
+ def _scan_comment(source: SourceMap, start: int) -> tuple[Token, int]:
101
+ text = source.text
102
+ newline = text.find("\n", start)
103
+ end = newline if newline != -1 else len(text)
104
+ return _make_token(source, start, end, TokenType.COMMENT, text[start:end]), end
105
+
106
+
107
+ def _scan_math_shift(source: SourceMap, start: int) -> tuple[Token, int]:
108
+ display = source.text.startswith("$$", start)
109
+ if display:
110
+ return (
111
+ _make_token(source, start, start + 2, TokenType.MATH_SHIFT_DISPLAY, "$$"),
112
+ start + 2,
113
+ )
114
+ return (
115
+ _make_token(source, start, start + 1, TokenType.MATH_SHIFT_INLINE, "$"),
116
+ start + 1,
117
+ )
118
+
119
+
120
+ def _scan_whitespace_run(source: SourceMap, start: int) -> tuple[Token, int]:
121
+ text = source.text
122
+ end = start + 1
123
+ while end < len(text) and text[end] in _WHITESPACE:
124
+ end += 1
125
+ raw = text[start:end]
126
+ token_type = (
127
+ TokenType.PARAGRAPH_BREAK if raw.count("\n") >= 2 else TokenType.WHITESPACE
128
+ )
129
+ return _make_token(source, start, end, token_type, raw), end
latexwalker/nodes.py ADDED
@@ -0,0 +1,123 @@
1
+ """Immutable syntax tree schemas produced by the parser.
2
+
3
+ Every node kind is a frozen ``msgspec.Struct`` carrying the source extent it
4
+ was built from, so trees stay lossless and cheap to share across stages.
5
+ """
6
+
7
+ from __future__ import annotations
8
+
9
+ from enum import StrEnum
10
+
11
+ import msgspec
12
+
13
+ from .source import SourceMap, SourceSpan
14
+
15
+
16
+ class MathMode(StrEnum):
17
+ """Whether a math region is inline or display."""
18
+
19
+ INLINE = "inline"
20
+ DISPLAY = "display"
21
+
22
+
23
+ class Node(msgspec.Struct, frozen=True, kw_only=True):
24
+ """Base schema shared by every syntax node; anchors it in the source."""
25
+
26
+ span: SourceSpan
27
+
28
+
29
+ class Text(Node, frozen=True):
30
+ """A run of ordinary characters, including interior spacing."""
31
+
32
+ value: str
33
+
34
+
35
+ class Whitespace(Node, frozen=True):
36
+ """A run of spaces or newlines that does not separate paragraphs."""
37
+
38
+ value: str
39
+
40
+
41
+ class ParBreak(Node, frozen=True):
42
+ """One or more blank lines separating paragraphs."""
43
+
44
+
45
+ class Comment(Node, frozen=True):
46
+ """A percent-introduced comment up to but excluding its newline."""
47
+
48
+ text: str
49
+
50
+
51
+ class SpecialChar(Node, frozen=True):
52
+ """A category-code character such as ``&``, ``#``, ``_``, or ``^``."""
53
+
54
+ char: str
55
+
56
+
57
+ class BraceGroup(Node, frozen=True):
58
+ """A brace-delimited ``{body}``."""
59
+
60
+ body: tuple[Content, ...]
61
+
62
+
63
+ class BracketGroup(Node, frozen=True):
64
+ """A bracket-delimited ``[body]``."""
65
+
66
+ body: tuple[Content, ...]
67
+
68
+
69
+ type MacroArgument = BraceGroup | BracketGroup
70
+
71
+
72
+ class Macro(Node, frozen=True):
73
+ """A control sequence with its immediately attached group arguments."""
74
+
75
+ name: str
76
+ arguments: tuple[MacroArgument, ...]
77
+
78
+
79
+ class Environment(Node, frozen=True):
80
+ """A ``\\begin{name}...\\end{name}`` region with its begin-side arguments."""
81
+
82
+ name: str
83
+ arguments: tuple[MacroArgument, ...]
84
+ body: tuple[Content, ...]
85
+
86
+
87
+ class Math(Node, frozen=True):
88
+ """An inline or display math region delimited by shifts or ``\\[``/``\\]``."""
89
+
90
+ mode: MathMode
91
+ body: tuple[Content, ...]
92
+
93
+
94
+ class Document(Node, frozen=True):
95
+ """Root of a parsed file, owning the offset index over its exact source."""
96
+
97
+ children: tuple[Content, ...]
98
+ source: SourceMap
99
+
100
+ @property
101
+ def text(self) -> str:
102
+ return self.source.text
103
+
104
+ def __str__(self) -> str:
105
+ return self.source.text
106
+
107
+ def source_of(self, node: Node) -> str:
108
+ """Return the exact original text the given node was built from."""
109
+ return self.source.extract(node.span)
110
+
111
+
112
+ type Content = (
113
+ Text
114
+ | Whitespace
115
+ | ParBreak
116
+ | Comment
117
+ | SpecialChar
118
+ | BraceGroup
119
+ | BracketGroup
120
+ | Macro
121
+ | Environment
122
+ | Math
123
+ )
@@ -0,0 +1,46 @@
1
+ """Public entry point reducing LaTeX source toward its canonical form."""
2
+
3
+ from functools import lru_cache
4
+ from importlib import resources
5
+
6
+ from .errors import RulesError
7
+ from .rewrite import CompiledRuleSet
8
+ from .rules import RuleScope, RuleSet, decode_rules
9
+ from .zones import ZoneKind, classify
10
+
11
+ _ALLOWED_SCOPES: dict[ZoneKind, frozenset[RuleScope]] = {
12
+ ZoneKind.PRESERVE: frozenset(),
13
+ ZoneKind.MATH: frozenset((RuleScope.ANY, RuleScope.MATH)),
14
+ ZoneKind.TEXT: frozenset((RuleScope.ANY, RuleScope.TEXT)),
15
+ }
16
+
17
+
18
+ @lru_cache(maxsize=1)
19
+ def default_compiled() -> CompiledRuleSet:
20
+ resource = resources.files(__package__).joinpath("resources", "rules.yaml")
21
+ try:
22
+ raw_yaml = resource.read_text(encoding="utf-8")
23
+ except OSError as exc:
24
+ raise RulesError("packaged default rule set is unreadable") from exc
25
+ return CompiledRuleSet(decode_rules(raw_yaml))
26
+
27
+
28
+ def normalize(
29
+ source_text: str,
30
+ *,
31
+ rules: RuleSet | CompiledRuleSet | None = None,
32
+ ) -> str:
33
+ """Rewrite ``source_text`` with canonicalization rules over safe regions.
34
+
35
+ ``rules`` replaces the packaged defaults entirely; pass a pre-built
36
+ :class:`CompiledRuleSet` to amortize compilation across many snippets.
37
+ """
38
+ compiled = (
39
+ default_compiled()
40
+ if rules is None
41
+ else (rules if isinstance(rules, CompiledRuleSet) else CompiledRuleSet(rules))
42
+ )
43
+ return "".join(
44
+ compiled.rewrite(zone.text, allowed=_ALLOWED_SCOPES[zone.kind])
45
+ for zone in classify(source_text)
46
+ )