infinity_pyscanf 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,125 @@
1
+ Metadata-Version: 2.3
2
+ Name: infinity_pyscanf
3
+ Version: 0.1.0
4
+ Summary: Type-safe, scanf-style string parsing.
5
+ Keywords: scanf,parsing,parser,regex,type-safe
6
+ Classifier: Intended Audience :: Developers
7
+ Classifier: Operating System :: OS Independent
8
+ Classifier: Programming Language :: Python :: 3
9
+ Classifier: Programming Language :: Python :: 3.14
10
+ Classifier: Topic :: Software Development :: Libraries :: Python Modules
11
+ Classifier: Topic :: Text Processing
12
+ Requires-Dist: interrogate ; extra == 'dev'
13
+ Requires-Dist: ruff ; extra == 'dev'
14
+ Requires-Dist: pyright ; extra == 'dev'
15
+ Requires-Dist: pre-commit ; extra == 'dev'
16
+ Requires-Dist: infinity-pyscanf[test] ; extra == 'dev'
17
+ Requires-Dist: pytest ; extra == 'test'
18
+ Requires-Dist: pytest-cov ; extra == 'test'
19
+ Requires-Python: >=3.14
20
+ Provides-Extra: dev
21
+ Provides-Extra: test
22
+ Description-Content-Type: text/markdown
23
+
24
+ # infinity_pyscanf
25
+
26
+ Type-safe, `scanf`-style string parsing for Python 3.14+.
27
+ Distributed as `infinity_pyscanf`; imported as `scanf`.
28
+
29
+ Supply one converter type per `{}` placeholder as class type arguments, then
30
+ the template and the input string as call arguments:
31
+
32
+ ```python
33
+ from scanf import scanf
34
+
35
+ a, b, c = scanf[int, float, str]("{} {} {}")("1 2.5 hello")
36
+ # a=1 (int), b=2.5 (float), c="hello" (str)
37
+ ```
38
+
39
+ Parsers are reusable:
40
+
41
+ ```python
42
+ parse_pair = scanf[int, int]("{} {}")
43
+ parse_pair("7 8") # (7, 8)
44
+ ```
45
+
46
+ ## Built-in converters
47
+
48
+ Built-in generic containers (`dict`, `list`, `tuple`, `set` and `frozenset`)
49
+ are parsed as Python literals via `ast.literal_eval`, and `bool` accepts
50
+ `true` / `false` / `1` / `0` (case-insensitive):
51
+
52
+ ```python
53
+ from scanf import scanf
54
+
55
+ scanf[bool]("{}")("true") # (True,)
56
+ scanf[list[int]]("{}")("[1, 2, 3]") # ([1, 2, 3],)
57
+ ```
58
+
59
+ For JSON-specific syntax (`true`, `false`, `null`) or any custom conversion,
60
+ attach a `Converter` spec:
61
+
62
+ ```python
63
+ import json
64
+ from typing import Annotated
65
+
66
+ from scanf import Converter, scanf
67
+
68
+ parse_json = scanf[Annotated[dict[str, str], Converter(json.loads)]]("{}")
69
+ parse_json('{"key": "value"}') # ({"key": "value"},)
70
+ ```
71
+
72
+ ## Per-field configuration
73
+
74
+ Per-field configuration rides along as `Converter` metadata inside a
75
+ `typing.Annotated` converter: a conversion callable, a capture pattern, a
76
+ strip flag, `re` flags and a label:
77
+
78
+ ```python
79
+ from typing import Annotated
80
+
81
+ from scanf import Converter, scanf
82
+
83
+ parse = scanf[
84
+ Annotated[int, Converter(int, pattern=r"\d+", name="age")],
85
+ Annotated[str, Converter(str, pattern=r'"[^"]*"', name="name")],
86
+ ]("age={} name={}")
87
+ parse('age=42 name="kim"') # (42, '"kim"')
88
+ ```
89
+
90
+ `typing.Annotated` wrappers around converters are stripped; their metadata is
91
+ ignored unless it contains a `Converter` spec. Several specs on one field
92
+ compete in declaration order: each branch carries its own pattern, the first
93
+ branch whose pattern matches wins, and if its converter then fails the whole
94
+ scan fails — a failing converter never falls back to a later branch.
95
+
96
+ ## Template semantics
97
+
98
+ Each `{}` captures with the default `.+?` pattern:
99
+
100
+ * non-empty: at least one character;
101
+ * single-line: never matches a newline;
102
+ * lazy: takes the shortest text that still lets the rest of the template match.
103
+
104
+ A `Converter` spec can replace the pattern per field.
105
+
106
+ Whitespace runs in the template match any whitespace run in the input, and
107
+ surrounding whitespace of the input (and of each captured field) is ignored.
108
+ Literal braces in a template are written `{{` and `}}`; a single brace is an
109
+ error.
110
+
111
+ ## Development
112
+
113
+ ```bash
114
+ uv sync --extra dev
115
+ uv run pre-commit install # optional: run the gates on every commit
116
+ uv run pre-commit run --all-files # full local gate
117
+ ```
118
+
119
+ The gate runs ruff (lint + format), pyright (strict), interrogate (docstring
120
+ coverage), pytest with a coverage floor, and an `example/` smoke run. CI
121
+ (GitHub Actions) runs the same gates plus a 3-OS test matrix.
122
+
123
+ `example/example.py` contains worked examples: field customization, JSON
124
+ parsing and a recursive scanner built with fixpoint / `alt` / `sep_by`
125
+ combinators.
@@ -0,0 +1,102 @@
1
+ # infinity_pyscanf
2
+
3
+ Type-safe, `scanf`-style string parsing for Python 3.14+.
4
+ Distributed as `infinity_pyscanf`; imported as `scanf`.
5
+
6
+ Supply one converter type per `{}` placeholder as class type arguments, then
7
+ the template and the input string as call arguments:
8
+
9
+ ```python
10
+ from scanf import scanf
11
+
12
+ a, b, c = scanf[int, float, str]("{} {} {}")("1 2.5 hello")
13
+ # a=1 (int), b=2.5 (float), c="hello" (str)
14
+ ```
15
+
16
+ Parsers are reusable:
17
+
18
+ ```python
19
+ parse_pair = scanf[int, int]("{} {}")
20
+ parse_pair("7 8") # (7, 8)
21
+ ```
22
+
23
+ ## Built-in converters
24
+
25
+ Built-in generic containers (`dict`, `list`, `tuple`, `set` and `frozenset`)
26
+ are parsed as Python literals via `ast.literal_eval`, and `bool` accepts
27
+ `true` / `false` / `1` / `0` (case-insensitive):
28
+
29
+ ```python
30
+ from scanf import scanf
31
+
32
+ scanf[bool]("{}")("true") # (True,)
33
+ scanf[list[int]]("{}")("[1, 2, 3]") # ([1, 2, 3],)
34
+ ```
35
+
36
+ For JSON-specific syntax (`true`, `false`, `null`) or any custom conversion,
37
+ attach a `Converter` spec:
38
+
39
+ ```python
40
+ import json
41
+ from typing import Annotated
42
+
43
+ from scanf import Converter, scanf
44
+
45
+ parse_json = scanf[Annotated[dict[str, str], Converter(json.loads)]]("{}")
46
+ parse_json('{"key": "value"}') # ({"key": "value"},)
47
+ ```
48
+
49
+ ## Per-field configuration
50
+
51
+ Per-field configuration rides along as `Converter` metadata inside a
52
+ `typing.Annotated` converter: a conversion callable, a capture pattern, a
53
+ strip flag, `re` flags and a label:
54
+
55
+ ```python
56
+ from typing import Annotated
57
+
58
+ from scanf import Converter, scanf
59
+
60
+ parse = scanf[
61
+ Annotated[int, Converter(int, pattern=r"\d+", name="age")],
62
+ Annotated[str, Converter(str, pattern=r'"[^"]*"', name="name")],
63
+ ]("age={} name={}")
64
+ parse('age=42 name="kim"') # (42, '"kim"')
65
+ ```
66
+
67
+ `typing.Annotated` wrappers around converters are stripped; their metadata is
68
+ ignored unless it contains a `Converter` spec. Several specs on one field
69
+ compete in declaration order: each branch carries its own pattern, the first
70
+ branch whose pattern matches wins, and if its converter then fails the whole
71
+ scan fails — a failing converter never falls back to a later branch.
72
+
73
+ ## Template semantics
74
+
75
+ Each `{}` captures with the default `.+?` pattern:
76
+
77
+ * non-empty: at least one character;
78
+ * single-line: never matches a newline;
79
+ * lazy: takes the shortest text that still lets the rest of the template match.
80
+
81
+ A `Converter` spec can replace the pattern per field.
82
+
83
+ Whitespace runs in the template match any whitespace run in the input, and
84
+ surrounding whitespace of the input (and of each captured field) is ignored.
85
+ Literal braces in a template are written `{{` and `}}`; a single brace is an
86
+ error.
87
+
88
+ ## Development
89
+
90
+ ```bash
91
+ uv sync --extra dev
92
+ uv run pre-commit install # optional: run the gates on every commit
93
+ uv run pre-commit run --all-files # full local gate
94
+ ```
95
+
96
+ The gate runs ruff (lint + format), pyright (strict), interrogate (docstring
97
+ coverage), pytest with a coverage floor, and an `example/` smoke run. CI
98
+ (GitHub Actions) runs the same gates plus a 3-OS test matrix.
99
+
100
+ `example/example.py` contains worked examples: field customization, JSON
101
+ parsing and a recursive scanner built with fixpoint / `alt` / `sep_by`
102
+ combinators.
@@ -0,0 +1,86 @@
1
+ [build-system]
2
+ requires = ["uv_build>=0.11.21,<0.12.0"]
3
+ build-backend = "uv_build"
4
+
5
+ [project]
6
+ name = "infinity_pyscanf"
7
+ version = "0.1.0"
8
+ description = "Type-safe, scanf-style string parsing."
9
+ readme = "README.md"
10
+ keywords = [
11
+ "scanf",
12
+ "parsing",
13
+ "parser",
14
+ "regex",
15
+ "type-safe",
16
+ ]
17
+ classifiers = [
18
+ "Intended Audience :: Developers",
19
+ "Operating System :: OS Independent",
20
+ "Programming Language :: Python :: 3",
21
+ "Programming Language :: Python :: 3.14",
22
+ "Topic :: Software Development :: Libraries :: Python Modules",
23
+ "Topic :: Text Processing",
24
+ ]
25
+ requires-python = ">=3.14"
26
+ dependencies = []
27
+
28
+ [project.optional-dependencies]
29
+ test = [
30
+ "pytest",
31
+ "pytest-cov",
32
+ ]
33
+ dev = [
34
+ "interrogate",
35
+ "ruff",
36
+ "pyright",
37
+ "pre-commit",
38
+ "infinity_pyscanf[test]",
39
+ ]
40
+
41
+ [tool.ruff]
42
+ target-version = "py314"
43
+ line-length = 88
44
+ src = ["src"]
45
+
46
+ [tool.ruff.lint]
47
+ select = [
48
+ "B",
49
+ "C4",
50
+ "E",
51
+ "F",
52
+ "I",
53
+ "RUF",
54
+ "SIM",
55
+ "UP",
56
+ "W",
57
+ ]
58
+
59
+ [tool.ruff.lint.isort]
60
+ known-first-party = ["scanf"]
61
+
62
+ [tool.pytest.ini_options]
63
+ testpaths = ["tests"]
64
+
65
+ [tool.pyright]
66
+ typeCheckingMode = "strict"
67
+ pythonVersion = "3.14"
68
+ venvPath = "."
69
+ venv = ".venv"
70
+
71
+ [tool.interrogate]
72
+ fail-under = 90
73
+ ignore-init-method = true
74
+ ignore-init-module = true
75
+ ignore-magic = true
76
+ ignore-private = true
77
+ ignore-semiprivate = true
78
+ ignore-nested-functions = true
79
+ exclude = [
80
+ "tests",
81
+ "example",
82
+ ]
83
+
84
+ [tool.uv.build-backend]
85
+ module-root = "src"
86
+ module-name = "scanf"
@@ -0,0 +1,83 @@
1
+ [build-system]
2
+ requires = ["uv_build>=0.11.21,<0.12.0"]
3
+ build-backend = "uv_build"
4
+
5
+ [project]
6
+ name = "infinity_pyscanf"
7
+ version = "0.1.0"
8
+ description = "Type-safe, scanf-style string parsing."
9
+ readme = "README.md"
10
+ keywords = [
11
+ "scanf",
12
+ "parsing",
13
+ "parser",
14
+ "regex",
15
+ "type-safe",
16
+ ]
17
+ classifiers = [
18
+ "Intended Audience :: Developers",
19
+ "Operating System :: OS Independent",
20
+ "Programming Language :: Python :: 3",
21
+ "Programming Language :: Python :: 3.14",
22
+ "Topic :: Software Development :: Libraries :: Python Modules",
23
+ "Topic :: Text Processing",
24
+ ]
25
+ requires-python = ">=3.14"
26
+ dependencies = []
27
+
28
+ [project.optional-dependencies]
29
+ test = [
30
+ "pytest",
31
+ "pytest-cov",
32
+ ]
33
+ dev = [
34
+ "interrogate",
35
+ "ruff",
36
+ "pyright",
37
+ "pre-commit",
38
+ "infinity_pyscanf[test]",
39
+ ]
40
+
41
+ [tool.ruff]
42
+ target-version = "py314"
43
+ line-length = 88
44
+ src = ["src"]
45
+
46
+ [tool.ruff.lint]
47
+ select = [
48
+ "B", # flake8-bugbear
49
+ "C4", # flake8-comprehensions
50
+ "E", # pycodestyle (errors)
51
+ "F", # pyflakes
52
+ "I", # isort
53
+ "RUF", # Ruff-specific rules
54
+ "SIM", # flake8-simplify
55
+ "UP", # pyupgrade
56
+ "W", # pycodestyle (warnings)
57
+ ]
58
+
59
+ [tool.ruff.lint.isort]
60
+ known-first-party = ["scanf"]
61
+
62
+ [tool.pytest.ini_options]
63
+ testpaths = ["tests"]
64
+
65
+ [tool.pyright]
66
+ typeCheckingMode = "strict"
67
+ pythonVersion = "3.14"
68
+ venvPath = "."
69
+ venv = ".venv"
70
+
71
+ [tool.interrogate]
72
+ fail-under = 90
73
+ ignore-init-method = true
74
+ ignore-init-module = true
75
+ ignore-magic = true
76
+ ignore-private = true
77
+ ignore-semiprivate = true
78
+ ignore-nested-functions = true
79
+ exclude = ["tests", "example"]
80
+
81
+ [tool.uv.build-backend]
82
+ module-root = "src"
83
+ module-name = "scanf"
@@ -0,0 +1,5 @@
1
+ """Strictly typed, ``scanf``-style parsing for Python."""
2
+
3
+ from scanf.scanf import Converter, ConvertError, MatchError, ScanError, scanf
4
+
5
+ __all__ = ["ConvertError", "Converter", "MatchError", "ScanError", "scanf"]
@@ -0,0 +1,301 @@
1
+ """Runtime engine for :mod:`scanf`: resolve field metas, then scan input.
2
+
3
+ Two responsibilities, with no type-level semantics:
4
+
5
+ * :func:`resolve_field` -- turn one field's meta (a bare declared converter
6
+ plus an optional :class:`Converter` spec) into a concrete converter configuration
7
+ (:class:`Field`);
8
+ * :class:`Template` -- validate a configuration, compile the template regex
9
+ and ``scan`` input strings into tuples.
10
+ """
11
+
12
+ import ast
13
+ import re
14
+ from collections.abc import Callable
15
+ from dataclasses import dataclass
16
+ from functools import lru_cache
17
+ from typing import Any, get_origin
18
+
19
+ # The default field pattern: non-empty, single-line (no DOTALL) and lazy.
20
+ DEFAULT_PATTERN = r".+?"
21
+ _PLACEHOLDER = "{}"
22
+ _WHITESPACE = re.compile(r"\s+")
23
+
24
+
25
+ class ScanError(ValueError):
26
+ """Base failure while scanning: a ``MatchError`` or a ``ConvertError``."""
27
+
28
+
29
+ class MatchError(ScanError):
30
+ """Input did not match the template (a soft failure).
31
+
32
+ Ordered-choice combinators fall through to the next alternative on this
33
+ one. A branch that matched but failed to convert raises ``ConvertError``
34
+ instead, which aborts the choice so its diagnostics are never swallowed.
35
+ """
36
+
37
+
38
+ class ConvertError(ScanError):
39
+ """A field matched but its converter failed (a hard failure).
40
+
41
+ The message carries the field number, the text and its offset; the
42
+ underlying error is chained through ``__cause__``.
43
+ """
44
+
45
+
46
+ @dataclass(frozen=True)
47
+ class Converter[T]:
48
+ """Per-field converter spec, passed as ``Annotated[T, Converter(...)]`` metadata.
49
+
50
+ ``fn`` converts the matched text; ``pattern`` is the regex used to capture
51
+ this field (inserted into the template regex as-is); ``strip`` strips the
52
+ matched text before conversion; ``name`` labels the field in error messages
53
+ (defaults to the callable's name); ``flags`` are ``re`` flags applied to
54
+ ``pattern``. Several specs on one field compete in declaration order: the
55
+ first branch whose pattern matches wins, and if its ``fn`` then raises, the
56
+ scan fails as a whole (no fallback to the remaining branches).
57
+ """
58
+
59
+ fn: Callable[[str], T]
60
+ pattern: str = DEFAULT_PATTERN
61
+ strip: bool = True
62
+ name: str | None = None
63
+ flags: int = 0
64
+
65
+
66
+ @dataclass(frozen=True)
67
+ class Field:
68
+ """One placeholder: converter branches competing in declaration order."""
69
+
70
+ converters: tuple[Converter[Any], ...]
71
+
72
+
73
+ def _converter_name(converter: object) -> str | None:
74
+ """Return the display name of ``converter`` (used to label fields)."""
75
+ name = getattr(converter, "__name__", None)
76
+ if isinstance(name, str):
77
+ return name
78
+ origin = get_origin(converter)
79
+ if origin is not None:
80
+ name = getattr(origin, "__name__", None)
81
+ if isinstance(name, str):
82
+ return name
83
+ return None
84
+
85
+
86
+ def _label(converter: Converter[Any]) -> str:
87
+ """Display name of one converter branch in error messages."""
88
+ return converter.name or _converter_name(converter.fn) or repr(converter.fn)
89
+
90
+
91
+ def _literal_converter(origin: type[Any]) -> Callable[[str], Any]:
92
+ return lambda text: origin(ast.literal_eval(text))
93
+
94
+
95
+ _BOOL_VALUES = {"true": True, "false": False, "1": True, "0": False}
96
+
97
+
98
+ def _bool_converter(text: str) -> bool:
99
+ value = _BOOL_VALUES.get(text.casefold())
100
+ if value is None:
101
+ raise ValueError(f"cannot parse {text!r} as bool")
102
+ return value
103
+
104
+
105
+ # Keyed by the origin *object* (not by name), so user classes that happen to be
106
+ # named ``dict``, ``list``, ``bool``, ... never pick up a built-in default.
107
+ _BUILTIN_CONVERTERS: dict[object, Callable[[str], Any]] = {
108
+ bool: _bool_converter,
109
+ dict: _literal_converter(dict),
110
+ frozenset: _literal_converter(frozenset),
111
+ list: _literal_converter(list),
112
+ set: _literal_converter(set),
113
+ tuple: _literal_converter(tuple),
114
+ }
115
+
116
+
117
+ def _default_converter(converter: object) -> Callable[[str], Any] | None:
118
+ """Return the built-in default converter for supported built-in types."""
119
+ origin = get_origin(converter)
120
+ return _BUILTIN_CONVERTERS.get(origin if origin is not None else converter)
121
+
122
+
123
+ def resolve_field(declared: Any, convs: list[Converter[Any]]) -> Field:
124
+ """Resolve one field's meta into its converter branches (meta -> converters).
125
+
126
+ User-provided ``Converter`` specs become the branches, in order. Without
127
+ any, one branch is synthesized from the declared type: the built-in default
128
+ for supported containers, otherwise the declared type used as a callable.
129
+ """
130
+ if convs:
131
+ return Field(converters=tuple(convs))
132
+ branch: Converter[Any] = Converter(
133
+ fn=_default_converter(declared) or declared,
134
+ name=_converter_name(declared),
135
+ )
136
+ return Field(converters=(branch,))
137
+
138
+
139
+ def _literal_to_pattern(text: str) -> str:
140
+ """Escape literal template text; whitespace runs become ``\\s+``."""
141
+ pieces: list[str] = []
142
+ position = 0
143
+ for match in _WHITESPACE.finditer(text):
144
+ pieces.append(re.escape(text[position : match.start()]))
145
+ pieces.append(r"\s+")
146
+ position = match.end()
147
+ pieces.append(re.escape(text[position:]))
148
+ return "".join(pieces)
149
+
150
+
151
+ _INLINE_FLAGS: tuple[tuple[int, str], ...] = (
152
+ (re.IGNORECASE, "i"),
153
+ (re.MULTILINE, "m"),
154
+ (re.DOTALL, "s"),
155
+ (re.VERBOSE, "x"),
156
+ (re.ASCII, "a"),
157
+ )
158
+ _INLINE_FLAG_MASK = re.IGNORECASE | re.MULTILINE | re.DOTALL | re.VERBOSE | re.ASCII
159
+
160
+
161
+ def _scoped(pattern: str, flags: int) -> str:
162
+ """Wrap ``pattern`` in scoped inline flags (e.g. ``(?i:...)``)."""
163
+ if not flags:
164
+ return pattern
165
+ if flags & ~_INLINE_FLAG_MASK:
166
+ raise ValueError(f"unsupported re flags for a field pattern: {flags:#x}")
167
+ letters = "".join(letter for bit, letter in _INLINE_FLAGS if flags & bit)
168
+ return f"(?{letters}:{pattern})"
169
+
170
+
171
+ @lru_cache(maxsize=1024)
172
+ def _segments(template: str) -> tuple[str | None, ...]:
173
+ """Split ``template`` into literal segments and ``None`` placeholders.
174
+
175
+ ``{{`` and ``}}`` escape literal braces; a single brace is an error.
176
+ """
177
+ segments: list[str | None] = []
178
+ literal: list[str] = []
179
+ text = template.strip()
180
+ position = 0
181
+ while position < len(text):
182
+ character = text[position]
183
+ if character == "{":
184
+ if text.startswith("{{", position):
185
+ literal.append("{")
186
+ position += 2
187
+ elif text.startswith(_PLACEHOLDER, position):
188
+ segments.append("".join(literal))
189
+ literal.clear()
190
+ segments.append(None)
191
+ position += 2
192
+ else:
193
+ raise ValueError(
194
+ f"single '{{' in template {template!r}; escape literal "
195
+ "braces as '{{' and '}}'"
196
+ )
197
+ elif character == "}":
198
+ if text.startswith("}}", position):
199
+ literal.append("}")
200
+ position += 2
201
+ else:
202
+ raise ValueError(
203
+ f"single '}}' in template {template!r}; escape literal "
204
+ "braces as '{{' and '}}'"
205
+ )
206
+ else:
207
+ literal.append(character)
208
+ position += 1
209
+ segments.append("".join(literal))
210
+ return tuple(segments)
211
+
212
+
213
+ def _branch_group(index: int, branch: int) -> str:
214
+ """Regex group name capturing one branch's own match for a field."""
215
+ return f"s{index}b{branch}"
216
+
217
+
218
+ def _field_chunk(index: int, field: Field) -> str:
219
+ """One field's regex: its converter branches as an ordered alternation."""
220
+ converters = field.converters
221
+ if len(converters) == 1:
222
+ single = converters[0]
223
+ return f"(?P<s{index}>{_scoped(single.pattern, single.flags)})"
224
+ branches = "|".join(
225
+ f"(?P<{_branch_group(index, branch)}>{_scoped(conv.pattern, conv.flags)})"
226
+ for branch, conv in enumerate(converters)
227
+ )
228
+ return f"(?P<s{index}>{branches})"
229
+
230
+
231
+ @lru_cache(maxsize=1024)
232
+ def _compile(template: str, fields: tuple[Field, ...]) -> re.Pattern[str]:
233
+ """Compile ``template`` into a regex with one named group per placeholder."""
234
+ chunks: list[str] = []
235
+ placeholder = 0
236
+ for segment in _segments(template):
237
+ if segment is None:
238
+ chunks.append(_field_chunk(placeholder, fields[placeholder]))
239
+ placeholder += 1
240
+ else:
241
+ chunks.append(_literal_to_pattern(segment))
242
+ return re.compile("".join(chunks))
243
+
244
+
245
+ def _winner(match: re.Match[str], index: int, field: Field) -> int:
246
+ """Index of the branch whose pattern produced the field's match."""
247
+ if len(field.converters) == 1:
248
+ return 0
249
+ for branch, _ in enumerate(field.converters):
250
+ if match.group(_branch_group(index, branch)) is not None:
251
+ return branch
252
+ raise AssertionError(f"no converter branch matched field {index}")
253
+
254
+
255
+ class Template:
256
+ """A validated, compiled template, ready to scan input."""
257
+
258
+ def __init__(self, template: str, fields: tuple[Field, ...]) -> None:
259
+ # Fail fast on converters that cannot go into the compile cache.
260
+ hash(fields)
261
+ for field in fields:
262
+ if not field.converters:
263
+ raise ValueError("each field requires at least one converter")
264
+ for converter in field.converters:
265
+ if not callable(converter.fn):
266
+ raise TypeError(
267
+ f"converters must be callable, got {converter.fn!r}"
268
+ )
269
+ placeholders = sum(segment is None for segment in _segments(template))
270
+ if placeholders != len(fields):
271
+ raise ValueError(
272
+ f"template {template!r} has {placeholders} placeholder(s) "
273
+ f"but {len(fields)} converter type(s) were given"
274
+ )
275
+ self.template = template
276
+ self.fields = fields
277
+ self._pattern = _compile(template, fields)
278
+
279
+ def scan(self, input_value: str) -> tuple[Any, ...]:
280
+ """Scan ``input_value``: match the template and convert each field."""
281
+ match = self._pattern.fullmatch(input_value.strip())
282
+ if match is None:
283
+ raise MatchError(
284
+ f"input {input_value!r} does not match template {self.template!r}"
285
+ )
286
+ values: list[Any] = []
287
+ for index, field in enumerate(self.fields):
288
+ converter = field.converters[_winner(match, index, field)]
289
+ raw = match.group(f"s{index}")
290
+ text = raw.strip() if converter.strip else raw
291
+ try:
292
+ values.append(converter.fn(text))
293
+ except Exception as error:
294
+ # Report the offset within the original input, not the stripped one.
295
+ shift = len(input_value) - len(input_value.lstrip())
296
+ offset = shift + match.start(f"s{index}")
297
+ raise ConvertError(
298
+ f"cannot convert field {index + 1} ({text!r}) at offset "
299
+ f"{offset} using {_label(converter)}"
300
+ ) from error
301
+ return tuple(values)
File without changes
@@ -0,0 +1,111 @@
1
+ r"""Type-safe, ``scanf``-style string parsing.
2
+
3
+ Supply one converter type per ``{}`` placeholder as class type arguments, then
4
+ the template and the input string as call arguments::
5
+
6
+ from scanf import scanf
7
+
8
+ a, b, c = scanf[int, float, str]("{} {} {}")("1 2.5 hello")
9
+ # a=1 (int), b=2.5 (float), c="hello" (str)
10
+
11
+ Parsers are reusable. Fields accept per-field ``Converter`` specs and built-in
12
+ defaults (containers via ``ast.literal_eval``, ``bool`` literals); see the
13
+ project README for the full guide -- usage, semantics and examples.
14
+ """
15
+
16
+ import types
17
+ from typing import Annotated, Any, cast, get_args, get_origin
18
+
19
+ from scanf.engine import (
20
+ Converter,
21
+ ConvertError,
22
+ Field,
23
+ MatchError,
24
+ ScanError,
25
+ Template,
26
+ resolve_field,
27
+ )
28
+
29
+ __all__ = ["ConvertError", "Converter", "MatchError", "ScanError", "scanf"]
30
+
31
+
32
+ def flatten_annotated(meta: Any) -> tuple[Any, list[Any]]:
33
+ """Flatten nested ``Annotated`` wrappers; metadata ordered deep to shallow
34
+ (which is also the order converter branches compete in)."""
35
+ if get_origin(meta) is not Annotated:
36
+ return meta, []
37
+ args = get_args(meta)
38
+ base, metadata = flatten_annotated(args[0])
39
+ return base, [*metadata, *args[1:]]
40
+
41
+
42
+ def build_field(meta: Any) -> Field:
43
+ """Extract one declaration's Converter specs and resolve them via the engine."""
44
+ declared, metadata = flatten_annotated(meta)
45
+ converters: list[Converter[Any]] = [
46
+ cast(Converter[Any], item) for item in metadata if isinstance(item, Converter)
47
+ ]
48
+ return resolve_field(declared, converters)
49
+
50
+
51
+ class scanf[*Meta]:
52
+ """Typed ``scanf``: ``scanf[Meta1, Meta2, ...](template)(input_value)``.
53
+
54
+ Each ``Meta`` declares one field -- a converter type, or
55
+ ``Annotated[converter, Converter(...)]`` for per-field configuration --
56
+ and determines the corresponding result element type. Several ``Converter``
57
+ specs on one field compete in order: the first matching branch wins. The
58
+ template and the input string are call arguments. Parsers are reusable.
59
+
60
+ The converter setup is validated eagerly when the parser is constructed:
61
+ non-callable converters and placeholder/converter-count mismatches raise
62
+ immediately.
63
+ """
64
+
65
+ _engine: Template
66
+
67
+ def __init__(self, template: str, _engine: Template | None = None) -> None:
68
+ # Parsers are built through ``scanf[...]`` (see _ScanfAlias), which
69
+ # passes the ready engine in; direct construction is a misuse.
70
+ if _engine is None:
71
+ raise TypeError(
72
+ "scanf requires at least one converter type, e.g. scanf[int]('{}')"
73
+ )
74
+ self._engine = _engine
75
+
76
+ @classmethod
77
+ def __class_getitem__(cls, item: Any) -> Any:
78
+ """Bind the field metas; see the class docstring.
79
+
80
+ Type checkers ignore this hook for generic classes and keep treating
81
+ ``scanf[int, float]`` as a specialization. At runtime it returns a
82
+ ``types.GenericAlias`` subclass carrying the metas; calling that alias
83
+ with the template is what validates and builds the parser.
84
+ """
85
+ metas = cast(tuple[Any, ...], item if isinstance(item, tuple) else (item,))
86
+ if not metas:
87
+ raise TypeError(
88
+ "scanf requires at least one converter type, e.g. scanf[int]('{}')"
89
+ )
90
+ return _ScanfAlias(cls, metas)
91
+
92
+ def __call__(self, input_value: str) -> tuple[*Meta]:
93
+ """Parse ``input_value``."""
94
+ return self._engine.scan(input_value)
95
+
96
+
97
+ class _ScanfAlias(types.GenericAlias):
98
+ """Subclass of ``types.GenericAlias`` produced by ``scanf[...]``.
99
+
100
+ The field metas travel in the alias itself (``__args__``). Calling the
101
+ alias with the template is where both sides meet: it resolves each meta
102
+ into its converter, validates the configuration and constructs the parser,
103
+ setting ``__orig_class__`` the way typing would.
104
+ """
105
+
106
+ def __call__(self, template: str) -> Any:
107
+ origin = cast(Any, self.__origin__)
108
+ fields: tuple[Field, ...] = tuple(build_field(meta) for meta in self.__args__)
109
+ instance = origin(template, Template(template, fields))
110
+ instance.__orig_class__ = self
111
+ return instance