infinity_pyscanf 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- infinity_pyscanf-0.1.0/PKG-INFO +125 -0
- infinity_pyscanf-0.1.0/README.md +102 -0
- infinity_pyscanf-0.1.0/pyproject.toml +86 -0
- infinity_pyscanf-0.1.0/pyproject.toml.orig +83 -0
- infinity_pyscanf-0.1.0/src/scanf/__init__.py +5 -0
- infinity_pyscanf-0.1.0/src/scanf/engine.py +301 -0
- infinity_pyscanf-0.1.0/src/scanf/py.typed +0 -0
- infinity_pyscanf-0.1.0/src/scanf/scanf.py +111 -0
|
@@ -0,0 +1,125 @@
|
|
|
1
|
+
Metadata-Version: 2.3
|
|
2
|
+
Name: infinity_pyscanf
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Type-safe, scanf-style string parsing.
|
|
5
|
+
Keywords: scanf,parsing,parser,regex,type-safe
|
|
6
|
+
Classifier: Intended Audience :: Developers
|
|
7
|
+
Classifier: Operating System :: OS Independent
|
|
8
|
+
Classifier: Programming Language :: Python :: 3
|
|
9
|
+
Classifier: Programming Language :: Python :: 3.14
|
|
10
|
+
Classifier: Topic :: Software Development :: Libraries :: Python Modules
|
|
11
|
+
Classifier: Topic :: Text Processing
|
|
12
|
+
Requires-Dist: interrogate ; extra == 'dev'
|
|
13
|
+
Requires-Dist: ruff ; extra == 'dev'
|
|
14
|
+
Requires-Dist: pyright ; extra == 'dev'
|
|
15
|
+
Requires-Dist: pre-commit ; extra == 'dev'
|
|
16
|
+
Requires-Dist: infinity-pyscanf[test] ; extra == 'dev'
|
|
17
|
+
Requires-Dist: pytest ; extra == 'test'
|
|
18
|
+
Requires-Dist: pytest-cov ; extra == 'test'
|
|
19
|
+
Requires-Python: >=3.14
|
|
20
|
+
Provides-Extra: dev
|
|
21
|
+
Provides-Extra: test
|
|
22
|
+
Description-Content-Type: text/markdown
|
|
23
|
+
|
|
24
|
+
# infinity_pyscanf
|
|
25
|
+
|
|
26
|
+
Type-safe, `scanf`-style string parsing for Python 3.14+.
|
|
27
|
+
Distributed as `infinity_pyscanf`; imported as `scanf`.
|
|
28
|
+
|
|
29
|
+
Supply one converter type per `{}` placeholder as class type arguments, then
|
|
30
|
+
the template and the input string as call arguments:
|
|
31
|
+
|
|
32
|
+
```python
|
|
33
|
+
from scanf import scanf
|
|
34
|
+
|
|
35
|
+
a, b, c = scanf[int, float, str]("{} {} {}")("1 2.5 hello")
|
|
36
|
+
# a=1 (int), b=2.5 (float), c="hello" (str)
|
|
37
|
+
```
|
|
38
|
+
|
|
39
|
+
Parsers are reusable:
|
|
40
|
+
|
|
41
|
+
```python
|
|
42
|
+
parse_pair = scanf[int, int]("{} {}")
|
|
43
|
+
parse_pair("7 8") # (7, 8)
|
|
44
|
+
```
|
|
45
|
+
|
|
46
|
+
## Built-in converters
|
|
47
|
+
|
|
48
|
+
Built-in generic containers (`dict`, `list`, `tuple`, `set` and `frozenset`)
|
|
49
|
+
are parsed as Python literals via `ast.literal_eval`, and `bool` accepts
|
|
50
|
+
`true` / `false` / `1` / `0` (case-insensitive):
|
|
51
|
+
|
|
52
|
+
```python
|
|
53
|
+
from scanf import scanf
|
|
54
|
+
|
|
55
|
+
scanf[bool]("{}")("true") # (True,)
|
|
56
|
+
scanf[list[int]]("{}")("[1, 2, 3]") # ([1, 2, 3],)
|
|
57
|
+
```
|
|
58
|
+
|
|
59
|
+
For JSON-specific syntax (`true`, `false`, `null`) or any custom conversion,
|
|
60
|
+
attach a `Converter` spec:
|
|
61
|
+
|
|
62
|
+
```python
|
|
63
|
+
import json
|
|
64
|
+
from typing import Annotated
|
|
65
|
+
|
|
66
|
+
from scanf import Converter, scanf
|
|
67
|
+
|
|
68
|
+
parse_json = scanf[Annotated[dict[str, str], Converter(json.loads)]]("{}")
|
|
69
|
+
parse_json('{"key": "value"}') # ({"key": "value"},)
|
|
70
|
+
```
|
|
71
|
+
|
|
72
|
+
## Per-field configuration
|
|
73
|
+
|
|
74
|
+
Per-field configuration rides along as `Converter` metadata inside a
|
|
75
|
+
`typing.Annotated` converter: a conversion callable, a capture pattern, a
|
|
76
|
+
strip flag, `re` flags and a label:
|
|
77
|
+
|
|
78
|
+
```python
|
|
79
|
+
from typing import Annotated
|
|
80
|
+
|
|
81
|
+
from scanf import Converter, scanf
|
|
82
|
+
|
|
83
|
+
parse = scanf[
|
|
84
|
+
Annotated[int, Converter(int, pattern=r"\d+", name="age")],
|
|
85
|
+
Annotated[str, Converter(str, pattern=r'"[^"]*"', name="name")],
|
|
86
|
+
]("age={} name={}")
|
|
87
|
+
parse('age=42 name="kim"') # (42, '"kim"')
|
|
88
|
+
```
|
|
89
|
+
|
|
90
|
+
`typing.Annotated` wrappers around converters are stripped; their metadata is
|
|
91
|
+
ignored unless it contains a `Converter` spec. Several specs on one field
|
|
92
|
+
compete in declaration order: each branch carries its own pattern, the first
|
|
93
|
+
branch whose pattern matches wins, and if its converter then fails the whole
|
|
94
|
+
scan fails — a failing converter never falls back to a later branch.
|
|
95
|
+
|
|
96
|
+
## Template semantics
|
|
97
|
+
|
|
98
|
+
Each `{}` captures with the default `.+?` pattern:
|
|
99
|
+
|
|
100
|
+
* non-empty: at least one character;
|
|
101
|
+
* single-line: never matches a newline;
|
|
102
|
+
* lazy: takes the shortest text that still lets the rest of the template match.
|
|
103
|
+
|
|
104
|
+
A `Converter` spec can replace the pattern per field.
|
|
105
|
+
|
|
106
|
+
Whitespace runs in the template match any whitespace run in the input, and
|
|
107
|
+
surrounding whitespace of the input (and of each captured field) is ignored.
|
|
108
|
+
Literal braces in a template are written `{{` and `}}`; a single brace is an
|
|
109
|
+
error.
|
|
110
|
+
|
|
111
|
+
## Development
|
|
112
|
+
|
|
113
|
+
```bash
|
|
114
|
+
uv sync --extra dev
|
|
115
|
+
uv run pre-commit install # optional: run the gates on every commit
|
|
116
|
+
uv run pre-commit run --all-files # full local gate
|
|
117
|
+
```
|
|
118
|
+
|
|
119
|
+
The gate runs ruff (lint + format), pyright (strict), interrogate (docstring
|
|
120
|
+
coverage), pytest with a coverage floor, and an `example/` smoke run. CI
|
|
121
|
+
(GitHub Actions) runs the same gates plus a 3-OS test matrix.
|
|
122
|
+
|
|
123
|
+
`example/example.py` contains worked examples: field customization, JSON
|
|
124
|
+
parsing and a recursive scanner built with fixpoint / `alt` / `sep_by`
|
|
125
|
+
combinators.
|
|
@@ -0,0 +1,102 @@
|
|
|
1
|
+
# infinity_pyscanf
|
|
2
|
+
|
|
3
|
+
Type-safe, `scanf`-style string parsing for Python 3.14+.
|
|
4
|
+
Distributed as `infinity_pyscanf`; imported as `scanf`.
|
|
5
|
+
|
|
6
|
+
Supply one converter type per `{}` placeholder as class type arguments, then
|
|
7
|
+
the template and the input string as call arguments:
|
|
8
|
+
|
|
9
|
+
```python
|
|
10
|
+
from scanf import scanf
|
|
11
|
+
|
|
12
|
+
a, b, c = scanf[int, float, str]("{} {} {}")("1 2.5 hello")
|
|
13
|
+
# a=1 (int), b=2.5 (float), c="hello" (str)
|
|
14
|
+
```
|
|
15
|
+
|
|
16
|
+
Parsers are reusable:
|
|
17
|
+
|
|
18
|
+
```python
|
|
19
|
+
parse_pair = scanf[int, int]("{} {}")
|
|
20
|
+
parse_pair("7 8") # (7, 8)
|
|
21
|
+
```
|
|
22
|
+
|
|
23
|
+
## Built-in converters
|
|
24
|
+
|
|
25
|
+
Built-in generic containers (`dict`, `list`, `tuple`, `set` and `frozenset`)
|
|
26
|
+
are parsed as Python literals via `ast.literal_eval`, and `bool` accepts
|
|
27
|
+
`true` / `false` / `1` / `0` (case-insensitive):
|
|
28
|
+
|
|
29
|
+
```python
|
|
30
|
+
from scanf import scanf
|
|
31
|
+
|
|
32
|
+
scanf[bool]("{}")("true") # (True,)
|
|
33
|
+
scanf[list[int]]("{}")("[1, 2, 3]") # ([1, 2, 3],)
|
|
34
|
+
```
|
|
35
|
+
|
|
36
|
+
For JSON-specific syntax (`true`, `false`, `null`) or any custom conversion,
|
|
37
|
+
attach a `Converter` spec:
|
|
38
|
+
|
|
39
|
+
```python
|
|
40
|
+
import json
|
|
41
|
+
from typing import Annotated
|
|
42
|
+
|
|
43
|
+
from scanf import Converter, scanf
|
|
44
|
+
|
|
45
|
+
parse_json = scanf[Annotated[dict[str, str], Converter(json.loads)]]("{}")
|
|
46
|
+
parse_json('{"key": "value"}') # ({"key": "value"},)
|
|
47
|
+
```
|
|
48
|
+
|
|
49
|
+
## Per-field configuration
|
|
50
|
+
|
|
51
|
+
Per-field configuration rides along as `Converter` metadata inside a
|
|
52
|
+
`typing.Annotated` converter: a conversion callable, a capture pattern, a
|
|
53
|
+
strip flag, `re` flags and a label:
|
|
54
|
+
|
|
55
|
+
```python
|
|
56
|
+
from typing import Annotated
|
|
57
|
+
|
|
58
|
+
from scanf import Converter, scanf
|
|
59
|
+
|
|
60
|
+
parse = scanf[
|
|
61
|
+
Annotated[int, Converter(int, pattern=r"\d+", name="age")],
|
|
62
|
+
Annotated[str, Converter(str, pattern=r'"[^"]*"', name="name")],
|
|
63
|
+
]("age={} name={}")
|
|
64
|
+
parse('age=42 name="kim"') # (42, '"kim"')
|
|
65
|
+
```
|
|
66
|
+
|
|
67
|
+
`typing.Annotated` wrappers around converters are stripped; their metadata is
|
|
68
|
+
ignored unless it contains a `Converter` spec. Several specs on one field
|
|
69
|
+
compete in declaration order: each branch carries its own pattern, the first
|
|
70
|
+
branch whose pattern matches wins, and if its converter then fails the whole
|
|
71
|
+
scan fails — a failing converter never falls back to a later branch.
|
|
72
|
+
|
|
73
|
+
## Template semantics
|
|
74
|
+
|
|
75
|
+
Each `{}` captures with the default `.+?` pattern:
|
|
76
|
+
|
|
77
|
+
* non-empty: at least one character;
|
|
78
|
+
* single-line: never matches a newline;
|
|
79
|
+
* lazy: takes the shortest text that still lets the rest of the template match.
|
|
80
|
+
|
|
81
|
+
A `Converter` spec can replace the pattern per field.
|
|
82
|
+
|
|
83
|
+
Whitespace runs in the template match any whitespace run in the input, and
|
|
84
|
+
surrounding whitespace of the input (and of each captured field) is ignored.
|
|
85
|
+
Literal braces in a template are written `{{` and `}}`; a single brace is an
|
|
86
|
+
error.
|
|
87
|
+
|
|
88
|
+
## Development
|
|
89
|
+
|
|
90
|
+
```bash
|
|
91
|
+
uv sync --extra dev
|
|
92
|
+
uv run pre-commit install # optional: run the gates on every commit
|
|
93
|
+
uv run pre-commit run --all-files # full local gate
|
|
94
|
+
```
|
|
95
|
+
|
|
96
|
+
The gate runs ruff (lint + format), pyright (strict), interrogate (docstring
|
|
97
|
+
coverage), pytest with a coverage floor, and an `example/` smoke run. CI
|
|
98
|
+
(GitHub Actions) runs the same gates plus a 3-OS test matrix.
|
|
99
|
+
|
|
100
|
+
`example/example.py` contains worked examples: field customization, JSON
|
|
101
|
+
parsing and a recursive scanner built with fixpoint / `alt` / `sep_by`
|
|
102
|
+
combinators.
|
|
@@ -0,0 +1,86 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["uv_build>=0.11.21,<0.12.0"]
|
|
3
|
+
build-backend = "uv_build"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "infinity_pyscanf"
|
|
7
|
+
version = "0.1.0"
|
|
8
|
+
description = "Type-safe, scanf-style string parsing."
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
keywords = [
|
|
11
|
+
"scanf",
|
|
12
|
+
"parsing",
|
|
13
|
+
"parser",
|
|
14
|
+
"regex",
|
|
15
|
+
"type-safe",
|
|
16
|
+
]
|
|
17
|
+
classifiers = [
|
|
18
|
+
"Intended Audience :: Developers",
|
|
19
|
+
"Operating System :: OS Independent",
|
|
20
|
+
"Programming Language :: Python :: 3",
|
|
21
|
+
"Programming Language :: Python :: 3.14",
|
|
22
|
+
"Topic :: Software Development :: Libraries :: Python Modules",
|
|
23
|
+
"Topic :: Text Processing",
|
|
24
|
+
]
|
|
25
|
+
requires-python = ">=3.14"
|
|
26
|
+
dependencies = []
|
|
27
|
+
|
|
28
|
+
[project.optional-dependencies]
|
|
29
|
+
test = [
|
|
30
|
+
"pytest",
|
|
31
|
+
"pytest-cov",
|
|
32
|
+
]
|
|
33
|
+
dev = [
|
|
34
|
+
"interrogate",
|
|
35
|
+
"ruff",
|
|
36
|
+
"pyright",
|
|
37
|
+
"pre-commit",
|
|
38
|
+
"infinity_pyscanf[test]",
|
|
39
|
+
]
|
|
40
|
+
|
|
41
|
+
[tool.ruff]
|
|
42
|
+
target-version = "py314"
|
|
43
|
+
line-length = 88
|
|
44
|
+
src = ["src"]
|
|
45
|
+
|
|
46
|
+
[tool.ruff.lint]
|
|
47
|
+
select = [
|
|
48
|
+
"B",
|
|
49
|
+
"C4",
|
|
50
|
+
"E",
|
|
51
|
+
"F",
|
|
52
|
+
"I",
|
|
53
|
+
"RUF",
|
|
54
|
+
"SIM",
|
|
55
|
+
"UP",
|
|
56
|
+
"W",
|
|
57
|
+
]
|
|
58
|
+
|
|
59
|
+
[tool.ruff.lint.isort]
|
|
60
|
+
known-first-party = ["scanf"]
|
|
61
|
+
|
|
62
|
+
[tool.pytest.ini_options]
|
|
63
|
+
testpaths = ["tests"]
|
|
64
|
+
|
|
65
|
+
[tool.pyright]
|
|
66
|
+
typeCheckingMode = "strict"
|
|
67
|
+
pythonVersion = "3.14"
|
|
68
|
+
venvPath = "."
|
|
69
|
+
venv = ".venv"
|
|
70
|
+
|
|
71
|
+
[tool.interrogate]
|
|
72
|
+
fail-under = 90
|
|
73
|
+
ignore-init-method = true
|
|
74
|
+
ignore-init-module = true
|
|
75
|
+
ignore-magic = true
|
|
76
|
+
ignore-private = true
|
|
77
|
+
ignore-semiprivate = true
|
|
78
|
+
ignore-nested-functions = true
|
|
79
|
+
exclude = [
|
|
80
|
+
"tests",
|
|
81
|
+
"example",
|
|
82
|
+
]
|
|
83
|
+
|
|
84
|
+
[tool.uv.build-backend]
|
|
85
|
+
module-root = "src"
|
|
86
|
+
module-name = "scanf"
|
|
@@ -0,0 +1,83 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["uv_build>=0.11.21,<0.12.0"]
|
|
3
|
+
build-backend = "uv_build"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "infinity_pyscanf"
|
|
7
|
+
version = "0.1.0"
|
|
8
|
+
description = "Type-safe, scanf-style string parsing."
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
keywords = [
|
|
11
|
+
"scanf",
|
|
12
|
+
"parsing",
|
|
13
|
+
"parser",
|
|
14
|
+
"regex",
|
|
15
|
+
"type-safe",
|
|
16
|
+
]
|
|
17
|
+
classifiers = [
|
|
18
|
+
"Intended Audience :: Developers",
|
|
19
|
+
"Operating System :: OS Independent",
|
|
20
|
+
"Programming Language :: Python :: 3",
|
|
21
|
+
"Programming Language :: Python :: 3.14",
|
|
22
|
+
"Topic :: Software Development :: Libraries :: Python Modules",
|
|
23
|
+
"Topic :: Text Processing",
|
|
24
|
+
]
|
|
25
|
+
requires-python = ">=3.14"
|
|
26
|
+
dependencies = []
|
|
27
|
+
|
|
28
|
+
[project.optional-dependencies]
|
|
29
|
+
test = [
|
|
30
|
+
"pytest",
|
|
31
|
+
"pytest-cov",
|
|
32
|
+
]
|
|
33
|
+
dev = [
|
|
34
|
+
"interrogate",
|
|
35
|
+
"ruff",
|
|
36
|
+
"pyright",
|
|
37
|
+
"pre-commit",
|
|
38
|
+
"infinity_pyscanf[test]",
|
|
39
|
+
]
|
|
40
|
+
|
|
41
|
+
[tool.ruff]
|
|
42
|
+
target-version = "py314"
|
|
43
|
+
line-length = 88
|
|
44
|
+
src = ["src"]
|
|
45
|
+
|
|
46
|
+
[tool.ruff.lint]
|
|
47
|
+
select = [
|
|
48
|
+
"B", # flake8-bugbear
|
|
49
|
+
"C4", # flake8-comprehensions
|
|
50
|
+
"E", # pycodestyle (errors)
|
|
51
|
+
"F", # pyflakes
|
|
52
|
+
"I", # isort
|
|
53
|
+
"RUF", # Ruff-specific rules
|
|
54
|
+
"SIM", # flake8-simplify
|
|
55
|
+
"UP", # pyupgrade
|
|
56
|
+
"W", # pycodestyle (warnings)
|
|
57
|
+
]
|
|
58
|
+
|
|
59
|
+
[tool.ruff.lint.isort]
|
|
60
|
+
known-first-party = ["scanf"]
|
|
61
|
+
|
|
62
|
+
[tool.pytest.ini_options]
|
|
63
|
+
testpaths = ["tests"]
|
|
64
|
+
|
|
65
|
+
[tool.pyright]
|
|
66
|
+
typeCheckingMode = "strict"
|
|
67
|
+
pythonVersion = "3.14"
|
|
68
|
+
venvPath = "."
|
|
69
|
+
venv = ".venv"
|
|
70
|
+
|
|
71
|
+
[tool.interrogate]
|
|
72
|
+
fail-under = 90
|
|
73
|
+
ignore-init-method = true
|
|
74
|
+
ignore-init-module = true
|
|
75
|
+
ignore-magic = true
|
|
76
|
+
ignore-private = true
|
|
77
|
+
ignore-semiprivate = true
|
|
78
|
+
ignore-nested-functions = true
|
|
79
|
+
exclude = ["tests", "example"]
|
|
80
|
+
|
|
81
|
+
[tool.uv.build-backend]
|
|
82
|
+
module-root = "src"
|
|
83
|
+
module-name = "scanf"
|
|
@@ -0,0 +1,301 @@
|
|
|
1
|
+
"""Runtime engine for :mod:`scanf`: resolve field metas, then scan input.
|
|
2
|
+
|
|
3
|
+
Two responsibilities, with no type-level semantics:
|
|
4
|
+
|
|
5
|
+
* :func:`resolve_field` -- turn one field's meta (a bare declared converter
|
|
6
|
+
plus an optional :class:`Converter` spec) into a concrete converter configuration
|
|
7
|
+
(:class:`Field`);
|
|
8
|
+
* :class:`Template` -- validate a configuration, compile the template regex
|
|
9
|
+
and ``scan`` input strings into tuples.
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
import ast
|
|
13
|
+
import re
|
|
14
|
+
from collections.abc import Callable
|
|
15
|
+
from dataclasses import dataclass
|
|
16
|
+
from functools import lru_cache
|
|
17
|
+
from typing import Any, get_origin
|
|
18
|
+
|
|
19
|
+
# The default field pattern: non-empty, single-line (no DOTALL) and lazy.
|
|
20
|
+
DEFAULT_PATTERN = r".+?"
|
|
21
|
+
_PLACEHOLDER = "{}"
|
|
22
|
+
_WHITESPACE = re.compile(r"\s+")
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
class ScanError(ValueError):
|
|
26
|
+
"""Base failure while scanning: a ``MatchError`` or a ``ConvertError``."""
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
class MatchError(ScanError):
|
|
30
|
+
"""Input did not match the template (a soft failure).
|
|
31
|
+
|
|
32
|
+
Ordered-choice combinators fall through to the next alternative on this
|
|
33
|
+
one. A branch that matched but failed to convert raises ``ConvertError``
|
|
34
|
+
instead, which aborts the choice so its diagnostics are never swallowed.
|
|
35
|
+
"""
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
class ConvertError(ScanError):
|
|
39
|
+
"""A field matched but its converter failed (a hard failure).
|
|
40
|
+
|
|
41
|
+
The message carries the field number, the text and its offset; the
|
|
42
|
+
underlying error is chained through ``__cause__``.
|
|
43
|
+
"""
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
@dataclass(frozen=True)
|
|
47
|
+
class Converter[T]:
|
|
48
|
+
"""Per-field converter spec, passed as ``Annotated[T, Converter(...)]`` metadata.
|
|
49
|
+
|
|
50
|
+
``fn`` converts the matched text; ``pattern`` is the regex used to capture
|
|
51
|
+
this field (inserted into the template regex as-is); ``strip`` strips the
|
|
52
|
+
matched text before conversion; ``name`` labels the field in error messages
|
|
53
|
+
(defaults to the callable's name); ``flags`` are ``re`` flags applied to
|
|
54
|
+
``pattern``. Several specs on one field compete in declaration order: the
|
|
55
|
+
first branch whose pattern matches wins, and if its ``fn`` then raises, the
|
|
56
|
+
scan fails as a whole (no fallback to the remaining branches).
|
|
57
|
+
"""
|
|
58
|
+
|
|
59
|
+
fn: Callable[[str], T]
|
|
60
|
+
pattern: str = DEFAULT_PATTERN
|
|
61
|
+
strip: bool = True
|
|
62
|
+
name: str | None = None
|
|
63
|
+
flags: int = 0
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
@dataclass(frozen=True)
|
|
67
|
+
class Field:
|
|
68
|
+
"""One placeholder: converter branches competing in declaration order."""
|
|
69
|
+
|
|
70
|
+
converters: tuple[Converter[Any], ...]
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
def _converter_name(converter: object) -> str | None:
|
|
74
|
+
"""Return the display name of ``converter`` (used to label fields)."""
|
|
75
|
+
name = getattr(converter, "__name__", None)
|
|
76
|
+
if isinstance(name, str):
|
|
77
|
+
return name
|
|
78
|
+
origin = get_origin(converter)
|
|
79
|
+
if origin is not None:
|
|
80
|
+
name = getattr(origin, "__name__", None)
|
|
81
|
+
if isinstance(name, str):
|
|
82
|
+
return name
|
|
83
|
+
return None
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
def _label(converter: Converter[Any]) -> str:
|
|
87
|
+
"""Display name of one converter branch in error messages."""
|
|
88
|
+
return converter.name or _converter_name(converter.fn) or repr(converter.fn)
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
def _literal_converter(origin: type[Any]) -> Callable[[str], Any]:
|
|
92
|
+
return lambda text: origin(ast.literal_eval(text))
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
_BOOL_VALUES = {"true": True, "false": False, "1": True, "0": False}
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
def _bool_converter(text: str) -> bool:
|
|
99
|
+
value = _BOOL_VALUES.get(text.casefold())
|
|
100
|
+
if value is None:
|
|
101
|
+
raise ValueError(f"cannot parse {text!r} as bool")
|
|
102
|
+
return value
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
# Keyed by the origin *object* (not by name), so user classes that happen to be
|
|
106
|
+
# named ``dict``, ``list``, ``bool``, ... never pick up a built-in default.
|
|
107
|
+
_BUILTIN_CONVERTERS: dict[object, Callable[[str], Any]] = {
|
|
108
|
+
bool: _bool_converter,
|
|
109
|
+
dict: _literal_converter(dict),
|
|
110
|
+
frozenset: _literal_converter(frozenset),
|
|
111
|
+
list: _literal_converter(list),
|
|
112
|
+
set: _literal_converter(set),
|
|
113
|
+
tuple: _literal_converter(tuple),
|
|
114
|
+
}
|
|
115
|
+
|
|
116
|
+
|
|
117
|
+
def _default_converter(converter: object) -> Callable[[str], Any] | None:
|
|
118
|
+
"""Return the built-in default converter for supported built-in types."""
|
|
119
|
+
origin = get_origin(converter)
|
|
120
|
+
return _BUILTIN_CONVERTERS.get(origin if origin is not None else converter)
|
|
121
|
+
|
|
122
|
+
|
|
123
|
+
def resolve_field(declared: Any, convs: list[Converter[Any]]) -> Field:
|
|
124
|
+
"""Resolve one field's meta into its converter branches (meta -> converters).
|
|
125
|
+
|
|
126
|
+
User-provided ``Converter`` specs become the branches, in order. Without
|
|
127
|
+
any, one branch is synthesized from the declared type: the built-in default
|
|
128
|
+
for supported containers, otherwise the declared type used as a callable.
|
|
129
|
+
"""
|
|
130
|
+
if convs:
|
|
131
|
+
return Field(converters=tuple(convs))
|
|
132
|
+
branch: Converter[Any] = Converter(
|
|
133
|
+
fn=_default_converter(declared) or declared,
|
|
134
|
+
name=_converter_name(declared),
|
|
135
|
+
)
|
|
136
|
+
return Field(converters=(branch,))
|
|
137
|
+
|
|
138
|
+
|
|
139
|
+
def _literal_to_pattern(text: str) -> str:
|
|
140
|
+
"""Escape literal template text; whitespace runs become ``\\s+``."""
|
|
141
|
+
pieces: list[str] = []
|
|
142
|
+
position = 0
|
|
143
|
+
for match in _WHITESPACE.finditer(text):
|
|
144
|
+
pieces.append(re.escape(text[position : match.start()]))
|
|
145
|
+
pieces.append(r"\s+")
|
|
146
|
+
position = match.end()
|
|
147
|
+
pieces.append(re.escape(text[position:]))
|
|
148
|
+
return "".join(pieces)
|
|
149
|
+
|
|
150
|
+
|
|
151
|
+
_INLINE_FLAGS: tuple[tuple[int, str], ...] = (
|
|
152
|
+
(re.IGNORECASE, "i"),
|
|
153
|
+
(re.MULTILINE, "m"),
|
|
154
|
+
(re.DOTALL, "s"),
|
|
155
|
+
(re.VERBOSE, "x"),
|
|
156
|
+
(re.ASCII, "a"),
|
|
157
|
+
)
|
|
158
|
+
_INLINE_FLAG_MASK = re.IGNORECASE | re.MULTILINE | re.DOTALL | re.VERBOSE | re.ASCII
|
|
159
|
+
|
|
160
|
+
|
|
161
|
+
def _scoped(pattern: str, flags: int) -> str:
|
|
162
|
+
"""Wrap ``pattern`` in scoped inline flags (e.g. ``(?i:...)``)."""
|
|
163
|
+
if not flags:
|
|
164
|
+
return pattern
|
|
165
|
+
if flags & ~_INLINE_FLAG_MASK:
|
|
166
|
+
raise ValueError(f"unsupported re flags for a field pattern: {flags:#x}")
|
|
167
|
+
letters = "".join(letter for bit, letter in _INLINE_FLAGS if flags & bit)
|
|
168
|
+
return f"(?{letters}:{pattern})"
|
|
169
|
+
|
|
170
|
+
|
|
171
|
+
@lru_cache(maxsize=1024)
|
|
172
|
+
def _segments(template: str) -> tuple[str | None, ...]:
|
|
173
|
+
"""Split ``template`` into literal segments and ``None`` placeholders.
|
|
174
|
+
|
|
175
|
+
``{{`` and ``}}`` escape literal braces; a single brace is an error.
|
|
176
|
+
"""
|
|
177
|
+
segments: list[str | None] = []
|
|
178
|
+
literal: list[str] = []
|
|
179
|
+
text = template.strip()
|
|
180
|
+
position = 0
|
|
181
|
+
while position < len(text):
|
|
182
|
+
character = text[position]
|
|
183
|
+
if character == "{":
|
|
184
|
+
if text.startswith("{{", position):
|
|
185
|
+
literal.append("{")
|
|
186
|
+
position += 2
|
|
187
|
+
elif text.startswith(_PLACEHOLDER, position):
|
|
188
|
+
segments.append("".join(literal))
|
|
189
|
+
literal.clear()
|
|
190
|
+
segments.append(None)
|
|
191
|
+
position += 2
|
|
192
|
+
else:
|
|
193
|
+
raise ValueError(
|
|
194
|
+
f"single '{{' in template {template!r}; escape literal "
|
|
195
|
+
"braces as '{{' and '}}'"
|
|
196
|
+
)
|
|
197
|
+
elif character == "}":
|
|
198
|
+
if text.startswith("}}", position):
|
|
199
|
+
literal.append("}")
|
|
200
|
+
position += 2
|
|
201
|
+
else:
|
|
202
|
+
raise ValueError(
|
|
203
|
+
f"single '}}' in template {template!r}; escape literal "
|
|
204
|
+
"braces as '{{' and '}}'"
|
|
205
|
+
)
|
|
206
|
+
else:
|
|
207
|
+
literal.append(character)
|
|
208
|
+
position += 1
|
|
209
|
+
segments.append("".join(literal))
|
|
210
|
+
return tuple(segments)
|
|
211
|
+
|
|
212
|
+
|
|
213
|
+
def _branch_group(index: int, branch: int) -> str:
|
|
214
|
+
"""Regex group name capturing one branch's own match for a field."""
|
|
215
|
+
return f"s{index}b{branch}"
|
|
216
|
+
|
|
217
|
+
|
|
218
|
+
def _field_chunk(index: int, field: Field) -> str:
|
|
219
|
+
"""One field's regex: its converter branches as an ordered alternation."""
|
|
220
|
+
converters = field.converters
|
|
221
|
+
if len(converters) == 1:
|
|
222
|
+
single = converters[0]
|
|
223
|
+
return f"(?P<s{index}>{_scoped(single.pattern, single.flags)})"
|
|
224
|
+
branches = "|".join(
|
|
225
|
+
f"(?P<{_branch_group(index, branch)}>{_scoped(conv.pattern, conv.flags)})"
|
|
226
|
+
for branch, conv in enumerate(converters)
|
|
227
|
+
)
|
|
228
|
+
return f"(?P<s{index}>{branches})"
|
|
229
|
+
|
|
230
|
+
|
|
231
|
+
@lru_cache(maxsize=1024)
|
|
232
|
+
def _compile(template: str, fields: tuple[Field, ...]) -> re.Pattern[str]:
|
|
233
|
+
"""Compile ``template`` into a regex with one named group per placeholder."""
|
|
234
|
+
chunks: list[str] = []
|
|
235
|
+
placeholder = 0
|
|
236
|
+
for segment in _segments(template):
|
|
237
|
+
if segment is None:
|
|
238
|
+
chunks.append(_field_chunk(placeholder, fields[placeholder]))
|
|
239
|
+
placeholder += 1
|
|
240
|
+
else:
|
|
241
|
+
chunks.append(_literal_to_pattern(segment))
|
|
242
|
+
return re.compile("".join(chunks))
|
|
243
|
+
|
|
244
|
+
|
|
245
|
+
def _winner(match: re.Match[str], index: int, field: Field) -> int:
|
|
246
|
+
"""Index of the branch whose pattern produced the field's match."""
|
|
247
|
+
if len(field.converters) == 1:
|
|
248
|
+
return 0
|
|
249
|
+
for branch, _ in enumerate(field.converters):
|
|
250
|
+
if match.group(_branch_group(index, branch)) is not None:
|
|
251
|
+
return branch
|
|
252
|
+
raise AssertionError(f"no converter branch matched field {index}")
|
|
253
|
+
|
|
254
|
+
|
|
255
|
+
class Template:
|
|
256
|
+
"""A validated, compiled template, ready to scan input."""
|
|
257
|
+
|
|
258
|
+
def __init__(self, template: str, fields: tuple[Field, ...]) -> None:
|
|
259
|
+
# Fail fast on converters that cannot go into the compile cache.
|
|
260
|
+
hash(fields)
|
|
261
|
+
for field in fields:
|
|
262
|
+
if not field.converters:
|
|
263
|
+
raise ValueError("each field requires at least one converter")
|
|
264
|
+
for converter in field.converters:
|
|
265
|
+
if not callable(converter.fn):
|
|
266
|
+
raise TypeError(
|
|
267
|
+
f"converters must be callable, got {converter.fn!r}"
|
|
268
|
+
)
|
|
269
|
+
placeholders = sum(segment is None for segment in _segments(template))
|
|
270
|
+
if placeholders != len(fields):
|
|
271
|
+
raise ValueError(
|
|
272
|
+
f"template {template!r} has {placeholders} placeholder(s) "
|
|
273
|
+
f"but {len(fields)} converter type(s) were given"
|
|
274
|
+
)
|
|
275
|
+
self.template = template
|
|
276
|
+
self.fields = fields
|
|
277
|
+
self._pattern = _compile(template, fields)
|
|
278
|
+
|
|
279
|
+
def scan(self, input_value: str) -> tuple[Any, ...]:
|
|
280
|
+
"""Scan ``input_value``: match the template and convert each field."""
|
|
281
|
+
match = self._pattern.fullmatch(input_value.strip())
|
|
282
|
+
if match is None:
|
|
283
|
+
raise MatchError(
|
|
284
|
+
f"input {input_value!r} does not match template {self.template!r}"
|
|
285
|
+
)
|
|
286
|
+
values: list[Any] = []
|
|
287
|
+
for index, field in enumerate(self.fields):
|
|
288
|
+
converter = field.converters[_winner(match, index, field)]
|
|
289
|
+
raw = match.group(f"s{index}")
|
|
290
|
+
text = raw.strip() if converter.strip else raw
|
|
291
|
+
try:
|
|
292
|
+
values.append(converter.fn(text))
|
|
293
|
+
except Exception as error:
|
|
294
|
+
# Report the offset within the original input, not the stripped one.
|
|
295
|
+
shift = len(input_value) - len(input_value.lstrip())
|
|
296
|
+
offset = shift + match.start(f"s{index}")
|
|
297
|
+
raise ConvertError(
|
|
298
|
+
f"cannot convert field {index + 1} ({text!r}) at offset "
|
|
299
|
+
f"{offset} using {_label(converter)}"
|
|
300
|
+
) from error
|
|
301
|
+
return tuple(values)
|
|
File without changes
|
|
@@ -0,0 +1,111 @@
|
|
|
1
|
+
r"""Type-safe, ``scanf``-style string parsing.
|
|
2
|
+
|
|
3
|
+
Supply one converter type per ``{}`` placeholder as class type arguments, then
|
|
4
|
+
the template and the input string as call arguments::
|
|
5
|
+
|
|
6
|
+
from scanf import scanf
|
|
7
|
+
|
|
8
|
+
a, b, c = scanf[int, float, str]("{} {} {}")("1 2.5 hello")
|
|
9
|
+
# a=1 (int), b=2.5 (float), c="hello" (str)
|
|
10
|
+
|
|
11
|
+
Parsers are reusable. Fields accept per-field ``Converter`` specs and built-in
|
|
12
|
+
defaults (containers via ``ast.literal_eval``, ``bool`` literals); see the
|
|
13
|
+
project README for the full guide -- usage, semantics and examples.
|
|
14
|
+
"""
|
|
15
|
+
|
|
16
|
+
import types
|
|
17
|
+
from typing import Annotated, Any, cast, get_args, get_origin
|
|
18
|
+
|
|
19
|
+
from scanf.engine import (
|
|
20
|
+
Converter,
|
|
21
|
+
ConvertError,
|
|
22
|
+
Field,
|
|
23
|
+
MatchError,
|
|
24
|
+
ScanError,
|
|
25
|
+
Template,
|
|
26
|
+
resolve_field,
|
|
27
|
+
)
|
|
28
|
+
|
|
29
|
+
__all__ = ["ConvertError", "Converter", "MatchError", "ScanError", "scanf"]
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
def flatten_annotated(meta: Any) -> tuple[Any, list[Any]]:
|
|
33
|
+
"""Flatten nested ``Annotated`` wrappers; metadata ordered deep to shallow
|
|
34
|
+
(which is also the order converter branches compete in)."""
|
|
35
|
+
if get_origin(meta) is not Annotated:
|
|
36
|
+
return meta, []
|
|
37
|
+
args = get_args(meta)
|
|
38
|
+
base, metadata = flatten_annotated(args[0])
|
|
39
|
+
return base, [*metadata, *args[1:]]
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def build_field(meta: Any) -> Field:
|
|
43
|
+
"""Extract one declaration's Converter specs and resolve them via the engine."""
|
|
44
|
+
declared, metadata = flatten_annotated(meta)
|
|
45
|
+
converters: list[Converter[Any]] = [
|
|
46
|
+
cast(Converter[Any], item) for item in metadata if isinstance(item, Converter)
|
|
47
|
+
]
|
|
48
|
+
return resolve_field(declared, converters)
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
class scanf[*Meta]:
|
|
52
|
+
"""Typed ``scanf``: ``scanf[Meta1, Meta2, ...](template)(input_value)``.
|
|
53
|
+
|
|
54
|
+
Each ``Meta`` declares one field -- a converter type, or
|
|
55
|
+
``Annotated[converter, Converter(...)]`` for per-field configuration --
|
|
56
|
+
and determines the corresponding result element type. Several ``Converter``
|
|
57
|
+
specs on one field compete in order: the first matching branch wins. The
|
|
58
|
+
template and the input string are call arguments. Parsers are reusable.
|
|
59
|
+
|
|
60
|
+
The converter setup is validated eagerly when the parser is constructed:
|
|
61
|
+
non-callable converters and placeholder/converter-count mismatches raise
|
|
62
|
+
immediately.
|
|
63
|
+
"""
|
|
64
|
+
|
|
65
|
+
_engine: Template
|
|
66
|
+
|
|
67
|
+
def __init__(self, template: str, _engine: Template | None = None) -> None:
|
|
68
|
+
# Parsers are built through ``scanf[...]`` (see _ScanfAlias), which
|
|
69
|
+
# passes the ready engine in; direct construction is a misuse.
|
|
70
|
+
if _engine is None:
|
|
71
|
+
raise TypeError(
|
|
72
|
+
"scanf requires at least one converter type, e.g. scanf[int]('{}')"
|
|
73
|
+
)
|
|
74
|
+
self._engine = _engine
|
|
75
|
+
|
|
76
|
+
@classmethod
|
|
77
|
+
def __class_getitem__(cls, item: Any) -> Any:
|
|
78
|
+
"""Bind the field metas; see the class docstring.
|
|
79
|
+
|
|
80
|
+
Type checkers ignore this hook for generic classes and keep treating
|
|
81
|
+
``scanf[int, float]`` as a specialization. At runtime it returns a
|
|
82
|
+
``types.GenericAlias`` subclass carrying the metas; calling that alias
|
|
83
|
+
with the template is what validates and builds the parser.
|
|
84
|
+
"""
|
|
85
|
+
metas = cast(tuple[Any, ...], item if isinstance(item, tuple) else (item,))
|
|
86
|
+
if not metas:
|
|
87
|
+
raise TypeError(
|
|
88
|
+
"scanf requires at least one converter type, e.g. scanf[int]('{}')"
|
|
89
|
+
)
|
|
90
|
+
return _ScanfAlias(cls, metas)
|
|
91
|
+
|
|
92
|
+
def __call__(self, input_value: str) -> tuple[*Meta]:
|
|
93
|
+
"""Parse ``input_value``."""
|
|
94
|
+
return self._engine.scan(input_value)
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
class _ScanfAlias(types.GenericAlias):
|
|
98
|
+
"""Subclass of ``types.GenericAlias`` produced by ``scanf[...]``.
|
|
99
|
+
|
|
100
|
+
The field metas travel in the alias itself (``__args__``). Calling the
|
|
101
|
+
alias with the template is where both sides meet: it resolves each meta
|
|
102
|
+
into its converter, validates the configuration and constructs the parser,
|
|
103
|
+
setting ``__orig_class__`` the way typing would.
|
|
104
|
+
"""
|
|
105
|
+
|
|
106
|
+
def __call__(self, template: str) -> Any:
|
|
107
|
+
origin = cast(Any, self.__origin__)
|
|
108
|
+
fields: tuple[Field, ...] = tuple(build_field(meta) for meta in self.__args__)
|
|
109
|
+
instance = origin(template, Template(template, fields))
|
|
110
|
+
instance.__orig_class__ = self
|
|
111
|
+
return instance
|