infinity-data 1.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- infinity_data/__init__.py +41 -0
- infinity_data/emit/__init__.py +9 -0
- infinity_data/emit/converter.py +55 -0
- infinity_data/frontend.py +37 -0
- infinity_data/infra/__init__.py +4 -0
- infinity_data/infra/diagnostics.py +212 -0
- infinity_data/infra/file.py +83 -0
- infinity_data/infra/ll1_stream.py +71 -0
- infinity_data/infra/location.py +80 -0
- infinity_data/infra/path.py +46 -0
- infinity_data/parser/__init__.py +79 -0
- infinity_data/parser/diagnostics.py +92 -0
- infinity_data/parser/models.py +295 -0
- infinity_data/parser/parser.py +988 -0
- infinity_data/parser/token_stream.py +128 -0
- infinity_data/pipeline.py +252 -0
- infinity_data/sandbox/__init__.py +32 -0
- infinity_data/sandbox/config.py +61 -0
- infinity_data/sandbox/errors.py +106 -0
- infinity_data/sandbox/mediator.py +214 -0
- infinity_data/sandbox/schema.py +21 -0
- infinity_data/semantic/__init__.py +59 -0
- infinity_data/semantic/builder/__init__.py +29 -0
- infinity_data/semantic/builder/builder.py +523 -0
- infinity_data/semantic/builder/models.py +163 -0
- infinity_data/semantic/constraints.py +163 -0
- infinity_data/semantic/diagnostics.py +172 -0
- infinity_data/semantic/executor/__init__.py +10 -0
- infinity_data/semantic/executor/executor.py +259 -0
- infinity_data/semantic/registry/__init__.py +168 -0
- infinity_data/semantic/registry/_core.py +212 -0
- infinity_data/semantic/registry/dict_constraints.py +56 -0
- infinity_data/semantic/registry/general.py +362 -0
- infinity_data/semantic/registry/logic.py +134 -0
- infinity_data/semantic/registry/types.py +136 -0
- infinity_data/semantic/resolver/__init__.py +20 -0
- infinity_data/semantic/resolver/imports.py +191 -0
- infinity_data/semantic/resolver/models.py +67 -0
- infinity_data/semantic/resolver/resolver.py +345 -0
- infinity_data/tokenizer/__init__.py +16 -0
- infinity_data/tokenizer/char_stream.py +72 -0
- infinity_data/tokenizer/diagnostics.py +62 -0
- infinity_data/tokenizer/finalizer.py +271 -0
- infinity_data/tokenizer/models/__init__.py +17 -0
- infinity_data/tokenizer/models/raw_tokens.py +61 -0
- infinity_data/tokenizer/models/tokens.py +245 -0
- infinity_data/tokenizer/tokenizer.py +531 -0
- infinity_data-1.0.0.dist-info/METADATA +60 -0
- infinity_data-1.0.0.dist-info/RECORD +52 -0
- infinity_data-1.0.0.dist-info/WHEEL +5 -0
- infinity_data-1.0.0.dist-info/licenses/LICENSE +21 -0
- infinity_data-1.0.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,41 @@
|
|
|
1
|
+
"""InfinityData —— 声明式配置语言(.infd/.inft)的 Python 编译器库。
|
|
2
|
+
|
|
3
|
+
编译流水线:RawTokenizer → FinalTokenizer → Parser → AstBuilder → Executor → Converter
|
|
4
|
+
|
|
5
|
+
用法::
|
|
6
|
+
|
|
7
|
+
from infinity_data import load, safe_load, SandboxConfig, Schema
|
|
8
|
+
|
|
9
|
+
result = safe_load("app.infd")
|
|
10
|
+
if result.has_errors:
|
|
11
|
+
for d in result.diagnostics:
|
|
12
|
+
print(d.location, d.message)
|
|
13
|
+
else:
|
|
14
|
+
print(result.value)
|
|
15
|
+
"""
|
|
16
|
+
|
|
17
|
+
from infinity_data.infra.diagnostics import Diagnostic, Severity
|
|
18
|
+
from infinity_data.pipeline import (
|
|
19
|
+
CompilationResult,
|
|
20
|
+
check,
|
|
21
|
+
compile_document,
|
|
22
|
+
compile_source,
|
|
23
|
+
load,
|
|
24
|
+
safe_load,
|
|
25
|
+
)
|
|
26
|
+
from infinity_data.sandbox import SandboxConfig, SandboxError, Schema, SchemaError
|
|
27
|
+
|
|
28
|
+
__all__ = [
|
|
29
|
+
'compile_source',
|
|
30
|
+
'load',
|
|
31
|
+
'safe_load',
|
|
32
|
+
'check',
|
|
33
|
+
'compile_document',
|
|
34
|
+
'CompilationResult',
|
|
35
|
+
'SandboxConfig',
|
|
36
|
+
'SandboxError',
|
|
37
|
+
'Schema',
|
|
38
|
+
'SchemaError',
|
|
39
|
+
'Diagnostic',
|
|
40
|
+
'Severity',
|
|
41
|
+
]
|
|
@@ -0,0 +1,9 @@
|
|
|
1
|
+
"""产物发射层:StdAst → 宿主表示(Python dict / list / 标量)。
|
|
2
|
+
|
|
3
|
+
语义分析(semantic/)负责"值是否正确",本层负责"产物长什么样"。
|
|
4
|
+
M4 的 JSON/YAML/TOML 转换、M5 的 JSON Schema 生成均在此层扩展。
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from infinity_data.emit.converter import reduce_array, reduce_object, reduce_value
|
|
8
|
+
|
|
9
|
+
__all__ = ['reduce_array', 'reduce_object', 'reduce_value']
|
|
@@ -0,0 +1,55 @@
|
|
|
1
|
+
"""降维器:StdAst → 纯 Python 值(dict / list / 标量)。
|
|
2
|
+
|
|
3
|
+
- ``noexist`` 字段不出现在输出中
|
|
4
|
+
- ``null`` 字段保留键(值为 None),``keep_null=False`` 时跳过
|
|
5
|
+
- 浮点保持 :class:`decimal.Decimal`(无限精度);NaN / ±Infinity 以 Decimal 表示,
|
|
6
|
+
JSON/YAML 序列化时的特殊编码由 M4 转换层负责
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
from typing import Any
|
|
12
|
+
|
|
13
|
+
from infinity_data.semantic.builder.models import StdArray, StdLiteral, StdObject, StdValue
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
def reduce_object(obj: StdObject, *, keep_null: bool = True) -> dict[str, Any]:
|
|
17
|
+
"""将 StdObject 降维为 Python dict。"""
|
|
18
|
+
result: dict[str, Any] = {}
|
|
19
|
+
for f in obj.fields:
|
|
20
|
+
if f.value is None or f.is_noexist:
|
|
21
|
+
continue
|
|
22
|
+
if f.is_null:
|
|
23
|
+
if keep_null:
|
|
24
|
+
result[f.name] = None
|
|
25
|
+
continue
|
|
26
|
+
result[f.name] = reduce_value(f.value, keep_null=keep_null)
|
|
27
|
+
return result
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def reduce_array(arr: StdArray, *, keep_null: bool = True) -> list[Any]:
|
|
31
|
+
"""将 StdArray 降维为 Python list。"""
|
|
32
|
+
return [reduce_value(v, keep_null=keep_null) for v in arr.elements]
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def reduce_value(val: StdValue, *, keep_null: bool = True) -> Any:
|
|
36
|
+
"""将任意 StdValue 降维为 Python 原生值。"""
|
|
37
|
+
match val:
|
|
38
|
+
case StdLiteral():
|
|
39
|
+
return _reduce_literal(val)
|
|
40
|
+
case StdArray():
|
|
41
|
+
return reduce_array(val, keep_null=keep_null)
|
|
42
|
+
case StdObject():
|
|
43
|
+
return reduce_object(val, keep_null=keep_null)
|
|
44
|
+
raise TypeError(f'未知 StdValue 类型: {type(val)}')
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
# ── 内部 ────────────────────────────────────────────────
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def _reduce_literal(lit: StdLiteral) -> Any:
|
|
51
|
+
match lit.kind:
|
|
52
|
+
case 'null' | 'noexist':
|
|
53
|
+
return None
|
|
54
|
+
case _:
|
|
55
|
+
return lit.value
|
|
@@ -0,0 +1,37 @@
|
|
|
1
|
+
"""前端流水线:源码 File → RawAst Document + 前端诊断(容错收集)。
|
|
2
|
+
|
|
3
|
+
供 :mod:`pipeline`(主文件)与 :mod:`semantic.resolver`(外部模板文件)共用,
|
|
4
|
+
消除两处重复的 RawTokenizer → FinalTokenizer → Parser 组装。
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from infinity_data.infra.diagnostics import DiagnosticCollector
|
|
8
|
+
from infinity_data.infra.file import File
|
|
9
|
+
from infinity_data.parser import Document
|
|
10
|
+
from infinity_data.parser.parser import Parser
|
|
11
|
+
from infinity_data.tokenizer.finalizer import FinalTokenizer
|
|
12
|
+
from infinity_data.tokenizer.tokenizer import RawTokenizer
|
|
13
|
+
|
|
14
|
+
__all__ = ['parse_source']
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
def parse_source(
|
|
18
|
+
file: File,
|
|
19
|
+
collector: DiagnosticCollector | None = None,
|
|
20
|
+
) -> tuple[Document, DiagnosticCollector]:
|
|
21
|
+
"""词法 + 语法分析,返回 RawAst Document 与生效收集器。
|
|
22
|
+
|
|
23
|
+
容错:词法/语法错误经 :class:`DiagnosticCollector` 收集而非抛出。
|
|
24
|
+
传入 ``collector`` 时三阶段(RawTokenizer / FinalTokenizer / Parser)全程
|
|
25
|
+
复用同一收集器并**原样返回**(非副本);缺省时内部新建并返回。
|
|
26
|
+
返回值第二元素为生效收集器,可直接查询 errors / warnings / has_errors。
|
|
27
|
+
"""
|
|
28
|
+
if collector is None:
|
|
29
|
+
collector = DiagnosticCollector()
|
|
30
|
+
|
|
31
|
+
raw_tokens = RawTokenizer(file=file, error_collector=collector)
|
|
32
|
+
tokens = FinalTokenizer(raw_tokens, error_collector=collector)
|
|
33
|
+
parser = Parser(tokens, error_collector=collector)
|
|
34
|
+
doc = parser.parse()
|
|
35
|
+
|
|
36
|
+
# 词法/语法错误统一为 Diagnostic(纯数据,直接聚合)
|
|
37
|
+
return doc, collector
|
|
@@ -0,0 +1,212 @@
|
|
|
1
|
+
"""统一诊断模型(所有阶段共用)与容错收集器。
|
|
2
|
+
|
|
3
|
+
- :class:`Severity` / :class:`Diagnostic`:稳定错误码 + 结构化参数 + 渲染消息。
|
|
4
|
+
词法/语法/语义阶段统一使用;沙盒异常(见 :mod:`infinity_data.sandbox.errors`)
|
|
5
|
+
经 ``check()`` 边界转换为 Diagnostic。
|
|
6
|
+
- :class:`DiagnosticCollector`:前端容错收集器(词法/语法错误作为 Diagnostic
|
|
7
|
+
收集,从不抛异常)。
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
from collections.abc import Iterable, Iterator, Mapping
|
|
13
|
+
from dataclasses import dataclass, field
|
|
14
|
+
from enum import Enum
|
|
15
|
+
from typing import Any
|
|
16
|
+
|
|
17
|
+
from infinity_data.infra.location import SourceRange, format_location
|
|
18
|
+
|
|
19
|
+
__all__ = [
|
|
20
|
+
'DEFAULT_LANG',
|
|
21
|
+
'Severity',
|
|
22
|
+
'Diagnostic',
|
|
23
|
+
'DiagnosticDefine',
|
|
24
|
+
'diagnostic_define',
|
|
25
|
+
'register_diagnostic_define',
|
|
26
|
+
'render_message',
|
|
27
|
+
'DiagnosticCollector',
|
|
28
|
+
]
|
|
29
|
+
|
|
30
|
+
DEFAULT_LANG = 'zh'
|
|
31
|
+
"""默认渲染语言。"""
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
class Severity(Enum):
|
|
35
|
+
"""诊断严重级别。"""
|
|
36
|
+
|
|
37
|
+
ERROR = 'error'
|
|
38
|
+
WARNING = 'warning'
|
|
39
|
+
INFO = 'info'
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
@dataclass(frozen=True)
|
|
43
|
+
class Diagnostic:
|
|
44
|
+
"""统一诊断:稳定错误码 + 结构化参数 + 渲染消息。
|
|
45
|
+
|
|
46
|
+
- ``code``:稳定错误码(如 ``"template.undefined"``),测试/工具据此匹配
|
|
47
|
+
- ``params``:结构化参数;``message`` 为派生属性,由注册表按语言渲染
|
|
48
|
+
- ``lang``:渲染语言(None = 默认语言);模板缺失时回退默认语言
|
|
49
|
+
"""
|
|
50
|
+
|
|
51
|
+
severity: Severity
|
|
52
|
+
code: str
|
|
53
|
+
params: Mapping[str, Any] = field(default_factory=dict[str, Any])
|
|
54
|
+
source: SourceRange | None = None
|
|
55
|
+
path: str = ''
|
|
56
|
+
lang: str | None = None
|
|
57
|
+
|
|
58
|
+
@property
|
|
59
|
+
def location(self) -> str:
|
|
60
|
+
return format_location(self.source)
|
|
61
|
+
|
|
62
|
+
@property
|
|
63
|
+
def message(self) -> str:
|
|
64
|
+
"""按错误码 + 参数 + 语言渲染的人类可读消息。"""
|
|
65
|
+
return render_message(
|
|
66
|
+
self.code, self.params, location=self.location, path=self.path, lang=self.lang or DEFAULT_LANG
|
|
67
|
+
)
|
|
68
|
+
|
|
69
|
+
def sort_key(self) -> tuple[str, int, int]:
|
|
70
|
+
"""按源码位置排序。"""
|
|
71
|
+
if self.source is None:
|
|
72
|
+
return ('\uffff', 0, 0)
|
|
73
|
+
s = self.source.start
|
|
74
|
+
return (self.source.file.name, s.line, s.col)
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
@dataclass(frozen=True)
|
|
78
|
+
class DiagnosticDefine:
|
|
79
|
+
"""诊断定义:稳定错误码 + 语言模板。
|
|
80
|
+
|
|
81
|
+
- ``code``:稳定错误码(如 ``"template.undefined"``)
|
|
82
|
+
- ``template``:默认语言(``DEFAULT_LANG``)模板
|
|
83
|
+
- ``translations``:其他语言模板(语言码 → 模板),缺失时回退默认模板
|
|
84
|
+
"""
|
|
85
|
+
|
|
86
|
+
code: str
|
|
87
|
+
template: str
|
|
88
|
+
translations: Mapping[str, str] = field(default_factory=dict[str, str])
|
|
89
|
+
|
|
90
|
+
def template_for(self, lang: str = DEFAULT_LANG) -> str:
|
|
91
|
+
"""取指定语言的模板;缺失回退默认模板。"""
|
|
92
|
+
if lang == DEFAULT_LANG:
|
|
93
|
+
return self.template
|
|
94
|
+
return self.translations.get(lang, self.template)
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
_DIAGNOSTIC_DEFINE_REGISTRY: dict[str, DiagnosticDefine] = {}
|
|
98
|
+
"""诊断定义注册表(code → 定义)。"""
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
def diagnostic_define(code: str, template: str, **translations: str) -> DiagnosticDefine:
|
|
102
|
+
"""构造诊断定义(``translations`` 为其他语言模板,如 ``en=...``)。"""
|
|
103
|
+
return DiagnosticDefine(code=code, template=template, translations=translations)
|
|
104
|
+
|
|
105
|
+
|
|
106
|
+
def register_diagnostic_define(*defines: DiagnosticDefine) -> None:
|
|
107
|
+
"""注册诊断定义(重复 code 后者覆盖)。"""
|
|
108
|
+
for d in defines:
|
|
109
|
+
_DIAGNOSTIC_DEFINE_REGISTRY[d.code] = d
|
|
110
|
+
|
|
111
|
+
|
|
112
|
+
def registered_diagnostic_defines() -> Mapping[str, DiagnosticDefine]:
|
|
113
|
+
"""已注册定义表(只读视图)。"""
|
|
114
|
+
return _DIAGNOSTIC_DEFINE_REGISTRY
|
|
115
|
+
|
|
116
|
+
|
|
117
|
+
def render_message(
|
|
118
|
+
code: str,
|
|
119
|
+
params: Mapping[str, Any],
|
|
120
|
+
*,
|
|
121
|
+
location: str = '<unknown>',
|
|
122
|
+
path: str = '',
|
|
123
|
+
lang: str = DEFAULT_LANG,
|
|
124
|
+
) -> str:
|
|
125
|
+
"""按错误码 + 参数 + 语言渲染消息;未知错误码原样返回错误码本身。"""
|
|
126
|
+
d = _DIAGNOSTIC_DEFINE_REGISTRY.get(code)
|
|
127
|
+
if d is None:
|
|
128
|
+
return code
|
|
129
|
+
template = d.template_for(lang)
|
|
130
|
+
context: dict[str, Any] = {
|
|
131
|
+
'location': location,
|
|
132
|
+
'path': path,
|
|
133
|
+
'path_prefix': f'{path}: ' if path else '',
|
|
134
|
+
**params,
|
|
135
|
+
}
|
|
136
|
+
try:
|
|
137
|
+
return template.format(**context)
|
|
138
|
+
except (KeyError, IndexError, ValueError, AttributeError):
|
|
139
|
+
return code
|
|
140
|
+
|
|
141
|
+
|
|
142
|
+
class DiagnosticCollector:
|
|
143
|
+
"""诊断收集器:收集 :class:`Diagnostic`(词法/语法/语义阶段容错收集)。
|
|
144
|
+
|
|
145
|
+
词法/语法/语义阶段不抛异常:错误以 Diagnostic 形式收集,边界处直接聚合。
|
|
146
|
+
只读查询按 severity 分离:``errors`` / ``warnings`` / ``diagnostics``,
|
|
147
|
+
``has_errors`` 仅当含 ERROR 级别时成立(warning 不算错误)。
|
|
148
|
+
|
|
149
|
+
用法::
|
|
150
|
+
|
|
151
|
+
collector = DiagnosticCollector()
|
|
152
|
+
tokenizer = RawTokenizer(file, error_collector=collector)
|
|
153
|
+
...
|
|
154
|
+
for err in collector:
|
|
155
|
+
print(err.code)
|
|
156
|
+
"""
|
|
157
|
+
|
|
158
|
+
def __init__(self) -> None:
|
|
159
|
+
self._errors: list[Diagnostic] = []
|
|
160
|
+
|
|
161
|
+
# ── 写入 ──────────────────────────────────────────
|
|
162
|
+
|
|
163
|
+
def add(self, error: Diagnostic) -> None:
|
|
164
|
+
"""添加一个诊断。"""
|
|
165
|
+
self._errors.append(error)
|
|
166
|
+
|
|
167
|
+
def extend(self, errors: Iterable[Diagnostic]) -> None:
|
|
168
|
+
"""批量添加诊断。"""
|
|
169
|
+
self._errors.extend(errors)
|
|
170
|
+
|
|
171
|
+
# ── 只读查询(severity 感知:warning 与 error 语义分离) ──
|
|
172
|
+
|
|
173
|
+
@property
|
|
174
|
+
def diagnostics(self) -> list[Diagnostic]:
|
|
175
|
+
"""全部诊断的副本(含 ERROR / WARNING)。"""
|
|
176
|
+
return self._errors.copy()
|
|
177
|
+
|
|
178
|
+
@property
|
|
179
|
+
def errors(self) -> list[Diagnostic]:
|
|
180
|
+
"""仅 ERROR 级别诊断的副本(warning 不算错误)。"""
|
|
181
|
+
return [d for d in self._errors if d.severity is Severity.ERROR]
|
|
182
|
+
|
|
183
|
+
@property
|
|
184
|
+
def warnings(self) -> list[Diagnostic]:
|
|
185
|
+
"""仅 WARNING 级别诊断的副本。"""
|
|
186
|
+
return [d for d in self._errors if d.severity is Severity.WARNING]
|
|
187
|
+
|
|
188
|
+
@property
|
|
189
|
+
def has_errors(self) -> bool:
|
|
190
|
+
"""是否含 ERROR 级别诊断(与 :class:`CompilationResult` 的 has_errors 语义一致)。"""
|
|
191
|
+
return any(d.severity is Severity.ERROR for d in self._errors)
|
|
192
|
+
|
|
193
|
+
@property
|
|
194
|
+
def has_warnings(self) -> bool:
|
|
195
|
+
"""是否含 WARNING 级别诊断。"""
|
|
196
|
+
return any(d.severity is Severity.WARNING for d in self._errors)
|
|
197
|
+
|
|
198
|
+
# ── 容器协议 ──────────────────────────────────────
|
|
199
|
+
|
|
200
|
+
def __iter__(self) -> Iterator[Diagnostic]:
|
|
201
|
+
"""迭代所有已收集的诊断。"""
|
|
202
|
+
return iter(self._errors)
|
|
203
|
+
|
|
204
|
+
def __len__(self) -> int:
|
|
205
|
+
"""已收集诊断数量(容器协议)。
|
|
206
|
+
|
|
207
|
+
空收集器为 falsy(``bool(collector)`` = 是否收集到任何诊断,含 warning;
|
|
208
|
+
区别于 ``has_errors`` 的"是否含 ERROR")——
|
|
209
|
+
因此缺省构造**不可用 ``error_collector or DiagnosticCollector()``**:
|
|
210
|
+
空收集器会被 `or` 判定为假而静默替换,丢弃调用方传入的收集器。
|
|
211
|
+
"""
|
|
212
|
+
return len(self._errors)
|
|
@@ -0,0 +1,83 @@
|
|
|
1
|
+
"""源码来源抽象:磁盘文件与内存源码统一为 :class:`File`。
|
|
2
|
+
|
|
3
|
+
编译入口(``load`` / ``compile_source``)与模板导入链(``!from``)都消费 File:
|
|
4
|
+
|
|
5
|
+
- ``name``:诊断显示名(``file:line:col`` 中的 file)
|
|
6
|
+
- ``root_path``:相对导入解析基准(所在目录)
|
|
7
|
+
- ``read()``:源码内容
|
|
8
|
+
- ``chars()``:逐字符迭代流(词法分析输入)
|
|
9
|
+
- ``identity``:唯一身份(磁盘 = resolve 后绝对路径;内存 = ``路径:mem:内容hash``);模板身份(TemplateKey)基于它,含来源路径
|
|
10
|
+
- ``content_hash()``:内容 sha256 前缀(MemFile 身份的一部分 / 内容校验)
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
import hashlib
|
|
14
|
+
from collections.abc import Iterable
|
|
15
|
+
from dataclasses import dataclass
|
|
16
|
+
from pathlib import Path
|
|
17
|
+
|
|
18
|
+
__all__ = ['File', 'DiskFile', 'MemFile']
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
@dataclass(frozen=True)
|
|
22
|
+
class File:
|
|
23
|
+
"""源码来源基类。"""
|
|
24
|
+
|
|
25
|
+
name: str
|
|
26
|
+
root_path: Path
|
|
27
|
+
|
|
28
|
+
def read(self) -> str:
|
|
29
|
+
"""读取源码内容。"""
|
|
30
|
+
raise NotImplementedError
|
|
31
|
+
|
|
32
|
+
def chars(self) -> Iterable[str]:
|
|
33
|
+
"""逐字符迭代流(词法分析输入)。"""
|
|
34
|
+
return iter(self.read())
|
|
35
|
+
|
|
36
|
+
@property
|
|
37
|
+
def identity(self) -> str:
|
|
38
|
+
"""唯一身份(循环导入防护 / 模板身份(TemplateKey)基础,含来源路径)。"""
|
|
39
|
+
raise NotImplementedError
|
|
40
|
+
|
|
41
|
+
def content_hash(self) -> str:
|
|
42
|
+
"""内容 sha256 前缀(MemFile 身份的一部分;内容校验,非模板身份本身)。"""
|
|
43
|
+
return hashlib.sha256(self.read().encode('utf-8')).hexdigest()[:12]
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
@dataclass(frozen=True)
|
|
47
|
+
class DiskFile(File):
|
|
48
|
+
"""磁盘文件。path 为单一事实来源;name/root_path 由构造方从它派生。"""
|
|
49
|
+
|
|
50
|
+
@property
|
|
51
|
+
def path(self) -> Path:
|
|
52
|
+
return Path(self.name)
|
|
53
|
+
|
|
54
|
+
def read(self) -> str:
|
|
55
|
+
return self.path.read_text(encoding='utf-8')
|
|
56
|
+
|
|
57
|
+
@property
|
|
58
|
+
def identity(self) -> str:
|
|
59
|
+
return str(self.path.resolve())
|
|
60
|
+
|
|
61
|
+
@classmethod
|
|
62
|
+
def from_fullpath(cls, fullpath: str | Path) -> 'DiskFile':
|
|
63
|
+
"""从完整路径构造(与调用点一致的规范化入口)。"""
|
|
64
|
+
p = Path(fullpath)
|
|
65
|
+
return cls(name=str(p), root_path=p.parent)
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
@dataclass(frozen=True)
|
|
69
|
+
class MemFile(File):
|
|
70
|
+
"""内存源码(测试/嵌入式场景)。身份 = 根路径:mem:内容hash(含路径)。"""
|
|
71
|
+
|
|
72
|
+
content: str
|
|
73
|
+
|
|
74
|
+
def read(self) -> str:
|
|
75
|
+
return self.content
|
|
76
|
+
|
|
77
|
+
def chars(self) -> Iterable[str]:
|
|
78
|
+
"""逐字符迭代流(O(1) 构造,直连内容)。"""
|
|
79
|
+
return iter(self.content)
|
|
80
|
+
|
|
81
|
+
@property
|
|
82
|
+
def identity(self) -> str:
|
|
83
|
+
return str(self.root_path.resolve()) + ':mem:' + self.content_hash()
|
|
@@ -0,0 +1,71 @@
|
|
|
1
|
+
"""泛型 LL(1) 流包装器 —— 对任意 Iterable[T] 提供单元素预读能力。"""
|
|
2
|
+
|
|
3
|
+
from collections.abc import Iterable, Iterator
|
|
4
|
+
from typing import Generic, TypeVar
|
|
5
|
+
|
|
6
|
+
|
|
7
|
+
class UnSetType:
|
|
8
|
+
def __repr__(self) -> str:
|
|
9
|
+
return 'UnSet'
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
UnSet = UnSetType()
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
class NoNextType:
|
|
16
|
+
def __repr__(self) -> str:
|
|
17
|
+
return 'NoNext'
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
NoNext = NoNextType()
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
T = TypeVar('T')
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
class LL1Stream(Generic[T]):
|
|
27
|
+
"""泛型 LL(1) 流:对任意 Iterable[T] 提供单元素预读"""
|
|
28
|
+
|
|
29
|
+
def __init__(self, source: Iterable[T]) -> None:
|
|
30
|
+
self._iter: Iterator[T] | None = None
|
|
31
|
+
self._next: T | NoNextType | UnSetType = UnSet
|
|
32
|
+
self._source: Iterable[T] = source
|
|
33
|
+
|
|
34
|
+
def peek(self) -> T | NoNextType:
|
|
35
|
+
"""返回当前预读元素,首次访问时自动懒初始化。"""
|
|
36
|
+
self._ensure_buf()
|
|
37
|
+
assert not isinstance(self._next, UnSetType)
|
|
38
|
+
return self._next
|
|
39
|
+
|
|
40
|
+
def eof(self) -> bool:
|
|
41
|
+
"""是否已到达末尾(首次访问时自动懒初始化)。"""
|
|
42
|
+
self._ensure_buf()
|
|
43
|
+
return isinstance(self._next, NoNextType)
|
|
44
|
+
|
|
45
|
+
def advance(self) -> T:
|
|
46
|
+
"""消费当前元素,步进并预读下一个。"""
|
|
47
|
+
self._ensure_buf()
|
|
48
|
+
item = self._next
|
|
49
|
+
if isinstance(item, NoNextType):
|
|
50
|
+
raise IndexError('无法在 EOF 之后继续推进流')
|
|
51
|
+
self._pre_read()
|
|
52
|
+
assert not isinstance(item, UnSetType)
|
|
53
|
+
self._on_advance(item)
|
|
54
|
+
return item
|
|
55
|
+
|
|
56
|
+
def _ensure_buf(self) -> None:
|
|
57
|
+
"""懒初始化:首次访问时创建迭代器并预读第一个元素。"""
|
|
58
|
+
if self._iter is None:
|
|
59
|
+
self._iter = iter(self._source)
|
|
60
|
+
if self._next is UnSet:
|
|
61
|
+
self._pre_read()
|
|
62
|
+
|
|
63
|
+
def _pre_read(self) -> None:
|
|
64
|
+
"""从 _iter 预读下一个元素到 _next。"""
|
|
65
|
+
assert self._iter is not None
|
|
66
|
+
try:
|
|
67
|
+
self._next = next(self._iter)
|
|
68
|
+
except StopIteration:
|
|
69
|
+
self._next = NoNext
|
|
70
|
+
|
|
71
|
+
def _on_advance(self, item: T) -> None: ...
|
|
@@ -0,0 +1,80 @@
|
|
|
1
|
+
"""源码位置类型(所有阶段共用)。"""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from dataclasses import dataclass
|
|
6
|
+
from pathlib import Path
|
|
7
|
+
|
|
8
|
+
from infinity_data.infra.file import File
|
|
9
|
+
|
|
10
|
+
__all__ = ['SourceInfo', 'SourceRange', 'format_location']
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
def format_location(source: SourceRange | None) -> str:
|
|
14
|
+
"""格式化源码位置(无位置时为 ``<unknown>``)。
|
|
15
|
+
|
|
16
|
+
- 零宽 range(start == end)→ ``file:line:col``
|
|
17
|
+
- 非零宽 range → ``file:line:col-line:col``(起止区间)
|
|
18
|
+
|
|
19
|
+
:class:`Diagnostic` 与沙盒异常共用,消除重复的格式化逻辑。
|
|
20
|
+
"""
|
|
21
|
+
if source is None:
|
|
22
|
+
return '<unknown>'
|
|
23
|
+
s = source.start
|
|
24
|
+
base = f'{source.file.name}:{s.line}:{s.col}'
|
|
25
|
+
if s == source.end:
|
|
26
|
+
return base
|
|
27
|
+
e = source.end
|
|
28
|
+
return f'{base}-{e.line}:{e.col}'
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
class _UnknownFile(File):
|
|
32
|
+
"""无源码来源的占位(仅 :meth:`SourceRange.empty` 使用,位置域私有概念)。"""
|
|
33
|
+
|
|
34
|
+
def read(self) -> str:
|
|
35
|
+
return ''
|
|
36
|
+
|
|
37
|
+
@property
|
|
38
|
+
def identity(self) -> str:
|
|
39
|
+
return '<unknown>'
|
|
40
|
+
|
|
41
|
+
def content_hash(self) -> str:
|
|
42
|
+
return '<unknown>'
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
_UNKNOWN_FILE = _UnknownFile(name='<unknown>', root_path=Path())
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
@dataclass
|
|
49
|
+
class SourceInfo:
|
|
50
|
+
"""源码位置信息(纯位置,不含来源)。"""
|
|
51
|
+
|
|
52
|
+
line: int
|
|
53
|
+
col: int
|
|
54
|
+
index: int
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
@dataclass
|
|
58
|
+
class SourceRange:
|
|
59
|
+
"""源码位置范围(区间):来源文件 + 起止位置。
|
|
60
|
+
|
|
61
|
+
词法阶段使用零宽 range(start == end)。
|
|
62
|
+
"""
|
|
63
|
+
|
|
64
|
+
file: File
|
|
65
|
+
start: SourceInfo
|
|
66
|
+
end: SourceInfo
|
|
67
|
+
|
|
68
|
+
@classmethod
|
|
69
|
+
def empty(cls) -> SourceRange:
|
|
70
|
+
"""无来源占位(错误恢复、合成 token)。"""
|
|
71
|
+
return cls(
|
|
72
|
+
file=_UNKNOWN_FILE,
|
|
73
|
+
start=SourceInfo(line=0, col=0, index=0),
|
|
74
|
+
end=SourceInfo(line=0, col=0, index=0),
|
|
75
|
+
)
|
|
76
|
+
|
|
77
|
+
@classmethod
|
|
78
|
+
def at(cls, file: File, pos: SourceInfo) -> SourceRange:
|
|
79
|
+
"""单点位置 → 零宽 range(词法阶段错误定位用)。"""
|
|
80
|
+
return cls(file=file, start=pos, end=pos)
|
|
@@ -0,0 +1,46 @@
|
|
|
1
|
+
"""跨平台路径通行:语言内 POSIX 路径 ↔ 当前平台原生路径。
|
|
2
|
+
|
|
3
|
+
语言约定(见 neo_desg.md 3.2):
|
|
4
|
+
- 语言内**只允许** POSIX 风格路径(``/`` 分割),Windows 盘符写作 ``/c/...``(小写单字母)
|
|
5
|
+
- 访问文件系统时按当前平台**自动映射**,同一份 .infd 在 linux/windows/mac 均可用
|
|
6
|
+
|
|
7
|
+
函数:
|
|
8
|
+
- :func:`posix_to_native`:语言内 POSIX 路径 → 原生 :class:`Path`(Windows 上盘符映射)
|
|
9
|
+
- :func:`native_to_posix`:原生路径 → 语言内 POSIX 形式(glob 白名单匹配用)
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
from __future__ import annotations
|
|
13
|
+
|
|
14
|
+
import sys
|
|
15
|
+
from pathlib import Path, PurePosixPath, PureWindowsPath
|
|
16
|
+
|
|
17
|
+
__all__ = ['posix_to_native', 'native_to_posix']
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def posix_to_native(path: str, *, platform: str | None = None) -> Path:
|
|
21
|
+
"""语言内 POSIX 路径 → 当前平台原生 Path。
|
|
22
|
+
|
|
23
|
+
- POSIX(linux/mac):原样使用
|
|
24
|
+
- Windows:``/c/foo/bar`` → ``C:/foo/bar``(盘符转大写);相对路径由 pathlib 转换分隔符
|
|
25
|
+
- ``platform`` 可注入(默认 ``sys.platform``),便于跨平台单测
|
|
26
|
+
"""
|
|
27
|
+
if (platform or sys.platform).lower().startswith('win'):
|
|
28
|
+
pure = PurePosixPath(path)
|
|
29
|
+
if pure.is_absolute() and len(pure.parts) >= 2 and len(pure.parts[1]) == 1 and pure.parts[1].isalpha():
|
|
30
|
+
drive = pure.parts[1].upper() + ':'
|
|
31
|
+
rest = PurePosixPath(*pure.parts[2:]).as_posix()
|
|
32
|
+
return Path(f'{drive}/{rest}')
|
|
33
|
+
return Path(path)
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def native_to_posix(path: Path) -> str:
|
|
37
|
+
"""原生路径 → 语言内 POSIX 形式(``/`` 分割;Windows 盘符转 ``/c/`` 小写)。
|
|
38
|
+
|
|
39
|
+
glob 白名单按语言内 POSIX 形式匹配,保证跨平台一致。
|
|
40
|
+
"""
|
|
41
|
+
win = PureWindowsPath(path)
|
|
42
|
+
if win.drive:
|
|
43
|
+
drive_letter = win.drive[0].lower()
|
|
44
|
+
rest = PurePosixPath(*win.parts[1:]).as_posix()
|
|
45
|
+
return f'/{drive_letter}/{rest}'
|
|
46
|
+
return path.as_posix()
|