infinity-data 1.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- infinity_data/__init__.py +41 -0
- infinity_data/emit/__init__.py +9 -0
- infinity_data/emit/converter.py +55 -0
- infinity_data/frontend.py +37 -0
- infinity_data/infra/__init__.py +4 -0
- infinity_data/infra/diagnostics.py +212 -0
- infinity_data/infra/file.py +83 -0
- infinity_data/infra/ll1_stream.py +71 -0
- infinity_data/infra/location.py +80 -0
- infinity_data/infra/path.py +46 -0
- infinity_data/parser/__init__.py +79 -0
- infinity_data/parser/diagnostics.py +92 -0
- infinity_data/parser/models.py +295 -0
- infinity_data/parser/parser.py +988 -0
- infinity_data/parser/token_stream.py +128 -0
- infinity_data/pipeline.py +252 -0
- infinity_data/sandbox/__init__.py +32 -0
- infinity_data/sandbox/config.py +61 -0
- infinity_data/sandbox/errors.py +106 -0
- infinity_data/sandbox/mediator.py +214 -0
- infinity_data/sandbox/schema.py +21 -0
- infinity_data/semantic/__init__.py +59 -0
- infinity_data/semantic/builder/__init__.py +29 -0
- infinity_data/semantic/builder/builder.py +523 -0
- infinity_data/semantic/builder/models.py +163 -0
- infinity_data/semantic/constraints.py +163 -0
- infinity_data/semantic/diagnostics.py +172 -0
- infinity_data/semantic/executor/__init__.py +10 -0
- infinity_data/semantic/executor/executor.py +259 -0
- infinity_data/semantic/registry/__init__.py +168 -0
- infinity_data/semantic/registry/_core.py +212 -0
- infinity_data/semantic/registry/dict_constraints.py +56 -0
- infinity_data/semantic/registry/general.py +362 -0
- infinity_data/semantic/registry/logic.py +134 -0
- infinity_data/semantic/registry/types.py +136 -0
- infinity_data/semantic/resolver/__init__.py +20 -0
- infinity_data/semantic/resolver/imports.py +191 -0
- infinity_data/semantic/resolver/models.py +67 -0
- infinity_data/semantic/resolver/resolver.py +345 -0
- infinity_data/tokenizer/__init__.py +16 -0
- infinity_data/tokenizer/char_stream.py +72 -0
- infinity_data/tokenizer/diagnostics.py +62 -0
- infinity_data/tokenizer/finalizer.py +271 -0
- infinity_data/tokenizer/models/__init__.py +17 -0
- infinity_data/tokenizer/models/raw_tokens.py +61 -0
- infinity_data/tokenizer/models/tokens.py +245 -0
- infinity_data/tokenizer/tokenizer.py +531 -0
- infinity_data-1.0.0.dist-info/METADATA +60 -0
- infinity_data-1.0.0.dist-info/RECORD +52 -0
- infinity_data-1.0.0.dist-info/WHEEL +5 -0
- infinity_data-1.0.0.dist-info/licenses/LICENSE +21 -0
- infinity_data-1.0.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,191 @@
|
|
|
1
|
+
"""导入语句求解:``!env`` / ``!file`` / ``!from`` 的系统访问。
|
|
2
|
+
|
|
3
|
+
所有系统访问经 :class:`Sandbox` 中介:
|
|
4
|
+
|
|
5
|
+
- ``!env import``:变量经 ``Sandbox.getenv`` 授权查询
|
|
6
|
+
- ``!file``:数据文件经 ``Sandbox.open_file`` 产出 File 后解析
|
|
7
|
+
- ``!from``(模板导入):模板文件经 ``Sandbox.open_template`` 产出 File,
|
|
8
|
+
模板定义的实际加载由 Phase 1 的 :class:`TemplateGraphResolver` 完成
|
|
9
|
+
|
|
10
|
+
本层产出 ``$`` 引用命名空间(alias → Python 值);诊断直接写入调用方注入的
|
|
11
|
+
共享 :class:`DiagnosticCollector`(与 resolver / builder 的收集器模式统一)。
|
|
12
|
+
纯数据依赖:不引用任何 Phase 2 对象。
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
from __future__ import annotations
|
|
16
|
+
|
|
17
|
+
import json
|
|
18
|
+
import tomllib
|
|
19
|
+
from pathlib import Path
|
|
20
|
+
from typing import Any
|
|
21
|
+
|
|
22
|
+
from infinity_data.infra.diagnostics import Diagnostic, DiagnosticCollector, Severity
|
|
23
|
+
from infinity_data.infra.file import File
|
|
24
|
+
from infinity_data.parser import (
|
|
25
|
+
Document,
|
|
26
|
+
EnvImportStmt,
|
|
27
|
+
FileImportStmt,
|
|
28
|
+
JsonPathIndex,
|
|
29
|
+
JsonPathKey,
|
|
30
|
+
)
|
|
31
|
+
from infinity_data.sandbox import Sandbox, SandboxConfig
|
|
32
|
+
from infinity_data.tokenizer.models.raw_tokens import SourceRange
|
|
33
|
+
|
|
34
|
+
_FORMAT_MAP: dict[str, str] = {
|
|
35
|
+
'.json': 'json',
|
|
36
|
+
'.yaml': 'yaml',
|
|
37
|
+
'.yml': 'yaml',
|
|
38
|
+
'.toml': 'toml',
|
|
39
|
+
}
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
class ImportResolver:
|
|
43
|
+
"""解析导入语句,产出 ``$`` 引用命名空间(alias → Python 值)。
|
|
44
|
+
|
|
45
|
+
Args:
|
|
46
|
+
sandbox: 沙盒中介(授权 / 拒绝一切系统访问)。None = 零信任 deny_all。
|
|
47
|
+
"""
|
|
48
|
+
|
|
49
|
+
def __init__(self, *, sandbox: Sandbox | None = None) -> None:
|
|
50
|
+
# 零信任默认:未提供沙盒时拒绝一切系统访问(库默认 deny_all)
|
|
51
|
+
self._sandbox = sandbox or Sandbox(config=SandboxConfig.deny_all(), base_dir=Path.cwd())
|
|
52
|
+
|
|
53
|
+
@property
|
|
54
|
+
def sandbox(self) -> Sandbox:
|
|
55
|
+
return self._sandbox
|
|
56
|
+
|
|
57
|
+
@property
|
|
58
|
+
def base_dir(self) -> Path:
|
|
59
|
+
return self._sandbox.base_dir
|
|
60
|
+
|
|
61
|
+
def resolve(self, doc: Document, collector: DiagnosticCollector) -> dict[str, Any]:
|
|
62
|
+
"""解析所有导入语句(env/file),返回 namespace;诊断写入 ``collector``。"""
|
|
63
|
+
namespace: dict[str, Any] = {}
|
|
64
|
+
for stmt in doc.statements:
|
|
65
|
+
if isinstance(stmt, EnvImportStmt):
|
|
66
|
+
self._resolve_env(stmt, namespace, collector)
|
|
67
|
+
elif isinstance(stmt, FileImportStmt):
|
|
68
|
+
self._resolve_file(stmt, namespace, collector)
|
|
69
|
+
return namespace
|
|
70
|
+
|
|
71
|
+
def _bind(
|
|
72
|
+
self,
|
|
73
|
+
namespace: dict[str, Any],
|
|
74
|
+
name: str,
|
|
75
|
+
value: Any,
|
|
76
|
+
collector: DiagnosticCollector,
|
|
77
|
+
source: SourceRange | None,
|
|
78
|
+
) -> None:
|
|
79
|
+
"""绑定 ``$`` 命名空间条目;重复 alias → ERROR 并拒绝覆盖(保留先到者)。
|
|
80
|
+
|
|
81
|
+
与模板 scope 一致:``$`` 命名空间内不允许隐式的"后者覆盖前者"。
|
|
82
|
+
"""
|
|
83
|
+
if name in namespace:
|
|
84
|
+
collector.add(Diagnostic(Severity.ERROR, 'namespace.duplicate', {'name': name}, source))
|
|
85
|
+
return
|
|
86
|
+
namespace[name] = value
|
|
87
|
+
|
|
88
|
+
# ── 各类导入 ──────────────────────────────────────
|
|
89
|
+
|
|
90
|
+
def _resolve_env(
|
|
91
|
+
self,
|
|
92
|
+
stmt: EnvImportStmt,
|
|
93
|
+
namespace: dict[str, Any],
|
|
94
|
+
collector: DiagnosticCollector,
|
|
95
|
+
) -> None:
|
|
96
|
+
"""!env import NAME [as NEW_NAME]
|
|
97
|
+
|
|
98
|
+
未授权环境变量**总是失败**(无论 strict):Sandbox.getenv 直接抛
|
|
99
|
+
:class:`SandboxError`,不会退化为空字符串注入。
|
|
100
|
+
"""
|
|
101
|
+
name = stmt.alias or stmt.name
|
|
102
|
+
self._bind(namespace, name, self._sandbox.getenv(stmt.name, source=stmt.source), collector, stmt.source)
|
|
103
|
+
|
|
104
|
+
def _resolve_file(
|
|
105
|
+
self,
|
|
106
|
+
stmt: FileImportStmt,
|
|
107
|
+
namespace: dict[str, Any],
|
|
108
|
+
collector: DiagnosticCollector,
|
|
109
|
+
) -> None:
|
|
110
|
+
"""!file "path" [as fmt] import .path.to.key as alias, ..."""
|
|
111
|
+
file = self._sandbox.open_file(stmt.file_path, source=stmt.source)
|
|
112
|
+
if file is None:
|
|
113
|
+
collector.add(Diagnostic(Severity.WARNING, 'import.file_denied', {'path_src': stmt.file_path}, stmt.source))
|
|
114
|
+
return
|
|
115
|
+
|
|
116
|
+
fmt = stmt.format or _FORMAT_MAP.get(Path(stmt.file_path).suffix.lower(), 'json')
|
|
117
|
+
try:
|
|
118
|
+
text = file.read()
|
|
119
|
+
except OSError:
|
|
120
|
+
collector.add(Diagnostic(Severity.WARNING, 'import.file_missing', {'name': file.name}, stmt.source))
|
|
121
|
+
return
|
|
122
|
+
|
|
123
|
+
data = self._parse_data(text, fmt, collector, stmt.source)
|
|
124
|
+
if data is None:
|
|
125
|
+
return
|
|
126
|
+
|
|
127
|
+
for item in stmt.imports:
|
|
128
|
+
try:
|
|
129
|
+
value = self._resolve_json_path(data, item.json_path)
|
|
130
|
+
except (KeyError, IndexError, TypeError):
|
|
131
|
+
collector.add(Diagnostic(Severity.WARNING, 'import.path_failed', {'name': file.name}, item.source))
|
|
132
|
+
continue
|
|
133
|
+
self._bind(namespace, item.alias, value, collector, item.source)
|
|
134
|
+
|
|
135
|
+
# ── 模板导入路径解析(!from 由 TemplateGraphResolver 使用)──
|
|
136
|
+
|
|
137
|
+
def resolve_template_path(
|
|
138
|
+
self,
|
|
139
|
+
from_path: str,
|
|
140
|
+
*,
|
|
141
|
+
base_dir: Path | None,
|
|
142
|
+
source: SourceRange | None,
|
|
143
|
+
collector: DiagnosticCollector,
|
|
144
|
+
) -> File | None:
|
|
145
|
+
"""!from 目标:经沙盒授权产出 File(相对路径以导入所在文件目录解析)。"""
|
|
146
|
+
file = self._sandbox.open_template(from_path, base_dir=base_dir, source=source)
|
|
147
|
+
if file is None:
|
|
148
|
+
collector.add(Diagnostic(Severity.WARNING, 'import.template_denied', {'path_src': from_path}, source))
|
|
149
|
+
return file
|
|
150
|
+
|
|
151
|
+
# ── 辅助 ──────────────────────────────────────────
|
|
152
|
+
|
|
153
|
+
def _parse_data(
|
|
154
|
+
self,
|
|
155
|
+
text: str,
|
|
156
|
+
fmt: str,
|
|
157
|
+
collector: DiagnosticCollector,
|
|
158
|
+
source: SourceRange,
|
|
159
|
+
) -> Any | None:
|
|
160
|
+
"""按格式解析数据内容(文本 loads)。"""
|
|
161
|
+
try:
|
|
162
|
+
if fmt == 'json':
|
|
163
|
+
return json.loads(text)
|
|
164
|
+
if fmt in ('yaml', 'yml'):
|
|
165
|
+
try:
|
|
166
|
+
import yaml # pyright: ignore[reportMissingModuleSource]
|
|
167
|
+
except ImportError:
|
|
168
|
+
collector.add(Diagnostic(Severity.WARNING, 'import.yaml_missing', {}, source))
|
|
169
|
+
return None
|
|
170
|
+
return yaml.safe_load(text)
|
|
171
|
+
if fmt == 'toml':
|
|
172
|
+
return tomllib.loads(text)
|
|
173
|
+
collector.add(Diagnostic(Severity.WARNING, 'import.unsupported_format', {'format': fmt}, source))
|
|
174
|
+
return None
|
|
175
|
+
except Exception as e:
|
|
176
|
+
collector.add(Diagnostic(Severity.ERROR, 'import.parse_failed', {'error': e}, source))
|
|
177
|
+
return None
|
|
178
|
+
|
|
179
|
+
# ── 辅助 ──────────────────────────────────────────
|
|
180
|
+
|
|
181
|
+
@staticmethod
|
|
182
|
+
def _resolve_json_path(data: Any, segments: list[JsonPathKey | JsonPathIndex]) -> Any:
|
|
183
|
+
"""按结构化路径段定位数据;空路径 = 整个文件。"""
|
|
184
|
+
current = data
|
|
185
|
+
for seg in segments:
|
|
186
|
+
match seg:
|
|
187
|
+
case JsonPathKey(key=k):
|
|
188
|
+
current = current[k]
|
|
189
|
+
case JsonPathIndex(index=i):
|
|
190
|
+
current = current[i]
|
|
191
|
+
return current
|
|
@@ -0,0 +1,67 @@
|
|
|
1
|
+
"""Phase 1(导入求解)数据模型:模板身份、可见名表与解析上下文。
|
|
2
|
+
|
|
3
|
+
本子模块**只定义数据**,不包含任何解析逻辑(解析器见 :mod:`resolver`)。
|
|
4
|
+
Phase 2(构建 / 执行)通过 :class:`ResolvedContext` 消费本层产物——
|
|
5
|
+
子模块间仅经数据模型依赖,无对象引用。
|
|
6
|
+
|
|
7
|
+
- ``TemplateKey``:模板真名(来源文件身份 + 本地名)
|
|
8
|
+
- ``Scope``:文件级可见名表(可见名 → 真名)
|
|
9
|
+
- ``ResolvedContext``:Phase 1 完整产物(模板图 + 可见名表 + 数据命名空间)
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
from __future__ import annotations
|
|
13
|
+
|
|
14
|
+
from dataclasses import dataclass
|
|
15
|
+
from typing import Any
|
|
16
|
+
|
|
17
|
+
from infinity_data.parser import TemplateDef
|
|
18
|
+
|
|
19
|
+
__all__ = ['ResolvedContext', 'Scope', 'TemplateKey']
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
@dataclass(frozen=True)
|
|
23
|
+
class TemplateKey:
|
|
24
|
+
"""模板唯一身份:来源文件身份 + 模板本地名。
|
|
25
|
+
|
|
26
|
+
- ``identity``:来源文件身份(磁盘 = resolve 绝对路径;内存 = ``路径:mem:内容hash``)
|
|
27
|
+
- ``name``:模板在来源文件中的本地名(诊断显示用)
|
|
28
|
+
|
|
29
|
+
身份含来源路径:不同路径的文件即使内容相同也是不同模板身份——模板内部
|
|
30
|
+
``!from`` 按定义文件所在目录解析,内容相同的文件其依赖语义可能不同,
|
|
31
|
+
不能互相覆盖(纯内容寻址无法表达这一区别)。
|
|
32
|
+
|
|
33
|
+
frozen 保证可哈希,直接作为模板表等映射的键。
|
|
34
|
+
"""
|
|
35
|
+
|
|
36
|
+
identity: str
|
|
37
|
+
name: str
|
|
38
|
+
|
|
39
|
+
def __str__(self) -> str:
|
|
40
|
+
return f'{self.identity}:{self.name}'
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
Scope = dict[str, TemplateKey]
|
|
44
|
+
"""文件级可见名表:可见名 → 模板真名(:class:`TemplateKey`)。"""
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
@dataclass(frozen=True)
|
|
48
|
+
class ResolvedContext:
|
|
49
|
+
"""导入求解(Phase 1)产物:模板图 + 可见名表 + 数据命名空间。
|
|
50
|
+
|
|
51
|
+
由 :class:`infinity_data.semantic.resolver.TemplateGraphResolver` 产出,
|
|
52
|
+
供 Phase 2a(构建)经数据模型消费。
|
|
53
|
+
只含名字与模板定义,不含任何约束执行结果(约束求值属 Phase 2),
|
|
54
|
+
也不含诊断——诊断经共享 :class:`DiagnosticCollector` 收集(流水线单一收集器)。
|
|
55
|
+
|
|
56
|
+
- ``templates``:全部已加载模板(本地 + ``!from`` 导入)
|
|
57
|
+
- ``template_scopes``:每个模板定义点的可见名表(展开/校验按定义点可见性解析)
|
|
58
|
+
- ``root_scope``:入口文件可见名表(可见名 → :class:`TemplateKey`)
|
|
59
|
+
- ``schema_scope``:schema.from_file 隐式导入的可见名表(无则 None)
|
|
60
|
+
- ``namespace``:``$`` 引用命名空间(``!env`` / ``!file`` 解析结果)
|
|
61
|
+
"""
|
|
62
|
+
|
|
63
|
+
templates: dict[TemplateKey, TemplateDef]
|
|
64
|
+
template_scopes: dict[TemplateKey, Scope]
|
|
65
|
+
root_scope: Scope
|
|
66
|
+
schema_scope: Scope | None
|
|
67
|
+
namespace: dict[str, Any]
|
|
@@ -0,0 +1,345 @@
|
|
|
1
|
+
"""模板图求解器(Phase 1):构建模板图、可见名表与数据命名空间。
|
|
2
|
+
|
|
3
|
+
将「模板导入(``!from``)、数据导入(``!env`` / ``!file``)、模板定义收集」
|
|
4
|
+
从语义分析中独立出来,产出不可变的 :class:`ResolvedContext` 供
|
|
5
|
+
Phase 2a(:class:`~infinity_data.semantic.builder.AstBuilder`)消费。
|
|
6
|
+
|
|
7
|
+
- 本层**不执行任何约束**:只解析名字、加载模板定义、构建 scope;
|
|
8
|
+
模板展开 / 约束求值 / schema 校验全部留在 Phase 2。
|
|
9
|
+
- ``resolve()`` 幂等:同一输入产出等价上下文,不依赖调用历史;
|
|
10
|
+
外部文件解析结果可经 ``parse_cache`` 跨调用复用(增量编译 / LSP)。
|
|
11
|
+
- 遮蔽检查只读查询注册表的内置约束名(不触发约束执行)。
|
|
12
|
+
- 诊断写入调用方注入的共享 :class:`DiagnosticCollector`(流水线单一收集器),
|
|
13
|
+
本层不持有诊断列表、不产出诊断数据。
|
|
14
|
+
"""
|
|
15
|
+
|
|
16
|
+
from __future__ import annotations
|
|
17
|
+
|
|
18
|
+
from pathlib import Path
|
|
19
|
+
|
|
20
|
+
from infinity_data.frontend import parse_source
|
|
21
|
+
from infinity_data.infra.diagnostics import Diagnostic, DiagnosticCollector, Severity
|
|
22
|
+
from infinity_data.infra.file import File
|
|
23
|
+
from infinity_data.parser import (
|
|
24
|
+
Document,
|
|
25
|
+
TemplateDef,
|
|
26
|
+
TemplateImportItem,
|
|
27
|
+
TemplateImportStmt,
|
|
28
|
+
)
|
|
29
|
+
from infinity_data.sandbox import Schema
|
|
30
|
+
from infinity_data.semantic.registry import ConstraintRegistry
|
|
31
|
+
from infinity_data.semantic.resolver.imports import ImportResolver
|
|
32
|
+
from infinity_data.semantic.resolver.models import ResolvedContext, Scope, TemplateKey
|
|
33
|
+
from infinity_data.tokenizer.models.raw_tokens import SourceRange
|
|
34
|
+
|
|
35
|
+
MAX_IMPORT_DEPTH = 32
|
|
36
|
+
"""模板导入递归深度上限(防止循环导入无限递归)。"""
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
class TemplateGraphResolver:
|
|
40
|
+
"""模板图求解器(Phase 1):递归加载导入、构建 scope、解析数据导入。
|
|
41
|
+
|
|
42
|
+
Args:
|
|
43
|
+
registry: 约束注册表(仅用于内置约束名的遮蔽检查,不执行约束)
|
|
44
|
+
import_resolver: 数据 / 模板导入路径解析(沙盒授权)
|
|
45
|
+
schema: 顶层 schema(``from_file`` 隐式导入)
|
|
46
|
+
parse_cache: 可选外部文件解析缓存(identity → Document)。
|
|
47
|
+
传入后跨 ``resolve()`` 复用,文件不变时跳过重复词法/语法分析。
|
|
48
|
+
"""
|
|
49
|
+
|
|
50
|
+
def __init__(
|
|
51
|
+
self,
|
|
52
|
+
*,
|
|
53
|
+
registry: ConstraintRegistry | None = None,
|
|
54
|
+
import_resolver: ImportResolver | None = None,
|
|
55
|
+
schema: Schema | None = None,
|
|
56
|
+
parse_cache: dict[str, Document] | None = None,
|
|
57
|
+
) -> None:
|
|
58
|
+
self._registry = registry or ConstraintRegistry()
|
|
59
|
+
self._imports = import_resolver or ImportResolver()
|
|
60
|
+
self._schema = schema
|
|
61
|
+
self._parse_cache = parse_cache
|
|
62
|
+
# 工作状态(每次 resolve 重置)
|
|
63
|
+
self._templates: dict[TemplateKey, TemplateDef] = {}
|
|
64
|
+
self._template_scopes: dict[TemplateKey, Scope] = {}
|
|
65
|
+
self._scopes_by_file: dict[str, Scope] = {} # 文件 identity → 已构建 scope(循环导入防护)
|
|
66
|
+
self._root_file: File | None = None
|
|
67
|
+
self._root_scope: Scope = {}
|
|
68
|
+
self._root_local_names: set[str] = set()
|
|
69
|
+
self._schema_scope: Scope | None = None
|
|
70
|
+
# 本次 resolve 的共享诊断收集器(流水线单一收集器,resolve() 注入)
|
|
71
|
+
self._collector: DiagnosticCollector = DiagnosticCollector()
|
|
72
|
+
|
|
73
|
+
def resolve(self, doc: Document, file: File, collector: DiagnosticCollector) -> ResolvedContext:
|
|
74
|
+
"""求解导入,返回不可变上下文(幂等:同一输入产出等价结果)。
|
|
75
|
+
|
|
76
|
+
诊断(``import.*`` / ``template.*`` 域)写入 ``collector``。
|
|
77
|
+
"""
|
|
78
|
+
self._reset(file, collector)
|
|
79
|
+
|
|
80
|
+
# 收集本地模板定义(key 键控,身份含来源文件路径)
|
|
81
|
+
self._collect_templates(doc, file.identity)
|
|
82
|
+
|
|
83
|
+
# 解析模板导入(!from,含 schema.from_file 隐式导入)→ 构建主文件 scope
|
|
84
|
+
root_scope = self._load_imported_templates(doc)
|
|
85
|
+
|
|
86
|
+
# 解析数据导入语句(!env / !file)→ $ 引用命名空间
|
|
87
|
+
namespace = self._imports.resolve(doc, self._collector)
|
|
88
|
+
|
|
89
|
+
return ResolvedContext(
|
|
90
|
+
templates=dict(self._templates),
|
|
91
|
+
template_scopes=dict(self._template_scopes),
|
|
92
|
+
root_scope=root_scope, # 与 template_scopes 内的定义点 scope 同一对象
|
|
93
|
+
schema_scope=self._schema_scope,
|
|
94
|
+
namespace=dict(namespace),
|
|
95
|
+
)
|
|
96
|
+
|
|
97
|
+
@property
|
|
98
|
+
def registry(self) -> ConstraintRegistry:
|
|
99
|
+
"""共享约束注册表(Phase 2 复用同一注册表:模板即约束注册 / 执行)。"""
|
|
100
|
+
return self._registry
|
|
101
|
+
|
|
102
|
+
@property
|
|
103
|
+
def schema(self) -> Schema | None:
|
|
104
|
+
"""顶层 schema(Phase 2b 顶层校验复用同一实例)。"""
|
|
105
|
+
return self._schema
|
|
106
|
+
|
|
107
|
+
def _reset(self, file: File, collector: DiagnosticCollector) -> None:
|
|
108
|
+
self._templates = {}
|
|
109
|
+
self._template_scopes = {}
|
|
110
|
+
self._scopes_by_file = {}
|
|
111
|
+
self._root_file = file
|
|
112
|
+
self._root_scope = {}
|
|
113
|
+
self._root_local_names = set()
|
|
114
|
+
self._schema_scope = None
|
|
115
|
+
self._collector = collector
|
|
116
|
+
|
|
117
|
+
# ═══════════════════════════════════════════════════════
|
|
118
|
+
# 模板收集
|
|
119
|
+
# ═══════════════════════════════════════════════════════
|
|
120
|
+
|
|
121
|
+
def _check_template_name_conflict(self, name: str, source: SourceRange | None) -> bool:
|
|
122
|
+
"""模板名与已注册约束(内置/自定义)同名 → ERROR 并返回 True。
|
|
123
|
+
|
|
124
|
+
模板即约束:定义 ``~int`` / ``~range`` 会遮蔽同名内置约束(``int`` 类型
|
|
125
|
+
标注、``range(1, 100)`` 调用等语义被静默劫持),因此同名模板禁止定义,
|
|
126
|
+
内置约束保持可用。
|
|
127
|
+
"""
|
|
128
|
+
if name in self._registry.names:
|
|
129
|
+
self._collector.add(Diagnostic(Severity.ERROR, 'template.shadows_builtin', {'template': name}, source))
|
|
130
|
+
return True
|
|
131
|
+
return False
|
|
132
|
+
|
|
133
|
+
def _check_required_order(self, stmt: TemplateDef) -> None:
|
|
134
|
+
"""模板内部校验:必填字段必须全部在可选字段之前。
|
|
135
|
+
|
|
136
|
+
例外:``positional=false`` 的模板不接受位置参数,字段顺序不影响绑定,
|
|
137
|
+
允许必填与可选交错。
|
|
138
|
+
"""
|
|
139
|
+
if not stmt.config.positional:
|
|
140
|
+
return
|
|
141
|
+
seen_optional = False
|
|
142
|
+
for tf in stmt.fields:
|
|
143
|
+
if tf.default_value is None:
|
|
144
|
+
if seen_optional:
|
|
145
|
+
self._collector.add(
|
|
146
|
+
Diagnostic(
|
|
147
|
+
Severity.ERROR,
|
|
148
|
+
'template.required_order',
|
|
149
|
+
{'template': stmt.name, 'field': tf.name},
|
|
150
|
+
tf.source,
|
|
151
|
+
)
|
|
152
|
+
)
|
|
153
|
+
else:
|
|
154
|
+
seen_optional = True
|
|
155
|
+
|
|
156
|
+
def _collect_templates(self, doc: Document, root_identity: str) -> None:
|
|
157
|
+
for stmt in doc.statements:
|
|
158
|
+
if not isinstance(stmt, TemplateDef):
|
|
159
|
+
continue
|
|
160
|
+
rejected = self._check_template_name_conflict(stmt.name, stmt.source)
|
|
161
|
+
key = TemplateKey(identity=root_identity, name=stmt.name)
|
|
162
|
+
if not rejected and key in self._templates:
|
|
163
|
+
self._collector.add(
|
|
164
|
+
Diagnostic(Severity.ERROR, 'template.duplicate', {'template': stmt.name}, stmt.source)
|
|
165
|
+
)
|
|
166
|
+
rejected = True
|
|
167
|
+
# 无论是否被拒绝都校验内部(一次暴露所有错误,避免多轮修复)
|
|
168
|
+
self._check_required_order(stmt)
|
|
169
|
+
if rejected:
|
|
170
|
+
continue # 保留首次定义,拒绝隐式的"后者覆盖前者"
|
|
171
|
+
self._templates[key] = stmt
|
|
172
|
+
self._root_local_names.add(stmt.name)
|
|
173
|
+
|
|
174
|
+
# ═══════════════════════════════════════════════════════
|
|
175
|
+
# 模板导入(!from)
|
|
176
|
+
# ═══════════════════════════════════════════════════════
|
|
177
|
+
|
|
178
|
+
def _map_import_items(
|
|
179
|
+
self,
|
|
180
|
+
items: list[TemplateImportItem],
|
|
181
|
+
dep_scope: Scope,
|
|
182
|
+
scope: Scope,
|
|
183
|
+
local_names: set[str],
|
|
184
|
+
) -> None:
|
|
185
|
+
"""把 ``!from`` 的导入项映射进目标 scope;冲突一律 ERROR。
|
|
186
|
+
|
|
187
|
+
- 导入文件中不存在该模板 → ERROR
|
|
188
|
+
- 可见名与文件内定义同名 → ERROR(与文件内定义冲突)
|
|
189
|
+
- 可见名已存在(重复导入)→ ERROR,保留先到者(拒绝隐式覆盖)
|
|
190
|
+
"""
|
|
191
|
+
for item in items:
|
|
192
|
+
dep_key = dep_scope.get(item.name)
|
|
193
|
+
if dep_key is None:
|
|
194
|
+
self._collector.add(
|
|
195
|
+
Diagnostic(Severity.ERROR, 'template.import_not_found', {'template': item.name}, item.source)
|
|
196
|
+
)
|
|
197
|
+
continue
|
|
198
|
+
visible = item.alias or item.name
|
|
199
|
+
if visible in scope:
|
|
200
|
+
if visible in local_names:
|
|
201
|
+
self._collector.add(
|
|
202
|
+
Diagnostic(Severity.ERROR, 'template.import_conflict_local', {'visible': visible}, item.source)
|
|
203
|
+
)
|
|
204
|
+
else:
|
|
205
|
+
self._collector.add(
|
|
206
|
+
Diagnostic(Severity.ERROR, 'template.import_duplicate', {'visible': visible}, item.source)
|
|
207
|
+
)
|
|
208
|
+
else:
|
|
209
|
+
scope[visible] = dep_key
|
|
210
|
+
|
|
211
|
+
def _load_imported_templates(self, doc: Document) -> Scope:
|
|
212
|
+
"""构建主文件 scope(含 schema.from_file 隐式导入)。"""
|
|
213
|
+
assert self._root_file is not None
|
|
214
|
+
root_id = self._root_file.identity
|
|
215
|
+
loaded: set[str] = set()
|
|
216
|
+
root_scope: Scope = {tpl.name: key for key, tpl in self._templates.items() if key.identity == root_id}
|
|
217
|
+
|
|
218
|
+
# schema.from_file 隐式导入:独立 scope 供顶层校验使用
|
|
219
|
+
if self._schema is not None and self._schema.from_file:
|
|
220
|
+
self._schema_scope = self._import_template_path(
|
|
221
|
+
self._schema.from_file,
|
|
222
|
+
base_dir=self._imports.base_dir,
|
|
223
|
+
source=None,
|
|
224
|
+
loaded=loaded,
|
|
225
|
+
depth=0,
|
|
226
|
+
)
|
|
227
|
+
|
|
228
|
+
for stmt in doc.statements:
|
|
229
|
+
if not isinstance(stmt, TemplateImportStmt):
|
|
230
|
+
continue
|
|
231
|
+
dep_scope = self._import_template_path(
|
|
232
|
+
stmt.from_path,
|
|
233
|
+
base_dir=self._imports.base_dir,
|
|
234
|
+
source=stmt.source,
|
|
235
|
+
loaded=loaded,
|
|
236
|
+
depth=0,
|
|
237
|
+
)
|
|
238
|
+
self._map_import_items(stmt.items, dep_scope, root_scope, self._root_local_names)
|
|
239
|
+
# 主文件本地模板的 scope 登记(模板展开/约束校验按此解析名字)
|
|
240
|
+
for key in self._templates:
|
|
241
|
+
if key.identity == root_id:
|
|
242
|
+
self._template_scopes[key] = root_scope
|
|
243
|
+
return root_scope
|
|
244
|
+
|
|
245
|
+
def _import_template_path(
|
|
246
|
+
self,
|
|
247
|
+
from_path: str,
|
|
248
|
+
*,
|
|
249
|
+
base_dir: Path,
|
|
250
|
+
source: SourceRange | None,
|
|
251
|
+
loaded: set[str],
|
|
252
|
+
depth: int,
|
|
253
|
+
) -> Scope:
|
|
254
|
+
"""加载单个模板文件,返回该文件的可见 scope(递归解析嵌套 !from)。"""
|
|
255
|
+
if depth > MAX_IMPORT_DEPTH:
|
|
256
|
+
self._collector.add(
|
|
257
|
+
Diagnostic(
|
|
258
|
+
Severity.ERROR, 'template.import_depth', {'max': MAX_IMPORT_DEPTH, 'path_src': from_path}, source
|
|
259
|
+
)
|
|
260
|
+
)
|
|
261
|
+
return {}
|
|
262
|
+
|
|
263
|
+
file = self._imports.resolve_template_path(
|
|
264
|
+
from_path,
|
|
265
|
+
base_dir=base_dir,
|
|
266
|
+
source=source,
|
|
267
|
+
collector=self._collector,
|
|
268
|
+
)
|
|
269
|
+
if file is None:
|
|
270
|
+
return {}
|
|
271
|
+
|
|
272
|
+
file_id = file.identity
|
|
273
|
+
if file_id in loaded:
|
|
274
|
+
# 循环导入:返回已构建的本地名部分(本地模板先注册)
|
|
275
|
+
return self._scopes_by_file.get(file_id, {})
|
|
276
|
+
loaded.add(file_id)
|
|
277
|
+
|
|
278
|
+
try:
|
|
279
|
+
_ = file.content_hash() # 触发内容读取;身份不含内容,仍需校验文件可读
|
|
280
|
+
except OSError as e:
|
|
281
|
+
self._collector.add(
|
|
282
|
+
Diagnostic(Severity.ERROR, 'template.read_failed', {'file': file.name, 'error': e}, source)
|
|
283
|
+
)
|
|
284
|
+
return {}
|
|
285
|
+
|
|
286
|
+
imported_doc = self._parse_document(file)
|
|
287
|
+
|
|
288
|
+
# 1) 本地模板:先注册(循环导入时依赖文件的本地名部分已可见)
|
|
289
|
+
# 身份含来源文件路径:不同路径的文件即使内容相同也是不同模板身份——
|
|
290
|
+
# 模板内部 !from 按定义文件所在目录解析,内容相同的文件其依赖语义
|
|
291
|
+
# 可能不同,不能互相覆盖(纯内容寻址无法表达这一区别)
|
|
292
|
+
scope: Scope = {}
|
|
293
|
+
local_names: set[str] = set()
|
|
294
|
+
for s in imported_doc.statements:
|
|
295
|
+
if not isinstance(s, TemplateDef):
|
|
296
|
+
continue
|
|
297
|
+
if self._check_template_name_conflict(s.name, s.source):
|
|
298
|
+
continue
|
|
299
|
+
key = TemplateKey(identity=file_id, name=s.name)
|
|
300
|
+
self._templates[key] = s
|
|
301
|
+
scope[s.name] = key
|
|
302
|
+
local_names.add(s.name)
|
|
303
|
+
self._scopes_by_file[file_id] = scope
|
|
304
|
+
|
|
305
|
+
# 2) 嵌套 !from:可见名映射
|
|
306
|
+
for s in imported_doc.statements:
|
|
307
|
+
if not isinstance(s, TemplateImportStmt):
|
|
308
|
+
continue
|
|
309
|
+
dep_scope = self._import_template_path(
|
|
310
|
+
s.from_path,
|
|
311
|
+
base_dir=file.root_path,
|
|
312
|
+
source=s.source,
|
|
313
|
+
loaded=loaded,
|
|
314
|
+
depth=depth + 1,
|
|
315
|
+
)
|
|
316
|
+
self._map_import_items(s.items, dep_scope, scope, local_names)
|
|
317
|
+
|
|
318
|
+
# 3) 非模板语句校验 + 模板 scope 登记
|
|
319
|
+
for s in imported_doc.statements:
|
|
320
|
+
match s:
|
|
321
|
+
case TemplateDef():
|
|
322
|
+
key = TemplateKey(identity=file_id, name=s.name)
|
|
323
|
+
if key in self._templates: # 同名冲突被拒绝的模板不登记 scope
|
|
324
|
+
self._template_scopes[key] = scope
|
|
325
|
+
case TemplateImportStmt():
|
|
326
|
+
pass
|
|
327
|
+
case _:
|
|
328
|
+
if file.name.endswith('.inft'):
|
|
329
|
+
self._collector.add(Diagnostic(Severity.ERROR, 'inft.not_allowed', {}, s.source))
|
|
330
|
+
|
|
331
|
+
return scope
|
|
332
|
+
|
|
333
|
+
def _parse_document(self, file: File) -> Document:
|
|
334
|
+
"""词法 + 语法分析一段源码(用于外部模板文件)。
|
|
335
|
+
|
|
336
|
+
启用 ``parse_cache`` 时按文件 identity 复用(文件不变跳过重复分析)。
|
|
337
|
+
"""
|
|
338
|
+
if self._parse_cache is not None:
|
|
339
|
+
cached = self._parse_cache.get(file.identity)
|
|
340
|
+
if cached is not None:
|
|
341
|
+
return cached
|
|
342
|
+
doc, _ = parse_source(file, self._collector)
|
|
343
|
+
if self._parse_cache is not None:
|
|
344
|
+
self._parse_cache[file.identity] = doc
|
|
345
|
+
return doc
|
|
@@ -0,0 +1,16 @@
|
|
|
1
|
+
"""词法分析:字符流与两遍词法分析器。
|
|
2
|
+
|
|
3
|
+
- :class:`RawTokenizer`:容错第一遍,产出 ``RawToken`` 并收集词法错误
|
|
4
|
+
- :class:`FinalTokenizer`:值语义第二遍,产出 ``Token``(转义解析、数值转换)
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from infinity_data.tokenizer.char_stream import CharStream, LineCounter
|
|
8
|
+
from infinity_data.tokenizer.finalizer import FinalTokenizer
|
|
9
|
+
from infinity_data.tokenizer.tokenizer import RawTokenizer
|
|
10
|
+
|
|
11
|
+
__all__ = [
|
|
12
|
+
'CharStream',
|
|
13
|
+
'LineCounter',
|
|
14
|
+
'FinalTokenizer',
|
|
15
|
+
'RawTokenizer',
|
|
16
|
+
]
|
|
@@ -0,0 +1,72 @@
|
|
|
1
|
+
from collections.abc import Iterable
|
|
2
|
+
|
|
3
|
+
from infinity_data.infra.ll1_stream import LL1Stream
|
|
4
|
+
from infinity_data.tokenizer.models.raw_tokens import SourceInfo
|
|
5
|
+
|
|
6
|
+
|
|
7
|
+
class LineCounter:
|
|
8
|
+
"""行号/列号/字符序号计数器。"""
|
|
9
|
+
|
|
10
|
+
def __init__(self) -> None:
|
|
11
|
+
self._index: int = 0
|
|
12
|
+
self._line: int = 1
|
|
13
|
+
self._col: int = 1
|
|
14
|
+
self._last_was_cr: bool = False
|
|
15
|
+
|
|
16
|
+
def step(self, ch: str) -> None:
|
|
17
|
+
"""根据当前消费的字符推进 index/line/col。"""
|
|
18
|
+
for c in ch:
|
|
19
|
+
self._index += 1
|
|
20
|
+
if c == '\n':
|
|
21
|
+
self._line += 1
|
|
22
|
+
self._col = 1
|
|
23
|
+
else:
|
|
24
|
+
self._col += 1
|
|
25
|
+
|
|
26
|
+
@property
|
|
27
|
+
def index(self) -> int:
|
|
28
|
+
return self._index
|
|
29
|
+
|
|
30
|
+
@property
|
|
31
|
+
def line(self) -> int:
|
|
32
|
+
return self._line
|
|
33
|
+
|
|
34
|
+
@property
|
|
35
|
+
def col(self) -> int:
|
|
36
|
+
return self._col
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
class CharStream(LL1Stream[str]):
|
|
40
|
+
"""字符流:在 LL(1) 流基础上附加行列位置跟踪。"""
|
|
41
|
+
|
|
42
|
+
def __init__(self, source: Iterable[str]) -> None:
|
|
43
|
+
super().__init__(source)
|
|
44
|
+
self._counter: LineCounter = LineCounter()
|
|
45
|
+
|
|
46
|
+
# ── 同步行列属性 ──────────────────────────────────────
|
|
47
|
+
|
|
48
|
+
@property
|
|
49
|
+
def index(self) -> int:
|
|
50
|
+
return self._counter.index
|
|
51
|
+
|
|
52
|
+
@property
|
|
53
|
+
def line(self) -> int:
|
|
54
|
+
return self._counter.line
|
|
55
|
+
|
|
56
|
+
@property
|
|
57
|
+
def col(self) -> int:
|
|
58
|
+
return self._counter.col
|
|
59
|
+
|
|
60
|
+
def info(self) -> SourceInfo:
|
|
61
|
+
"""当前位置(纯位置,来源由 tokenizer 持有)。"""
|
|
62
|
+
return SourceInfo(
|
|
63
|
+
index=self.index,
|
|
64
|
+
line=self.line,
|
|
65
|
+
col=self.col,
|
|
66
|
+
)
|
|
67
|
+
|
|
68
|
+
# ── 内部钩子 ──────────────────────────────────────────
|
|
69
|
+
|
|
70
|
+
def _on_advance(self, item: str) -> None:
|
|
71
|
+
"""消费字符时同步更新行列计数器。"""
|
|
72
|
+
self._counter.step(item)
|