infinity-data 1.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (52) hide show
  1. infinity_data/__init__.py +41 -0
  2. infinity_data/emit/__init__.py +9 -0
  3. infinity_data/emit/converter.py +55 -0
  4. infinity_data/frontend.py +37 -0
  5. infinity_data/infra/__init__.py +4 -0
  6. infinity_data/infra/diagnostics.py +212 -0
  7. infinity_data/infra/file.py +83 -0
  8. infinity_data/infra/ll1_stream.py +71 -0
  9. infinity_data/infra/location.py +80 -0
  10. infinity_data/infra/path.py +46 -0
  11. infinity_data/parser/__init__.py +79 -0
  12. infinity_data/parser/diagnostics.py +92 -0
  13. infinity_data/parser/models.py +295 -0
  14. infinity_data/parser/parser.py +988 -0
  15. infinity_data/parser/token_stream.py +128 -0
  16. infinity_data/pipeline.py +252 -0
  17. infinity_data/sandbox/__init__.py +32 -0
  18. infinity_data/sandbox/config.py +61 -0
  19. infinity_data/sandbox/errors.py +106 -0
  20. infinity_data/sandbox/mediator.py +214 -0
  21. infinity_data/sandbox/schema.py +21 -0
  22. infinity_data/semantic/__init__.py +59 -0
  23. infinity_data/semantic/builder/__init__.py +29 -0
  24. infinity_data/semantic/builder/builder.py +523 -0
  25. infinity_data/semantic/builder/models.py +163 -0
  26. infinity_data/semantic/constraints.py +163 -0
  27. infinity_data/semantic/diagnostics.py +172 -0
  28. infinity_data/semantic/executor/__init__.py +10 -0
  29. infinity_data/semantic/executor/executor.py +259 -0
  30. infinity_data/semantic/registry/__init__.py +168 -0
  31. infinity_data/semantic/registry/_core.py +212 -0
  32. infinity_data/semantic/registry/dict_constraints.py +56 -0
  33. infinity_data/semantic/registry/general.py +362 -0
  34. infinity_data/semantic/registry/logic.py +134 -0
  35. infinity_data/semantic/registry/types.py +136 -0
  36. infinity_data/semantic/resolver/__init__.py +20 -0
  37. infinity_data/semantic/resolver/imports.py +191 -0
  38. infinity_data/semantic/resolver/models.py +67 -0
  39. infinity_data/semantic/resolver/resolver.py +345 -0
  40. infinity_data/tokenizer/__init__.py +16 -0
  41. infinity_data/tokenizer/char_stream.py +72 -0
  42. infinity_data/tokenizer/diagnostics.py +62 -0
  43. infinity_data/tokenizer/finalizer.py +271 -0
  44. infinity_data/tokenizer/models/__init__.py +17 -0
  45. infinity_data/tokenizer/models/raw_tokens.py +61 -0
  46. infinity_data/tokenizer/models/tokens.py +245 -0
  47. infinity_data/tokenizer/tokenizer.py +531 -0
  48. infinity_data-1.0.0.dist-info/METADATA +60 -0
  49. infinity_data-1.0.0.dist-info/RECORD +52 -0
  50. infinity_data-1.0.0.dist-info/WHEEL +5 -0
  51. infinity_data-1.0.0.dist-info/licenses/LICENSE +21 -0
  52. infinity_data-1.0.0.dist-info/top_level.txt +1 -0
@@ -0,0 +1,191 @@
1
+ """导入语句求解:``!env`` / ``!file`` / ``!from`` 的系统访问。
2
+
3
+ 所有系统访问经 :class:`Sandbox` 中介:
4
+
5
+ - ``!env import``:变量经 ``Sandbox.getenv`` 授权查询
6
+ - ``!file``:数据文件经 ``Sandbox.open_file`` 产出 File 后解析
7
+ - ``!from``(模板导入):模板文件经 ``Sandbox.open_template`` 产出 File,
8
+ 模板定义的实际加载由 Phase 1 的 :class:`TemplateGraphResolver` 完成
9
+
10
+ 本层产出 ``$`` 引用命名空间(alias → Python 值);诊断直接写入调用方注入的
11
+ 共享 :class:`DiagnosticCollector`(与 resolver / builder 的收集器模式统一)。
12
+ 纯数据依赖:不引用任何 Phase 2 对象。
13
+ """
14
+
15
+ from __future__ import annotations
16
+
17
+ import json
18
+ import tomllib
19
+ from pathlib import Path
20
+ from typing import Any
21
+
22
+ from infinity_data.infra.diagnostics import Diagnostic, DiagnosticCollector, Severity
23
+ from infinity_data.infra.file import File
24
+ from infinity_data.parser import (
25
+ Document,
26
+ EnvImportStmt,
27
+ FileImportStmt,
28
+ JsonPathIndex,
29
+ JsonPathKey,
30
+ )
31
+ from infinity_data.sandbox import Sandbox, SandboxConfig
32
+ from infinity_data.tokenizer.models.raw_tokens import SourceRange
33
+
34
+ _FORMAT_MAP: dict[str, str] = {
35
+ '.json': 'json',
36
+ '.yaml': 'yaml',
37
+ '.yml': 'yaml',
38
+ '.toml': 'toml',
39
+ }
40
+
41
+
42
+ class ImportResolver:
43
+ """解析导入语句,产出 ``$`` 引用命名空间(alias → Python 值)。
44
+
45
+ Args:
46
+ sandbox: 沙盒中介(授权 / 拒绝一切系统访问)。None = 零信任 deny_all。
47
+ """
48
+
49
+ def __init__(self, *, sandbox: Sandbox | None = None) -> None:
50
+ # 零信任默认:未提供沙盒时拒绝一切系统访问(库默认 deny_all)
51
+ self._sandbox = sandbox or Sandbox(config=SandboxConfig.deny_all(), base_dir=Path.cwd())
52
+
53
+ @property
54
+ def sandbox(self) -> Sandbox:
55
+ return self._sandbox
56
+
57
+ @property
58
+ def base_dir(self) -> Path:
59
+ return self._sandbox.base_dir
60
+
61
+ def resolve(self, doc: Document, collector: DiagnosticCollector) -> dict[str, Any]:
62
+ """解析所有导入语句(env/file),返回 namespace;诊断写入 ``collector``。"""
63
+ namespace: dict[str, Any] = {}
64
+ for stmt in doc.statements:
65
+ if isinstance(stmt, EnvImportStmt):
66
+ self._resolve_env(stmt, namespace, collector)
67
+ elif isinstance(stmt, FileImportStmt):
68
+ self._resolve_file(stmt, namespace, collector)
69
+ return namespace
70
+
71
+ def _bind(
72
+ self,
73
+ namespace: dict[str, Any],
74
+ name: str,
75
+ value: Any,
76
+ collector: DiagnosticCollector,
77
+ source: SourceRange | None,
78
+ ) -> None:
79
+ """绑定 ``$`` 命名空间条目;重复 alias → ERROR 并拒绝覆盖(保留先到者)。
80
+
81
+ 与模板 scope 一致:``$`` 命名空间内不允许隐式的"后者覆盖前者"。
82
+ """
83
+ if name in namespace:
84
+ collector.add(Diagnostic(Severity.ERROR, 'namespace.duplicate', {'name': name}, source))
85
+ return
86
+ namespace[name] = value
87
+
88
+ # ── 各类导入 ──────────────────────────────────────
89
+
90
+ def _resolve_env(
91
+ self,
92
+ stmt: EnvImportStmt,
93
+ namespace: dict[str, Any],
94
+ collector: DiagnosticCollector,
95
+ ) -> None:
96
+ """!env import NAME [as NEW_NAME]
97
+
98
+ 未授权环境变量**总是失败**(无论 strict):Sandbox.getenv 直接抛
99
+ :class:`SandboxError`,不会退化为空字符串注入。
100
+ """
101
+ name = stmt.alias or stmt.name
102
+ self._bind(namespace, name, self._sandbox.getenv(stmt.name, source=stmt.source), collector, stmt.source)
103
+
104
+ def _resolve_file(
105
+ self,
106
+ stmt: FileImportStmt,
107
+ namespace: dict[str, Any],
108
+ collector: DiagnosticCollector,
109
+ ) -> None:
110
+ """!file "path" [as fmt] import .path.to.key as alias, ..."""
111
+ file = self._sandbox.open_file(stmt.file_path, source=stmt.source)
112
+ if file is None:
113
+ collector.add(Diagnostic(Severity.WARNING, 'import.file_denied', {'path_src': stmt.file_path}, stmt.source))
114
+ return
115
+
116
+ fmt = stmt.format or _FORMAT_MAP.get(Path(stmt.file_path).suffix.lower(), 'json')
117
+ try:
118
+ text = file.read()
119
+ except OSError:
120
+ collector.add(Diagnostic(Severity.WARNING, 'import.file_missing', {'name': file.name}, stmt.source))
121
+ return
122
+
123
+ data = self._parse_data(text, fmt, collector, stmt.source)
124
+ if data is None:
125
+ return
126
+
127
+ for item in stmt.imports:
128
+ try:
129
+ value = self._resolve_json_path(data, item.json_path)
130
+ except (KeyError, IndexError, TypeError):
131
+ collector.add(Diagnostic(Severity.WARNING, 'import.path_failed', {'name': file.name}, item.source))
132
+ continue
133
+ self._bind(namespace, item.alias, value, collector, item.source)
134
+
135
+ # ── 模板导入路径解析(!from 由 TemplateGraphResolver 使用)──
136
+
137
+ def resolve_template_path(
138
+ self,
139
+ from_path: str,
140
+ *,
141
+ base_dir: Path | None,
142
+ source: SourceRange | None,
143
+ collector: DiagnosticCollector,
144
+ ) -> File | None:
145
+ """!from 目标:经沙盒授权产出 File(相对路径以导入所在文件目录解析)。"""
146
+ file = self._sandbox.open_template(from_path, base_dir=base_dir, source=source)
147
+ if file is None:
148
+ collector.add(Diagnostic(Severity.WARNING, 'import.template_denied', {'path_src': from_path}, source))
149
+ return file
150
+
151
+ # ── 辅助 ──────────────────────────────────────────
152
+
153
+ def _parse_data(
154
+ self,
155
+ text: str,
156
+ fmt: str,
157
+ collector: DiagnosticCollector,
158
+ source: SourceRange,
159
+ ) -> Any | None:
160
+ """按格式解析数据内容(文本 loads)。"""
161
+ try:
162
+ if fmt == 'json':
163
+ return json.loads(text)
164
+ if fmt in ('yaml', 'yml'):
165
+ try:
166
+ import yaml # pyright: ignore[reportMissingModuleSource]
167
+ except ImportError:
168
+ collector.add(Diagnostic(Severity.WARNING, 'import.yaml_missing', {}, source))
169
+ return None
170
+ return yaml.safe_load(text)
171
+ if fmt == 'toml':
172
+ return tomllib.loads(text)
173
+ collector.add(Diagnostic(Severity.WARNING, 'import.unsupported_format', {'format': fmt}, source))
174
+ return None
175
+ except Exception as e:
176
+ collector.add(Diagnostic(Severity.ERROR, 'import.parse_failed', {'error': e}, source))
177
+ return None
178
+
179
+ # ── 辅助 ──────────────────────────────────────────
180
+
181
+ @staticmethod
182
+ def _resolve_json_path(data: Any, segments: list[JsonPathKey | JsonPathIndex]) -> Any:
183
+ """按结构化路径段定位数据;空路径 = 整个文件。"""
184
+ current = data
185
+ for seg in segments:
186
+ match seg:
187
+ case JsonPathKey(key=k):
188
+ current = current[k]
189
+ case JsonPathIndex(index=i):
190
+ current = current[i]
191
+ return current
@@ -0,0 +1,67 @@
1
+ """Phase 1(导入求解)数据模型:模板身份、可见名表与解析上下文。
2
+
3
+ 本子模块**只定义数据**,不包含任何解析逻辑(解析器见 :mod:`resolver`)。
4
+ Phase 2(构建 / 执行)通过 :class:`ResolvedContext` 消费本层产物——
5
+ 子模块间仅经数据模型依赖,无对象引用。
6
+
7
+ - ``TemplateKey``:模板真名(来源文件身份 + 本地名)
8
+ - ``Scope``:文件级可见名表(可见名 → 真名)
9
+ - ``ResolvedContext``:Phase 1 完整产物(模板图 + 可见名表 + 数据命名空间)
10
+ """
11
+
12
+ from __future__ import annotations
13
+
14
+ from dataclasses import dataclass
15
+ from typing import Any
16
+
17
+ from infinity_data.parser import TemplateDef
18
+
19
+ __all__ = ['ResolvedContext', 'Scope', 'TemplateKey']
20
+
21
+
22
+ @dataclass(frozen=True)
23
+ class TemplateKey:
24
+ """模板唯一身份:来源文件身份 + 模板本地名。
25
+
26
+ - ``identity``:来源文件身份(磁盘 = resolve 绝对路径;内存 = ``路径:mem:内容hash``)
27
+ - ``name``:模板在来源文件中的本地名(诊断显示用)
28
+
29
+ 身份含来源路径:不同路径的文件即使内容相同也是不同模板身份——模板内部
30
+ ``!from`` 按定义文件所在目录解析,内容相同的文件其依赖语义可能不同,
31
+ 不能互相覆盖(纯内容寻址无法表达这一区别)。
32
+
33
+ frozen 保证可哈希,直接作为模板表等映射的键。
34
+ """
35
+
36
+ identity: str
37
+ name: str
38
+
39
+ def __str__(self) -> str:
40
+ return f'{self.identity}:{self.name}'
41
+
42
+
43
+ Scope = dict[str, TemplateKey]
44
+ """文件级可见名表:可见名 → 模板真名(:class:`TemplateKey`)。"""
45
+
46
+
47
+ @dataclass(frozen=True)
48
+ class ResolvedContext:
49
+ """导入求解(Phase 1)产物:模板图 + 可见名表 + 数据命名空间。
50
+
51
+ 由 :class:`infinity_data.semantic.resolver.TemplateGraphResolver` 产出,
52
+ 供 Phase 2a(构建)经数据模型消费。
53
+ 只含名字与模板定义,不含任何约束执行结果(约束求值属 Phase 2),
54
+ 也不含诊断——诊断经共享 :class:`DiagnosticCollector` 收集(流水线单一收集器)。
55
+
56
+ - ``templates``:全部已加载模板(本地 + ``!from`` 导入)
57
+ - ``template_scopes``:每个模板定义点的可见名表(展开/校验按定义点可见性解析)
58
+ - ``root_scope``:入口文件可见名表(可见名 → :class:`TemplateKey`)
59
+ - ``schema_scope``:schema.from_file 隐式导入的可见名表(无则 None)
60
+ - ``namespace``:``$`` 引用命名空间(``!env`` / ``!file`` 解析结果)
61
+ """
62
+
63
+ templates: dict[TemplateKey, TemplateDef]
64
+ template_scopes: dict[TemplateKey, Scope]
65
+ root_scope: Scope
66
+ schema_scope: Scope | None
67
+ namespace: dict[str, Any]
@@ -0,0 +1,345 @@
1
+ """模板图求解器(Phase 1):构建模板图、可见名表与数据命名空间。
2
+
3
+ 将「模板导入(``!from``)、数据导入(``!env`` / ``!file``)、模板定义收集」
4
+ 从语义分析中独立出来,产出不可变的 :class:`ResolvedContext` 供
5
+ Phase 2a(:class:`~infinity_data.semantic.builder.AstBuilder`)消费。
6
+
7
+ - 本层**不执行任何约束**:只解析名字、加载模板定义、构建 scope;
8
+ 模板展开 / 约束求值 / schema 校验全部留在 Phase 2。
9
+ - ``resolve()`` 幂等:同一输入产出等价上下文,不依赖调用历史;
10
+ 外部文件解析结果可经 ``parse_cache`` 跨调用复用(增量编译 / LSP)。
11
+ - 遮蔽检查只读查询注册表的内置约束名(不触发约束执行)。
12
+ - 诊断写入调用方注入的共享 :class:`DiagnosticCollector`(流水线单一收集器),
13
+ 本层不持有诊断列表、不产出诊断数据。
14
+ """
15
+
16
+ from __future__ import annotations
17
+
18
+ from pathlib import Path
19
+
20
+ from infinity_data.frontend import parse_source
21
+ from infinity_data.infra.diagnostics import Diagnostic, DiagnosticCollector, Severity
22
+ from infinity_data.infra.file import File
23
+ from infinity_data.parser import (
24
+ Document,
25
+ TemplateDef,
26
+ TemplateImportItem,
27
+ TemplateImportStmt,
28
+ )
29
+ from infinity_data.sandbox import Schema
30
+ from infinity_data.semantic.registry import ConstraintRegistry
31
+ from infinity_data.semantic.resolver.imports import ImportResolver
32
+ from infinity_data.semantic.resolver.models import ResolvedContext, Scope, TemplateKey
33
+ from infinity_data.tokenizer.models.raw_tokens import SourceRange
34
+
35
+ MAX_IMPORT_DEPTH = 32
36
+ """模板导入递归深度上限(防止循环导入无限递归)。"""
37
+
38
+
39
+ class TemplateGraphResolver:
40
+ """模板图求解器(Phase 1):递归加载导入、构建 scope、解析数据导入。
41
+
42
+ Args:
43
+ registry: 约束注册表(仅用于内置约束名的遮蔽检查,不执行约束)
44
+ import_resolver: 数据 / 模板导入路径解析(沙盒授权)
45
+ schema: 顶层 schema(``from_file`` 隐式导入)
46
+ parse_cache: 可选外部文件解析缓存(identity → Document)。
47
+ 传入后跨 ``resolve()`` 复用,文件不变时跳过重复词法/语法分析。
48
+ """
49
+
50
+ def __init__(
51
+ self,
52
+ *,
53
+ registry: ConstraintRegistry | None = None,
54
+ import_resolver: ImportResolver | None = None,
55
+ schema: Schema | None = None,
56
+ parse_cache: dict[str, Document] | None = None,
57
+ ) -> None:
58
+ self._registry = registry or ConstraintRegistry()
59
+ self._imports = import_resolver or ImportResolver()
60
+ self._schema = schema
61
+ self._parse_cache = parse_cache
62
+ # 工作状态(每次 resolve 重置)
63
+ self._templates: dict[TemplateKey, TemplateDef] = {}
64
+ self._template_scopes: dict[TemplateKey, Scope] = {}
65
+ self._scopes_by_file: dict[str, Scope] = {} # 文件 identity → 已构建 scope(循环导入防护)
66
+ self._root_file: File | None = None
67
+ self._root_scope: Scope = {}
68
+ self._root_local_names: set[str] = set()
69
+ self._schema_scope: Scope | None = None
70
+ # 本次 resolve 的共享诊断收集器(流水线单一收集器,resolve() 注入)
71
+ self._collector: DiagnosticCollector = DiagnosticCollector()
72
+
73
+ def resolve(self, doc: Document, file: File, collector: DiagnosticCollector) -> ResolvedContext:
74
+ """求解导入,返回不可变上下文(幂等:同一输入产出等价结果)。
75
+
76
+ 诊断(``import.*`` / ``template.*`` 域)写入 ``collector``。
77
+ """
78
+ self._reset(file, collector)
79
+
80
+ # 收集本地模板定义(key 键控,身份含来源文件路径)
81
+ self._collect_templates(doc, file.identity)
82
+
83
+ # 解析模板导入(!from,含 schema.from_file 隐式导入)→ 构建主文件 scope
84
+ root_scope = self._load_imported_templates(doc)
85
+
86
+ # 解析数据导入语句(!env / !file)→ $ 引用命名空间
87
+ namespace = self._imports.resolve(doc, self._collector)
88
+
89
+ return ResolvedContext(
90
+ templates=dict(self._templates),
91
+ template_scopes=dict(self._template_scopes),
92
+ root_scope=root_scope, # 与 template_scopes 内的定义点 scope 同一对象
93
+ schema_scope=self._schema_scope,
94
+ namespace=dict(namespace),
95
+ )
96
+
97
+ @property
98
+ def registry(self) -> ConstraintRegistry:
99
+ """共享约束注册表(Phase 2 复用同一注册表:模板即约束注册 / 执行)。"""
100
+ return self._registry
101
+
102
+ @property
103
+ def schema(self) -> Schema | None:
104
+ """顶层 schema(Phase 2b 顶层校验复用同一实例)。"""
105
+ return self._schema
106
+
107
+ def _reset(self, file: File, collector: DiagnosticCollector) -> None:
108
+ self._templates = {}
109
+ self._template_scopes = {}
110
+ self._scopes_by_file = {}
111
+ self._root_file = file
112
+ self._root_scope = {}
113
+ self._root_local_names = set()
114
+ self._schema_scope = None
115
+ self._collector = collector
116
+
117
+ # ═══════════════════════════════════════════════════════
118
+ # 模板收集
119
+ # ═══════════════════════════════════════════════════════
120
+
121
+ def _check_template_name_conflict(self, name: str, source: SourceRange | None) -> bool:
122
+ """模板名与已注册约束(内置/自定义)同名 → ERROR 并返回 True。
123
+
124
+ 模板即约束:定义 ``~int`` / ``~range`` 会遮蔽同名内置约束(``int`` 类型
125
+ 标注、``range(1, 100)`` 调用等语义被静默劫持),因此同名模板禁止定义,
126
+ 内置约束保持可用。
127
+ """
128
+ if name in self._registry.names:
129
+ self._collector.add(Diagnostic(Severity.ERROR, 'template.shadows_builtin', {'template': name}, source))
130
+ return True
131
+ return False
132
+
133
+ def _check_required_order(self, stmt: TemplateDef) -> None:
134
+ """模板内部校验:必填字段必须全部在可选字段之前。
135
+
136
+ 例外:``positional=false`` 的模板不接受位置参数,字段顺序不影响绑定,
137
+ 允许必填与可选交错。
138
+ """
139
+ if not stmt.config.positional:
140
+ return
141
+ seen_optional = False
142
+ for tf in stmt.fields:
143
+ if tf.default_value is None:
144
+ if seen_optional:
145
+ self._collector.add(
146
+ Diagnostic(
147
+ Severity.ERROR,
148
+ 'template.required_order',
149
+ {'template': stmt.name, 'field': tf.name},
150
+ tf.source,
151
+ )
152
+ )
153
+ else:
154
+ seen_optional = True
155
+
156
+ def _collect_templates(self, doc: Document, root_identity: str) -> None:
157
+ for stmt in doc.statements:
158
+ if not isinstance(stmt, TemplateDef):
159
+ continue
160
+ rejected = self._check_template_name_conflict(stmt.name, stmt.source)
161
+ key = TemplateKey(identity=root_identity, name=stmt.name)
162
+ if not rejected and key in self._templates:
163
+ self._collector.add(
164
+ Diagnostic(Severity.ERROR, 'template.duplicate', {'template': stmt.name}, stmt.source)
165
+ )
166
+ rejected = True
167
+ # 无论是否被拒绝都校验内部(一次暴露所有错误,避免多轮修复)
168
+ self._check_required_order(stmt)
169
+ if rejected:
170
+ continue # 保留首次定义,拒绝隐式的"后者覆盖前者"
171
+ self._templates[key] = stmt
172
+ self._root_local_names.add(stmt.name)
173
+
174
+ # ═══════════════════════════════════════════════════════
175
+ # 模板导入(!from)
176
+ # ═══════════════════════════════════════════════════════
177
+
178
+ def _map_import_items(
179
+ self,
180
+ items: list[TemplateImportItem],
181
+ dep_scope: Scope,
182
+ scope: Scope,
183
+ local_names: set[str],
184
+ ) -> None:
185
+ """把 ``!from`` 的导入项映射进目标 scope;冲突一律 ERROR。
186
+
187
+ - 导入文件中不存在该模板 → ERROR
188
+ - 可见名与文件内定义同名 → ERROR(与文件内定义冲突)
189
+ - 可见名已存在(重复导入)→ ERROR,保留先到者(拒绝隐式覆盖)
190
+ """
191
+ for item in items:
192
+ dep_key = dep_scope.get(item.name)
193
+ if dep_key is None:
194
+ self._collector.add(
195
+ Diagnostic(Severity.ERROR, 'template.import_not_found', {'template': item.name}, item.source)
196
+ )
197
+ continue
198
+ visible = item.alias or item.name
199
+ if visible in scope:
200
+ if visible in local_names:
201
+ self._collector.add(
202
+ Diagnostic(Severity.ERROR, 'template.import_conflict_local', {'visible': visible}, item.source)
203
+ )
204
+ else:
205
+ self._collector.add(
206
+ Diagnostic(Severity.ERROR, 'template.import_duplicate', {'visible': visible}, item.source)
207
+ )
208
+ else:
209
+ scope[visible] = dep_key
210
+
211
+ def _load_imported_templates(self, doc: Document) -> Scope:
212
+ """构建主文件 scope(含 schema.from_file 隐式导入)。"""
213
+ assert self._root_file is not None
214
+ root_id = self._root_file.identity
215
+ loaded: set[str] = set()
216
+ root_scope: Scope = {tpl.name: key for key, tpl in self._templates.items() if key.identity == root_id}
217
+
218
+ # schema.from_file 隐式导入:独立 scope 供顶层校验使用
219
+ if self._schema is not None and self._schema.from_file:
220
+ self._schema_scope = self._import_template_path(
221
+ self._schema.from_file,
222
+ base_dir=self._imports.base_dir,
223
+ source=None,
224
+ loaded=loaded,
225
+ depth=0,
226
+ )
227
+
228
+ for stmt in doc.statements:
229
+ if not isinstance(stmt, TemplateImportStmt):
230
+ continue
231
+ dep_scope = self._import_template_path(
232
+ stmt.from_path,
233
+ base_dir=self._imports.base_dir,
234
+ source=stmt.source,
235
+ loaded=loaded,
236
+ depth=0,
237
+ )
238
+ self._map_import_items(stmt.items, dep_scope, root_scope, self._root_local_names)
239
+ # 主文件本地模板的 scope 登记(模板展开/约束校验按此解析名字)
240
+ for key in self._templates:
241
+ if key.identity == root_id:
242
+ self._template_scopes[key] = root_scope
243
+ return root_scope
244
+
245
+ def _import_template_path(
246
+ self,
247
+ from_path: str,
248
+ *,
249
+ base_dir: Path,
250
+ source: SourceRange | None,
251
+ loaded: set[str],
252
+ depth: int,
253
+ ) -> Scope:
254
+ """加载单个模板文件,返回该文件的可见 scope(递归解析嵌套 !from)。"""
255
+ if depth > MAX_IMPORT_DEPTH:
256
+ self._collector.add(
257
+ Diagnostic(
258
+ Severity.ERROR, 'template.import_depth', {'max': MAX_IMPORT_DEPTH, 'path_src': from_path}, source
259
+ )
260
+ )
261
+ return {}
262
+
263
+ file = self._imports.resolve_template_path(
264
+ from_path,
265
+ base_dir=base_dir,
266
+ source=source,
267
+ collector=self._collector,
268
+ )
269
+ if file is None:
270
+ return {}
271
+
272
+ file_id = file.identity
273
+ if file_id in loaded:
274
+ # 循环导入:返回已构建的本地名部分(本地模板先注册)
275
+ return self._scopes_by_file.get(file_id, {})
276
+ loaded.add(file_id)
277
+
278
+ try:
279
+ _ = file.content_hash() # 触发内容读取;身份不含内容,仍需校验文件可读
280
+ except OSError as e:
281
+ self._collector.add(
282
+ Diagnostic(Severity.ERROR, 'template.read_failed', {'file': file.name, 'error': e}, source)
283
+ )
284
+ return {}
285
+
286
+ imported_doc = self._parse_document(file)
287
+
288
+ # 1) 本地模板:先注册(循环导入时依赖文件的本地名部分已可见)
289
+ # 身份含来源文件路径:不同路径的文件即使内容相同也是不同模板身份——
290
+ # 模板内部 !from 按定义文件所在目录解析,内容相同的文件其依赖语义
291
+ # 可能不同,不能互相覆盖(纯内容寻址无法表达这一区别)
292
+ scope: Scope = {}
293
+ local_names: set[str] = set()
294
+ for s in imported_doc.statements:
295
+ if not isinstance(s, TemplateDef):
296
+ continue
297
+ if self._check_template_name_conflict(s.name, s.source):
298
+ continue
299
+ key = TemplateKey(identity=file_id, name=s.name)
300
+ self._templates[key] = s
301
+ scope[s.name] = key
302
+ local_names.add(s.name)
303
+ self._scopes_by_file[file_id] = scope
304
+
305
+ # 2) 嵌套 !from:可见名映射
306
+ for s in imported_doc.statements:
307
+ if not isinstance(s, TemplateImportStmt):
308
+ continue
309
+ dep_scope = self._import_template_path(
310
+ s.from_path,
311
+ base_dir=file.root_path,
312
+ source=s.source,
313
+ loaded=loaded,
314
+ depth=depth + 1,
315
+ )
316
+ self._map_import_items(s.items, dep_scope, scope, local_names)
317
+
318
+ # 3) 非模板语句校验 + 模板 scope 登记
319
+ for s in imported_doc.statements:
320
+ match s:
321
+ case TemplateDef():
322
+ key = TemplateKey(identity=file_id, name=s.name)
323
+ if key in self._templates: # 同名冲突被拒绝的模板不登记 scope
324
+ self._template_scopes[key] = scope
325
+ case TemplateImportStmt():
326
+ pass
327
+ case _:
328
+ if file.name.endswith('.inft'):
329
+ self._collector.add(Diagnostic(Severity.ERROR, 'inft.not_allowed', {}, s.source))
330
+
331
+ return scope
332
+
333
+ def _parse_document(self, file: File) -> Document:
334
+ """词法 + 语法分析一段源码(用于外部模板文件)。
335
+
336
+ 启用 ``parse_cache`` 时按文件 identity 复用(文件不变跳过重复分析)。
337
+ """
338
+ if self._parse_cache is not None:
339
+ cached = self._parse_cache.get(file.identity)
340
+ if cached is not None:
341
+ return cached
342
+ doc, _ = parse_source(file, self._collector)
343
+ if self._parse_cache is not None:
344
+ self._parse_cache[file.identity] = doc
345
+ return doc
@@ -0,0 +1,16 @@
1
+ """词法分析:字符流与两遍词法分析器。
2
+
3
+ - :class:`RawTokenizer`:容错第一遍,产出 ``RawToken`` 并收集词法错误
4
+ - :class:`FinalTokenizer`:值语义第二遍,产出 ``Token``(转义解析、数值转换)
5
+ """
6
+
7
+ from infinity_data.tokenizer.char_stream import CharStream, LineCounter
8
+ from infinity_data.tokenizer.finalizer import FinalTokenizer
9
+ from infinity_data.tokenizer.tokenizer import RawTokenizer
10
+
11
+ __all__ = [
12
+ 'CharStream',
13
+ 'LineCounter',
14
+ 'FinalTokenizer',
15
+ 'RawTokenizer',
16
+ ]
@@ -0,0 +1,72 @@
1
+ from collections.abc import Iterable
2
+
3
+ from infinity_data.infra.ll1_stream import LL1Stream
4
+ from infinity_data.tokenizer.models.raw_tokens import SourceInfo
5
+
6
+
7
+ class LineCounter:
8
+ """行号/列号/字符序号计数器。"""
9
+
10
+ def __init__(self) -> None:
11
+ self._index: int = 0
12
+ self._line: int = 1
13
+ self._col: int = 1
14
+ self._last_was_cr: bool = False
15
+
16
+ def step(self, ch: str) -> None:
17
+ """根据当前消费的字符推进 index/line/col。"""
18
+ for c in ch:
19
+ self._index += 1
20
+ if c == '\n':
21
+ self._line += 1
22
+ self._col = 1
23
+ else:
24
+ self._col += 1
25
+
26
+ @property
27
+ def index(self) -> int:
28
+ return self._index
29
+
30
+ @property
31
+ def line(self) -> int:
32
+ return self._line
33
+
34
+ @property
35
+ def col(self) -> int:
36
+ return self._col
37
+
38
+
39
+ class CharStream(LL1Stream[str]):
40
+ """字符流:在 LL(1) 流基础上附加行列位置跟踪。"""
41
+
42
+ def __init__(self, source: Iterable[str]) -> None:
43
+ super().__init__(source)
44
+ self._counter: LineCounter = LineCounter()
45
+
46
+ # ── 同步行列属性 ──────────────────────────────────────
47
+
48
+ @property
49
+ def index(self) -> int:
50
+ return self._counter.index
51
+
52
+ @property
53
+ def line(self) -> int:
54
+ return self._counter.line
55
+
56
+ @property
57
+ def col(self) -> int:
58
+ return self._counter.col
59
+
60
+ def info(self) -> SourceInfo:
61
+ """当前位置(纯位置,来源由 tokenizer 持有)。"""
62
+ return SourceInfo(
63
+ index=self.index,
64
+ line=self.line,
65
+ col=self.col,
66
+ )
67
+
68
+ # ── 内部钩子 ──────────────────────────────────────────
69
+
70
+ def _on_advance(self, item: str) -> None:
71
+ """消费字符时同步更新行列计数器。"""
72
+ self._counter.step(item)