infinity-data 1.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- infinity_data/__init__.py +41 -0
- infinity_data/emit/__init__.py +9 -0
- infinity_data/emit/converter.py +55 -0
- infinity_data/frontend.py +37 -0
- infinity_data/infra/__init__.py +4 -0
- infinity_data/infra/diagnostics.py +212 -0
- infinity_data/infra/file.py +83 -0
- infinity_data/infra/ll1_stream.py +71 -0
- infinity_data/infra/location.py +80 -0
- infinity_data/infra/path.py +46 -0
- infinity_data/parser/__init__.py +79 -0
- infinity_data/parser/diagnostics.py +92 -0
- infinity_data/parser/models.py +295 -0
- infinity_data/parser/parser.py +988 -0
- infinity_data/parser/token_stream.py +128 -0
- infinity_data/pipeline.py +252 -0
- infinity_data/sandbox/__init__.py +32 -0
- infinity_data/sandbox/config.py +61 -0
- infinity_data/sandbox/errors.py +106 -0
- infinity_data/sandbox/mediator.py +214 -0
- infinity_data/sandbox/schema.py +21 -0
- infinity_data/semantic/__init__.py +59 -0
- infinity_data/semantic/builder/__init__.py +29 -0
- infinity_data/semantic/builder/builder.py +523 -0
- infinity_data/semantic/builder/models.py +163 -0
- infinity_data/semantic/constraints.py +163 -0
- infinity_data/semantic/diagnostics.py +172 -0
- infinity_data/semantic/executor/__init__.py +10 -0
- infinity_data/semantic/executor/executor.py +259 -0
- infinity_data/semantic/registry/__init__.py +168 -0
- infinity_data/semantic/registry/_core.py +212 -0
- infinity_data/semantic/registry/dict_constraints.py +56 -0
- infinity_data/semantic/registry/general.py +362 -0
- infinity_data/semantic/registry/logic.py +134 -0
- infinity_data/semantic/registry/types.py +136 -0
- infinity_data/semantic/resolver/__init__.py +20 -0
- infinity_data/semantic/resolver/imports.py +191 -0
- infinity_data/semantic/resolver/models.py +67 -0
- infinity_data/semantic/resolver/resolver.py +345 -0
- infinity_data/tokenizer/__init__.py +16 -0
- infinity_data/tokenizer/char_stream.py +72 -0
- infinity_data/tokenizer/diagnostics.py +62 -0
- infinity_data/tokenizer/finalizer.py +271 -0
- infinity_data/tokenizer/models/__init__.py +17 -0
- infinity_data/tokenizer/models/raw_tokens.py +61 -0
- infinity_data/tokenizer/models/tokens.py +245 -0
- infinity_data/tokenizer/tokenizer.py +531 -0
- infinity_data-1.0.0.dist-info/METADATA +60 -0
- infinity_data-1.0.0.dist-info/RECORD +52 -0
- infinity_data-1.0.0.dist-info/WHEEL +5 -0
- infinity_data-1.0.0.dist-info/licenses/LICENSE +21 -0
- infinity_data-1.0.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,128 @@
|
|
|
1
|
+
"""LL(1) Token 流包装器,继承 LL1Stream[Token],自动追踪 source range。
|
|
2
|
+
|
|
3
|
+
全链路流式:CharStream → RawToken → Token → AST。
|
|
4
|
+
"""
|
|
5
|
+
|
|
6
|
+
from __future__ import annotations
|
|
7
|
+
|
|
8
|
+
from collections.abc import Iterable
|
|
9
|
+
from typing import TypeVar
|
|
10
|
+
|
|
11
|
+
from infinity_data.infra.diagnostics import DiagnosticCollector
|
|
12
|
+
from infinity_data.infra.ll1_stream import LL1Stream, NoNextType
|
|
13
|
+
from infinity_data.parser.diagnostics import diag
|
|
14
|
+
from infinity_data.tokenizer.models.raw_tokens import (
|
|
15
|
+
RawToken,
|
|
16
|
+
RawTokenType,
|
|
17
|
+
SourceRange,
|
|
18
|
+
)
|
|
19
|
+
from infinity_data.tokenizer.models.tokens import (
|
|
20
|
+
CommaToken,
|
|
21
|
+
EofToken,
|
|
22
|
+
NewlineToken,
|
|
23
|
+
Token,
|
|
24
|
+
)
|
|
25
|
+
|
|
26
|
+
_TToken = TypeVar('_TToken', bound=Token)
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
class TokenStream(LL1Stream[Token]):
|
|
30
|
+
"""LL(1) Token 流,继承 LL1Stream[Token]"""
|
|
31
|
+
|
|
32
|
+
def __init__(
|
|
33
|
+
self,
|
|
34
|
+
source: Iterable[Token],
|
|
35
|
+
error_collector: DiagnosticCollector,
|
|
36
|
+
) -> None:
|
|
37
|
+
super().__init__(source)
|
|
38
|
+
self._errors: DiagnosticCollector = error_collector
|
|
39
|
+
self._last: Token | None = None
|
|
40
|
+
|
|
41
|
+
# ── LL1Stream 钩子 ────────────────────────────────────
|
|
42
|
+
|
|
43
|
+
def _on_advance(self, item: Token) -> None:
|
|
44
|
+
"""消费每个 token 时记录,用于 range 追踪。"""
|
|
45
|
+
self._last = item
|
|
46
|
+
|
|
47
|
+
def check(self, expect: RawTokenType) -> bool:
|
|
48
|
+
token = self.peek()
|
|
49
|
+
if isinstance(token, NoNextType):
|
|
50
|
+
return False
|
|
51
|
+
return token.raw.type == expect
|
|
52
|
+
|
|
53
|
+
def eof(self) -> bool:
|
|
54
|
+
"""结束判定:当前 token 为 EofToken,或流已物理耗尽(哨兵被消费后)。
|
|
55
|
+
|
|
56
|
+
EofToken 是 FinalTokenizer 产出的哨兵 token;基类 :class:`LL1Stream` 的
|
|
57
|
+
``eof()`` 只认物理耗尽(哨兵被消费后才为 True),此处统一为「哨兵即结束」
|
|
58
|
+
——解析循环无需再区分 ``check(EOF)`` 与 ``eof()``。
|
|
59
|
+
"""
|
|
60
|
+
return isinstance(self.peek(), (EofToken, NoNextType))
|
|
61
|
+
|
|
62
|
+
# ── 跳过 ───────────────────────────────────────
|
|
63
|
+
|
|
64
|
+
def skip_newlines(self) -> None:
|
|
65
|
+
"""跳过连续的换行"""
|
|
66
|
+
while not self.eof() and isinstance(self.peek(), NewlineToken):
|
|
67
|
+
self.advance()
|
|
68
|
+
|
|
69
|
+
def skip_separators(self) -> bool:
|
|
70
|
+
"""跳过逗号和换行(元素分隔符)。
|
|
71
|
+
|
|
72
|
+
Returns:
|
|
73
|
+
是否消费了至少一个分隔符——区分「有分隔符」与「无分隔符」:
|
|
74
|
+
元素之间必须显式分隔,空格不构成分隔符。
|
|
75
|
+
"""
|
|
76
|
+
saw = False
|
|
77
|
+
while not self.eof() and isinstance(self.peek(), (CommaToken, NewlineToken)):
|
|
78
|
+
self.advance()
|
|
79
|
+
saw = True
|
|
80
|
+
return saw
|
|
81
|
+
|
|
82
|
+
# ── Range 追踪 ────────────────────────────────────────
|
|
83
|
+
|
|
84
|
+
def span_from(self, first: Token | NoNextType | None) -> SourceRange:
|
|
85
|
+
"""计算从 first 到当前最后消费 token 的 SourceRange。"""
|
|
86
|
+
if isinstance(first, NoNextType) or first is None:
|
|
87
|
+
first = self._last
|
|
88
|
+
if isinstance(first, NoNextType) or first is None:
|
|
89
|
+
return SourceRange.empty()
|
|
90
|
+
last = self._last if self._last else first
|
|
91
|
+
return SourceRange(file=first.raw.source.file, start=first.raw.source.start, end=last.raw.source.end)
|
|
92
|
+
|
|
93
|
+
@staticmethod
|
|
94
|
+
def single_span(token: Token) -> SourceRange:
|
|
95
|
+
"""为单个 token 创建 SourceRange。"""
|
|
96
|
+
return SourceRange(file=token.raw.source.file, start=token.raw.source.start, end=token.raw.source.end)
|
|
97
|
+
|
|
98
|
+
# ── 期望 / 错误恢复 ─────────────────────────────
|
|
99
|
+
|
|
100
|
+
def expect(self, token_cls: type[_TToken]) -> _TToken:
|
|
101
|
+
"""期望当前 token 为指定类型,否则收集错误并插入合成 token。
|
|
102
|
+
|
|
103
|
+
错误恢复策略:记录 UnexpectedTokenError → 消费意外 token → 返回合成 token。
|
|
104
|
+
合成 token 保证调用方拿到类型安全的对象,解析器始终前进,避免级联崩溃。
|
|
105
|
+
"""
|
|
106
|
+
tok = self.peek()
|
|
107
|
+
if isinstance(tok, NoNextType):
|
|
108
|
+
rng = self.span_from(None)
|
|
109
|
+
self._errors.add(diag('parse.unexpected_token', {'expected': token_cls.__name__, 'actual': 'EOF'}, rng))
|
|
110
|
+
return self._synthetic(token_cls, source=rng)
|
|
111
|
+
if not isinstance(tok, token_cls):
|
|
112
|
+
self._errors.add(
|
|
113
|
+
diag(
|
|
114
|
+
'parse.unexpected_token',
|
|
115
|
+
{'expected': token_cls.__name__, 'actual': tok.raw.type.name},
|
|
116
|
+
tok.raw.source,
|
|
117
|
+
)
|
|
118
|
+
)
|
|
119
|
+
self.advance()
|
|
120
|
+
return self._synthetic(token_cls, source=tok.raw.source)
|
|
121
|
+
self.advance()
|
|
122
|
+
return tok
|
|
123
|
+
|
|
124
|
+
def _synthetic(self, token_cls: type[_TToken], *, source: SourceRange | None = None) -> _TToken:
|
|
125
|
+
"""构造合成 token(错误恢复用)。"""
|
|
126
|
+
rng = source or self.span_from(None)
|
|
127
|
+
raw = RawToken(type=RawTokenType.EOF, raw='', source=rng)
|
|
128
|
+
return token_cls(raw=raw)
|
|
@@ -0,0 +1,252 @@
|
|
|
1
|
+
"""编译流水线:源码字符串 → StdDocument / Python dict。
|
|
2
|
+
|
|
3
|
+
全链路流式:chars → RawTokenizer → FinalTokenizer → Parser → AstBuilder → Executor → Converter。
|
|
4
|
+
|
|
5
|
+
公共 API:
|
|
6
|
+
- :func:`load` / :func:`compile_source`:编译入口,返回 :class:`CompilationResult`
|
|
7
|
+
- :func:`safe_load`:零信任加载(deny_all 沙盒,禁止一切导入)
|
|
8
|
+
- :func:`check`:仅校验,返回诊断列表
|
|
9
|
+
- :func:`compile_document`:编译为 StdDocument(不降维)
|
|
10
|
+
|
|
11
|
+
共享选项(env/sandbox/registry/schema)统一由 :class:`CompileOptions` 承载,
|
|
12
|
+
各入口不再逐参数重复声明/转发。
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
from __future__ import annotations
|
|
16
|
+
|
|
17
|
+
from collections.abc import Mapping
|
|
18
|
+
from dataclasses import dataclass, field, replace
|
|
19
|
+
from functools import cached_property
|
|
20
|
+
from pathlib import Path
|
|
21
|
+
from typing import Any
|
|
22
|
+
|
|
23
|
+
from infinity_data.emit import reduce_object
|
|
24
|
+
from infinity_data.frontend import parse_source
|
|
25
|
+
from infinity_data.infra.diagnostics import Diagnostic, DiagnosticCollector, Severity
|
|
26
|
+
from infinity_data.infra.file import DiskFile, File, MemFile
|
|
27
|
+
from infinity_data.sandbox import Sandbox, SandboxConfig, SandboxError, Schema, SchemaError
|
|
28
|
+
from infinity_data.semantic.builder import AstBuilder, StdDocument
|
|
29
|
+
from infinity_data.semantic.executor import ConstraintExecutor
|
|
30
|
+
from infinity_data.semantic.registry import ConstraintRegistry
|
|
31
|
+
from infinity_data.semantic.resolver import ImportResolver, TemplateGraphResolver
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
@dataclass(frozen=True)
|
|
35
|
+
class CompileOptions:
|
|
36
|
+
"""一次编译的共享选项(各公共入口的统一参数载体)。
|
|
37
|
+
|
|
38
|
+
- ``env``:环境变量便捷授权,等价于 ``SandboxConfig(env=...)``,与 ``sandbox`` 合并
|
|
39
|
+
- ``sandbox``:沙盒配置;``None`` = 零信任(deny_all,库默认)
|
|
40
|
+
- ``registry``:自定义约束注册表;``None`` = 内置
|
|
41
|
+
- ``schema``:顶层模板约束;``None`` = 不校验
|
|
42
|
+
"""
|
|
43
|
+
|
|
44
|
+
env: Mapping[str, str] | None = None
|
|
45
|
+
sandbox: SandboxConfig | None = None
|
|
46
|
+
registry: ConstraintRegistry | None = None
|
|
47
|
+
schema: Schema | None = None
|
|
48
|
+
|
|
49
|
+
def effective_sandbox(self) -> SandboxConfig:
|
|
50
|
+
"""env 合并进 sandbox 得实际生效配置;两者均缺省 → deny_all(零信任,库默认)。"""
|
|
51
|
+
if self.sandbox is None:
|
|
52
|
+
return SandboxConfig(env=dict(self.env)) if self.env is not None else SandboxConfig.deny_all()
|
|
53
|
+
if self.env is not None:
|
|
54
|
+
return replace(self.sandbox, env={**self.sandbox.env, **dict(self.env)})
|
|
55
|
+
return self.sandbox
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
@dataclass
|
|
59
|
+
class CompilationResult:
|
|
60
|
+
"""一次编译的完整产物(根产物 + 诊断)。
|
|
61
|
+
|
|
62
|
+
- ``document``::class:`StdDocument`(纯数据:root / templates / scope;诊断见 ``diagnostics``)
|
|
63
|
+
- ``root`` / ``value``:由 ``document`` 派生(惰性)——降维属 emit 层职责,
|
|
64
|
+
编译阶段不急于产出,访问时经 :mod:`infinity_data.emit` 计算
|
|
65
|
+
"""
|
|
66
|
+
|
|
67
|
+
document: StdDocument
|
|
68
|
+
diagnostics: list[Diagnostic] = field(default_factory=lambda: [])
|
|
69
|
+
|
|
70
|
+
@cached_property
|
|
71
|
+
def value(self) -> dict[str, Any]:
|
|
72
|
+
"""降维后的纯 Python dict(惰性,由 emit 层负责;尽力而为)。"""
|
|
73
|
+
return reduce_object(self.document.root)
|
|
74
|
+
|
|
75
|
+
@property
|
|
76
|
+
def has_errors(self) -> bool:
|
|
77
|
+
return any(d.severity is Severity.ERROR for d in self.diagnostics)
|
|
78
|
+
|
|
79
|
+
@property
|
|
80
|
+
def warnings(self) -> list[Diagnostic]:
|
|
81
|
+
return [d for d in self.diagnostics if d.severity is Severity.WARNING]
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
def _sorted_diagnostics(collector: DiagnosticCollector) -> list[Diagnostic]:
|
|
85
|
+
"""收集器快照 → 按位置排序(末端归一化,成功与异常路径共用)。"""
|
|
86
|
+
return sorted(collector, key=lambda d: d.sort_key())
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
def _compile(file: File, options: CompileOptions) -> CompilationResult:
|
|
90
|
+
"""统一编译核心:File + CompileOptions → CompilationResult。
|
|
91
|
+
|
|
92
|
+
三阶段语义流水线在此**顶层组装**(各阶段互相零耦合,仅经数据模型):
|
|
93
|
+
|
|
94
|
+
Phase 1(导入求解)→ :class:`ResolvedContext`
|
|
95
|
+
→ Phase 2a(AST 构建)→ :class:`StdDocument`
|
|
96
|
+
→ Phase 2b(约束执行 + schema 校验)→ 校验后的 root
|
|
97
|
+
|
|
98
|
+
诊断:单一 :class:`DiagnosticCollector` 从词法到语义全程复用,
|
|
99
|
+
末端仅做位置排序快照,无多集合事后合并。
|
|
100
|
+
"""
|
|
101
|
+
text = file.read()
|
|
102
|
+
# 空源码 → 空配置
|
|
103
|
+
if not text.strip():
|
|
104
|
+
return CompilationResult(document=StdDocument())
|
|
105
|
+
|
|
106
|
+
# 单一诊断收集器:词法 → 语法 → 语义(Phase 1/2a/2b)全程复用
|
|
107
|
+
collector = DiagnosticCollector()
|
|
108
|
+
doc, _ = parse_source(file, collector)
|
|
109
|
+
|
|
110
|
+
sandbox_impl = Sandbox(
|
|
111
|
+
config=options.effective_sandbox(),
|
|
112
|
+
base_dir=file.root_path,
|
|
113
|
+
)
|
|
114
|
+
import_resolver = ImportResolver(sandbox=sandbox_impl)
|
|
115
|
+
resolver = TemplateGraphResolver(
|
|
116
|
+
registry=options.registry,
|
|
117
|
+
import_resolver=import_resolver,
|
|
118
|
+
schema=options.schema,
|
|
119
|
+
)
|
|
120
|
+
try:
|
|
121
|
+
# Phase 1:导入求解(模板图 / 可见名表 / 数据命名空间)
|
|
122
|
+
context = resolver.resolve(doc, file, collector)
|
|
123
|
+
|
|
124
|
+
# Phase 2a:AST 构建(约束挂载未执行)
|
|
125
|
+
std = AstBuilder().build(doc, context, collector)
|
|
126
|
+
|
|
127
|
+
# Phase 2b:约束执行 + 顶层 schema 校验(两阶段共享同一注册表实例)
|
|
128
|
+
executor = ConstraintExecutor(
|
|
129
|
+
registry=resolver.registry,
|
|
130
|
+
templates=std.templates,
|
|
131
|
+
template_scopes=context.template_scopes,
|
|
132
|
+
)
|
|
133
|
+
executor.validate(std.root, collector)
|
|
134
|
+
if resolver.schema is not None:
|
|
135
|
+
scope = (
|
|
136
|
+
context.schema_scope
|
|
137
|
+
if resolver.schema.from_file and context.schema_scope is not None
|
|
138
|
+
else context.root_scope
|
|
139
|
+
)
|
|
140
|
+
key = scope.get(resolver.schema.template)
|
|
141
|
+
if key is None:
|
|
142
|
+
raise SchemaError('schema.undefined_template', {'template': resolver.schema.template})
|
|
143
|
+
tpl = std.templates[key]
|
|
144
|
+
root = executor.apply_schema(std.root, resolver.schema, tpl, context.template_scopes[key], collector)
|
|
145
|
+
std = replace(std, root=root)
|
|
146
|
+
except SandboxError as e:
|
|
147
|
+
# 沙盒/schema 违规 → 追加到共享收集器(保留此前已收集的诊断),返回空文档(不抛出)
|
|
148
|
+
collector.add(Diagnostic(Severity.ERROR, e.code, dict(e.params), e.source))
|
|
149
|
+
return CompilationResult(
|
|
150
|
+
document=StdDocument(),
|
|
151
|
+
diagnostics=_sorted_diagnostics(collector),
|
|
152
|
+
)
|
|
153
|
+
|
|
154
|
+
# 末端:StdDocument 纯数据(不携带诊断);诊断全部由收集器快照承载
|
|
155
|
+
document = StdDocument(
|
|
156
|
+
root=std.root,
|
|
157
|
+
templates=std.templates,
|
|
158
|
+
scope=std.scope,
|
|
159
|
+
)
|
|
160
|
+
return CompilationResult(
|
|
161
|
+
document=document,
|
|
162
|
+
diagnostics=_sorted_diagnostics(collector),
|
|
163
|
+
)
|
|
164
|
+
|
|
165
|
+
|
|
166
|
+
def load(
|
|
167
|
+
path: str | Path,
|
|
168
|
+
*,
|
|
169
|
+
env: Mapping[str, str] | None = None,
|
|
170
|
+
sandbox: SandboxConfig | None = None,
|
|
171
|
+
registry: ConstraintRegistry | None = None,
|
|
172
|
+
schema: Schema | None = None,
|
|
173
|
+
) -> CompilationResult:
|
|
174
|
+
"""加载 .infd/.inft 文件并编译。
|
|
175
|
+
|
|
176
|
+
Args:
|
|
177
|
+
path: 文件路径(相对导入以此为基准)
|
|
178
|
+
其余选项(env/sandbox/registry/schema)见 :class:`CompileOptions`
|
|
179
|
+
|
|
180
|
+
沙盒/schema 违规不抛出:由编译核心转为 ERROR 诊断,返回空文档。
|
|
181
|
+
"""
|
|
182
|
+
return _compile(
|
|
183
|
+
DiskFile.from_fullpath(path),
|
|
184
|
+
CompileOptions(env=env, sandbox=sandbox, registry=registry, schema=schema),
|
|
185
|
+
)
|
|
186
|
+
|
|
187
|
+
|
|
188
|
+
def compile_source(
|
|
189
|
+
source: str,
|
|
190
|
+
*,
|
|
191
|
+
file_path: str = 'unknown',
|
|
192
|
+
env: Mapping[str, str] | None = None,
|
|
193
|
+
sandbox: SandboxConfig | None = None,
|
|
194
|
+
registry: ConstraintRegistry | None = None,
|
|
195
|
+
schema: Schema | None = None,
|
|
196
|
+
) -> CompilationResult:
|
|
197
|
+
"""编译源码字符串,返回 CompilationResult(选项见 :class:`CompileOptions`)。"""
|
|
198
|
+
file = MemFile(name=file_path, root_path=Path(file_path).parent, content=source)
|
|
199
|
+
return _compile(
|
|
200
|
+
file,
|
|
201
|
+
CompileOptions(env=env, sandbox=sandbox, registry=registry, schema=schema),
|
|
202
|
+
)
|
|
203
|
+
|
|
204
|
+
|
|
205
|
+
def safe_load(
|
|
206
|
+
path: str | Path,
|
|
207
|
+
*,
|
|
208
|
+
registry: ConstraintRegistry | None = None,
|
|
209
|
+
schema: Schema | None = None,
|
|
210
|
+
) -> CompilationResult:
|
|
211
|
+
"""零信任加载:等价于 ``load(path, sandbox=SandboxConfig.deny_all())``。
|
|
212
|
+
|
|
213
|
+
所有导入语句(``!env`` / ``!file`` / ``!from``)均报错。
|
|
214
|
+
只允许纯字段定义、模板定义与字面量值。
|
|
215
|
+
|
|
216
|
+
用途:
|
|
217
|
+
- 读取沙盒配置文件(自举:SandboxConfig.from_dict)
|
|
218
|
+
- 读取纯模板文件 (.inft)
|
|
219
|
+
- 读取不需要外部资源的配置
|
|
220
|
+
"""
|
|
221
|
+
return _compile(
|
|
222
|
+
DiskFile.from_fullpath(path),
|
|
223
|
+
CompileOptions(sandbox=SandboxConfig.deny_all(), registry=registry, schema=schema),
|
|
224
|
+
)
|
|
225
|
+
|
|
226
|
+
|
|
227
|
+
def check(
|
|
228
|
+
path: str | Path,
|
|
229
|
+
*,
|
|
230
|
+
env: Mapping[str, str] | None = None,
|
|
231
|
+
sandbox: SandboxConfig | None = None,
|
|
232
|
+
registry: ConstraintRegistry | None = None,
|
|
233
|
+
schema: Schema | None = None,
|
|
234
|
+
) -> list[Diagnostic]:
|
|
235
|
+
"""仅校验,不输出。沙盒/schema 违规已由编译核心转为 ERROR 诊断。"""
|
|
236
|
+
return load(path, env=env, sandbox=sandbox, registry=registry, schema=schema).diagnostics
|
|
237
|
+
|
|
238
|
+
|
|
239
|
+
def compile_document(
|
|
240
|
+
path: str | Path,
|
|
241
|
+
*,
|
|
242
|
+
env: Mapping[str, str] | None = None,
|
|
243
|
+
sandbox: SandboxConfig | None = None,
|
|
244
|
+
registry: ConstraintRegistry | None = None,
|
|
245
|
+
schema: Schema | None = None,
|
|
246
|
+
) -> StdDocument:
|
|
247
|
+
"""编译为 StdDocument(不经过降维):root / templates / scope(纯数据,不携带诊断)。
|
|
248
|
+
|
|
249
|
+
沙盒/schema 违规时返回空文档;诊断见 :meth:`load` 结果(``CompilationResult.diagnostics``)。
|
|
250
|
+
"""
|
|
251
|
+
result = load(path, env=env, sandbox=sandbox, registry=registry, schema=schema)
|
|
252
|
+
return result.document
|
|
@@ -0,0 +1,32 @@
|
|
|
1
|
+
"""M3 安全模型:沙盒配置、沙盒中介、顶层 Schema 约束与安全异常。
|
|
2
|
+
|
|
3
|
+
子模块:
|
|
4
|
+
- :mod:`config`:SandboxConfig(授权数据 + 工厂)
|
|
5
|
+
- :mod:`mediator`:Sandbox(系统访问中介,File 的唯一产出者)
|
|
6
|
+
- :mod:`errors`:SandboxError / SchemaError
|
|
7
|
+
- :mod:`schema`:Schema(顶层模板约束)
|
|
8
|
+
|
|
9
|
+
公共 API 从此包再导出,兼容 ``from infinity_data.sandbox import ...``。
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
from infinity_data.sandbox.config import SandboxConfig
|
|
13
|
+
from infinity_data.sandbox.errors import (
|
|
14
|
+
AccessDeniedError,
|
|
15
|
+
EnvNotAuthorizedError,
|
|
16
|
+
EnvNotSetError,
|
|
17
|
+
SandboxError,
|
|
18
|
+
SchemaError,
|
|
19
|
+
)
|
|
20
|
+
from infinity_data.sandbox.mediator import Sandbox
|
|
21
|
+
from infinity_data.sandbox.schema import Schema
|
|
22
|
+
|
|
23
|
+
__all__ = [
|
|
24
|
+
'Sandbox',
|
|
25
|
+
'SandboxConfig',
|
|
26
|
+
'SandboxError',
|
|
27
|
+
'EnvNotAuthorizedError',
|
|
28
|
+
'EnvNotSetError',
|
|
29
|
+
'AccessDeniedError',
|
|
30
|
+
'SchemaError',
|
|
31
|
+
'Schema',
|
|
32
|
+
]
|
|
@@ -0,0 +1,61 @@
|
|
|
1
|
+
"""沙盒配置:授权数据(env / allow_env / allow_files / allow_templates / strict)
|
|
2
|
+
与工厂方法。
|
|
3
|
+
|
|
4
|
+
纯数据零行为:授权匹配与访问行为见 :mod:`infinity_data.sandbox.mediator`。
|
|
5
|
+
自举场景可直接 ``SandboxConfig(**safe_load(...).value)`` 构造。
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
from dataclasses import dataclass, field
|
|
11
|
+
|
|
12
|
+
__all__ = ['SandboxConfig']
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
@dataclass
|
|
16
|
+
class SandboxConfig:
|
|
17
|
+
"""控制 .infd 文件的导入权限。默认零信任。"""
|
|
18
|
+
|
|
19
|
+
# ── 环境变量注入:key → value。命中即返回,优先于 allow_env ──
|
|
20
|
+
env: dict[str, str] = field(default_factory=lambda: {})
|
|
21
|
+
|
|
22
|
+
# ── 环境变量读取白名单:授权从真实 OS 环境(os.environ)实时读取。
|
|
23
|
+
# None = 全部允许;[] = 全部禁止(默认,零信任)──
|
|
24
|
+
allow_env: list[str] | None = field(default_factory=lambda: [])
|
|
25
|
+
|
|
26
|
+
# ── 文件导入白名单(glob 模式;None = 全部允许)──
|
|
27
|
+
allow_files: list[str] | None = field(default_factory=lambda: [])
|
|
28
|
+
|
|
29
|
+
# ── 模板导入白名单(glob 模式;None = 全部允许)──
|
|
30
|
+
allow_templates: list[str] | None = field(default_factory=lambda: [])
|
|
31
|
+
|
|
32
|
+
# ── 严格模式:True 白名单外导入抛 SandboxError;False 仅警告 ──
|
|
33
|
+
strict: bool = True
|
|
34
|
+
|
|
35
|
+
# ── 工厂方法 ──────────────────────────────────────
|
|
36
|
+
|
|
37
|
+
@staticmethod
|
|
38
|
+
def deny_all() -> SandboxConfig:
|
|
39
|
+
"""零信任"""
|
|
40
|
+
return SandboxConfig()
|
|
41
|
+
|
|
42
|
+
@staticmethod
|
|
43
|
+
def full_access() -> SandboxConfig:
|
|
44
|
+
"""全权限:全部环境变量实时读取 + 任意文件/模板。"""
|
|
45
|
+
return SandboxConfig(
|
|
46
|
+
allow_env=None,
|
|
47
|
+
allow_files=None,
|
|
48
|
+
allow_templates=None,
|
|
49
|
+
)
|
|
50
|
+
|
|
51
|
+
@staticmethod
|
|
52
|
+
def development() -> SandboxConfig:
|
|
53
|
+
"""开发模式:当前目录全权限 + 全部环境变量实时读取。
|
|
54
|
+
|
|
55
|
+
``**/*`` 匹配任意深度(``**`` 含零段,因此也能命中根级文件)。
|
|
56
|
+
"""
|
|
57
|
+
return SandboxConfig(
|
|
58
|
+
allow_env=None,
|
|
59
|
+
allow_files=['**/*'],
|
|
60
|
+
allow_templates=['**/*'],
|
|
61
|
+
)
|
|
@@ -0,0 +1,106 @@
|
|
|
1
|
+
"""沙盒安全异常体系(自有,独立于统一诊断模型)。
|
|
2
|
+
|
|
3
|
+
错误模型:
|
|
4
|
+
- 词法/语法/语义错误统一为 :class:`Diagnostic`(纯数据,从不抛异常)
|
|
5
|
+
- 沙盒错误必须**中止编译**(控制流),故保留异常;携带 ``code/params/source``,
|
|
6
|
+
由编译核心(``infinity_data.pipeline``)捕获并转换为 ERROR 诊断,返回空文档
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
from typing import Any
|
|
12
|
+
|
|
13
|
+
from infinity_data.infra.diagnostics import diagnostic_define, register_diagnostic_define, render_message
|
|
14
|
+
from infinity_data.infra.location import SourceRange, format_location
|
|
15
|
+
|
|
16
|
+
__all__ = [
|
|
17
|
+
'SandboxError',
|
|
18
|
+
'EnvNotAuthorizedError',
|
|
19
|
+
'EnvNotSetError',
|
|
20
|
+
'AccessDeniedError',
|
|
21
|
+
'SchemaError',
|
|
22
|
+
]
|
|
23
|
+
|
|
24
|
+
register_diagnostic_define(
|
|
25
|
+
diagnostic_define(
|
|
26
|
+
'sandbox.env_unauthorized',
|
|
27
|
+
'[{location}] 环境变量 {name!r} 未在沙盒授权(!env import)',
|
|
28
|
+
en='[{location}] environment variable {name!r} is not authorized (!env import)',
|
|
29
|
+
),
|
|
30
|
+
diagnostic_define(
|
|
31
|
+
'sandbox.env_not_set',
|
|
32
|
+
'[{location}] 环境变量 {name!r} 已授权但当前进程未设置',
|
|
33
|
+
en='[{location}] environment variable {name!r} is authorized but not set in this process',
|
|
34
|
+
),
|
|
35
|
+
diagnostic_define(
|
|
36
|
+
'sandbox.access_denied',
|
|
37
|
+
'[{location}] {label}超出沙盒授权: {path_src}',
|
|
38
|
+
en='[{location}] {label} denied by sandbox: {path_src}',
|
|
39
|
+
),
|
|
40
|
+
diagnostic_define(
|
|
41
|
+
'schema.undefined_template', '未定义的 schema 模板 {template!r}', en='undefined schema template {template!r}'
|
|
42
|
+
),
|
|
43
|
+
diagnostic_define(
|
|
44
|
+
'schema.failed', '顶层 schema 校验失败: {detail}', en='top-level schema validation failed: {detail}'
|
|
45
|
+
),
|
|
46
|
+
diagnostic_define(
|
|
47
|
+
'schema.extra_fields',
|
|
48
|
+
'顶层 schema 不允许额外字段: {fields}',
|
|
49
|
+
en='top-level schema does not allow extra fields: {fields}',
|
|
50
|
+
),
|
|
51
|
+
diagnostic_define('schema.extra_fields_lenient', '顶层 schema 存在额外字段(已保留): {fields}'),
|
|
52
|
+
diagnostic_define('schema.missing_required', '{path_prefix}顶层 schema 缺少必填字段 {field!r}(模板 {template})'),
|
|
53
|
+
)
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
class SandboxError(Exception):
|
|
57
|
+
"""沙盒安全异常基类(自有体系)。"""
|
|
58
|
+
|
|
59
|
+
def __init__(
|
|
60
|
+
self,
|
|
61
|
+
code: str,
|
|
62
|
+
params: dict[str, Any] | None = None,
|
|
63
|
+
source: SourceRange | None = None,
|
|
64
|
+
) -> None:
|
|
65
|
+
super().__init__(code)
|
|
66
|
+
self.code: str = code
|
|
67
|
+
self.params: dict[str, Any] = params or {}
|
|
68
|
+
self.source: SourceRange | None = source
|
|
69
|
+
|
|
70
|
+
@property
|
|
71
|
+
def location(self) -> str:
|
|
72
|
+
return format_location(self.source)
|
|
73
|
+
|
|
74
|
+
@property
|
|
75
|
+
def message(self) -> str:
|
|
76
|
+
return render_message(self.code, self.params, location=self.location)
|
|
77
|
+
|
|
78
|
+
def __str__(self) -> str:
|
|
79
|
+
return self.message
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
class EnvNotAuthorizedError(SandboxError):
|
|
83
|
+
"""``!env`` 引用的变量未在沙盒授权。"""
|
|
84
|
+
|
|
85
|
+
def __init__(self, name: str, source: SourceRange | None = None) -> None:
|
|
86
|
+
super().__init__('sandbox.env_unauthorized', {'name': name}, source)
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
class EnvNotSetError(SandboxError):
|
|
90
|
+
"""变量已授权但当前进程未设置。"""
|
|
91
|
+
|
|
92
|
+
def __init__(self, name: str, source: SourceRange | None = None) -> None:
|
|
93
|
+
super().__init__('sandbox.env_not_set', {'name': name}, source)
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
class AccessDeniedError(SandboxError):
|
|
97
|
+
"""文件/模板导入超出 glob 白名单。"""
|
|
98
|
+
|
|
99
|
+
def __init__(self, label: str, path: str, source: SourceRange | None = None) -> None:
|
|
100
|
+
super().__init__('sandbox.access_denied', {'label': label, 'path_src': path}, source)
|
|
101
|
+
|
|
102
|
+
|
|
103
|
+
class SchemaError(SandboxError):
|
|
104
|
+
"""顶层 schema 校验失败(中止编译);以 ``code`` 区分具体原因。"""
|
|
105
|
+
|
|
106
|
+
pass
|