infinity-data 1.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (52) hide show
  1. infinity_data/__init__.py +41 -0
  2. infinity_data/emit/__init__.py +9 -0
  3. infinity_data/emit/converter.py +55 -0
  4. infinity_data/frontend.py +37 -0
  5. infinity_data/infra/__init__.py +4 -0
  6. infinity_data/infra/diagnostics.py +212 -0
  7. infinity_data/infra/file.py +83 -0
  8. infinity_data/infra/ll1_stream.py +71 -0
  9. infinity_data/infra/location.py +80 -0
  10. infinity_data/infra/path.py +46 -0
  11. infinity_data/parser/__init__.py +79 -0
  12. infinity_data/parser/diagnostics.py +92 -0
  13. infinity_data/parser/models.py +295 -0
  14. infinity_data/parser/parser.py +988 -0
  15. infinity_data/parser/token_stream.py +128 -0
  16. infinity_data/pipeline.py +252 -0
  17. infinity_data/sandbox/__init__.py +32 -0
  18. infinity_data/sandbox/config.py +61 -0
  19. infinity_data/sandbox/errors.py +106 -0
  20. infinity_data/sandbox/mediator.py +214 -0
  21. infinity_data/sandbox/schema.py +21 -0
  22. infinity_data/semantic/__init__.py +59 -0
  23. infinity_data/semantic/builder/__init__.py +29 -0
  24. infinity_data/semantic/builder/builder.py +523 -0
  25. infinity_data/semantic/builder/models.py +163 -0
  26. infinity_data/semantic/constraints.py +163 -0
  27. infinity_data/semantic/diagnostics.py +172 -0
  28. infinity_data/semantic/executor/__init__.py +10 -0
  29. infinity_data/semantic/executor/executor.py +259 -0
  30. infinity_data/semantic/registry/__init__.py +168 -0
  31. infinity_data/semantic/registry/_core.py +212 -0
  32. infinity_data/semantic/registry/dict_constraints.py +56 -0
  33. infinity_data/semantic/registry/general.py +362 -0
  34. infinity_data/semantic/registry/logic.py +134 -0
  35. infinity_data/semantic/registry/types.py +136 -0
  36. infinity_data/semantic/resolver/__init__.py +20 -0
  37. infinity_data/semantic/resolver/imports.py +191 -0
  38. infinity_data/semantic/resolver/models.py +67 -0
  39. infinity_data/semantic/resolver/resolver.py +345 -0
  40. infinity_data/tokenizer/__init__.py +16 -0
  41. infinity_data/tokenizer/char_stream.py +72 -0
  42. infinity_data/tokenizer/diagnostics.py +62 -0
  43. infinity_data/tokenizer/finalizer.py +271 -0
  44. infinity_data/tokenizer/models/__init__.py +17 -0
  45. infinity_data/tokenizer/models/raw_tokens.py +61 -0
  46. infinity_data/tokenizer/models/tokens.py +245 -0
  47. infinity_data/tokenizer/tokenizer.py +531 -0
  48. infinity_data-1.0.0.dist-info/METADATA +60 -0
  49. infinity_data-1.0.0.dist-info/RECORD +52 -0
  50. infinity_data-1.0.0.dist-info/WHEEL +5 -0
  51. infinity_data-1.0.0.dist-info/licenses/LICENSE +21 -0
  52. infinity_data-1.0.0.dist-info/top_level.txt +1 -0
@@ -0,0 +1,988 @@
1
+ """递归下降语法分析器,将 Token 流转换为 RawAst"""
2
+
3
+ from collections.abc import Iterable
4
+ from typing import Any, TypeVar
5
+
6
+ from infinity_data.infra.diagnostics import DiagnosticCollector
7
+ from infinity_data.infra.ll1_stream import NoNextType
8
+ from infinity_data.parser.diagnostics import diag
9
+ from infinity_data.parser.models import (
10
+ ArrayValue,
11
+ Constraint,
12
+ ConstraintCall,
13
+ ConstraintIdent,
14
+ ConstraintLiteral,
15
+ Constraints,
16
+ ConstraintStmt,
17
+ DictValue,
18
+ Document,
19
+ DollarValue,
20
+ EnvImportStmt,
21
+ ErrorConstraint,
22
+ ErrorStatement,
23
+ ErrorValue,
24
+ Field,
25
+ FileImportItem,
26
+ FileImportStmt,
27
+ JsonPathIndex,
28
+ JsonPathKey,
29
+ JsonPathSegment,
30
+ LiteralValue,
31
+ Statement,
32
+ TemplateCallValue,
33
+ TemplateConfig,
34
+ TemplateDef,
35
+ TemplateField,
36
+ TemplateImportItem,
37
+ TemplateImportStmt,
38
+ Value,
39
+ )
40
+ from infinity_data.parser.token_stream import TokenStream
41
+ from infinity_data.tokenizer.models.raw_tokens import RawTokenType, SourceRange
42
+ from infinity_data.tokenizer.models.tokens import (
43
+ BoolToken,
44
+ ColonToken,
45
+ CommaToken,
46
+ DollarToken,
47
+ DotToken,
48
+ EnvImportToken,
49
+ EqualsToken,
50
+ FileImportToken,
51
+ FloatToken,
52
+ FromImportToken,
53
+ IdentifierToken,
54
+ IntegerToken,
55
+ LangleToken,
56
+ LbraceToken,
57
+ LbracketToken,
58
+ LparenToken,
59
+ NewlineToken,
60
+ NoexistToken,
61
+ NullToken,
62
+ QuestionToken,
63
+ RangleToken,
64
+ RbraceToken,
65
+ RbracketToken,
66
+ RparenToken,
67
+ SinglelineStringToken,
68
+ StringToken,
69
+ TildeToken,
70
+ Token,
71
+ )
72
+
73
+ _TToken = TypeVar('_TToken', bound=Token)
74
+
75
+ # 模板配置项白名单(dataclass 字段即白名单,此处按类型分组)
76
+ _CONFIG_BOOL_KEYS = frozenset({'allow_extra', 'positional'})
77
+ _CONFIG_STR_KEYS = frozenset({'description'})
78
+ _TEMPLATE_CONFIG_VALID = 'allow_extra, positional, description'
79
+
80
+
81
+ def _py_describe(v: Any) -> str:
82
+ """config 值类型的人类可读描述(诊断用)。"""
83
+ if isinstance(v, bool):
84
+ return 'bool'
85
+ if isinstance(v, str):
86
+ return 'string'
87
+ if isinstance(v, int):
88
+ return 'integer'
89
+ if isinstance(v, float):
90
+ return 'float'
91
+ return type(v).__name__
92
+
93
+
94
+ class Parser:
95
+ """递归下降解析器。"""
96
+
97
+ def __init__(
98
+ self,
99
+ source: Iterable[Token],
100
+ error_collector: DiagnosticCollector | None = None,
101
+ ) -> None:
102
+ self._errors = error_collector if error_collector is not None else DiagnosticCollector()
103
+ self._stream: TokenStream = TokenStream(source, self._errors)
104
+
105
+ @property
106
+ def error_collector(self) -> DiagnosticCollector:
107
+ return self._errors
108
+
109
+ # ═══════════════════════════════════════════════════════
110
+ # 公开入口
111
+ # ═══════════════════════════════════════════════════════
112
+
113
+ def parse(self) -> Document:
114
+ # 懒初始化:预读第一个 token
115
+ first_tok = self._stream.peek()
116
+ if isinstance(first_tok, NoNextType) or self._stream.eof():
117
+ self._errors.add(diag('parse.empty_token_list', {}, SourceRange.empty()))
118
+ return Document(source=SourceRange.empty())
119
+ doc = Document(source=first_tok.raw.source)
120
+ while True:
121
+ stmt = self._parse_statement()
122
+ if stmt is None:
123
+ break
124
+ doc.statements.append(stmt)
125
+ return doc
126
+
127
+ # ═══════════════════════════════════════════════════════
128
+ # 顶层
129
+ # ═══════════════════════════════════════════════════════
130
+
131
+ def _parse_statement(self) -> Statement | None:
132
+ """解析一条顶层语句。"""
133
+ self._stream.skip_newlines()
134
+
135
+ # EofToken 或物理耗尽均视为流结束
136
+ if self._stream.eof():
137
+ return None
138
+
139
+ match self._stream.peek():
140
+ case EnvImportToken() | FileImportToken() | FromImportToken():
141
+ return self._parse_import_statement()
142
+ case TildeToken():
143
+ return self._parse_template_def()
144
+ case IdentifierToken():
145
+ return self._parse_field()
146
+ case ColonToken():
147
+ return self._parse_constraint_stmt()
148
+ case _:
149
+ bad_tok = self._stream.advance()
150
+ self._errors.add(
151
+ diag('parse.unrecognized_statement', {'name': bad_tok.raw.type.name}, bad_tok.raw.source)
152
+ )
153
+ return ErrorStatement(
154
+ source=bad_tok.raw.source,
155
+ message=f'无法识别的顶层 token: {bad_tok.raw.type.name}',
156
+ )
157
+
158
+ # ═══════════════════════════════════════════════════════
159
+ # 导入语句
160
+ # ═══════════════════════════════════════════════════════
161
+
162
+ def _parse_import_statement(self) -> Statement:
163
+ """分发 !env / !file / !from 导入语句"""
164
+ match self._stream.peek():
165
+ case EnvImportToken() as tok:
166
+ self._stream.advance()
167
+ return self._parse_env_import(tok)
168
+ case FileImportToken() as tok:
169
+ self._stream.advance()
170
+ return self._parse_file_import(tok)
171
+ case FromImportToken() as tok:
172
+ self._stream.advance()
173
+ return self._parse_template_import(tok)
174
+ case _:
175
+ first = self._stream.peek()
176
+ return ErrorStatement(source=self._stream.span_from(first), message='无法识别的导入语句')
177
+
178
+ def _peek_keyword(self, name: str) -> bool:
179
+ """当前 token 是否为名为 ``name`` 的标识符(import/as 已降级为标识符)。"""
180
+ tok = self._stream.peek()
181
+ return isinstance(tok, IdentifierToken) and tok.name == name
182
+
183
+ def _expect_keyword(self, name: str) -> None:
184
+ """期望语法位置上的关键字名(import/as),含错误恢复。"""
185
+ if self._peek_keyword(name):
186
+ self._stream.advance()
187
+ return
188
+ tok = self._stream.peek()
189
+ if isinstance(tok, NoNextType):
190
+ self._errors.add(
191
+ diag(
192
+ 'parse.unexpected_token',
193
+ {'expected': f'关键字 {name!r}', 'actual': 'EOF'},
194
+ self._stream.span_from(None),
195
+ )
196
+ )
197
+ return
198
+ self._errors.add(
199
+ diag(
200
+ 'parse.unexpected_token', {'expected': f'关键字 {name!r}', 'actual': tok.raw.type.name}, tok.raw.source
201
+ )
202
+ )
203
+ self._stream.advance()
204
+
205
+ def _parse_env_import(self, kw_tok: EnvImportToken) -> EnvImportStmt:
206
+ """!env import NAME [as NEW_NAME]"""
207
+ self._expect_keyword('import')
208
+ name_tok = self._stream.expect(IdentifierToken)
209
+
210
+ alias = None
211
+ if self._peek_keyword('as'):
212
+ self._stream.advance()
213
+ alias = self._stream.expect(IdentifierToken).name
214
+
215
+ self._stream.skip_newlines()
216
+ return EnvImportStmt(
217
+ source=self._stream.span_from(kw_tok),
218
+ name=name_tok.name,
219
+ alias=alias,
220
+ )
221
+
222
+ def _parse_file_import(self, kw_tok: FileImportToken) -> FileImportStmt:
223
+ """!file "path" [as <format>] import .path.to.key as alias, ..."""
224
+ path_tok = self._stream.expect(SinglelineStringToken)
225
+
226
+ # 可选 as <format>
227
+ fmt = None
228
+ if self._peek_keyword('as'):
229
+ self._stream.advance()
230
+ fmt = self._stream.expect(IdentifierToken).name
231
+
232
+ self._expect_keyword('import')
233
+
234
+ # 导入项列表
235
+ items: list[FileImportItem] = []
236
+ items.append(self._parse_file_import_item())
237
+
238
+ while isinstance(self._stream.peek(), CommaToken):
239
+ self._stream.advance()
240
+ items.append(self._parse_file_import_item())
241
+
242
+ self._stream.skip_newlines()
243
+ return FileImportStmt(
244
+ source=self._stream.span_from(kw_tok),
245
+ file_path=path_tok.value,
246
+ format=fmt,
247
+ imports=items,
248
+ )
249
+
250
+ def _parse_file_import_item(self) -> FileImportItem:
251
+ """解析单个 .path.to.key as alias。
252
+
253
+ alias 必须提供——import 的值需要通过 $alias 引用。
254
+ """
255
+ first = self._stream.peek()
256
+ json_path = self._parse_json_path()
257
+
258
+ self._expect_keyword('as')
259
+ alias = self._stream.expect(IdentifierToken).name
260
+
261
+ self._stream.skip_newlines()
262
+ return FileImportItem(
263
+ source=self._stream.span_from(first),
264
+ json_path=json_path,
265
+ alias=alias,
266
+ )
267
+
268
+ def _parse_json_path(self) -> list[JsonPathSegment]:
269
+ """解析 JSON 路径为结构化段列表。
270
+
271
+ Token 序列示例:
272
+ .server.host → [DOT] [server] [DOT] [host]
273
+ .a.b[0]."c" → [DOT] [a] [DOT] [b] [[] [0] []] [DOT] ["c"]
274
+ . → [DOT]
275
+
276
+ 语法: "." identifier ( "." identifier | "[" integer "]" | "." string )*
277
+ 返回: 路径段列表;空列表表示导入整个文件(仅 .)
278
+ """
279
+ segments: list[JsonPathSegment] = []
280
+
281
+ # 路径必须以 . 起始
282
+ tok = self._stream.peek()
283
+ if not isinstance(tok, DotToken):
284
+ return []
285
+ self._stream.advance()
286
+
287
+ # 第一个段:标识符(非 as)或字符串;`.` 后是 as(别名关键字)→ 整文件导入
288
+ tok = self._stream.peek()
289
+ if isinstance(tok, IdentifierToken) and tok.name != 'as':
290
+ segments.append(JsonPathKey(source=tok.raw.source, key=tok.name))
291
+ self._stream.advance()
292
+ elif isinstance(tok, SinglelineStringToken):
293
+ segments.append(JsonPathKey(source=tok.raw.source, key=tok.value))
294
+ self._stream.advance()
295
+ else:
296
+ # 只有 .(或 . as alias)→ 导入整个文件
297
+ return []
298
+
299
+ # 后续段:".key" 或 "[index]"
300
+ while not self._stream.eof():
301
+ match self._stream.peek():
302
+ case DotToken():
303
+ self._stream.advance()
304
+ match self._stream.peek():
305
+ case IdentifierToken(name=name) as id_tok:
306
+ segments.append(JsonPathKey(source=self._stream.single_span(id_tok), key=name))
307
+ self._stream.advance()
308
+ case SinglelineStringToken(value=value) as str_tok:
309
+ segments.append(JsonPathKey(source=self._stream.single_span(str_tok), key=value))
310
+ self._stream.advance()
311
+ case _:
312
+ self._report_invalid_json_path(':. 后缺少段名')
313
+ break
314
+
315
+ case LbracketToken():
316
+ lbracket = self._stream.advance()
317
+ match self._stream.peek():
318
+ case IntegerToken(value=value):
319
+ self._stream.advance()
320
+ if isinstance(self._stream.peek(), RbracketToken):
321
+ self._stream.advance()
322
+ else:
323
+ self._report_invalid_json_path(':[ 下标后缺少 ]')
324
+ segments.append(JsonPathIndex(source=self._stream.span_from(lbracket), index=value))
325
+ case _:
326
+ self._report_invalid_json_path(':[ 后须为整数下标')
327
+ break
328
+
329
+ case _:
330
+ break
331
+
332
+ return segments
333
+
334
+ def _report_invalid_json_path(self, detail: str) -> None:
335
+ """JSON path 段无效 → 报 parse.invalid_json_path(指向当前 token)。"""
336
+ tok = self._stream.peek()
337
+ if isinstance(tok, NoNextType):
338
+ return
339
+ self._errors.add(diag('parse.invalid_json_path', {'detail': detail}, tok.raw.source))
340
+
341
+ def _parse_template_import(self, kw_tok: FromImportToken) -> TemplateImportStmt:
342
+ """!from "path" import Name1 [as Alias1], Name2, ..."""
343
+ path_tok = self._stream.expect(SinglelineStringToken)
344
+ self._expect_keyword('import')
345
+
346
+ items: list[TemplateImportItem] = []
347
+ items.append(self._parse_template_import_item())
348
+ while isinstance(self._stream.peek(), CommaToken):
349
+ self._stream.advance()
350
+ items.append(self._parse_template_import_item())
351
+
352
+ self._stream.skip_newlines()
353
+ return TemplateImportStmt(
354
+ source=self._stream.span_from(kw_tok),
355
+ from_path=path_tok.value,
356
+ items=items,
357
+ )
358
+
359
+ def _parse_template_import_item(self) -> TemplateImportItem:
360
+ """解析单个导入项: Name [as Alias]。"""
361
+ first = self._stream.peek()
362
+ name_tok = self._stream.expect(IdentifierToken)
363
+
364
+ alias = None
365
+ if self._peek_keyword('as'):
366
+ self._stream.advance()
367
+ alias = self._stream.expect(IdentifierToken).name
368
+
369
+ return TemplateImportItem(
370
+ source=self._stream.span_from(first),
371
+ name=name_tok.name,
372
+ alias=alias,
373
+ )
374
+
375
+ # ═══════════════════════════════════════════════════════
376
+ # 结构级约束语句(顶层)
377
+ # ═══════════════════════════════════════════════════════
378
+
379
+ def _parse_constraint_stmt(self) -> ConstraintStmt:
380
+ """顶层结构级约束: ``: <constraint, ...>`` 或 ``: constraint``。
381
+
382
+ 顶层是隐式 dict,``:`` 起始的语句约束编译产物 root 的整体。
383
+ """
384
+ first = self._stream.peek()
385
+ self._stream.advance() # 消费 ':'
386
+ parsed = self._parse_constraints()
387
+ self._stream.skip_separators()
388
+ return ConstraintStmt(
389
+ source=self._stream.span_from(first),
390
+ constraints=parsed.constraints,
391
+ )
392
+
393
+ # ═══════════════════════════════════════════════════════
394
+ # 模板定义
395
+ # ═══════════════════════════════════════════════════════
396
+
397
+ def _parse_template_def(self) -> TemplateDef:
398
+ """~Name [(config...)] { template_fields... }"""
399
+ first = self._stream.peek()
400
+ self._stream.expect(TildeToken)
401
+ name_tok = self._stream.expect(IdentifierToken)
402
+
403
+ # 可选模板配置参数: ~Name(allow_extra=true, ...)
404
+ # 语法层解析为类型化 TemplateConfig(未知键 / 类型错 / 非字面量 → 诊断)
405
+ config = TemplateConfig()
406
+ if isinstance(self._stream.peek(), LparenToken):
407
+ self._stream.advance()
408
+ self._stream.skip_newlines()
409
+ missing_sep_reported = [False]
410
+ while not self._stream.check(RawTokenType.RPAREN) and not self._stream.check(RawTokenType.EOF):
411
+ key_tok = self._stream.expect(IdentifierToken)
412
+ self._stream.expect(EqualsToken)
413
+ value = self._parse_value()
414
+ self._apply_template_config(config, key_tok, value)
415
+ had_sep = self._stream.skip_separators()
416
+ self._missing_separator(
417
+ had_sep,
418
+ isinstance(self._stream.peek(), IdentifierToken),
419
+ RawTokenType.RPAREN,
420
+ missing_sep_reported,
421
+ )
422
+ self._stream.expect(RparenToken)
423
+
424
+ self._stream.expect(LbraceToken)
425
+ self._stream.skip_newlines()
426
+
427
+ fields: list[TemplateField] = []
428
+ constraints: list[Constraint] = []
429
+ missing_sep_reported = [False]
430
+ while not self._stream.check(RawTokenType.RBRACE) and not self._stream.check(RawTokenType.EOF):
431
+ # 结构级约束: : <...>
432
+ if isinstance(self._stream.peek(), ColonToken):
433
+ self._stream.advance()
434
+ parsed = self._parse_constraints()
435
+ constraints.extend(parsed.constraints)
436
+ else:
437
+ fields.append(self._parse_template_field())
438
+ had_sep = self._stream.skip_separators()
439
+ self._missing_separator(
440
+ had_sep,
441
+ isinstance(self._stream.peek(), (IdentifierToken, ColonToken)),
442
+ RawTokenType.RBRACE,
443
+ missing_sep_reported,
444
+ )
445
+
446
+ self._stream.expect(RbraceToken)
447
+ self._stream.skip_newlines()
448
+
449
+ return TemplateDef(
450
+ source=self._stream.span_from(first),
451
+ name=name_tok.name,
452
+ fields=fields,
453
+ config=config,
454
+ constraints=constraints,
455
+ )
456
+
457
+ def _apply_template_config(
458
+ self,
459
+ config: TemplateConfig,
460
+ key_tok: IdentifierToken,
461
+ value: Value,
462
+ ) -> None:
463
+ """模板头部配置项 → 类型化字段;未知键 / 类型错 / 非字面量 → 语法诊断。
464
+
465
+ config 值是纯字面量(布尔 / 字符串 / 整数),不支持 ``$`` 引用等复杂值。
466
+ """
467
+ key = key_tok.name
468
+ if not isinstance(value, LiteralValue):
469
+ self._errors.add(diag('parse.template_config_value', {'key': key}, value.source))
470
+ return
471
+ py = self._literal_config_value(value)
472
+ if key in _CONFIG_BOOL_KEYS:
473
+ if isinstance(py, bool):
474
+ setattr(config, key, py)
475
+ else:
476
+ self._errors.add(
477
+ diag(
478
+ 'parse.template_config_type',
479
+ {'key': key, 'expected': 'bool', 'actual': _py_describe(py)},
480
+ key_tok.raw.source,
481
+ )
482
+ )
483
+ elif key in _CONFIG_STR_KEYS:
484
+ if isinstance(py, str):
485
+ setattr(config, key, py)
486
+ else:
487
+ self._errors.add(
488
+ diag(
489
+ 'parse.template_config_type',
490
+ {'key': key, 'expected': 'str', 'actual': _py_describe(py)},
491
+ key_tok.raw.source,
492
+ )
493
+ )
494
+ else:
495
+ self._errors.add(
496
+ diag(
497
+ 'parse.template_config_unknown',
498
+ {'key': key, 'valid': _TEMPLATE_CONFIG_VALID},
499
+ key_tok.raw.source,
500
+ )
501
+ )
502
+
503
+ @staticmethod
504
+ def _literal_config_value(lit: LiteralValue) -> Any:
505
+ """LiteralValue → Python 值(config 值限定为字面量)。"""
506
+ match lit.value:
507
+ case BoolToken(value=b):
508
+ return b
509
+ case StringToken(value=v):
510
+ return v
511
+ case IntegerToken(value=v):
512
+ return v
513
+ case FloatToken(value=v):
514
+ return v
515
+ case _:
516
+ return None
517
+
518
+ def _parse_template_field(self) -> TemplateField:
519
+ """解析模板内部字段:必须有类型标注,默认值可选。"""
520
+ first = self._stream.peek()
521
+ name_tok = self._stream.expect(IdentifierToken)
522
+
523
+ # 类型标注(模板字段必须):缺失或为空 → 报错并跳过该字段
524
+ if isinstance(self._stream.peek(), ColonToken):
525
+ self._stream.advance()
526
+ constraints = self._parse_constraints()
527
+ if not constraints.constraints:
528
+ self._errors.add(
529
+ diag(
530
+ 'parse.template_field_no_constraint',
531
+ {'field': name_tok.name},
532
+ self._stream.single_span(name_tok),
533
+ )
534
+ )
535
+ else:
536
+ self._errors.add(
537
+ diag(
538
+ 'parse.template_field_no_constraint',
539
+ {'field': name_tok.name},
540
+ self._stream.single_span(name_tok),
541
+ )
542
+ )
543
+ self._skip_to_field_boundary()
544
+ return TemplateField(
545
+ source=self._stream.single_span(name_tok),
546
+ name=name_tok.name,
547
+ constraints=Constraints(source=self._stream.single_span(name_tok)),
548
+ default_value=None,
549
+ )
550
+
551
+ # 默认值(可选,省略 = 必填)
552
+ default_value: Value | None = None
553
+ if isinstance(self._stream.peek(), EqualsToken):
554
+ self._stream.advance()
555
+ default_value = self._parse_value()
556
+ elif self._starts_value(self._stream.peek()):
557
+ default_value = self._parse_value()
558
+
559
+ return TemplateField(
560
+ source=self._stream.span_from(first),
561
+ name=name_tok.name,
562
+ constraints=constraints,
563
+ default_value=default_value,
564
+ )
565
+
566
+ def _skip_to_field_boundary(self) -> None:
567
+ """跳过当前模板字段的残余 token,直到分隔符或模板闭合符(错误恢复)。"""
568
+ while not self._stream.eof() and not isinstance(self._stream.peek(), (CommaToken, NewlineToken, RbraceToken)):
569
+ self._stream.advance()
570
+
571
+ # ═══════════════════════════════════════════════════════
572
+ # 字段
573
+ # ═══════════════════════════════════════════════════════
574
+
575
+ def _parse_field(self) -> Field:
576
+ """解析普通字段:name[: type] [= value]
577
+
578
+ 支持省略等号:name { ... }, name [ ... ], name Template(...)
579
+ """
580
+ first = self._stream.peek()
581
+ name_tok = self._stream.expect(IdentifierToken)
582
+
583
+ # 类型标注 name: <...> 或 name: type 或 name: type?
584
+ constraints: Constraints | None = None
585
+ if isinstance(self._stream.peek(), ColonToken):
586
+ self._stream.advance()
587
+ constraints = self._parse_constraints()
588
+
589
+ # 值:有 = 时直接解析;省略等号仅限复合值(dict/array)与模板调用
590
+ value: Value | None = None
591
+ tok = self._stream.peek()
592
+ if isinstance(tok, EqualsToken):
593
+ self._stream.advance()
594
+ value = self._parse_value()
595
+ elif isinstance(tok, (LbraceToken, LbracketToken, IdentifierToken)):
596
+ value = self._parse_value()
597
+ elif self._starts_value(tok):
598
+ # 字面量 / $ 引用省略等号 → 报错但仍解析(lint 式,尽力恢复)
599
+ self._errors.add(
600
+ diag('parse.field_requires_equals', {'name': name_tok.name}, self._stream.single_span(name_tok))
601
+ )
602
+ value = self._parse_value()
603
+
604
+ return Field(
605
+ source=self._stream.span_from(first),
606
+ name=name_tok.name,
607
+ constraints=constraints,
608
+ value=value,
609
+ )
610
+
611
+ @staticmethod
612
+ def _starts_value(tok: Token | NoNextType | None) -> bool:
613
+ """判断 token 是否可以起始一个值。"""
614
+ if isinstance(tok, NoNextType) or tok is None:
615
+ return False
616
+ return isinstance(
617
+ tok,
618
+ (
619
+ StringToken,
620
+ IntegerToken,
621
+ FloatToken,
622
+ BoolToken,
623
+ NullToken,
624
+ NoexistToken,
625
+ LbraceToken,
626
+ LbracketToken,
627
+ IdentifierToken,
628
+ DollarToken,
629
+ ),
630
+ )
631
+
632
+ @staticmethod
633
+ def _starts_constraint(tok: Token | NoNextType | None) -> bool:
634
+ """判断 token 是否可以起始一个约束(标识符 / ? / 字面量)。"""
635
+ if isinstance(tok, NoNextType) or tok is None:
636
+ return False
637
+ return isinstance(
638
+ tok,
639
+ (IdentifierToken, QuestionToken, StringToken, IntegerToken, FloatToken, BoolToken, NullToken),
640
+ )
641
+
642
+ # ═══════════════════════════════════════════════════════
643
+ # 分隔符检测(元素间必须显式分隔)
644
+ # ═══════════════════════════════════════════════════════
645
+
646
+ def _missing_separator(
647
+ self,
648
+ had_separator: bool,
649
+ starts_next: bool,
650
+ closing_type: RawTokenType,
651
+ reported: list[bool],
652
+ ) -> None:
653
+ """元素之间缺少显式分隔符(逗号/换行)→ 报 parse.missing_separator。
654
+
655
+ - ``had_separator``:刚消费过逗号/换行(有分隔则无事)
656
+ - ``starts_next``:当前 token 是否起始下一个元素(避免与其他恢复错误叠加)
657
+ - 每容器只报一次(``reported`` 标志),其余缺口静默恢复
658
+ """
659
+ if had_separator or not starts_next or reported[0]:
660
+ return
661
+ if self._stream.check(closing_type) or self._stream.check(RawTokenType.EOF):
662
+ return
663
+ tok = self._stream.peek()
664
+ if not isinstance(tok, NoNextType):
665
+ self._errors.add(diag('parse.missing_separator', {}, tok.raw.source))
666
+ reported[0] = True
667
+
668
+ # ═══════════════════════════════════════════════════════
669
+ # 约束列表
670
+ # ═══════════════════════════════════════════════════════
671
+
672
+ def _parse_constraints(self) -> Constraints:
673
+ """解析约束。
674
+
675
+ 支持:
676
+ - int, str, bool, float, list, dict, ?, object
677
+ - type? → one(type, ?)
678
+ - <constraint, constraint, ...> → all(constraint, ...)
679
+ - <any(...)>, <one(...)>, <not(...)>, <all(...)>
680
+ """
681
+ first = self._stream.peek()
682
+
683
+ match first:
684
+ case LangleToken():
685
+ self._stream.advance()
686
+ self._stream.skip_newlines()
687
+ constraints: list[Constraint] = []
688
+ missing_sep_reported = [False]
689
+ while not self._stream.check(RawTokenType.RANGLE) and not self._stream.check(RawTokenType.EOF):
690
+ constraints.append(self._parse_constraint())
691
+ had_sep = self._stream.skip_separators()
692
+ self._missing_separator(
693
+ had_sep,
694
+ self._starts_constraint(self._stream.peek()),
695
+ RawTokenType.RANGLE,
696
+ missing_sep_reported,
697
+ )
698
+ self._stream.expect(RangleToken)
699
+ return Constraints(source=self._stream.span_from(first), constraints=constraints)
700
+
701
+ case IdentifierToken(name=name):
702
+ self._stream.advance()
703
+ ident = ConstraintIdent(source=self._stream.single_span(first), name=name)
704
+
705
+ if isinstance(self._stream.peek(), QuestionToken):
706
+ self._stream.advance()
707
+ # 直接展开: type? → one(type, ?)
708
+ return Constraints(
709
+ source=self._stream.span_from(first),
710
+ constraints=[self._nullable(ident)],
711
+ )
712
+
713
+ # 单约束函数调用可省略尖括号: field: regex("re") = ...
714
+ if isinstance(self._stream.peek(), LparenToken):
715
+ call = self._parse_constraint_call(first)
716
+ # 调用后也可空: regex("re")? → one(regex("re"), ?)
717
+ if isinstance(self._stream.peek(), QuestionToken):
718
+ self._stream.advance()
719
+ call = self._nullable(call)
720
+ return Constraints(
721
+ source=self._stream.span_from(first),
722
+ constraints=[call],
723
+ )
724
+
725
+ return Constraints(
726
+ source=self._stream.span_from(first),
727
+ constraints=[ident],
728
+ )
729
+
730
+ case QuestionToken():
731
+ self._stream.advance()
732
+ return Constraints(
733
+ source=self._stream.span_from(first),
734
+ constraints=[ConstraintIdent(source=self._stream.single_span(first), name='?')],
735
+ )
736
+
737
+ case _:
738
+ # 无效约束起始:报错;不消费容器闭合符(避免吞掉 }/>/)破坏外层解析)
739
+ if not isinstance(self._stream.peek(), (RangleToken, RbraceToken, RparenToken)):
740
+ bad_tok = self._stream.advance()
741
+ self._errors.add(
742
+ diag('parse.unrecognized_constraint', {'name': bad_tok.raw.type.name}, bad_tok.raw.source)
743
+ )
744
+ return Constraints(source=self._stream.span_from(first))
745
+
746
+ @staticmethod
747
+ def _nullable(c: Constraint) -> ConstraintCall:
748
+ """可空包装:constraint? → one(constraint, ?)。"""
749
+ return ConstraintCall(
750
+ source=c.source,
751
+ name='one',
752
+ arguments=[c, ConstraintIdent(source=c.source, name='?')],
753
+ )
754
+
755
+ def _parse_constraint(self) -> Constraint:
756
+ """解析单个约束:标识符、函数调用或字面量(支持可空后缀 ?)。"""
757
+ tok = self._stream.peek()
758
+ base: Constraint
759
+
760
+ match tok:
761
+ case IdentifierToken() as name_tok:
762
+ self._stream.advance()
763
+ if isinstance(self._stream.peek(), LparenToken):
764
+ base = self._parse_constraint_call(name_tok)
765
+ else:
766
+ base = ConstraintIdent(source=self._stream.single_span(name_tok), name=name_tok.name)
767
+
768
+ case QuestionToken():
769
+ self._stream.advance()
770
+ return ConstraintIdent(source=tok.raw.source, name='?')
771
+
772
+ case StringToken() | IntegerToken() | FloatToken() | BoolToken() | NullToken():
773
+ base = self._parse_constraint_literal()
774
+
775
+ case _:
776
+ bad_tok = self._stream.advance()
777
+ return ErrorConstraint(
778
+ source=self._stream.single_span(bad_tok),
779
+ message=f'无法解析的约束: {bad_tok.raw.type.name}',
780
+ )
781
+
782
+ # 可空后缀: constraint? → one(constraint, ?)
783
+ if isinstance(self._stream.peek(), QuestionToken):
784
+ self._stream.advance()
785
+ return self._nullable(base)
786
+ return base
787
+
788
+ def _parse_constraint_call(self, name_tok: IdentifierToken) -> ConstraintCall:
789
+ """解析约束函数调用: name(arg, arg, ...)。"""
790
+ self._stream.expect(LparenToken)
791
+ self._stream.skip_newlines()
792
+ args: list[Constraint] = []
793
+ missing_sep_reported = [False]
794
+ while not self._stream.check(RawTokenType.RPAREN) and not self._stream.check(RawTokenType.EOF):
795
+ args.append(self._parse_constraint())
796
+ had_sep = self._stream.skip_separators()
797
+ self._missing_separator(
798
+ had_sep,
799
+ self._starts_constraint(self._stream.peek()),
800
+ RawTokenType.RPAREN,
801
+ missing_sep_reported,
802
+ )
803
+ self._stream.expect(RparenToken)
804
+ return ConstraintCall(
805
+ source=self._stream.span_from(name_tok),
806
+ name=name_tok.name,
807
+ arguments=args,
808
+ )
809
+
810
+ def _parse_constraint_literal(self) -> ConstraintLiteral:
811
+ """解析约束中的字面量参数,包装为 ConstraintLiteral。"""
812
+ tok = self._stream.peek()
813
+ self._stream.advance()
814
+ lit = self._wrap_literal(tok)
815
+ return ConstraintLiteral(source=lit.source, value=lit)
816
+
817
+ # ═══════════════════════════════════════════════════════
818
+ # 值
819
+ # ═══════════════════════════════════════════════════════
820
+
821
+ def _parse_value(self) -> Value:
822
+ """解析任意值。"""
823
+ match self._stream.peek():
824
+ # ── $ 导入空间引用 ──
825
+ case DollarToken():
826
+ return self._parse_dollar_value()
827
+
828
+ # ── 字面量(FloatToken 覆盖所有浮点值,含 NaN / ±Inf)──
829
+ case StringToken() | IntegerToken() | FloatToken() | BoolToken() | NullToken() | NoexistToken() as tok:
830
+ self._stream.advance()
831
+ return self._wrap_literal(tok)
832
+
833
+ # ── 复合值 ──
834
+ case LbraceToken():
835
+ return self._parse_object()
836
+ case LbracketToken():
837
+ return self._parse_array()
838
+
839
+ # ── 标识符 → 模板调用 ──
840
+ case IdentifierToken():
841
+ ident = self._stream.expect(IdentifierToken)
842
+ # 标识符后接 = / : → 不是模板调用而是新语句的字段定义:
843
+ # 说明外层数组/对象未闭合。报 parse.value_field 并停止,
844
+ # 避免把后续行误解析为模板调用(消除 template.undefined 级联)。
845
+ nxt = self._stream.peek()
846
+ if isinstance(nxt, (EqualsToken, ColonToken)):
847
+ self._errors.add(diag('parse.value_field', {'name': ident.name}, self._stream.single_span(ident)))
848
+ return ErrorValue(
849
+ source=self._stream.single_span(ident), message=f'值位置出现字段定义: {ident.name}'
850
+ )
851
+ return self._parse_template_call(ident)
852
+
853
+ case tok:
854
+ self._stream.advance()
855
+ name = 'EOF' if isinstance(tok, NoNextType) else tok.raw.type.name
856
+ source = SourceRange.empty() if isinstance(tok, NoNextType) else tok.raw.source
857
+ self._errors.add(diag('parse.unrecognized_value', {'name': name}, source))
858
+ return ErrorValue(source=source, message=f'无法解析的值: {name}')
859
+
860
+ def _wrap_literal(self, tok: Token | NoNextType | None) -> LiteralValue:
861
+ """将字面量 Token 包装为 LiteralValue。"""
862
+ assert not isinstance(tok, NoNextType) and tok is not None
863
+ return LiteralValue(source=tok.raw.source, value=tok) # type: ignore[arg-type]
864
+
865
+ def _parse_dollar_value(self) -> DollarValue:
866
+ """$name [as type] 导入空间引用。"""
867
+ dollar_tok = self._stream.expect(DollarToken)
868
+ name_tok = self._stream.expect(IdentifierToken)
869
+
870
+ type_cast = None
871
+ if self._peek_keyword('as'):
872
+ self._stream.advance()
873
+ cast_tok = self._stream.peek()
874
+ if isinstance(cast_tok, IdentifierToken):
875
+ name = cast_tok.name
876
+ if name in ('int', 'float', 'bool', 'str'):
877
+ type_cast = name
878
+ else:
879
+ self._errors.add(diag('parse.invalid_cast', {'type': name}, cast_tok.raw.source))
880
+ self._stream.advance()
881
+
882
+ return DollarValue(
883
+ source=self._stream.span_from(dollar_tok),
884
+ name=name_tok.name,
885
+ type_cast=type_cast,
886
+ )
887
+
888
+ def _parse_object(self) -> DictValue:
889
+ """{ field, ... }
890
+
891
+ dict 结构级约束: ``: <constraint, ...>`` 作用于该字面量 dict 的整体。
892
+ """
893
+ lbrace_tok = self._stream.expect(LbraceToken)
894
+ self._stream.skip_separators()
895
+
896
+ fields: list[Field] = []
897
+ constraints: list[Constraint] = []
898
+ missing_sep_reported = [False]
899
+ while not self._stream.check(RawTokenType.RBRACE) and not self._stream.check(RawTokenType.EOF):
900
+ # 结构级约束: : <...>
901
+ if isinstance(self._stream.peek(), ColonToken):
902
+ self._stream.advance()
903
+ parsed = self._parse_constraints()
904
+ constraints.extend(parsed.constraints)
905
+ else:
906
+ fields.append(self._parse_field())
907
+ had_sep = self._stream.skip_separators()
908
+ self._missing_separator(
909
+ had_sep,
910
+ isinstance(self._stream.peek(), (IdentifierToken, ColonToken)),
911
+ RawTokenType.RBRACE,
912
+ missing_sep_reported,
913
+ )
914
+
915
+ self._stream.expect(RbraceToken)
916
+ return DictValue(source=self._stream.span_from(lbrace_tok), fields=fields, constraints=constraints)
917
+
918
+ def _parse_array(self) -> ArrayValue:
919
+ """[ value, ... ] 逗号或换行分隔,元素间必须显式分隔。"""
920
+ lbracket_tok = self._stream.expect(LbracketToken)
921
+ self._stream.skip_newlines()
922
+
923
+ elements: list[Value] = []
924
+ missing_sep_reported = [False]
925
+ while not self._stream.check(RawTokenType.RBRACKET) and not self._stream.check(RawTokenType.EOF):
926
+ val = self._parse_value()
927
+ elements.append(val)
928
+ had_sep = self._stream.skip_separators()
929
+ self._missing_separator(
930
+ had_sep,
931
+ self._starts_value(self._stream.peek()),
932
+ RawTokenType.RBRACKET,
933
+ missing_sep_reported,
934
+ )
935
+
936
+ self._stream.expect(RbracketToken)
937
+ return ArrayValue(source=self._stream.span_from(lbracket_tok), elements=elements)
938
+
939
+ def _parse_template_call(self, name_tok: IdentifierToken) -> TemplateCallValue:
940
+ """Name(pos_args..., named_arg=value, ...)"""
941
+ self._stream.expect(LparenToken)
942
+ self._stream.skip_newlines()
943
+
944
+ positional: list[Value] = []
945
+ named: dict[str, Value] = {}
946
+ saw_named = False
947
+ missing_sep_reported = [False]
948
+
949
+ while not self._stream.check(RawTokenType.RPAREN) and not self._stream.eof():
950
+ self._stream.skip_newlines()
951
+ if self._stream.check(RawTokenType.RPAREN) or self._stream.eof():
952
+ break
953
+
954
+ tok = self._stream.peek()
955
+
956
+ # 分支 1:标识符 → 可能是命名参数或模板调用(位置参数)
957
+ if isinstance(tok, IdentifierToken):
958
+ ident: IdentifierToken = self._stream.expect(IdentifierToken) # 消费到缓冲区
959
+ nxt = self._stream.peek() # 下一个 token
960
+
961
+ if isinstance(nxt, EqualsToken):
962
+ self._stream.advance() # 消费 =
963
+ named[ident.name] = self._parse_value()
964
+ saw_named = True
965
+ else:
966
+ # 不是 = → 模板调用(位置参数),复用已消费的 ident
967
+ positional.append(self._parse_template_call(ident))
968
+ else:
969
+ # 分支 2:其他 token → 一定是位置参数
970
+ if saw_named:
971
+ self._errors.add(diag('parse.template_arg_order', {}, self._stream.single_span(name_tok)))
972
+ positional.append(self._parse_value())
973
+
974
+ had_sep = self._stream.skip_separators()
975
+ self._missing_separator(
976
+ had_sep,
977
+ self._starts_value(self._stream.peek()),
978
+ RawTokenType.RPAREN,
979
+ missing_sep_reported,
980
+ )
981
+
982
+ self._stream.expect(RparenToken)
983
+ return TemplateCallValue(
984
+ source=self._stream.span_from(name_tok),
985
+ template_name=name_tok.name,
986
+ positional_args=positional,
987
+ named_args=named,
988
+ )