millforge 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (116) hide show
  1. millforge/__init__.py +1174 -0
  2. millforge/_forge/LICENSE +21 -0
  3. millforge/_forge/PROVENANCE.json +295 -0
  4. millforge/_forge/UPDATE_POLICY.md +24 -0
  5. millforge/_forge/__init__.py +14 -0
  6. millforge/_forge/adapter.py +2232 -0
  7. millforge/_forge/base_runner.py +121 -0
  8. millforge/_forge/clients/__init__.py +10 -0
  9. millforge/_forge/clients/base.py +200 -0
  10. millforge/_forge/context/__init__.py +23 -0
  11. millforge/_forge/context/manager.py +178 -0
  12. millforge/_forge/context/strategies.py +335 -0
  13. millforge/_forge/core/__init__.py +16 -0
  14. millforge/_forge/core/inference.py +433 -0
  15. millforge/_forge/core/messages.py +119 -0
  16. millforge/_forge/core/runner.py +479 -0
  17. millforge/_forge/core/steps.py +108 -0
  18. millforge/_forge/core/workflow.py +400 -0
  19. millforge/_forge/errors.py +222 -0
  20. millforge/_forge/guardrails/__init__.py +21 -0
  21. millforge/_forge/guardrails/error_tracker.py +71 -0
  22. millforge/_forge/guardrails/guardrails.py +194 -0
  23. millforge/_forge/guardrails/nudge.py +47 -0
  24. millforge/_forge/guardrails/response_validator.py +119 -0
  25. millforge/_forge/guardrails/step_enforcer.py +183 -0
  26. millforge/_forge/prompts/__init__.py +16 -0
  27. millforge/_forge/prompts/nudges.py +95 -0
  28. millforge/_forge/prompts/templates.py +285 -0
  29. millforge/_version.py +3 -0
  30. millforge/artifacts.py +570 -0
  31. millforge/base/__init__.py +97 -0
  32. millforge/base/composition.py +402 -0
  33. millforge/base/context.py +285 -0
  34. millforge/base/harness.py +138 -0
  35. millforge/base/identity.py +465 -0
  36. millforge/base/options.py +34 -0
  37. millforge/base/platform.py +17 -0
  38. millforge/base/prompt.py +317 -0
  39. millforge/base/runner.py +546 -0
  40. millforge/compiled_plan.py +970 -0
  41. millforge/compiler/__init__.py +231 -0
  42. millforge/compiler/artifact_validation.py +257 -0
  43. millforge/compiler/canonicalization.py +169 -0
  44. millforge/compiler/capabilities.py +66 -0
  45. millforge/compiler/catalogs.py +500 -0
  46. millforge/compiler/diagnostics.py +491 -0
  47. millforge/compiler/graph.py +678 -0
  48. millforge/compiler/lowering.py +198 -0
  49. millforge/compiler/output.py +692 -0
  50. millforge/compiler/parsing.py +1424 -0
  51. millforge/compiler/requests.py +1180 -0
  52. millforge/compiler/schema_validation.py +272 -0
  53. millforge/compiler/semantic.py +490 -0
  54. millforge/compiler/service.py +448 -0
  55. millforge/compiler/source.py +375 -0
  56. millforge/compiler/validators.py +184 -0
  57. millforge/connectors/__init__.py +95 -0
  58. millforge/connectors/admission.py +801 -0
  59. millforge/connectors/broker.py +202 -0
  60. millforge/connectors/contracts.py +1159 -0
  61. millforge/connectors/diagnostics.py +189 -0
  62. millforge/connectors/fake.py +66 -0
  63. millforge/connectors/runtime.py +236 -0
  64. millforge/contracts.py +2860 -0
  65. millforge/custom_tools/__init__.py +67 -0
  66. millforge/custom_tools/compiler.py +724 -0
  67. millforge/custom_tools/contracts.py +1093 -0
  68. millforge/custom_tools/diagnostics.py +205 -0
  69. millforge/eval_artifacts.py +952 -0
  70. millforge/eval_boundary.py +2435 -0
  71. millforge/eval_fixtures/__init__.py +1 -0
  72. millforge/eval_fixtures/default_pack/__init__.py +1 -0
  73. millforge/eval_fixtures/default_pack/fixtures/fixture.08a.bug_diagnosis.traceback.v1.json +52 -0
  74. millforge/eval_fixtures/default_pack/fixtures/fixture.08a.direct_edit.import_sort.v1.json +52 -0
  75. millforge/eval_fixtures/default_pack/fixtures/fixture.08a.evidence_discipline.no_source_change.v1.json +51 -0
  76. millforge/eval_fixtures/default_pack/fixtures/fixture.08a.false_closure.visible_green.v1.json +52 -0
  77. millforge/eval_fixtures/default_pack/fixtures/fixture.08a.multi_file.api_contract.v1.json +54 -0
  78. millforge/eval_fixtures/default_pack/fixtures/fixture.08a.recovery.malformed_artifact.v1.json +54 -0
  79. millforge/eval_fixtures/default_pack/manifest.json +12 -0
  80. millforge/eval_modes.py +1282 -0
  81. millforge/eval_presets.py +1398 -0
  82. millforge/eval_reports.py +2517 -0
  83. millforge/eval_suite.py +2429 -0
  84. millforge/eval_trials.py +2632 -0
  85. millforge/eval_workflow.py +794 -0
  86. millforge/exceptions.py +122 -0
  87. millforge/model_backend.py +2098 -0
  88. millforge/protocols.py +340 -0
  89. millforge/py.typed +0 -0
  90. millforge/runtime.py +1791 -0
  91. millforge/testing/__init__.py +1089 -0
  92. millforge/tools/__init__.py +83 -0
  93. millforge/tools/builtin_runtime.py +1339 -0
  94. millforge/tools/builtins.py +773 -0
  95. millforge/tools/execution.py +1545 -0
  96. millforge/tools/path_policy.py +155 -0
  97. millforge/tools/pi_compat/PI_LICENSE +21 -0
  98. millforge/tools/pi_compat/PROVENANCE.json +55 -0
  99. millforge/tools/pi_compat/UPDATE_POLICY.md +36 -0
  100. millforge/tools/pi_compat/__init__.py +34 -0
  101. millforge/tools/pi_compat/contracts.py +49 -0
  102. millforge/tools/pi_compat/editing.py +390 -0
  103. millforge/tools/pi_compat/mutations.py +57 -0
  104. millforge/tools/pi_compat/operations.py +401 -0
  105. millforge/tools/pi_compat/paths.py +155 -0
  106. millforge/tools/pi_compat/process.py +1375 -0
  107. millforge/tools/pi_compat/search.py +738 -0
  108. millforge/tools/pi_compat/truncation.py +267 -0
  109. millforge/tools/pi_compat_catalog.py +396 -0
  110. millforge/tools/pi_compat_runtime.py +460 -0
  111. millforge/tools/registry.py +553 -0
  112. millforge/tools/results.py +533 -0
  113. millforge-0.1.0.dist-info/METADATA +844 -0
  114. millforge-0.1.0.dist-info/RECORD +116 -0
  115. millforge-0.1.0.dist-info/WHEEL +4 -0
  116. millforge-0.1.0.dist-info/licenses/LICENSE +201 -0
@@ -0,0 +1,1424 @@
1
+ """Parser boundary contracts and deterministic front ends for harness sources."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import hashlib
6
+ import json
7
+ import re
8
+ from collections.abc import Mapping, Sequence
9
+ from dataclasses import dataclass
10
+ from typing import Any, Protocol, TypeAlias, cast
11
+
12
+ from pydantic import (
13
+ BaseModel,
14
+ ConfigDict,
15
+ Field,
16
+ StrictBytes,
17
+ StrictStr,
18
+ ValidationError,
19
+ field_validator,
20
+ )
21
+
22
+ from millforge.compiler.diagnostics import (
23
+ bound_diagnostics,
24
+ CompilerDiagnostic,
25
+ CompilerPhase,
26
+ DiagnosticField,
27
+ DiagnosticSeverity,
28
+ SourceLocation,
29
+ SourceReference,
30
+ )
31
+ from millforge.compiler.source import HarnessSource
32
+ from millforge.compiler.validators import validate_sha256, validate_utf8_size
33
+
34
+ MAX_SOURCE_SIZE_BYTES = 1_048_576
35
+ MAX_NESTING_DEPTH = 32
36
+ MAX_TOTAL_ENTRIES = 10_000
37
+ MAX_SCALAR_UTF8_SIZE = 65_536
38
+ MAX_INTEGER_LEXEME_ASCII = 128
39
+ MAX_FLOAT_LEXEME_ASCII = 128
40
+ _ZERO_SHA256 = "0" * 64
41
+ _FORBIDDEN_CONTROL_RE = re.compile(r"[\x00-\x08\x0b\x0c\x0e-\x1f\x7f-\x9f]")
42
+ _JSON_INTEGER_RE = re.compile(r"-?(?:0|[1-9][0-9]*)\Z")
43
+ _JSON_FLOAT_RE = re.compile(
44
+ r"-?(?:0|[1-9][0-9]*)(?:\.[0-9]+)?(?:[eE][+-]?[0-9]+)\Z|"
45
+ r"-?(?:0|[1-9][0-9]*)\.[0-9]+(?:[eE][+-]?[0-9]+)?\Z"
46
+ )
47
+ _YAML_TAG_ANCHOR_ALIAS_RE = re.compile(r"(^|[\s\[{,])(?:[!&*][A-Za-z0-9_.:/-]*|!)")
48
+
49
+ JsonScalar: TypeAlias = str | int | float | bool | None
50
+ JsonValue: TypeAlias = JsonScalar | list["JsonValue"] | dict[str, "JsonValue"]
51
+
52
+
53
+ class ParserError(ValueError):
54
+ """Bounded parser failure that maps to a compiler diagnostic."""
55
+
56
+ def __init__(
57
+ self,
58
+ code: str,
59
+ message: str,
60
+ *,
61
+ line: int | None = None,
62
+ column: int | None = None,
63
+ field_path: str = "/",
64
+ fields: tuple[DiagnosticField, ...] = (),
65
+ ) -> None:
66
+ super().__init__(message)
67
+ self.code = code
68
+ self.message = message
69
+ self.line = line
70
+ self.column = column
71
+ self.field_path = field_path
72
+ self.fields = fields
73
+
74
+
75
+ @dataclass(frozen=True)
76
+ class _YamlLine:
77
+ number: int
78
+ indent: int
79
+ text: str
80
+
81
+
82
+ @dataclass
83
+ class _ParseState:
84
+ entries: int = 0
85
+
86
+ def note_entry(self) -> None:
87
+ self.entries += 1
88
+ if self.entries > MAX_TOTAL_ENTRIES:
89
+ raise ParserError(
90
+ "MF-S010", "Total mapping/list entries exceed parser limit."
91
+ )
92
+
93
+
94
+ class SourceDocument(BaseModel):
95
+ """Immutable parser input document; content is excluded from repr."""
96
+
97
+ model_config = ConfigDict(extra="forbid", frozen=True)
98
+
99
+ logical_path: StrictStr
100
+ format: StrictStr
101
+ content: StrictBytes = Field(repr=False)
102
+
103
+ @field_validator("logical_path")
104
+ @classmethod
105
+ def _logical_path_valid(cls, value: str) -> str:
106
+ if not value.strip():
107
+ raise ValueError("logical_path must be nonblank")
108
+ return value
109
+
110
+ @field_validator("format")
111
+ @classmethod
112
+ def _format_valid(cls, value: str) -> str:
113
+ if not value.strip():
114
+ raise ValueError("format must be nonblank")
115
+ return validate_utf8_size(value, "format", 64)
116
+
117
+
118
+ class ParsedHarnessSource(BaseModel):
119
+ """Immutable parser output boundary for a harness source document."""
120
+
121
+ model_config = ConfigDict(extra="forbid", frozen=True)
122
+
123
+ source: HarnessSource | None
124
+ source_document_sha256: StrictStr
125
+ diagnostics: tuple[CompilerDiagnostic, ...] = Field(default_factory=tuple)
126
+ location_index: tuple[SourceReference, ...] = Field(default_factory=tuple)
127
+
128
+ @field_validator("source_document_sha256")
129
+ @classmethod
130
+ def _source_document_sha256_valid(cls, value: str) -> str:
131
+ return validate_sha256(value, "source_document_sha256")
132
+
133
+
134
+ class HarnessSourceParserProtocol(Protocol):
135
+ """Protocol for deterministic harness source parsers."""
136
+
137
+ def parse(self, document: SourceDocument) -> ParsedHarnessSource:
138
+ """Parse a source document into the shared immutable source contract."""
139
+ ...
140
+
141
+
142
+ class HarnessSourceParser:
143
+ """Deterministic YAML/JSON parser front end for harness source documents."""
144
+
145
+ def parse(self, document: SourceDocument) -> ParsedHarnessSource:
146
+ try:
147
+ normalized_bytes, digest = _normalize_and_hash(document.content)
148
+ if document.format == "json":
149
+ payload = _parse_json(normalized_bytes)
150
+ elif document.format == "yaml":
151
+ payload = _parse_yaml(normalized_bytes)
152
+ else:
153
+ raise ParserError("MF-S005", "Unsupported source format.")
154
+ _validate_json_compatible(payload)
155
+ if not isinstance(payload, dict):
156
+ raise ParserError(
157
+ "MF-S012", "Top-level source document must be an object."
158
+ )
159
+ location_index = _build_location_index(document, normalized_bytes, payload)
160
+ source = HarnessSource.model_validate(payload)
161
+ return ParsedHarnessSource(
162
+ source=source,
163
+ source_document_sha256=digest,
164
+ diagnostics=(),
165
+ location_index=location_index,
166
+ )
167
+ except ParserError as exc:
168
+ return ParsedHarnessSource(
169
+ source=None,
170
+ source_document_sha256=(
171
+ digest if "digest" in locals() else _ZERO_SHA256
172
+ ),
173
+ diagnostics=(_diagnostic(document, exc),),
174
+ location_index=(),
175
+ )
176
+ except ValidationError as exc:
177
+ return ParsedHarnessSource(
178
+ source=None,
179
+ source_document_sha256=digest if "digest" in locals() else _ZERO_SHA256,
180
+ diagnostics=_schema_validation_diagnostics(
181
+ document,
182
+ exc,
183
+ location_index if "location_index" in locals() else (),
184
+ ),
185
+ location_index=location_index if "location_index" in locals() else (),
186
+ )
187
+
188
+
189
+ def _normalize_and_hash(content: bytes) -> tuple[bytes, str]:
190
+ if len(content) > MAX_SOURCE_SIZE_BYTES:
191
+ raise ParserError(
192
+ "MF-S003",
193
+ "Source document exceeds the maximum size.",
194
+ fields=(DiagnosticField(key="source_size", value=len(content)),),
195
+ )
196
+ if content.startswith(b"\xef\xbb\xbf"):
197
+ content = content[3:]
198
+ try:
199
+ text = content.decode("utf-8", "strict")
200
+ except UnicodeDecodeError as exc:
201
+ raise ParserError(
202
+ "MF-S004",
203
+ "Source document must be strict UTF-8.",
204
+ line=1,
205
+ column=exc.start + 1,
206
+ ) from exc
207
+ _reject_forbidden_text(text)
208
+ normalized = text.replace("\r\n", "\n").replace("\r", "\n").encode("utf-8")
209
+ return normalized, hashlib.sha256(normalized).hexdigest()
210
+
211
+
212
+ def _reject_forbidden_text(text: str) -> None:
213
+ for index, char in enumerate(text):
214
+ codepoint = ord(char)
215
+ if 0xD800 <= codepoint <= 0xDFFF:
216
+ line, column = _line_column(text, index)
217
+ raise ParserError(
218
+ "MF-S011",
219
+ "Source document contains an unpaired Unicode surrogate.",
220
+ line=line,
221
+ column=column,
222
+ )
223
+ match = _FORBIDDEN_CONTROL_RE.search(text)
224
+ if match is not None:
225
+ line, column = _line_column(text, match.start())
226
+ raise ParserError(
227
+ "MF-S011",
228
+ "Source document contains a forbidden control character.",
229
+ line=line,
230
+ column=column,
231
+ )
232
+
233
+
234
+ def _parse_json(content: bytes) -> JsonValue:
235
+ return _parse_json_text(content.decode("utf-8"))
236
+
237
+
238
+ def _parse_json_text(
239
+ text: str, *, state: _ParseState | None = None, depth: int = 1
240
+ ) -> JsonValue:
241
+ state = state if state is not None else _ParseState()
242
+
243
+ def skip_ws(index: int) -> int:
244
+ while index < len(text) and text[index] in " \t\r\n":
245
+ index += 1
246
+ return index
247
+
248
+ def parse_value(index: int, depth: int) -> tuple[JsonValue, int]:
249
+ index = skip_ws(index)
250
+ if index >= len(text):
251
+ raise ParserError("MF-S011", "Malformed JSON source document.")
252
+ if text.startswith(("NaN", "Infinity", "+Infinity", "-Infinity"), index):
253
+ line, column = _line_column(text, index)
254
+ raise ParserError(
255
+ "MF-S011",
256
+ "Non-finite number is not allowed.",
257
+ line=line,
258
+ column=column,
259
+ )
260
+ char = text[index]
261
+ if char == "{":
262
+ return parse_object(index, depth)
263
+ if char == "[":
264
+ return parse_array(index, depth)
265
+ if char == '"':
266
+ value, end = _decode_json_value(text, index)
267
+ if not isinstance(value, str):
268
+ raise ParserError("MF-S011", "Malformed JSON source document.")
269
+ _validate_scalar_text(value)
270
+ return value, end
271
+ if char in "-0123456789":
272
+ return parse_number(index)
273
+ if text.startswith("true", index):
274
+ return True, index + 4
275
+ if text.startswith("false", index):
276
+ return False, index + 5
277
+ if text.startswith("null", index):
278
+ return None, index + 4
279
+ line, column = _line_column(text, index)
280
+ raise ParserError(
281
+ "MF-S011",
282
+ "Malformed JSON source document.",
283
+ line=line,
284
+ column=column,
285
+ )
286
+
287
+ def parse_object(index: int, depth: int) -> tuple[dict[str, JsonValue], int]:
288
+ _check_nesting_depth(depth)
289
+ index += 1
290
+ index = skip_ws(index)
291
+ result: dict[str, JsonValue] = {}
292
+ if index < len(text) and text[index] == "}":
293
+ return result, index + 1
294
+ while True:
295
+ key_start = skip_ws(index)
296
+ if key_start >= len(text) or text[key_start] != '"':
297
+ line, column = _line_column(
298
+ text, key_start if key_start < len(text) else index
299
+ )
300
+ raise ParserError(
301
+ "MF-S011",
302
+ "Malformed JSON source document.",
303
+ line=line,
304
+ column=column,
305
+ )
306
+ key, key_end = _decode_json_value(text, key_start)
307
+ if not isinstance(key, str):
308
+ line, column = _line_column(text, key_start)
309
+ raise ParserError(
310
+ "MF-S011",
311
+ "Malformed JSON source document.",
312
+ line=line,
313
+ column=column,
314
+ )
315
+ _validate_scalar_text(key)
316
+ if key in result:
317
+ raise ParserError("MF-S006", "Duplicate object key.")
318
+ index = skip_ws(key_end)
319
+ if index >= len(text) or text[index] != ":":
320
+ line, column = _line_column(
321
+ text, index if index < len(text) else key_end
322
+ )
323
+ raise ParserError(
324
+ "MF-S011",
325
+ "Malformed JSON source document.",
326
+ line=line,
327
+ column=column,
328
+ )
329
+ state.note_entry()
330
+ value, index = parse_value(index + 1, depth + 1)
331
+ result[key] = value
332
+ index = skip_ws(index)
333
+ if index < len(text) and text[index] == ",":
334
+ index += 1
335
+ continue
336
+ if index < len(text) and text[index] == "}":
337
+ return result, index + 1
338
+ line, column = _line_column(text, index if index < len(text) else len(text))
339
+ raise ParserError(
340
+ "MF-S011",
341
+ "Malformed JSON source document.",
342
+ line=line,
343
+ column=column,
344
+ )
345
+
346
+ def parse_array(index: int, depth: int) -> tuple[list[JsonValue], int]:
347
+ _check_nesting_depth(depth)
348
+ index += 1
349
+ index = skip_ws(index)
350
+ result: list[JsonValue] = []
351
+ if index < len(text) and text[index] == "]":
352
+ return result, index + 1
353
+ while True:
354
+ item_start = skip_ws(index)
355
+ state.note_entry()
356
+ value, index = parse_value(item_start, depth + 1)
357
+ result.append(value)
358
+ index = skip_ws(index)
359
+ if index < len(text) and text[index] == ",":
360
+ index += 1
361
+ continue
362
+ if index < len(text) and text[index] == "]":
363
+ return result, index + 1
364
+ line, column = _line_column(text, index if index < len(text) else len(text))
365
+ raise ParserError(
366
+ "MF-S011",
367
+ "Malformed JSON source document.",
368
+ line=line,
369
+ column=column,
370
+ )
371
+
372
+ def parse_number(index: int) -> tuple[int | float, int]:
373
+ start = index
374
+ if text[index] == "-":
375
+ index += 1
376
+ if index >= len(text):
377
+ line, column = _line_column(text, start)
378
+ raise ParserError(
379
+ "MF-S011",
380
+ "Malformed JSON source document.",
381
+ line=line,
382
+ column=column,
383
+ )
384
+ if index >= len(text) or not text[index].isdigit():
385
+ line, column = _line_column(text, start)
386
+ raise ParserError(
387
+ "MF-S011",
388
+ "Malformed JSON source document.",
389
+ line=line,
390
+ column=column,
391
+ )
392
+ if text[index] == "0":
393
+ index += 1
394
+ if index < len(text) and text[index].isdigit():
395
+ line, column = _line_column(text, start)
396
+ raise ParserError(
397
+ "MF-S011",
398
+ "Malformed JSON source document.",
399
+ line=line,
400
+ column=column,
401
+ )
402
+ else:
403
+ while index < len(text) and text[index].isdigit():
404
+ index += 1
405
+ is_float = False
406
+ if index < len(text) and text[index] == ".":
407
+ is_float = True
408
+ index += 1
409
+ if index >= len(text) or not text[index].isdigit():
410
+ line, column = _line_column(text, start)
411
+ raise ParserError(
412
+ "MF-S011",
413
+ "Malformed JSON source document.",
414
+ line=line,
415
+ column=column,
416
+ )
417
+ while index < len(text) and text[index].isdigit():
418
+ index += 1
419
+ if index < len(text) and text[index] in "eE":
420
+ is_float = True
421
+ index += 1
422
+ if index < len(text) and text[index] in "+-":
423
+ index += 1
424
+ if index >= len(text) or not text[index].isdigit():
425
+ line, column = _line_column(text, start)
426
+ raise ParserError(
427
+ "MF-S011",
428
+ "Malformed JSON source document.",
429
+ line=line,
430
+ column=column,
431
+ )
432
+ while index < len(text) and text[index].isdigit():
433
+ index += 1
434
+ lexeme = text[start:index]
435
+ maximum = MAX_FLOAT_LEXEME_ASCII if is_float else MAX_INTEGER_LEXEME_ASCII
436
+ if len(lexeme.encode("ascii")) > maximum:
437
+ raise ParserError(
438
+ "MF-S010",
439
+ "Floating-point lexeme exceeds parser limit."
440
+ if is_float
441
+ else "Integer lexeme exceeds parser limit.",
442
+ )
443
+ return (float(lexeme) if is_float else int(lexeme)), index
444
+
445
+ value, end = parse_value(0, depth)
446
+ end = skip_ws(end)
447
+ if end != len(text):
448
+ line, column = _line_column(text, end)
449
+ raise ParserError(
450
+ "MF-S011",
451
+ "JSON source document has trailing non-whitespace content.",
452
+ line=line,
453
+ column=column,
454
+ )
455
+ return value
456
+
457
+
458
+ def _decode_json_value(text: str, index: int = 0) -> tuple[JsonValue, int]:
459
+ decoder = json.JSONDecoder(
460
+ object_pairs_hook=_json_object_pairs,
461
+ parse_int=_parse_json_int,
462
+ parse_float=_parse_json_float,
463
+ parse_constant=_reject_json_constant,
464
+ )
465
+ try:
466
+ value, end = decoder.raw_decode(text, index)
467
+ except json.JSONDecodeError as exc:
468
+ raise ParserError(
469
+ "MF-S011",
470
+ "Malformed JSON source document.",
471
+ line=exc.lineno,
472
+ column=exc.colno,
473
+ ) from exc
474
+ return cast(JsonValue, value), end
475
+
476
+
477
+ def _json_object_pairs(pairs: Sequence[tuple[str, JsonValue]]) -> dict[str, JsonValue]:
478
+ result: dict[str, JsonValue] = {}
479
+ for key, value in pairs:
480
+ _validate_scalar_text(key)
481
+ if key in result:
482
+ raise ParserError("MF-S006", "Duplicate object key.")
483
+ result[key] = value
484
+ return result
485
+
486
+
487
+ def _parse_json_int(value: str) -> int:
488
+ if len(value.encode("ascii")) > MAX_INTEGER_LEXEME_ASCII:
489
+ raise ParserError("MF-S010", "Integer lexeme exceeds parser limit.")
490
+ return int(value)
491
+
492
+
493
+ def _parse_json_float(value: str) -> float:
494
+ if len(value.encode("ascii")) > MAX_FLOAT_LEXEME_ASCII:
495
+ raise ParserError("MF-S010", "Floating-point lexeme exceeds parser limit.")
496
+ return float(value)
497
+
498
+
499
+ def _reject_json_constant(value: str) -> None:
500
+ raise ParserError("MF-S011", "Non-finite number is not allowed.")
501
+
502
+
503
+ def _check_nesting_depth(depth: int) -> None:
504
+ if depth > MAX_NESTING_DEPTH:
505
+ raise ParserError("MF-S010", "Nesting depth exceeds parser limit.")
506
+
507
+
508
+ def _parse_yaml(content: bytes) -> JsonValue:
509
+ text = content.decode("utf-8")
510
+ lines = _yaml_lines(text)
511
+ if not lines:
512
+ raise ParserError(
513
+ "MF-S012", "Top-level YAML source document must be an object."
514
+ )
515
+ state = _ParseState()
516
+ value, index = _parse_yaml_block(lines, 0, lines[0].indent, state, 1)
517
+ if index != len(lines):
518
+ line = lines[index]
519
+ raise ParserError(
520
+ "MF-S011",
521
+ "Malformed YAML indentation.",
522
+ line=line.number,
523
+ column=line.indent + 1,
524
+ )
525
+ if not isinstance(value, dict):
526
+ raise ParserError(
527
+ "MF-S012", "Top-level YAML source document must be an object."
528
+ )
529
+ return value
530
+
531
+
532
+ def _yaml_lines(text: str) -> list[_YamlLine]:
533
+ result: list[_YamlLine] = []
534
+ document_markers = 0
535
+ for line_number, raw_line in enumerate(text.split("\n"), start=1):
536
+ if "\t" in raw_line:
537
+ raise ParserError(
538
+ "MF-S011",
539
+ "YAML tabs are not accepted for indentation.",
540
+ line=line_number,
541
+ column=raw_line.index("\t") + 1,
542
+ )
543
+ stripped = _strip_yaml_comment(raw_line).rstrip()
544
+ if not stripped.strip():
545
+ continue
546
+ indent = len(stripped) - len(stripped.lstrip(" "))
547
+ if indent == 0 and stripped.strip() in {"---", "..."}:
548
+ document_markers += 1
549
+ if document_markers > 1 or result:
550
+ raise ParserError(
551
+ "MF-S009",
552
+ "YAML source must contain exactly one document.",
553
+ line=line_number,
554
+ column=raw_line.index(stripped.strip()) + 1,
555
+ )
556
+ continue
557
+ if _YAML_TAG_ANCHOR_ALIAS_RE.search(stripped) is not None:
558
+ code = (
559
+ "MF-S007" if any(mark in stripped for mark in ("&", "*")) else "MF-S008"
560
+ )
561
+ raise ParserError(
562
+ code,
563
+ "YAML anchors, aliases, and tags are not supported.",
564
+ line=line_number,
565
+ column=1,
566
+ )
567
+ if indent % 2:
568
+ raise ParserError(
569
+ "MF-S011",
570
+ "YAML indentation must use two-space levels.",
571
+ line=line_number,
572
+ column=indent + 1,
573
+ )
574
+ result.append(
575
+ _YamlLine(number=line_number, indent=indent, text=stripped[indent:])
576
+ )
577
+ return result
578
+
579
+
580
+ def _parse_yaml_block(
581
+ lines: Sequence[_YamlLine],
582
+ index: int,
583
+ indent: int,
584
+ state: _ParseState,
585
+ depth: int,
586
+ ) -> tuple[JsonValue, int]:
587
+ _check_nesting_depth(depth)
588
+ if index >= len(lines):
589
+ raise ParserError("MF-S011", "Expected YAML block.")
590
+ line = lines[index]
591
+ if line.indent != indent:
592
+ raise ParserError(
593
+ "MF-S011",
594
+ "Malformed YAML indentation.",
595
+ line=line.number,
596
+ column=line.indent + 1,
597
+ )
598
+ if line.text.startswith("- "):
599
+ return _parse_yaml_sequence(lines, index, indent, state, depth)
600
+ return _parse_yaml_mapping(lines, index, indent, state, depth)
601
+
602
+
603
+ def _parse_yaml_mapping(
604
+ lines: Sequence[_YamlLine],
605
+ index: int,
606
+ indent: int,
607
+ state: _ParseState,
608
+ depth: int,
609
+ ) -> tuple[dict[str, JsonValue], int]:
610
+ result: dict[str, JsonValue] = {}
611
+ while index < len(lines):
612
+ line = lines[index]
613
+ if line.indent < indent:
614
+ break
615
+ if line.indent != indent or line.text.startswith("- "):
616
+ break
617
+ key_text, value_text = _split_yaml_key_value(line)
618
+ key = _parse_yaml_key(key_text, line)
619
+ if key == "<<":
620
+ raise ParserError(
621
+ "MF-S008",
622
+ "YAML merge keys are not supported.",
623
+ line=line.number,
624
+ column=indent + 1,
625
+ )
626
+ if key in result:
627
+ raise ParserError(
628
+ "MF-S006",
629
+ "Duplicate mapping key.",
630
+ line=line.number,
631
+ column=indent + 1,
632
+ )
633
+ state.note_entry()
634
+ if value_text == "":
635
+ if index + 1 >= len(lines) or lines[index + 1].indent <= indent:
636
+ result[key] = {}
637
+ index += 1
638
+ else:
639
+ result[key], index = _parse_yaml_block(
640
+ lines, index + 1, lines[index + 1].indent, state, depth + 1
641
+ )
642
+ elif value_text == "|":
643
+ result[key], index = _parse_yaml_block_scalar(lines, index, indent, line)
644
+ elif value_text.startswith("|"):
645
+ raise ParserError(
646
+ "MF-S011",
647
+ "Unsupported YAML block scalar indicator.",
648
+ line=line.number,
649
+ column=line.indent + 1,
650
+ )
651
+ else:
652
+ result[key] = _parse_yaml_scalar(value_text, line, state, depth)
653
+ index += 1
654
+ return result, index
655
+
656
+
657
+ def _parse_yaml_sequence(
658
+ lines: Sequence[_YamlLine],
659
+ index: int,
660
+ indent: int,
661
+ state: _ParseState,
662
+ depth: int,
663
+ ) -> tuple[list[JsonValue], int]:
664
+ result: list[JsonValue] = []
665
+ while index < len(lines):
666
+ line = lines[index]
667
+ if line.indent < indent:
668
+ break
669
+ if line.indent != indent or not line.text.startswith("- "):
670
+ break
671
+ item_text = line.text[2:].strip()
672
+ state.note_entry()
673
+ if item_text == "":
674
+ if index + 1 >= len(lines) or lines[index + 1].indent <= indent:
675
+ result.append({})
676
+ index += 1
677
+ else:
678
+ block_item, index = _parse_yaml_block(
679
+ lines, index + 1, lines[index + 1].indent, state, depth + 1
680
+ )
681
+ result.append(block_item)
682
+ continue
683
+ if _looks_like_yaml_key_value(item_text):
684
+ key_text, value_text = _split_yaml_key_value(
685
+ _YamlLine(line.number, line.indent + 2, item_text)
686
+ )
687
+ key = _parse_yaml_key(key_text, line)
688
+ state.note_entry()
689
+ mapping_item: dict[str, JsonValue] = {
690
+ key: (
691
+ _parse_yaml_scalar(value_text, line, state, depth)
692
+ if value_text
693
+ else {}
694
+ )
695
+ }
696
+ index += 1
697
+ if index < len(lines) and lines[index].indent > indent:
698
+ extra, index = _parse_yaml_mapping(
699
+ lines, index, lines[index].indent, state, depth + 1
700
+ )
701
+ for extra_key, extra_value in extra.items():
702
+ if extra_key in mapping_item:
703
+ raise ParserError(
704
+ "MF-S006",
705
+ "Duplicate mapping key.",
706
+ line=line.number,
707
+ column=line.indent + 1,
708
+ )
709
+ mapping_item[extra_key] = extra_value
710
+ result.append(mapping_item)
711
+ continue
712
+ if item_text == "|":
713
+ block_value, index = _parse_yaml_block_scalar(lines, index, indent, line)
714
+ result.append(block_value)
715
+ continue
716
+ if item_text.startswith("|"):
717
+ raise ParserError(
718
+ "MF-S011",
719
+ "Unsupported YAML block scalar indicator.",
720
+ line=line.number,
721
+ column=line.indent + 1,
722
+ )
723
+ result.append(_parse_yaml_scalar(item_text, line, state, depth))
724
+ index += 1
725
+ return result, index
726
+
727
+
728
+ def _split_yaml_key_value(line: _YamlLine) -> tuple[str, str]:
729
+ separator = _find_yaml_separator(line.text)
730
+ if separator < 0:
731
+ raise ParserError(
732
+ "MF-S011",
733
+ "YAML mapping entries must contain a colon.",
734
+ line=line.number,
735
+ column=line.indent + 1,
736
+ )
737
+ key = line.text[:separator].strip()
738
+ value = line.text[separator + 1 :].strip()
739
+ if not key:
740
+ raise ParserError(
741
+ "MF-S011",
742
+ "YAML mapping key must be non-empty.",
743
+ line=line.number,
744
+ column=line.indent + 1,
745
+ )
746
+ return key, value
747
+
748
+
749
+ def _find_yaml_separator(text: str) -> int:
750
+ quote: str | None = None
751
+ for index, char in enumerate(text):
752
+ if quote is not None:
753
+ if char == quote:
754
+ quote = None
755
+ continue
756
+ if char in {"'", '"'}:
757
+ quote = char
758
+ continue
759
+ if char == ":" and (index + 1 == len(text) or text[index + 1].isspace()):
760
+ return index
761
+ return -1
762
+
763
+
764
+ def _looks_like_yaml_key_value(text: str) -> bool:
765
+ return _find_yaml_separator(text) >= 0
766
+
767
+
768
+ def _parse_yaml_key(key_text: str, line: _YamlLine) -> str:
769
+ if key_text.startswith(("'", '"')):
770
+ value = _parse_yaml_scalar(key_text, line)
771
+ if not isinstance(value, str):
772
+ raise ParserError("MF-S008", "YAML mapping keys must be strings.")
773
+ return value
774
+ if key_text.startswith(("[", "{", "?")) or key_text in {"true", "false", "null"}:
775
+ raise ParserError(
776
+ "MF-S008",
777
+ "YAML mapping keys must be strings.",
778
+ line=line.number,
779
+ column=line.indent + 1,
780
+ )
781
+ if _JSON_INTEGER_RE.fullmatch(key_text) or _JSON_FLOAT_RE.fullmatch(key_text):
782
+ raise ParserError(
783
+ "MF-S008",
784
+ "YAML mapping keys must be strings.",
785
+ line=line.number,
786
+ column=line.indent + 1,
787
+ )
788
+ _validate_scalar_text(key_text)
789
+ return key_text
790
+
791
+
792
+ def _parse_yaml_block_scalar(
793
+ lines: Sequence[_YamlLine], index: int, indent: int, line: _YamlLine
794
+ ) -> tuple[str, int]:
795
+ block_lines: list[_YamlLine] = []
796
+ end = index + 1
797
+ while end < len(lines):
798
+ next_line = lines[end]
799
+ if next_line.indent <= indent:
800
+ break
801
+ block_lines.append(next_line)
802
+ end += 1
803
+ if not block_lines:
804
+ raise ParserError(
805
+ "MF-S011",
806
+ "YAML block scalar must contain indented content.",
807
+ line=line.number,
808
+ column=line.indent + 1,
809
+ )
810
+ content_indent = min(item.indent for item in block_lines)
811
+ pieces: list[str] = []
812
+ for block_line in block_lines:
813
+ if block_line.indent < content_indent:
814
+ raise ParserError(
815
+ "MF-S011",
816
+ "Malformed YAML block scalar indentation.",
817
+ line=block_line.number,
818
+ column=block_line.indent + 1,
819
+ )
820
+ pieces.append(" " * (block_line.indent - content_indent) + block_line.text)
821
+ return "\n".join(pieces) + "\n", end
822
+
823
+
824
+ def _parse_yaml_scalar(
825
+ text: str,
826
+ line: _YamlLine,
827
+ state: _ParseState | None = None,
828
+ depth: int = 1,
829
+ ) -> JsonValue:
830
+ state = state if state is not None else _ParseState()
831
+ if text in {"null", "~"}:
832
+ return None
833
+ if text == "true":
834
+ return True
835
+ if text == "false":
836
+ return False
837
+ lowered = text.lower()
838
+ if lowered in {".nan", ".inf", "+.inf", "-.inf", "nan", "inf", "+inf", "-inf"}:
839
+ raise ParserError(
840
+ "MF-S011",
841
+ "Non-finite YAML numbers are not allowed.",
842
+ line=line.number,
843
+ column=line.indent + 1,
844
+ )
845
+ if _looks_like_timestamp(text):
846
+ raise ParserError(
847
+ "MF-S008",
848
+ "YAML timestamps are not supported.",
849
+ line=line.number,
850
+ column=line.indent + 1,
851
+ )
852
+ if text.startswith(("[", "{")):
853
+ try:
854
+ return _parse_json_text(text, state=state, depth=depth + 1)
855
+ except ParserError as exc:
856
+ if exc.line is not None and exc.column is not None:
857
+ raise ParserError(
858
+ exc.code,
859
+ exc.message,
860
+ line=line.number + exc.line - 1,
861
+ column=line.indent + exc.column,
862
+ ) from exc
863
+ raise
864
+ if text.startswith('"'):
865
+ try:
866
+ value = _parse_json_text(text, state=state, depth=depth + 1)
867
+ except ParserError as exc:
868
+ if exc.line is not None and exc.column is not None:
869
+ raise ParserError(
870
+ exc.code,
871
+ exc.message,
872
+ line=line.number + exc.line - 1,
873
+ column=line.indent + exc.column,
874
+ ) from exc
875
+ raise
876
+ if not isinstance(value, str):
877
+ raise ParserError(
878
+ "MF-S011",
879
+ "Malformed YAML quoted scalar.",
880
+ line=line.number,
881
+ column=line.indent + 1,
882
+ )
883
+ return value
884
+ if text.startswith("'") and text.endswith("'"):
885
+ value = text[1:-1].replace("''", "'")
886
+ _validate_scalar_text(value)
887
+ return value
888
+ if _JSON_INTEGER_RE.fullmatch(text):
889
+ if len(text.encode("ascii")) > MAX_INTEGER_LEXEME_ASCII:
890
+ raise ParserError("MF-S010", "Integer lexeme exceeds parser limit.")
891
+ return int(text)
892
+ if _JSON_FLOAT_RE.fullmatch(text):
893
+ if len(text.encode("ascii")) > MAX_FLOAT_LEXEME_ASCII:
894
+ raise ParserError("MF-S010", "Floating-point lexeme exceeds parser limit.")
895
+ return float(text)
896
+ _validate_scalar_text(text)
897
+ return text
898
+
899
+
900
+ def _strip_yaml_comment(line: str) -> str:
901
+ quote: str | None = None
902
+ index = 0
903
+ while index < len(line):
904
+ char = line[index]
905
+ if quote is not None:
906
+ if char == quote:
907
+ quote = None
908
+ index += 1
909
+ continue
910
+ if char in {"'", '"'}:
911
+ quote = char
912
+ elif char == "#" and (index == 0 or line[index - 1].isspace()):
913
+ return line[:index]
914
+ index += 1
915
+ return line
916
+
917
+
918
+ def _looks_like_timestamp(text: str) -> bool:
919
+ return re.fullmatch(r"[0-9]{4}-[0-9]{2}-[0-9]{2}(?:[Tt ].*)?", text) is not None
920
+
921
+
922
+ def _validate_json_compatible(value: JsonValue) -> None:
923
+ if isinstance(value, dict):
924
+ for key, item in value.items():
925
+ _validate_scalar_text(key)
926
+ _validate_json_compatible(item)
927
+ return
928
+ if isinstance(value, list):
929
+ for item in value:
930
+ _validate_json_compatible(item)
931
+ return
932
+ if isinstance(value, str):
933
+ _validate_scalar_text(value)
934
+ return
935
+ if isinstance(value, float) and not (float("-inf") < value < float("inf")):
936
+ raise ParserError("MF-S011", "Non-finite number is not allowed.")
937
+ if not isinstance(value, (int, bool, type(None), float)):
938
+ raise ParserError("MF-S008", "Only JSON-native YAML values are supported.")
939
+
940
+
941
+ def _validate_scalar_text(value: str) -> None:
942
+ _reject_forbidden_text(value)
943
+ if len(value.encode("utf-8")) > MAX_SCALAR_UTF8_SIZE:
944
+ raise ParserError("MF-S010", "Scalar exceeds parser UTF-8 size limit.")
945
+
946
+
947
+ def _enforce_limits(value: JsonValue) -> None:
948
+ entries = 0
949
+
950
+ def walk(item: JsonValue, depth: int) -> None:
951
+ nonlocal entries
952
+ if depth > MAX_NESTING_DEPTH:
953
+ raise ParserError("MF-S010", "Nesting depth exceeds parser limit.")
954
+ if isinstance(item, Mapping):
955
+ entries += len(item)
956
+ if entries > MAX_TOTAL_ENTRIES:
957
+ raise ParserError(
958
+ "MF-S010", "Total mapping/list entries exceed parser limit."
959
+ )
960
+ for child in item.values():
961
+ walk(child, depth + 1)
962
+ elif isinstance(item, list):
963
+ entries += len(item)
964
+ if entries > MAX_TOTAL_ENTRIES:
965
+ raise ParserError(
966
+ "MF-S010", "Total mapping/list entries exceed parser limit."
967
+ )
968
+ for child in item:
969
+ walk(child, depth + 1)
970
+
971
+ walk(value, 1)
972
+
973
+
974
+ def _build_location_index(
975
+ document: SourceDocument, content: bytes, payload: JsonValue
976
+ ) -> tuple[SourceReference, ...]:
977
+ if not isinstance(payload, Mapping):
978
+ return ()
979
+ text = content.decode("utf-8")
980
+ if document.format == "json":
981
+ references = _json_location_references(document.logical_path, text)
982
+ else:
983
+ references = _yaml_location_references(document.logical_path, text)
984
+ path_aliases = _source_path_aliases(payload)
985
+ known_payload_paths = {
986
+ _translate_source_path(path, path_aliases) for path in _payload_paths(payload)
987
+ }
988
+ translated: list[SourceReference] = []
989
+ for reference in references:
990
+ translated_path = _translate_source_path(reference.field_path, path_aliases)
991
+ if translated_path in known_payload_paths:
992
+ translated.append(
993
+ SourceReference(
994
+ logical_path=reference.logical_path,
995
+ field_path=translated_path,
996
+ location=reference.location,
997
+ )
998
+ )
999
+ return tuple(translated)
1000
+
1001
+
1002
+ def _payload_paths(value: JsonValue, path: str = "/") -> tuple[str, ...]:
1003
+ result = [path]
1004
+ if isinstance(value, Mapping):
1005
+ for key, item in value.items():
1006
+ result.extend(_payload_paths(item, _join_pointer(path, key)))
1007
+ elif isinstance(value, list):
1008
+ for index, item in enumerate(value):
1009
+ result.extend(_payload_paths(item, _join_pointer(path, str(index))))
1010
+ return tuple(result)
1011
+
1012
+
1013
+ def _source_path_aliases(
1014
+ value: JsonValue,
1015
+ *,
1016
+ raw_path: str = "/",
1017
+ model_path: str = "/",
1018
+ aliases: dict[str, str] | None = None,
1019
+ ) -> dict[str, str]:
1020
+ aliases = {} if aliases is None else aliases
1021
+ aliases[raw_path] = model_path
1022
+ if isinstance(value, Mapping):
1023
+ converted = raw_path.endswith(
1024
+ ("/nodes", "/argument_matches", "/required_by_terminal")
1025
+ )
1026
+ for index, (key, item) in enumerate(value.items()):
1027
+ child_raw_path = _join_pointer(raw_path, key)
1028
+ child_model_path = (
1029
+ _join_pointer(model_path, str(index))
1030
+ if converted
1031
+ else _join_pointer(model_path, key)
1032
+ )
1033
+ aliases[child_raw_path] = child_model_path
1034
+ _source_path_aliases(
1035
+ item,
1036
+ raw_path=child_raw_path,
1037
+ model_path=child_model_path,
1038
+ aliases=aliases,
1039
+ )
1040
+ elif isinstance(value, list):
1041
+ for index, item in enumerate(value):
1042
+ child_raw_path = _join_pointer(raw_path, str(index))
1043
+ child_model_path = _join_pointer(model_path, str(index))
1044
+ _source_path_aliases(
1045
+ item,
1046
+ raw_path=child_raw_path,
1047
+ model_path=child_model_path,
1048
+ aliases=aliases,
1049
+ )
1050
+ return aliases
1051
+
1052
+
1053
+ def _translate_source_path(path: str, aliases: Mapping[str, str]) -> str:
1054
+ candidate = path
1055
+ while candidate not in aliases:
1056
+ if candidate == "/":
1057
+ return path
1058
+ candidate = candidate.rsplit("/", 1)[0] or "/"
1059
+ translated = aliases[candidate]
1060
+ if candidate == path:
1061
+ return translated
1062
+ suffix = path[len(candidate) :]
1063
+ return translated + suffix if suffix else translated
1064
+
1065
+
1066
+ def _json_location_references(
1067
+ logical_path: str, text: str
1068
+ ) -> tuple[SourceReference, ...]:
1069
+ references: list[SourceReference] = []
1070
+
1071
+ def skip_ws(index: int) -> int:
1072
+ while index < len(text) and text[index] in " \t\r\n":
1073
+ index += 1
1074
+ return index
1075
+
1076
+ def parse_value(index: int, path: str) -> int:
1077
+ index = skip_ws(index)
1078
+ if index >= len(text):
1079
+ return index
1080
+ if text[index] == "{":
1081
+ return parse_object(index, path)
1082
+ if text[index] == "[":
1083
+ return parse_array(index, path)
1084
+ _, end = _decode_json_value(text, index)
1085
+ return end
1086
+
1087
+ def parse_object(index: int, path: str) -> int:
1088
+ index += 1
1089
+ index = skip_ws(index)
1090
+ if index < len(text) and text[index] == "}":
1091
+ return index + 1
1092
+ while index < len(text):
1093
+ key_start = skip_ws(index)
1094
+ key, key_end = _decode_json_value(text, key_start)
1095
+ if not isinstance(key, str):
1096
+ return key_end
1097
+ line, column = _line_column(text, key_start)
1098
+ key_path = _join_pointer(path, key)
1099
+ references.append(
1100
+ SourceReference(
1101
+ logical_path=logical_path,
1102
+ field_path=key_path,
1103
+ location=SourceLocation(line=line, column=column),
1104
+ )
1105
+ )
1106
+ index = skip_ws(key_end)
1107
+ if index >= len(text) or text[index] != ":":
1108
+ return index
1109
+ index = parse_value(index + 1, key_path)
1110
+ index = skip_ws(index)
1111
+ if index < len(text) and text[index] == ",":
1112
+ index += 1
1113
+ continue
1114
+ if index < len(text) and text[index] == "}":
1115
+ return index + 1
1116
+ return index
1117
+ return index
1118
+
1119
+ def parse_array(index: int, path: str) -> int:
1120
+ index += 1
1121
+ item_index = 0
1122
+ index = skip_ws(index)
1123
+ if index < len(text) and text[index] == "]":
1124
+ return index + 1
1125
+ while index < len(text):
1126
+ item_start = skip_ws(index)
1127
+ item_path = _join_pointer(path, str(item_index))
1128
+ line, column = _line_column(text, item_start)
1129
+ references.append(
1130
+ SourceReference(
1131
+ logical_path=logical_path,
1132
+ field_path=item_path,
1133
+ location=SourceLocation(line=line, column=column),
1134
+ )
1135
+ )
1136
+ index = parse_value(item_start, item_path)
1137
+ item_index += 1
1138
+ index = skip_ws(index)
1139
+ if index < len(text) and text[index] == ",":
1140
+ index += 1
1141
+ continue
1142
+ if index < len(text) and text[index] == "]":
1143
+ return index + 1
1144
+ return index
1145
+ return index
1146
+
1147
+ parse_value(0, "/")
1148
+ return tuple(references)
1149
+
1150
+
1151
+ def _yaml_location_references(
1152
+ logical_path: str, text: str
1153
+ ) -> tuple[SourceReference, ...]:
1154
+ references: list[SourceReference] = []
1155
+
1156
+ lines = _yaml_lines(text)
1157
+ if not lines:
1158
+ return ()
1159
+
1160
+ def record(field_path: str, line: int, column: int) -> None:
1161
+ references.append(
1162
+ SourceReference(
1163
+ logical_path=logical_path,
1164
+ field_path=field_path,
1165
+ location=SourceLocation(line=line, column=column),
1166
+ )
1167
+ )
1168
+
1169
+ def consume_block_scalar(index: int, indent: int) -> int:
1170
+ end = index + 1
1171
+ while end < len(lines):
1172
+ next_line = lines[end]
1173
+ if next_line.indent <= indent:
1174
+ break
1175
+ end += 1
1176
+ return end
1177
+
1178
+ def walk_block(index: int, indent: int, path: str) -> int:
1179
+ if index >= len(lines):
1180
+ return index
1181
+ line = lines[index]
1182
+ if line.indent != indent:
1183
+ return index
1184
+ if line.text.startswith("- "):
1185
+ return walk_sequence(index, indent, path)
1186
+ return walk_mapping(index, indent, path)
1187
+
1188
+ def walk_mapping(index: int, indent: int, path: str) -> int:
1189
+ while index < len(lines):
1190
+ line = lines[index]
1191
+ if line.indent < indent:
1192
+ break
1193
+ if line.indent != indent or line.text.startswith("- "):
1194
+ break
1195
+ key_text, value_text = _split_yaml_key_value(line)
1196
+ key = _parse_yaml_key(key_text, line)
1197
+ key_path = _join_pointer(path, key)
1198
+ key_column = line.indent + line.text.index(key_text) + 1
1199
+ record(key_path, line.number, key_column)
1200
+ if value_text == "":
1201
+ index += 1
1202
+ if index < len(lines) and lines[index].indent > indent:
1203
+ index = walk_block(index, lines[index].indent, key_path)
1204
+ continue
1205
+ if value_text == "|":
1206
+ index = consume_block_scalar(index, indent)
1207
+ continue
1208
+ index += 1
1209
+ return index
1210
+
1211
+ def walk_sequence(index: int, indent: int, path: str) -> int:
1212
+ item_index = 0
1213
+ while index < len(lines):
1214
+ line = lines[index]
1215
+ if line.indent < indent:
1216
+ break
1217
+ if line.indent != indent or not line.text.startswith("- "):
1218
+ break
1219
+ item_path = _join_pointer(path, str(item_index))
1220
+ record(item_path, line.number, line.indent + 1)
1221
+ item_text = line.text[2:].strip()
1222
+ if item_text == "":
1223
+ index += 1
1224
+ if index < len(lines) and lines[index].indent > indent:
1225
+ index = walk_block(index, lines[index].indent, item_path)
1226
+ item_index += 1
1227
+ continue
1228
+ if _looks_like_yaml_key_value(item_text):
1229
+ item_line = _YamlLine(line.number, line.indent + 2, item_text)
1230
+ key_text, _ = _split_yaml_key_value(item_line)
1231
+ key = _parse_yaml_key(key_text, line)
1232
+ key_path = _join_pointer(item_path, key)
1233
+ key_column = item_line.indent + item_line.text.index(key_text) + 1
1234
+ record(key_path, line.number, key_column)
1235
+ index += 1
1236
+ if index < len(lines) and lines[index].indent > indent:
1237
+ index = walk_mapping(index, lines[index].indent, item_path)
1238
+ item_index += 1
1239
+ continue
1240
+ if item_text == "|":
1241
+ index = consume_block_scalar(index, indent)
1242
+ item_index += 1
1243
+ continue
1244
+ index += 1
1245
+ item_index += 1
1246
+ return index
1247
+
1248
+ walk_block(0, lines[0].indent, "/")
1249
+ return tuple(references)
1250
+
1251
+
1252
+ def _nearest_source_reference(
1253
+ document: SourceDocument,
1254
+ field_path: str,
1255
+ location_index: tuple[SourceReference, ...],
1256
+ ) -> SourceReference:
1257
+ by_path = {reference.field_path: reference for reference in location_index}
1258
+ candidate = field_path
1259
+ while candidate:
1260
+ if candidate in by_path:
1261
+ return by_path[candidate]
1262
+ if candidate == "/":
1263
+ break
1264
+ candidate = candidate.rsplit("/", 1)[0] or "/"
1265
+ return SourceReference(logical_path=document.logical_path, field_path=field_path)
1266
+
1267
+
1268
+ def _join_pointer(parent: str, token: str) -> str:
1269
+ escaped = token.replace("~", "~0").replace("/", "~1")
1270
+ return f"/{escaped}" if parent == "/" else f"{parent}/{escaped}"
1271
+
1272
+
1273
+ def _diagnostic(
1274
+ document: SourceDocument,
1275
+ error: ParserError,
1276
+ location_index: tuple[SourceReference, ...] = (),
1277
+ ) -> CompilerDiagnostic:
1278
+ phase, severity = (
1279
+ CompilerPhase.SCHEMA
1280
+ if error.code.startswith("MF-S02")
1281
+ else CompilerPhase.PARSE,
1282
+ DiagnosticSeverity.ERROR,
1283
+ )
1284
+ source_reference = _nearest_source_reference(
1285
+ document, error.field_path, location_index
1286
+ )
1287
+ if error.line is not None and error.column is not None:
1288
+ source_reference = SourceReference(
1289
+ logical_path=document.logical_path,
1290
+ field_path=error.field_path,
1291
+ location=SourceLocation(line=error.line, column=error.column),
1292
+ )
1293
+ return CompilerDiagnostic(
1294
+ code=error.code,
1295
+ phase=phase,
1296
+ severity=severity,
1297
+ message=error.message,
1298
+ source_reference=source_reference,
1299
+ fields=error.fields,
1300
+ )
1301
+
1302
+
1303
+ def _schema_validation_diagnostics(
1304
+ document: SourceDocument,
1305
+ exc: ValidationError,
1306
+ location_index: tuple[SourceReference, ...],
1307
+ ) -> tuple[CompilerDiagnostic, ...]:
1308
+ diagnostics = []
1309
+ for error in exc.errors():
1310
+ code = _schema_validation_code(error)
1311
+ field_path = _validation_error_pointer(error)
1312
+ diagnostics.append(
1313
+ _diagnostic(
1314
+ document,
1315
+ ParserError(
1316
+ code,
1317
+ _schema_validation_message(code),
1318
+ field_path=field_path,
1319
+ fields=(DiagnosticField(key="field_path", value=field_path),),
1320
+ ),
1321
+ location_index,
1322
+ )
1323
+ )
1324
+ if not diagnostics:
1325
+ diagnostics.append(
1326
+ _diagnostic(
1327
+ document,
1328
+ ParserError(
1329
+ "MF-S020",
1330
+ _schema_validation_message("MF-S020"),
1331
+ field_path="/",
1332
+ fields=(DiagnosticField(key="field_path", value="/"),),
1333
+ ),
1334
+ location_index,
1335
+ )
1336
+ )
1337
+ return bound_diagnostics(diagnostics)
1338
+
1339
+
1340
+ def _schema_validation_code(error: Mapping[str, Any]) -> str:
1341
+ error_type = error.get("type")
1342
+ loc = error.get("loc")
1343
+ path = tuple(str(part) for part in loc) if isinstance(loc, tuple) else ()
1344
+ message = str(error.get("msg", ""))
1345
+
1346
+ if error_type == "extra_forbidden":
1347
+ return "MF-S021"
1348
+ if "tool_ref" in path or "tool_ref" in message:
1349
+ return "MF-S023"
1350
+ if path and path[0] == "budgets":
1351
+ return "MF-S024"
1352
+ if path and path[0] == "context":
1353
+ return "MF-S025"
1354
+ if _is_identifier_validation_error(path, message):
1355
+ return "MF-S022"
1356
+ return "MF-S020"
1357
+
1358
+
1359
+ def _is_identifier_validation_error(path: tuple[str, ...], message: str) -> bool:
1360
+ identifier_names = {
1361
+ "argument_name",
1362
+ "artifact_id",
1363
+ "current_argument",
1364
+ "declared_artifact_ids",
1365
+ "expected_harness_id",
1366
+ "harness_id",
1367
+ "model_profile_id",
1368
+ "node_id",
1369
+ "policy_id",
1370
+ "prior_argument",
1371
+ "produces",
1372
+ "profile_id",
1373
+ "required_by_terminal",
1374
+ "stage_kind_id",
1375
+ "stage_kind_ids",
1376
+ "terminal_result",
1377
+ }
1378
+ identifier_message_markers = (
1379
+ "argument_name ",
1380
+ "artifact_id ",
1381
+ "harness_id ",
1382
+ "node_id ",
1383
+ "policy_id ",
1384
+ "profile_id ",
1385
+ "stage_kind_id ",
1386
+ "terminal_result ",
1387
+ )
1388
+ return any(part in identifier_names for part in path) or any(
1389
+ marker in message for marker in identifier_message_markers
1390
+ )
1391
+
1392
+
1393
+ def _schema_validation_message(code: str) -> str:
1394
+ return {
1395
+ "MF-S021": "Source schema contains an unknown field.",
1396
+ "MF-S022": "Source identifier is invalid.",
1397
+ "MF-S023": "Source tool_ref must be an exact-version tool reference.",
1398
+ "MF-S024": "Source budget value is invalid.",
1399
+ "MF-S025": "Source context policy value is invalid.",
1400
+ }.get(code, "Source schema validation failed.")
1401
+
1402
+
1403
+ def _validation_pointer(exc: ValidationError) -> str:
1404
+ errors = exc.errors()
1405
+ if not errors:
1406
+ return "/"
1407
+ return _validation_error_pointer(errors[0])
1408
+
1409
+
1410
+ def _validation_error_pointer(error: Mapping[str, Any]) -> str:
1411
+ loc = error.get("loc")
1412
+ if not isinstance(loc, tuple) or not loc:
1413
+ return "/"
1414
+ parts = [str(part).replace("~", "~0").replace("/", "~1") for part in loc]
1415
+ return "/" + "/".join(parts)
1416
+
1417
+
1418
+ def _line_column(text: str, index: int) -> tuple[int, int]:
1419
+ line = text.count("\n", 0, index) + 1
1420
+ line_start = text.rfind("\n", 0, index) + 1
1421
+ return line, index - line_start + 1
1422
+
1423
+
1424
+ DefaultHarnessSourceParser = HarnessSourceParser