millforge 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (116) hide show
  1. millforge/__init__.py +1174 -0
  2. millforge/_forge/LICENSE +21 -0
  3. millforge/_forge/PROVENANCE.json +295 -0
  4. millforge/_forge/UPDATE_POLICY.md +24 -0
  5. millforge/_forge/__init__.py +14 -0
  6. millforge/_forge/adapter.py +2232 -0
  7. millforge/_forge/base_runner.py +121 -0
  8. millforge/_forge/clients/__init__.py +10 -0
  9. millforge/_forge/clients/base.py +200 -0
  10. millforge/_forge/context/__init__.py +23 -0
  11. millforge/_forge/context/manager.py +178 -0
  12. millforge/_forge/context/strategies.py +335 -0
  13. millforge/_forge/core/__init__.py +16 -0
  14. millforge/_forge/core/inference.py +433 -0
  15. millforge/_forge/core/messages.py +119 -0
  16. millforge/_forge/core/runner.py +479 -0
  17. millforge/_forge/core/steps.py +108 -0
  18. millforge/_forge/core/workflow.py +400 -0
  19. millforge/_forge/errors.py +222 -0
  20. millforge/_forge/guardrails/__init__.py +21 -0
  21. millforge/_forge/guardrails/error_tracker.py +71 -0
  22. millforge/_forge/guardrails/guardrails.py +194 -0
  23. millforge/_forge/guardrails/nudge.py +47 -0
  24. millforge/_forge/guardrails/response_validator.py +119 -0
  25. millforge/_forge/guardrails/step_enforcer.py +183 -0
  26. millforge/_forge/prompts/__init__.py +16 -0
  27. millforge/_forge/prompts/nudges.py +95 -0
  28. millforge/_forge/prompts/templates.py +285 -0
  29. millforge/_version.py +3 -0
  30. millforge/artifacts.py +570 -0
  31. millforge/base/__init__.py +97 -0
  32. millforge/base/composition.py +402 -0
  33. millforge/base/context.py +285 -0
  34. millforge/base/harness.py +138 -0
  35. millforge/base/identity.py +465 -0
  36. millforge/base/options.py +34 -0
  37. millforge/base/platform.py +17 -0
  38. millforge/base/prompt.py +317 -0
  39. millforge/base/runner.py +546 -0
  40. millforge/compiled_plan.py +970 -0
  41. millforge/compiler/__init__.py +231 -0
  42. millforge/compiler/artifact_validation.py +257 -0
  43. millforge/compiler/canonicalization.py +169 -0
  44. millforge/compiler/capabilities.py +66 -0
  45. millforge/compiler/catalogs.py +500 -0
  46. millforge/compiler/diagnostics.py +491 -0
  47. millforge/compiler/graph.py +678 -0
  48. millforge/compiler/lowering.py +198 -0
  49. millforge/compiler/output.py +692 -0
  50. millforge/compiler/parsing.py +1424 -0
  51. millforge/compiler/requests.py +1180 -0
  52. millforge/compiler/schema_validation.py +272 -0
  53. millforge/compiler/semantic.py +490 -0
  54. millforge/compiler/service.py +448 -0
  55. millforge/compiler/source.py +375 -0
  56. millforge/compiler/validators.py +184 -0
  57. millforge/connectors/__init__.py +95 -0
  58. millforge/connectors/admission.py +801 -0
  59. millforge/connectors/broker.py +202 -0
  60. millforge/connectors/contracts.py +1159 -0
  61. millforge/connectors/diagnostics.py +189 -0
  62. millforge/connectors/fake.py +66 -0
  63. millforge/connectors/runtime.py +236 -0
  64. millforge/contracts.py +2860 -0
  65. millforge/custom_tools/__init__.py +67 -0
  66. millforge/custom_tools/compiler.py +724 -0
  67. millforge/custom_tools/contracts.py +1093 -0
  68. millforge/custom_tools/diagnostics.py +205 -0
  69. millforge/eval_artifacts.py +952 -0
  70. millforge/eval_boundary.py +2435 -0
  71. millforge/eval_fixtures/__init__.py +1 -0
  72. millforge/eval_fixtures/default_pack/__init__.py +1 -0
  73. millforge/eval_fixtures/default_pack/fixtures/fixture.08a.bug_diagnosis.traceback.v1.json +52 -0
  74. millforge/eval_fixtures/default_pack/fixtures/fixture.08a.direct_edit.import_sort.v1.json +52 -0
  75. millforge/eval_fixtures/default_pack/fixtures/fixture.08a.evidence_discipline.no_source_change.v1.json +51 -0
  76. millforge/eval_fixtures/default_pack/fixtures/fixture.08a.false_closure.visible_green.v1.json +52 -0
  77. millforge/eval_fixtures/default_pack/fixtures/fixture.08a.multi_file.api_contract.v1.json +54 -0
  78. millforge/eval_fixtures/default_pack/fixtures/fixture.08a.recovery.malformed_artifact.v1.json +54 -0
  79. millforge/eval_fixtures/default_pack/manifest.json +12 -0
  80. millforge/eval_modes.py +1282 -0
  81. millforge/eval_presets.py +1398 -0
  82. millforge/eval_reports.py +2517 -0
  83. millforge/eval_suite.py +2429 -0
  84. millforge/eval_trials.py +2632 -0
  85. millforge/eval_workflow.py +794 -0
  86. millforge/exceptions.py +122 -0
  87. millforge/model_backend.py +2098 -0
  88. millforge/protocols.py +340 -0
  89. millforge/py.typed +0 -0
  90. millforge/runtime.py +1791 -0
  91. millforge/testing/__init__.py +1089 -0
  92. millforge/tools/__init__.py +83 -0
  93. millforge/tools/builtin_runtime.py +1339 -0
  94. millforge/tools/builtins.py +773 -0
  95. millforge/tools/execution.py +1545 -0
  96. millforge/tools/path_policy.py +155 -0
  97. millforge/tools/pi_compat/PI_LICENSE +21 -0
  98. millforge/tools/pi_compat/PROVENANCE.json +55 -0
  99. millforge/tools/pi_compat/UPDATE_POLICY.md +36 -0
  100. millforge/tools/pi_compat/__init__.py +34 -0
  101. millforge/tools/pi_compat/contracts.py +49 -0
  102. millforge/tools/pi_compat/editing.py +390 -0
  103. millforge/tools/pi_compat/mutations.py +57 -0
  104. millforge/tools/pi_compat/operations.py +401 -0
  105. millforge/tools/pi_compat/paths.py +155 -0
  106. millforge/tools/pi_compat/process.py +1375 -0
  107. millforge/tools/pi_compat/search.py +738 -0
  108. millforge/tools/pi_compat/truncation.py +267 -0
  109. millforge/tools/pi_compat_catalog.py +396 -0
  110. millforge/tools/pi_compat_runtime.py +460 -0
  111. millforge/tools/registry.py +553 -0
  112. millforge/tools/results.py +533 -0
  113. millforge-0.1.0.dist-info/METADATA +844 -0
  114. millforge-0.1.0.dist-info/RECORD +116 -0
  115. millforge-0.1.0.dist-info/WHEEL +4 -0
  116. millforge-0.1.0.dist-info/licenses/LICENSE +201 -0
@@ -0,0 +1,724 @@
1
+ """Deterministic offline compilation for custom-tool source manifests."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import re
6
+ from collections.abc import Mapping
7
+ from typing import Any, TypeVar
8
+
9
+ from pydantic import BaseModel, ValidationError
10
+
11
+ from millforge.compiler.diagnostics import detect_secret_candidate
12
+ from millforge.compiler.schema_validation import (
13
+ SchemaSubsetError,
14
+ normalized_schema_bytes,
15
+ )
16
+ from millforge.contracts import RedactionPolicy
17
+ from millforge.custom_tools.contracts import (
18
+ CustomToolApprovalPolicy,
19
+ CustomToolCompilationRecord,
20
+ CustomToolCompilationResult,
21
+ CustomToolCompilerPolicy,
22
+ CustomToolDeclaration,
23
+ CustomToolSourceManifest,
24
+ compilation_record_from_declaration,
25
+ tool_descriptor_from_declaration,
26
+ )
27
+ from millforge.custom_tools.diagnostics import (
28
+ CustomToolDiagnostic,
29
+ CustomToolDiagnosticCode,
30
+ CustomToolDiagnosticPhase,
31
+ custom_tool_diagnostic,
32
+ custom_tool_diagnostic_sort_key,
33
+ malformed_input_diagnostic,
34
+ )
35
+ from millforge.tools.registry import ToolDescriptor
36
+
37
+ _T = TypeVar("_T", bound=BaseModel)
38
+
39
+ _LIVE_URL_RE = re.compile(r"\bhttps?://[^\s<>'\"]+", re.IGNORECASE)
40
+ _ABSOLUTE_PATH_RE = re.compile(r"(?<![A-Za-z0-9_.-])(?:/[A-Za-z0-9_.-][^\s]*)")
41
+ _WINDOWS_ABSOLUTE_PATH_RE = re.compile(r"\b[A-Za-z]:\\[^\s]+")
42
+ _PARENT_TRAVERSAL_RE = re.compile(r"(^|[\s\\/])\.\.([\\/]|$)")
43
+ _SHELL_COMMAND_RE = re.compile(
44
+ r"(?i)(^|\s)(?:rm\s+-rf|curl\s+|wget\s+|bash\s+-c|sh\s+-c|"
45
+ r"python(?:3)?\s+-c|node\s+-e|powershell\b|cmd\.exe\b|chmod\s+\+x)\b"
46
+ )
47
+ _SCRIPT_BODY_RE = re.compile(
48
+ r"(?is)(^#!|<script\b|function\s+\w*\s*\(|def\s+\w+\s*\(|"
49
+ r"import\s+os\b|subprocess\.|eval\s*\(|exec\s*\()"
50
+ )
51
+ _TEMPLATE_INTERPOLATION_RE = re.compile(r"({{.*?}}|{%.+?%}|\$\{[^}]+})")
52
+ _INSTRUCTION_LIKE_RE = re.compile(
53
+ r"\b(ignore|override|system prompt|developer message|previous instructions|"
54
+ r"follow these instructions|you must|do not tell)\b",
55
+ re.IGNORECASE,
56
+ )
57
+ _EXECUTABLE_RUNTIME_KINDS = frozenset(
58
+ {
59
+ "shell",
60
+ "process",
61
+ "python",
62
+ "javascript",
63
+ "js",
64
+ "node",
65
+ "wasm",
66
+ "http",
67
+ "https",
68
+ "mcp",
69
+ "connector",
70
+ "connector_alias",
71
+ "filesystem",
72
+ "fs",
73
+ "terminal",
74
+ }
75
+ )
76
+
77
+
78
+ def compile_custom_tools(
79
+ source: CustomToolSourceManifest | Mapping[str, Any],
80
+ policy: CustomToolCompilerPolicy | Mapping[str, Any],
81
+ ) -> CustomToolCompilationResult:
82
+ """Validate and lower contract-only custom tools into descriptors.
83
+
84
+ Raw mappings are treated as untrusted source. Validation and lowering errors
85
+ are returned as stable custom-tool diagnostics, and any diagnostic rejects
86
+ the whole manifest with no partial descriptors or records.
87
+ """
88
+ source_hazards = _raw_source_hazards(source, phase=CustomToolDiagnosticPhase.SOURCE)
89
+ policy_hazards = _raw_source_hazards(policy, phase=CustomToolDiagnosticPhase.POLICY)
90
+ if source_hazards or policy_hazards:
91
+ return _rejected((*source_hazards, *policy_hazards))
92
+
93
+ valid_source, source_diagnostic = _validate_contract(
94
+ CustomToolSourceManifest,
95
+ source,
96
+ phase=CustomToolDiagnosticPhase.SOURCE,
97
+ )
98
+ valid_policy, policy_diagnostic = _validate_contract(
99
+ CustomToolCompilerPolicy,
100
+ policy,
101
+ phase=CustomToolDiagnosticPhase.POLICY,
102
+ )
103
+ diagnostics = tuple(
104
+ diagnostic
105
+ for diagnostic in (source_diagnostic, policy_diagnostic)
106
+ if diagnostic
107
+ )
108
+ if diagnostics:
109
+ return _rejected(diagnostics)
110
+ if not isinstance(valid_source, CustomToolSourceManifest):
111
+ raise AssertionError("source validation returned unexpected contract")
112
+ if not isinstance(valid_policy, CustomToolCompilerPolicy):
113
+ raise AssertionError("policy validation returned unexpected contract")
114
+
115
+ compiler = _CustomToolCompilation(valid_source, valid_policy)
116
+ return compiler.run()
117
+
118
+
119
+ class _CustomToolCompilation:
120
+ def __init__(
121
+ self,
122
+ source: CustomToolSourceManifest,
123
+ policy: CustomToolCompilerPolicy,
124
+ ) -> None:
125
+ self.source = source
126
+ self.policy = policy
127
+ self.diagnostics: list[CustomToolDiagnostic] = []
128
+
129
+ def run(self) -> CustomToolCompilationResult:
130
+ self._validate_source_policy()
131
+
132
+ lowered: list[tuple[ToolDescriptor, CustomToolCompilationRecord]] = []
133
+ for index, declaration in enumerate(self.source.tools):
134
+ compiled = self._lower_declaration(declaration, index=index)
135
+ if compiled is not None:
136
+ lowered.append(compiled)
137
+
138
+ if self.diagnostics:
139
+ return self._rejected()
140
+
141
+ return CustomToolCompilationResult(
142
+ accepted=True,
143
+ source_sha256=self.source.source_sha256,
144
+ descriptors=tuple(descriptor for descriptor, _ in _sort_lowered(lowered)),
145
+ records=tuple(record for _, record in _sort_lowered(lowered)),
146
+ )
147
+
148
+ def _validate_source_policy(self) -> None:
149
+ if len(self.source.tools) > self.policy.max_tools:
150
+ self._diagnose(
151
+ CustomToolDiagnosticCode.LIMIT_EXCEEDED,
152
+ phase=CustomToolDiagnosticPhase.POLICY,
153
+ path="/tools",
154
+ message="Custom-tool source exceeds the compiler policy tool limit.",
155
+ evidence={"limit": self.policy.max_tools},
156
+ )
157
+ self._check_hash(
158
+ supplied=self.source.expected_source_sha256,
159
+ actual=self.source.source_sha256,
160
+ path="/expected_source_sha256",
161
+ label="source_sha256",
162
+ )
163
+ if (
164
+ self.policy.require_expected_hashes
165
+ and self.source.expected_source_sha256 is None
166
+ ):
167
+ self._missing_hash("/expected_source_sha256", "source_sha256")
168
+
169
+ produced_artifacts: dict[str, str] = {}
170
+ for declaration in self.source.tools:
171
+ for artifact_id in declaration.produced_artifact_ids:
172
+ prior = produced_artifacts.get(artifact_id)
173
+ if prior is not None:
174
+ self._diagnose(
175
+ CustomToolDiagnosticCode.ARTIFACT_POLICY_INVALID,
176
+ phase=CustomToolDiagnosticPhase.SOURCE,
177
+ path="/tools",
178
+ message="Custom-tool source contains duplicate produced artifacts.",
179
+ evidence={"artifact_id": artifact_id, "first_tool_id": prior},
180
+ )
181
+ else:
182
+ produced_artifacts[artifact_id] = declaration.tool_id
183
+
184
+ def _lower_declaration(
185
+ self,
186
+ declaration: CustomToolDeclaration,
187
+ *,
188
+ index: int,
189
+ ) -> tuple[ToolDescriptor, CustomToolCompilationRecord] | None:
190
+ path = f"/tools/{index}"
191
+ if declaration.runtime_kind not in self.policy.allowed_runtime_kinds:
192
+ self._diagnose(
193
+ CustomToolDiagnosticCode.RUNTIME_KIND_UNSUPPORTED,
194
+ phase=CustomToolDiagnosticPhase.POLICY,
195
+ path=f"{path}/runtime_kind",
196
+ message="Custom-tool runtime kind is not allowed by compiler policy.",
197
+ evidence={"runtime_kind": declaration.runtime_kind.value},
198
+ )
199
+ if (
200
+ len(declaration.description.encode("utf-8"))
201
+ > self.policy.max_description_utf8
202
+ ):
203
+ self._diagnose(
204
+ CustomToolDiagnosticCode.LIMIT_EXCEEDED,
205
+ phase=CustomToolDiagnosticPhase.POLICY,
206
+ path=f"{path}/description",
207
+ message="Custom-tool description exceeds the compiler policy limit.",
208
+ evidence={"tool_id": declaration.tool_id},
209
+ )
210
+ self._validate_schema_bytes(
211
+ declaration.input_schema,
212
+ path=f"{path}/input_schema",
213
+ code=CustomToolDiagnosticCode.INPUT_SCHEMA_UNSUPPORTED,
214
+ )
215
+ self._validate_schema_bytes(
216
+ declaration.output_schema,
217
+ path=f"{path}/output_schema",
218
+ code=CustomToolDiagnosticCode.OUTPUT_SCHEMA_UNSUPPORTED,
219
+ )
220
+ self._validate_capabilities(declaration, path=path)
221
+ self._validate_approval(declaration, path=path)
222
+ self._check_hash(
223
+ supplied=declaration.expected_declaration_sha256,
224
+ actual=declaration.declaration_sha256,
225
+ path=f"{path}/expected_declaration_sha256",
226
+ label="declaration_sha256",
227
+ )
228
+ if (
229
+ self.policy.require_expected_hashes
230
+ and declaration.expected_declaration_sha256 is None
231
+ ):
232
+ self._missing_hash(
233
+ f"{path}/expected_declaration_sha256", "declaration_sha256"
234
+ )
235
+ try:
236
+ descriptor = tool_descriptor_from_declaration(declaration)
237
+ self._check_hash(
238
+ supplied=declaration.expected_descriptor_sha256,
239
+ actual=descriptor.descriptor_sha256,
240
+ path=f"{path}/expected_descriptor_sha256",
241
+ label="descriptor_sha256",
242
+ )
243
+ if (
244
+ self.policy.require_expected_hashes
245
+ and declaration.expected_descriptor_sha256 is None
246
+ ):
247
+ self._missing_hash(
248
+ f"{path}/expected_descriptor_sha256", "descriptor_sha256"
249
+ )
250
+ record = compilation_record_from_declaration(
251
+ self.source, declaration, descriptor
252
+ )
253
+ except Exception as exc:
254
+ self._diagnose(
255
+ CustomToolDiagnosticCode.DECLARATION_INVALID,
256
+ phase=CustomToolDiagnosticPhase.COMPILATION,
257
+ path=path,
258
+ message="Custom-tool descriptor construction failed.",
259
+ evidence={"error_type": type(exc).__name__},
260
+ )
261
+ return None
262
+
263
+ self._check_hash(
264
+ supplied=declaration.expected_compilation_record_sha256,
265
+ actual=record.compilation_record_sha256,
266
+ path=f"{path}/expected_compilation_record_sha256",
267
+ label="compilation_record_sha256",
268
+ )
269
+ if (
270
+ self.policy.require_expected_hashes
271
+ and declaration.expected_compilation_record_sha256 is None
272
+ ):
273
+ self._missing_hash(
274
+ f"{path}/expected_compilation_record_sha256",
275
+ "compilation_record_sha256",
276
+ )
277
+ if self.diagnostics:
278
+ return None
279
+ return descriptor, record
280
+
281
+ def _validate_schema_bytes(
282
+ self,
283
+ schema: Mapping[str, Any],
284
+ *,
285
+ path: str,
286
+ code: CustomToolDiagnosticCode,
287
+ ) -> None:
288
+ size = len(normalized_schema_bytes(schema))
289
+ if size <= self.policy.max_schema_bytes:
290
+ return
291
+ self._diagnose(
292
+ code,
293
+ phase=CustomToolDiagnosticPhase.POLICY,
294
+ path=path,
295
+ message="Custom-tool schema exceeds the compiler policy byte limit.",
296
+ evidence={"limit": self.policy.max_schema_bytes, "size": size},
297
+ )
298
+
299
+ def _validate_capabilities(
300
+ self, declaration: CustomToolDeclaration, *, path: str
301
+ ) -> None:
302
+ if (
303
+ declaration.side_effect_class.value != "read_only"
304
+ or declaration.produced_artifact_ids
305
+ ) and not declaration.required_capabilities:
306
+ self._diagnose(
307
+ CustomToolDiagnosticCode.CAPABILITY_MISSING,
308
+ phase=CustomToolDiagnosticPhase.POLICY,
309
+ path=f"{path}/required_capabilities",
310
+ message="Custom tool requires explicit capabilities.",
311
+ evidence={"tool_id": declaration.tool_id},
312
+ )
313
+ return
314
+ allowed = set(self.policy.allowed_capability_ids)
315
+ for capability_id in declaration.required_capabilities:
316
+ if capability_id not in allowed:
317
+ self._diagnose(
318
+ CustomToolDiagnosticCode.CAPABILITY_UNKNOWN,
319
+ phase=CustomToolDiagnosticPhase.POLICY,
320
+ path=f"{path}/required_capabilities",
321
+ message="Custom-tool capability is not allowed by compiler policy.",
322
+ evidence={"capability_id": capability_id},
323
+ )
324
+
325
+ def _validate_approval(
326
+ self, declaration: CustomToolDeclaration, *, path: str
327
+ ) -> None:
328
+ allowed = self.policy.side_effect_approval_matrix.get(
329
+ declaration.side_effect_class
330
+ )
331
+ if allowed is None or declaration.approval_policy not in allowed:
332
+ self._diagnose(
333
+ CustomToolDiagnosticCode.APPROVAL_POLICY_INVALID,
334
+ phase=CustomToolDiagnosticPhase.POLICY,
335
+ path=f"{path}/approval_policy",
336
+ message="Approval policy is not allowed for side-effect class.",
337
+ evidence={
338
+ "approval_policy": declaration.approval_policy.value,
339
+ "side_effect_class": declaration.side_effect_class.value,
340
+ },
341
+ )
342
+ if (
343
+ declaration.produced_artifact_ids
344
+ and declaration.approval_policy is CustomToolApprovalPolicy.NONE
345
+ ):
346
+ self._diagnose(
347
+ CustomToolDiagnosticCode.APPROVAL_POLICY_INVALID,
348
+ phase=CustomToolDiagnosticPhase.POLICY,
349
+ path=f"{path}/approval_policy",
350
+ message="Artifact-producing custom tools require explicit approval.",
351
+ evidence={
352
+ "approval_policy": declaration.approval_policy.value,
353
+ "tool_id": declaration.tool_id,
354
+ },
355
+ )
356
+
357
+ def _check_hash(
358
+ self,
359
+ *,
360
+ supplied: str | None,
361
+ actual: str,
362
+ path: str,
363
+ label: str,
364
+ ) -> None:
365
+ if supplied is not None and supplied != actual:
366
+ self._diagnose(
367
+ CustomToolDiagnosticCode.HASH_MISMATCH,
368
+ phase=CustomToolDiagnosticPhase.COMPILATION,
369
+ path=path,
370
+ message="Supplied custom-tool hash does not match recomputed hash.",
371
+ evidence={"hash": label},
372
+ )
373
+
374
+ def _missing_hash(self, path: str, label: str) -> None:
375
+ self._diagnose(
376
+ CustomToolDiagnosticCode.HASH_MISMATCH,
377
+ phase=CustomToolDiagnosticPhase.POLICY,
378
+ path=path,
379
+ message="Compiler policy requires expected custom-tool hashes.",
380
+ evidence={"hash": label},
381
+ )
382
+
383
+ def _diagnose(
384
+ self,
385
+ code: CustomToolDiagnosticCode,
386
+ *,
387
+ phase: CustomToolDiagnosticPhase,
388
+ message: str,
389
+ location: str | None = None,
390
+ path: str | None = None,
391
+ evidence: Mapping[str, Any] | None = None,
392
+ ) -> None:
393
+ self.diagnostics.append(
394
+ custom_tool_diagnostic(
395
+ code,
396
+ phase=phase,
397
+ message=message,
398
+ location=location,
399
+ path=path,
400
+ evidence=evidence,
401
+ )
402
+ )
403
+
404
+ def _rejected(self) -> CustomToolCompilationResult:
405
+ return _rejected(tuple(self.diagnostics))
406
+
407
+
408
+ def _validate_contract(
409
+ model: type[_T],
410
+ value: _T | Mapping[str, Any],
411
+ *,
412
+ phase: CustomToolDiagnosticPhase,
413
+ ) -> tuple[_T | None, CustomToolDiagnostic | None]:
414
+ if isinstance(value, model):
415
+ return value, None
416
+ try:
417
+ return model.model_validate(value), None
418
+ except ValidationError as exc:
419
+ schema_diagnostic = _schema_subset_validation_diagnostic(
420
+ exc,
421
+ model_name=model.__name__,
422
+ phase=phase,
423
+ )
424
+ if schema_diagnostic is not None:
425
+ return None, schema_diagnostic
426
+ return (
427
+ None,
428
+ malformed_input_diagnostic(
429
+ phase=phase,
430
+ model_name=model.__name__,
431
+ path=_validation_pointer(exc),
432
+ missing_field=_missing_field(exc),
433
+ code=_validation_code(exc),
434
+ ),
435
+ )
436
+ except Exception:
437
+ return (
438
+ None,
439
+ malformed_input_diagnostic(
440
+ phase=phase,
441
+ model_name=model.__name__,
442
+ code=CustomToolDiagnosticCode.SOURCE_INVALID,
443
+ ),
444
+ )
445
+
446
+
447
+ def _raw_source_hazards(
448
+ value: Any, *, phase: CustomToolDiagnosticPhase
449
+ ) -> tuple[CustomToolDiagnostic, ...]:
450
+ if isinstance(value, BaseModel):
451
+ try:
452
+ value = value.model_dump(mode="json")
453
+ except Exception:
454
+ return (_hazard_diagnostic("/", phase, "runtime_object"),)
455
+ diagnostics: list[CustomToolDiagnostic] = []
456
+ _scan_raw_value(
457
+ value,
458
+ path="",
459
+ phase=phase,
460
+ diagnostics=diagnostics,
461
+ active_container_ids=set(),
462
+ )
463
+ return tuple(diagnostics)
464
+
465
+
466
+ def _scan_raw_value(
467
+ value: Any,
468
+ *,
469
+ path: str,
470
+ phase: CustomToolDiagnosticPhase,
471
+ diagnostics: list[CustomToolDiagnostic],
472
+ active_container_ids: set[int],
473
+ ) -> None:
474
+ if isinstance(value, Mapping):
475
+ container_id = id(value)
476
+ if container_id in active_container_ids:
477
+ diagnostics.append(
478
+ _hazard_diagnostic(path or "/", phase, "recursive_reference")
479
+ )
480
+ return
481
+ active_container_ids.add(container_id)
482
+ try:
483
+ for key, item in value.items():
484
+ key_path = _join_pointer(path, str(key))
485
+ if not isinstance(key, str):
486
+ diagnostics.append(_hazard_diagnostic(key_path, phase, "non_json"))
487
+ continue
488
+ _scan_raw_value(
489
+ item,
490
+ path=key_path,
491
+ phase=phase,
492
+ diagnostics=diagnostics,
493
+ active_container_ids=active_container_ids,
494
+ )
495
+ finally:
496
+ active_container_ids.remove(container_id)
497
+ return
498
+ if isinstance(value, list | tuple):
499
+ container_id = id(value)
500
+ if container_id in active_container_ids:
501
+ diagnostics.append(
502
+ _hazard_diagnostic(path or "/", phase, "recursive_reference")
503
+ )
504
+ return
505
+ active_container_ids.add(container_id)
506
+ try:
507
+ for index, item in enumerate(value):
508
+ _scan_raw_value(
509
+ item,
510
+ path=_join_pointer(path, str(index)),
511
+ phase=phase,
512
+ diagnostics=diagnostics,
513
+ active_container_ids=active_container_ids,
514
+ )
515
+ finally:
516
+ active_container_ids.remove(container_id)
517
+ return
518
+ if isinstance(value, str):
519
+ hazard = _hazard_kind(value, field_name=path.rsplit("/", 1)[-1])
520
+ if hazard is not None:
521
+ diagnostics.append(_hazard_diagnostic(path or "/", phase, hazard))
522
+ return
523
+ if value is None or isinstance(value, bool | int | float):
524
+ return
525
+ diagnostics.append(_hazard_diagnostic(path or "/", phase, "runtime_object"))
526
+
527
+
528
+ def _hazard_kind(value: str, *, field_name: str) -> str | None:
529
+ stripped = value.strip()
530
+ if detect_secret_candidate(
531
+ field_path=f"/{field_name}",
532
+ field_name=field_name,
533
+ value=stripped,
534
+ policy=RedactionPolicy(),
535
+ ):
536
+ return "secret_material"
537
+ if field_name == "runtime_kind" and stripped != "contract_only":
538
+ if stripped.lower() in _EXECUTABLE_RUNTIME_KINDS:
539
+ return "runtime_kind"
540
+ if _LIVE_URL_RE.search(stripped):
541
+ return "live_endpoint_url"
542
+ if _PARENT_TRAVERSAL_RE.search(stripped):
543
+ return "parent_traversal"
544
+ if _SHELL_COMMAND_RE.search(stripped):
545
+ return "shell_command"
546
+ if _WINDOWS_ABSOLUTE_PATH_RE.search(stripped) or _ABSOLUTE_PATH_RE.search(stripped):
547
+ return "absolute_path"
548
+ if _SCRIPT_BODY_RE.search(stripped):
549
+ return "script_body"
550
+ if _TEMPLATE_INTERPOLATION_RE.search(stripped):
551
+ return "template_interpolation"
552
+ if field_name == "description" and _INSTRUCTION_LIKE_RE.search(stripped):
553
+ return "instruction_like"
554
+ return None
555
+
556
+
557
+ def _hazard_diagnostic(
558
+ path: str, phase: CustomToolDiagnosticPhase, hazard: str
559
+ ) -> CustomToolDiagnostic:
560
+ code = _hazard_code(hazard)
561
+ return custom_tool_diagnostic(
562
+ code,
563
+ phase=phase,
564
+ path=path or "/",
565
+ message="Custom tool source contains unsupported or hazardous material.",
566
+ evidence={"hazard": hazard},
567
+ )
568
+
569
+
570
+ def _hazard_code(hazard: str) -> CustomToolDiagnosticCode:
571
+ if hazard == "secret_material":
572
+ return CustomToolDiagnosticCode.SECRET_MATERIAL
573
+ if hazard == "runtime_kind":
574
+ return CustomToolDiagnosticCode.RUNTIME_KIND_UNSUPPORTED
575
+ if hazard in {
576
+ "absolute_path",
577
+ "live_endpoint_url",
578
+ "parent_traversal",
579
+ "script_body",
580
+ "shell_command",
581
+ "template_interpolation",
582
+ }:
583
+ return CustomToolDiagnosticCode.EXECUTABLE_MATERIAL
584
+ if hazard == "instruction_like":
585
+ return CustomToolDiagnosticCode.DESCRIPTION_UNSAFE
586
+ return CustomToolDiagnosticCode.SOURCE_MALFORMED
587
+
588
+
589
+ def _schema_subset_validation_diagnostic(
590
+ exc: ValidationError,
591
+ *,
592
+ model_name: str,
593
+ phase: CustomToolDiagnosticPhase,
594
+ ) -> CustomToolDiagnostic | None:
595
+ for error in exc.errors():
596
+ ctx = error.get("ctx")
597
+ if not isinstance(ctx, Mapping):
598
+ continue
599
+ schema_error = ctx.get("error")
600
+ if not isinstance(schema_error, SchemaSubsetError):
601
+ continue
602
+ loc = error.get("loc")
603
+ path = _validation_pointer_from_loc(loc)
604
+ code = (
605
+ CustomToolDiagnosticCode.OUTPUT_SCHEMA_UNSUPPORTED
606
+ if _validation_loc_has_field(loc, "output_schema")
607
+ else CustomToolDiagnosticCode.INPUT_SCHEMA_UNSUPPORTED
608
+ )
609
+ return custom_tool_diagnostic(
610
+ code,
611
+ phase=phase,
612
+ path=path,
613
+ message="Custom-tool schema is outside the accepted JSON Schema subset.",
614
+ evidence={
615
+ "model": model_name,
616
+ "error_type": type(schema_error).__name__,
617
+ "schema_error": str(schema_error),
618
+ },
619
+ )
620
+ return None
621
+
622
+
623
+ def _validation_code(exc: ValidationError) -> CustomToolDiagnosticCode:
624
+ errors = exc.errors()
625
+ if not errors:
626
+ return CustomToolDiagnosticCode.SOURCE_INVALID
627
+ first = errors[0]
628
+ loc = tuple(str(part) for part in first.get("loc", ()))
629
+ message = str(first.get("msg", "")).lower()
630
+ error_text = str(errors).lower()
631
+ if "secret material" in error_text:
632
+ return CustomToolDiagnosticCode.SECRET_MATERIAL
633
+ if "runtime_kind" in loc:
634
+ return CustomToolDiagnosticCode.RUNTIME_KIND_UNSUPPORTED
635
+ if "produced_artifact_ids" in loc:
636
+ return CustomToolDiagnosticCode.ARTIFACT_POLICY_INVALID
637
+ if (
638
+ "required_capabilities" in loc
639
+ or "require capabilities" in error_text
640
+ or "requires explicit capabilities" in error_text
641
+ ):
642
+ return CustomToolDiagnosticCode.CAPABILITY_MISSING
643
+ if "forbidden approval policy" in error_text:
644
+ return CustomToolDiagnosticCode.FORBIDDEN_TOOL_COMPILED
645
+ if "approval_policy" in loc or "side-effecting custom tools" in error_text:
646
+ return CustomToolDiagnosticCode.APPROVAL_POLICY_INVALID
647
+ if "timeout_policy" in loc:
648
+ return CustomToolDiagnosticCode.TIMEOUT_POLICY_INVALID
649
+ if "output_policy" in loc:
650
+ return CustomToolDiagnosticCode.OUTPUT_POLICY_INVALID
651
+ if "input schema" in message or "input_schema" in loc:
652
+ return CustomToolDiagnosticCode.INPUT_SCHEMA_UNSUPPORTED
653
+ if "output schema" in message or "output_schema" in loc:
654
+ return CustomToolDiagnosticCode.OUTPUT_SCHEMA_UNSUPPORTED
655
+ if "custom tool identities" in error_text:
656
+ return CustomToolDiagnosticCode.DUPLICATE_TOOL
657
+ if "custom tool model_tool_name" in error_text:
658
+ return CustomToolDiagnosticCode.DUPLICATE_MODEL_TOOL_NAME
659
+ if "custom tool implementation_id" in error_text:
660
+ return CustomToolDiagnosticCode.DUPLICATE_IMPLEMENTATION_ID
661
+ return CustomToolDiagnosticCode.SOURCE_INVALID
662
+
663
+
664
+ def _missing_field(exc: ValidationError) -> str | None:
665
+ errors = exc.errors()
666
+ if not errors or errors[0].get("type") != "missing":
667
+ return None
668
+ loc = errors[0].get("loc")
669
+ if not isinstance(loc, tuple | list) or not loc:
670
+ return None
671
+ return str(loc[-1])
672
+
673
+
674
+ def _validation_pointer(exc: ValidationError) -> str:
675
+ errors = exc.errors()
676
+ if not errors:
677
+ return "/"
678
+ return _validation_pointer_from_loc(errors[0].get("loc"))
679
+
680
+
681
+ def _validation_pointer_from_loc(loc: Any) -> str:
682
+ if not isinstance(loc, tuple | list) or not loc:
683
+ return "/"
684
+ parts = [str(part).replace("~", "~0").replace("/", "~1") for part in loc]
685
+ return "/" + "/".join(parts)
686
+
687
+
688
+ def _validation_loc_has_field(loc: Any, field_name: str) -> bool:
689
+ if not isinstance(loc, tuple | list):
690
+ return False
691
+ return any(str(part) == field_name for part in loc)
692
+
693
+
694
+ def _join_pointer(prefix: str, raw_part: str) -> str:
695
+ part = raw_part.replace("~", "~0").replace("/", "~1")
696
+ return f"{prefix}/{part}" if prefix else f"/{part}"
697
+
698
+
699
+ def _rejected(
700
+ diagnostics: tuple[CustomToolDiagnostic, ...],
701
+ ) -> CustomToolCompilationResult:
702
+ return CustomToolCompilationResult(
703
+ accepted=False,
704
+ diagnostics=tuple(sorted(diagnostics, key=custom_tool_diagnostic_sort_key)),
705
+ )
706
+
707
+
708
+ def _sort_lowered(
709
+ lowered: list[tuple[ToolDescriptor, CustomToolCompilationRecord]],
710
+ ) -> tuple[tuple[ToolDescriptor, CustomToolCompilationRecord], ...]:
711
+ return tuple(
712
+ sorted(
713
+ lowered,
714
+ key=lambda item: (
715
+ item[1].package_id,
716
+ item[0].tool_id,
717
+ item[0].tool_version,
718
+ item[0].model_tool_name,
719
+ item[0].implementation_id,
720
+ item[0].descriptor_sha256,
721
+ item[1].compilation_record_sha256,
722
+ ),
723
+ )
724
+ )