millforge 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- millforge/__init__.py +1174 -0
- millforge/_forge/LICENSE +21 -0
- millforge/_forge/PROVENANCE.json +295 -0
- millforge/_forge/UPDATE_POLICY.md +24 -0
- millforge/_forge/__init__.py +14 -0
- millforge/_forge/adapter.py +2232 -0
- millforge/_forge/base_runner.py +121 -0
- millforge/_forge/clients/__init__.py +10 -0
- millforge/_forge/clients/base.py +200 -0
- millforge/_forge/context/__init__.py +23 -0
- millforge/_forge/context/manager.py +178 -0
- millforge/_forge/context/strategies.py +335 -0
- millforge/_forge/core/__init__.py +16 -0
- millforge/_forge/core/inference.py +433 -0
- millforge/_forge/core/messages.py +119 -0
- millforge/_forge/core/runner.py +479 -0
- millforge/_forge/core/steps.py +108 -0
- millforge/_forge/core/workflow.py +400 -0
- millforge/_forge/errors.py +222 -0
- millforge/_forge/guardrails/__init__.py +21 -0
- millforge/_forge/guardrails/error_tracker.py +71 -0
- millforge/_forge/guardrails/guardrails.py +194 -0
- millforge/_forge/guardrails/nudge.py +47 -0
- millforge/_forge/guardrails/response_validator.py +119 -0
- millforge/_forge/guardrails/step_enforcer.py +183 -0
- millforge/_forge/prompts/__init__.py +16 -0
- millforge/_forge/prompts/nudges.py +95 -0
- millforge/_forge/prompts/templates.py +285 -0
- millforge/_version.py +3 -0
- millforge/artifacts.py +570 -0
- millforge/base/__init__.py +97 -0
- millforge/base/composition.py +402 -0
- millforge/base/context.py +285 -0
- millforge/base/harness.py +138 -0
- millforge/base/identity.py +465 -0
- millforge/base/options.py +34 -0
- millforge/base/platform.py +17 -0
- millforge/base/prompt.py +317 -0
- millforge/base/runner.py +546 -0
- millforge/compiled_plan.py +970 -0
- millforge/compiler/__init__.py +231 -0
- millforge/compiler/artifact_validation.py +257 -0
- millforge/compiler/canonicalization.py +169 -0
- millforge/compiler/capabilities.py +66 -0
- millforge/compiler/catalogs.py +500 -0
- millforge/compiler/diagnostics.py +491 -0
- millforge/compiler/graph.py +678 -0
- millforge/compiler/lowering.py +198 -0
- millforge/compiler/output.py +692 -0
- millforge/compiler/parsing.py +1424 -0
- millforge/compiler/requests.py +1180 -0
- millforge/compiler/schema_validation.py +272 -0
- millforge/compiler/semantic.py +490 -0
- millforge/compiler/service.py +448 -0
- millforge/compiler/source.py +375 -0
- millforge/compiler/validators.py +184 -0
- millforge/connectors/__init__.py +95 -0
- millforge/connectors/admission.py +801 -0
- millforge/connectors/broker.py +202 -0
- millforge/connectors/contracts.py +1159 -0
- millforge/connectors/diagnostics.py +189 -0
- millforge/connectors/fake.py +66 -0
- millforge/connectors/runtime.py +236 -0
- millforge/contracts.py +2860 -0
- millforge/custom_tools/__init__.py +67 -0
- millforge/custom_tools/compiler.py +724 -0
- millforge/custom_tools/contracts.py +1093 -0
- millforge/custom_tools/diagnostics.py +205 -0
- millforge/eval_artifacts.py +952 -0
- millforge/eval_boundary.py +2435 -0
- millforge/eval_fixtures/__init__.py +1 -0
- millforge/eval_fixtures/default_pack/__init__.py +1 -0
- millforge/eval_fixtures/default_pack/fixtures/fixture.08a.bug_diagnosis.traceback.v1.json +52 -0
- millforge/eval_fixtures/default_pack/fixtures/fixture.08a.direct_edit.import_sort.v1.json +52 -0
- millforge/eval_fixtures/default_pack/fixtures/fixture.08a.evidence_discipline.no_source_change.v1.json +51 -0
- millforge/eval_fixtures/default_pack/fixtures/fixture.08a.false_closure.visible_green.v1.json +52 -0
- millforge/eval_fixtures/default_pack/fixtures/fixture.08a.multi_file.api_contract.v1.json +54 -0
- millforge/eval_fixtures/default_pack/fixtures/fixture.08a.recovery.malformed_artifact.v1.json +54 -0
- millforge/eval_fixtures/default_pack/manifest.json +12 -0
- millforge/eval_modes.py +1282 -0
- millforge/eval_presets.py +1398 -0
- millforge/eval_reports.py +2517 -0
- millforge/eval_suite.py +2429 -0
- millforge/eval_trials.py +2632 -0
- millforge/eval_workflow.py +794 -0
- millforge/exceptions.py +122 -0
- millforge/model_backend.py +2098 -0
- millforge/protocols.py +340 -0
- millforge/py.typed +0 -0
- millforge/runtime.py +1791 -0
- millforge/testing/__init__.py +1089 -0
- millforge/tools/__init__.py +83 -0
- millforge/tools/builtin_runtime.py +1339 -0
- millforge/tools/builtins.py +773 -0
- millforge/tools/execution.py +1545 -0
- millforge/tools/path_policy.py +155 -0
- millforge/tools/pi_compat/PI_LICENSE +21 -0
- millforge/tools/pi_compat/PROVENANCE.json +55 -0
- millforge/tools/pi_compat/UPDATE_POLICY.md +36 -0
- millforge/tools/pi_compat/__init__.py +34 -0
- millforge/tools/pi_compat/contracts.py +49 -0
- millforge/tools/pi_compat/editing.py +390 -0
- millforge/tools/pi_compat/mutations.py +57 -0
- millforge/tools/pi_compat/operations.py +401 -0
- millforge/tools/pi_compat/paths.py +155 -0
- millforge/tools/pi_compat/process.py +1375 -0
- millforge/tools/pi_compat/search.py +738 -0
- millforge/tools/pi_compat/truncation.py +267 -0
- millforge/tools/pi_compat_catalog.py +396 -0
- millforge/tools/pi_compat_runtime.py +460 -0
- millforge/tools/registry.py +553 -0
- millforge/tools/results.py +533 -0
- millforge-0.1.0.dist-info/METADATA +844 -0
- millforge-0.1.0.dist-info/RECORD +116 -0
- millforge-0.1.0.dist-info/WHEEL +4 -0
- millforge-0.1.0.dist-info/licenses/LICENSE +201 -0
|
@@ -0,0 +1,724 @@
|
|
|
1
|
+
"""Deterministic offline compilation for custom-tool source manifests."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import re
|
|
6
|
+
from collections.abc import Mapping
|
|
7
|
+
from typing import Any, TypeVar
|
|
8
|
+
|
|
9
|
+
from pydantic import BaseModel, ValidationError
|
|
10
|
+
|
|
11
|
+
from millforge.compiler.diagnostics import detect_secret_candidate
|
|
12
|
+
from millforge.compiler.schema_validation import (
|
|
13
|
+
SchemaSubsetError,
|
|
14
|
+
normalized_schema_bytes,
|
|
15
|
+
)
|
|
16
|
+
from millforge.contracts import RedactionPolicy
|
|
17
|
+
from millforge.custom_tools.contracts import (
|
|
18
|
+
CustomToolApprovalPolicy,
|
|
19
|
+
CustomToolCompilationRecord,
|
|
20
|
+
CustomToolCompilationResult,
|
|
21
|
+
CustomToolCompilerPolicy,
|
|
22
|
+
CustomToolDeclaration,
|
|
23
|
+
CustomToolSourceManifest,
|
|
24
|
+
compilation_record_from_declaration,
|
|
25
|
+
tool_descriptor_from_declaration,
|
|
26
|
+
)
|
|
27
|
+
from millforge.custom_tools.diagnostics import (
|
|
28
|
+
CustomToolDiagnostic,
|
|
29
|
+
CustomToolDiagnosticCode,
|
|
30
|
+
CustomToolDiagnosticPhase,
|
|
31
|
+
custom_tool_diagnostic,
|
|
32
|
+
custom_tool_diagnostic_sort_key,
|
|
33
|
+
malformed_input_diagnostic,
|
|
34
|
+
)
|
|
35
|
+
from millforge.tools.registry import ToolDescriptor
|
|
36
|
+
|
|
37
|
+
_T = TypeVar("_T", bound=BaseModel)
|
|
38
|
+
|
|
39
|
+
_LIVE_URL_RE = re.compile(r"\bhttps?://[^\s<>'\"]+", re.IGNORECASE)
|
|
40
|
+
_ABSOLUTE_PATH_RE = re.compile(r"(?<![A-Za-z0-9_.-])(?:/[A-Za-z0-9_.-][^\s]*)")
|
|
41
|
+
_WINDOWS_ABSOLUTE_PATH_RE = re.compile(r"\b[A-Za-z]:\\[^\s]+")
|
|
42
|
+
_PARENT_TRAVERSAL_RE = re.compile(r"(^|[\s\\/])\.\.([\\/]|$)")
|
|
43
|
+
_SHELL_COMMAND_RE = re.compile(
|
|
44
|
+
r"(?i)(^|\s)(?:rm\s+-rf|curl\s+|wget\s+|bash\s+-c|sh\s+-c|"
|
|
45
|
+
r"python(?:3)?\s+-c|node\s+-e|powershell\b|cmd\.exe\b|chmod\s+\+x)\b"
|
|
46
|
+
)
|
|
47
|
+
_SCRIPT_BODY_RE = re.compile(
|
|
48
|
+
r"(?is)(^#!|<script\b|function\s+\w*\s*\(|def\s+\w+\s*\(|"
|
|
49
|
+
r"import\s+os\b|subprocess\.|eval\s*\(|exec\s*\()"
|
|
50
|
+
)
|
|
51
|
+
_TEMPLATE_INTERPOLATION_RE = re.compile(r"({{.*?}}|{%.+?%}|\$\{[^}]+})")
|
|
52
|
+
_INSTRUCTION_LIKE_RE = re.compile(
|
|
53
|
+
r"\b(ignore|override|system prompt|developer message|previous instructions|"
|
|
54
|
+
r"follow these instructions|you must|do not tell)\b",
|
|
55
|
+
re.IGNORECASE,
|
|
56
|
+
)
|
|
57
|
+
_EXECUTABLE_RUNTIME_KINDS = frozenset(
|
|
58
|
+
{
|
|
59
|
+
"shell",
|
|
60
|
+
"process",
|
|
61
|
+
"python",
|
|
62
|
+
"javascript",
|
|
63
|
+
"js",
|
|
64
|
+
"node",
|
|
65
|
+
"wasm",
|
|
66
|
+
"http",
|
|
67
|
+
"https",
|
|
68
|
+
"mcp",
|
|
69
|
+
"connector",
|
|
70
|
+
"connector_alias",
|
|
71
|
+
"filesystem",
|
|
72
|
+
"fs",
|
|
73
|
+
"terminal",
|
|
74
|
+
}
|
|
75
|
+
)
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
def compile_custom_tools(
|
|
79
|
+
source: CustomToolSourceManifest | Mapping[str, Any],
|
|
80
|
+
policy: CustomToolCompilerPolicy | Mapping[str, Any],
|
|
81
|
+
) -> CustomToolCompilationResult:
|
|
82
|
+
"""Validate and lower contract-only custom tools into descriptors.
|
|
83
|
+
|
|
84
|
+
Raw mappings are treated as untrusted source. Validation and lowering errors
|
|
85
|
+
are returned as stable custom-tool diagnostics, and any diagnostic rejects
|
|
86
|
+
the whole manifest with no partial descriptors or records.
|
|
87
|
+
"""
|
|
88
|
+
source_hazards = _raw_source_hazards(source, phase=CustomToolDiagnosticPhase.SOURCE)
|
|
89
|
+
policy_hazards = _raw_source_hazards(policy, phase=CustomToolDiagnosticPhase.POLICY)
|
|
90
|
+
if source_hazards or policy_hazards:
|
|
91
|
+
return _rejected((*source_hazards, *policy_hazards))
|
|
92
|
+
|
|
93
|
+
valid_source, source_diagnostic = _validate_contract(
|
|
94
|
+
CustomToolSourceManifest,
|
|
95
|
+
source,
|
|
96
|
+
phase=CustomToolDiagnosticPhase.SOURCE,
|
|
97
|
+
)
|
|
98
|
+
valid_policy, policy_diagnostic = _validate_contract(
|
|
99
|
+
CustomToolCompilerPolicy,
|
|
100
|
+
policy,
|
|
101
|
+
phase=CustomToolDiagnosticPhase.POLICY,
|
|
102
|
+
)
|
|
103
|
+
diagnostics = tuple(
|
|
104
|
+
diagnostic
|
|
105
|
+
for diagnostic in (source_diagnostic, policy_diagnostic)
|
|
106
|
+
if diagnostic
|
|
107
|
+
)
|
|
108
|
+
if diagnostics:
|
|
109
|
+
return _rejected(diagnostics)
|
|
110
|
+
if not isinstance(valid_source, CustomToolSourceManifest):
|
|
111
|
+
raise AssertionError("source validation returned unexpected contract")
|
|
112
|
+
if not isinstance(valid_policy, CustomToolCompilerPolicy):
|
|
113
|
+
raise AssertionError("policy validation returned unexpected contract")
|
|
114
|
+
|
|
115
|
+
compiler = _CustomToolCompilation(valid_source, valid_policy)
|
|
116
|
+
return compiler.run()
|
|
117
|
+
|
|
118
|
+
|
|
119
|
+
class _CustomToolCompilation:
|
|
120
|
+
def __init__(
|
|
121
|
+
self,
|
|
122
|
+
source: CustomToolSourceManifest,
|
|
123
|
+
policy: CustomToolCompilerPolicy,
|
|
124
|
+
) -> None:
|
|
125
|
+
self.source = source
|
|
126
|
+
self.policy = policy
|
|
127
|
+
self.diagnostics: list[CustomToolDiagnostic] = []
|
|
128
|
+
|
|
129
|
+
def run(self) -> CustomToolCompilationResult:
|
|
130
|
+
self._validate_source_policy()
|
|
131
|
+
|
|
132
|
+
lowered: list[tuple[ToolDescriptor, CustomToolCompilationRecord]] = []
|
|
133
|
+
for index, declaration in enumerate(self.source.tools):
|
|
134
|
+
compiled = self._lower_declaration(declaration, index=index)
|
|
135
|
+
if compiled is not None:
|
|
136
|
+
lowered.append(compiled)
|
|
137
|
+
|
|
138
|
+
if self.diagnostics:
|
|
139
|
+
return self._rejected()
|
|
140
|
+
|
|
141
|
+
return CustomToolCompilationResult(
|
|
142
|
+
accepted=True,
|
|
143
|
+
source_sha256=self.source.source_sha256,
|
|
144
|
+
descriptors=tuple(descriptor for descriptor, _ in _sort_lowered(lowered)),
|
|
145
|
+
records=tuple(record for _, record in _sort_lowered(lowered)),
|
|
146
|
+
)
|
|
147
|
+
|
|
148
|
+
def _validate_source_policy(self) -> None:
|
|
149
|
+
if len(self.source.tools) > self.policy.max_tools:
|
|
150
|
+
self._diagnose(
|
|
151
|
+
CustomToolDiagnosticCode.LIMIT_EXCEEDED,
|
|
152
|
+
phase=CustomToolDiagnosticPhase.POLICY,
|
|
153
|
+
path="/tools",
|
|
154
|
+
message="Custom-tool source exceeds the compiler policy tool limit.",
|
|
155
|
+
evidence={"limit": self.policy.max_tools},
|
|
156
|
+
)
|
|
157
|
+
self._check_hash(
|
|
158
|
+
supplied=self.source.expected_source_sha256,
|
|
159
|
+
actual=self.source.source_sha256,
|
|
160
|
+
path="/expected_source_sha256",
|
|
161
|
+
label="source_sha256",
|
|
162
|
+
)
|
|
163
|
+
if (
|
|
164
|
+
self.policy.require_expected_hashes
|
|
165
|
+
and self.source.expected_source_sha256 is None
|
|
166
|
+
):
|
|
167
|
+
self._missing_hash("/expected_source_sha256", "source_sha256")
|
|
168
|
+
|
|
169
|
+
produced_artifacts: dict[str, str] = {}
|
|
170
|
+
for declaration in self.source.tools:
|
|
171
|
+
for artifact_id in declaration.produced_artifact_ids:
|
|
172
|
+
prior = produced_artifacts.get(artifact_id)
|
|
173
|
+
if prior is not None:
|
|
174
|
+
self._diagnose(
|
|
175
|
+
CustomToolDiagnosticCode.ARTIFACT_POLICY_INVALID,
|
|
176
|
+
phase=CustomToolDiagnosticPhase.SOURCE,
|
|
177
|
+
path="/tools",
|
|
178
|
+
message="Custom-tool source contains duplicate produced artifacts.",
|
|
179
|
+
evidence={"artifact_id": artifact_id, "first_tool_id": prior},
|
|
180
|
+
)
|
|
181
|
+
else:
|
|
182
|
+
produced_artifacts[artifact_id] = declaration.tool_id
|
|
183
|
+
|
|
184
|
+
def _lower_declaration(
|
|
185
|
+
self,
|
|
186
|
+
declaration: CustomToolDeclaration,
|
|
187
|
+
*,
|
|
188
|
+
index: int,
|
|
189
|
+
) -> tuple[ToolDescriptor, CustomToolCompilationRecord] | None:
|
|
190
|
+
path = f"/tools/{index}"
|
|
191
|
+
if declaration.runtime_kind not in self.policy.allowed_runtime_kinds:
|
|
192
|
+
self._diagnose(
|
|
193
|
+
CustomToolDiagnosticCode.RUNTIME_KIND_UNSUPPORTED,
|
|
194
|
+
phase=CustomToolDiagnosticPhase.POLICY,
|
|
195
|
+
path=f"{path}/runtime_kind",
|
|
196
|
+
message="Custom-tool runtime kind is not allowed by compiler policy.",
|
|
197
|
+
evidence={"runtime_kind": declaration.runtime_kind.value},
|
|
198
|
+
)
|
|
199
|
+
if (
|
|
200
|
+
len(declaration.description.encode("utf-8"))
|
|
201
|
+
> self.policy.max_description_utf8
|
|
202
|
+
):
|
|
203
|
+
self._diagnose(
|
|
204
|
+
CustomToolDiagnosticCode.LIMIT_EXCEEDED,
|
|
205
|
+
phase=CustomToolDiagnosticPhase.POLICY,
|
|
206
|
+
path=f"{path}/description",
|
|
207
|
+
message="Custom-tool description exceeds the compiler policy limit.",
|
|
208
|
+
evidence={"tool_id": declaration.tool_id},
|
|
209
|
+
)
|
|
210
|
+
self._validate_schema_bytes(
|
|
211
|
+
declaration.input_schema,
|
|
212
|
+
path=f"{path}/input_schema",
|
|
213
|
+
code=CustomToolDiagnosticCode.INPUT_SCHEMA_UNSUPPORTED,
|
|
214
|
+
)
|
|
215
|
+
self._validate_schema_bytes(
|
|
216
|
+
declaration.output_schema,
|
|
217
|
+
path=f"{path}/output_schema",
|
|
218
|
+
code=CustomToolDiagnosticCode.OUTPUT_SCHEMA_UNSUPPORTED,
|
|
219
|
+
)
|
|
220
|
+
self._validate_capabilities(declaration, path=path)
|
|
221
|
+
self._validate_approval(declaration, path=path)
|
|
222
|
+
self._check_hash(
|
|
223
|
+
supplied=declaration.expected_declaration_sha256,
|
|
224
|
+
actual=declaration.declaration_sha256,
|
|
225
|
+
path=f"{path}/expected_declaration_sha256",
|
|
226
|
+
label="declaration_sha256",
|
|
227
|
+
)
|
|
228
|
+
if (
|
|
229
|
+
self.policy.require_expected_hashes
|
|
230
|
+
and declaration.expected_declaration_sha256 is None
|
|
231
|
+
):
|
|
232
|
+
self._missing_hash(
|
|
233
|
+
f"{path}/expected_declaration_sha256", "declaration_sha256"
|
|
234
|
+
)
|
|
235
|
+
try:
|
|
236
|
+
descriptor = tool_descriptor_from_declaration(declaration)
|
|
237
|
+
self._check_hash(
|
|
238
|
+
supplied=declaration.expected_descriptor_sha256,
|
|
239
|
+
actual=descriptor.descriptor_sha256,
|
|
240
|
+
path=f"{path}/expected_descriptor_sha256",
|
|
241
|
+
label="descriptor_sha256",
|
|
242
|
+
)
|
|
243
|
+
if (
|
|
244
|
+
self.policy.require_expected_hashes
|
|
245
|
+
and declaration.expected_descriptor_sha256 is None
|
|
246
|
+
):
|
|
247
|
+
self._missing_hash(
|
|
248
|
+
f"{path}/expected_descriptor_sha256", "descriptor_sha256"
|
|
249
|
+
)
|
|
250
|
+
record = compilation_record_from_declaration(
|
|
251
|
+
self.source, declaration, descriptor
|
|
252
|
+
)
|
|
253
|
+
except Exception as exc:
|
|
254
|
+
self._diagnose(
|
|
255
|
+
CustomToolDiagnosticCode.DECLARATION_INVALID,
|
|
256
|
+
phase=CustomToolDiagnosticPhase.COMPILATION,
|
|
257
|
+
path=path,
|
|
258
|
+
message="Custom-tool descriptor construction failed.",
|
|
259
|
+
evidence={"error_type": type(exc).__name__},
|
|
260
|
+
)
|
|
261
|
+
return None
|
|
262
|
+
|
|
263
|
+
self._check_hash(
|
|
264
|
+
supplied=declaration.expected_compilation_record_sha256,
|
|
265
|
+
actual=record.compilation_record_sha256,
|
|
266
|
+
path=f"{path}/expected_compilation_record_sha256",
|
|
267
|
+
label="compilation_record_sha256",
|
|
268
|
+
)
|
|
269
|
+
if (
|
|
270
|
+
self.policy.require_expected_hashes
|
|
271
|
+
and declaration.expected_compilation_record_sha256 is None
|
|
272
|
+
):
|
|
273
|
+
self._missing_hash(
|
|
274
|
+
f"{path}/expected_compilation_record_sha256",
|
|
275
|
+
"compilation_record_sha256",
|
|
276
|
+
)
|
|
277
|
+
if self.diagnostics:
|
|
278
|
+
return None
|
|
279
|
+
return descriptor, record
|
|
280
|
+
|
|
281
|
+
def _validate_schema_bytes(
|
|
282
|
+
self,
|
|
283
|
+
schema: Mapping[str, Any],
|
|
284
|
+
*,
|
|
285
|
+
path: str,
|
|
286
|
+
code: CustomToolDiagnosticCode,
|
|
287
|
+
) -> None:
|
|
288
|
+
size = len(normalized_schema_bytes(schema))
|
|
289
|
+
if size <= self.policy.max_schema_bytes:
|
|
290
|
+
return
|
|
291
|
+
self._diagnose(
|
|
292
|
+
code,
|
|
293
|
+
phase=CustomToolDiagnosticPhase.POLICY,
|
|
294
|
+
path=path,
|
|
295
|
+
message="Custom-tool schema exceeds the compiler policy byte limit.",
|
|
296
|
+
evidence={"limit": self.policy.max_schema_bytes, "size": size},
|
|
297
|
+
)
|
|
298
|
+
|
|
299
|
+
def _validate_capabilities(
|
|
300
|
+
self, declaration: CustomToolDeclaration, *, path: str
|
|
301
|
+
) -> None:
|
|
302
|
+
if (
|
|
303
|
+
declaration.side_effect_class.value != "read_only"
|
|
304
|
+
or declaration.produced_artifact_ids
|
|
305
|
+
) and not declaration.required_capabilities:
|
|
306
|
+
self._diagnose(
|
|
307
|
+
CustomToolDiagnosticCode.CAPABILITY_MISSING,
|
|
308
|
+
phase=CustomToolDiagnosticPhase.POLICY,
|
|
309
|
+
path=f"{path}/required_capabilities",
|
|
310
|
+
message="Custom tool requires explicit capabilities.",
|
|
311
|
+
evidence={"tool_id": declaration.tool_id},
|
|
312
|
+
)
|
|
313
|
+
return
|
|
314
|
+
allowed = set(self.policy.allowed_capability_ids)
|
|
315
|
+
for capability_id in declaration.required_capabilities:
|
|
316
|
+
if capability_id not in allowed:
|
|
317
|
+
self._diagnose(
|
|
318
|
+
CustomToolDiagnosticCode.CAPABILITY_UNKNOWN,
|
|
319
|
+
phase=CustomToolDiagnosticPhase.POLICY,
|
|
320
|
+
path=f"{path}/required_capabilities",
|
|
321
|
+
message="Custom-tool capability is not allowed by compiler policy.",
|
|
322
|
+
evidence={"capability_id": capability_id},
|
|
323
|
+
)
|
|
324
|
+
|
|
325
|
+
def _validate_approval(
|
|
326
|
+
self, declaration: CustomToolDeclaration, *, path: str
|
|
327
|
+
) -> None:
|
|
328
|
+
allowed = self.policy.side_effect_approval_matrix.get(
|
|
329
|
+
declaration.side_effect_class
|
|
330
|
+
)
|
|
331
|
+
if allowed is None or declaration.approval_policy not in allowed:
|
|
332
|
+
self._diagnose(
|
|
333
|
+
CustomToolDiagnosticCode.APPROVAL_POLICY_INVALID,
|
|
334
|
+
phase=CustomToolDiagnosticPhase.POLICY,
|
|
335
|
+
path=f"{path}/approval_policy",
|
|
336
|
+
message="Approval policy is not allowed for side-effect class.",
|
|
337
|
+
evidence={
|
|
338
|
+
"approval_policy": declaration.approval_policy.value,
|
|
339
|
+
"side_effect_class": declaration.side_effect_class.value,
|
|
340
|
+
},
|
|
341
|
+
)
|
|
342
|
+
if (
|
|
343
|
+
declaration.produced_artifact_ids
|
|
344
|
+
and declaration.approval_policy is CustomToolApprovalPolicy.NONE
|
|
345
|
+
):
|
|
346
|
+
self._diagnose(
|
|
347
|
+
CustomToolDiagnosticCode.APPROVAL_POLICY_INVALID,
|
|
348
|
+
phase=CustomToolDiagnosticPhase.POLICY,
|
|
349
|
+
path=f"{path}/approval_policy",
|
|
350
|
+
message="Artifact-producing custom tools require explicit approval.",
|
|
351
|
+
evidence={
|
|
352
|
+
"approval_policy": declaration.approval_policy.value,
|
|
353
|
+
"tool_id": declaration.tool_id,
|
|
354
|
+
},
|
|
355
|
+
)
|
|
356
|
+
|
|
357
|
+
def _check_hash(
|
|
358
|
+
self,
|
|
359
|
+
*,
|
|
360
|
+
supplied: str | None,
|
|
361
|
+
actual: str,
|
|
362
|
+
path: str,
|
|
363
|
+
label: str,
|
|
364
|
+
) -> None:
|
|
365
|
+
if supplied is not None and supplied != actual:
|
|
366
|
+
self._diagnose(
|
|
367
|
+
CustomToolDiagnosticCode.HASH_MISMATCH,
|
|
368
|
+
phase=CustomToolDiagnosticPhase.COMPILATION,
|
|
369
|
+
path=path,
|
|
370
|
+
message="Supplied custom-tool hash does not match recomputed hash.",
|
|
371
|
+
evidence={"hash": label},
|
|
372
|
+
)
|
|
373
|
+
|
|
374
|
+
def _missing_hash(self, path: str, label: str) -> None:
|
|
375
|
+
self._diagnose(
|
|
376
|
+
CustomToolDiagnosticCode.HASH_MISMATCH,
|
|
377
|
+
phase=CustomToolDiagnosticPhase.POLICY,
|
|
378
|
+
path=path,
|
|
379
|
+
message="Compiler policy requires expected custom-tool hashes.",
|
|
380
|
+
evidence={"hash": label},
|
|
381
|
+
)
|
|
382
|
+
|
|
383
|
+
def _diagnose(
|
|
384
|
+
self,
|
|
385
|
+
code: CustomToolDiagnosticCode,
|
|
386
|
+
*,
|
|
387
|
+
phase: CustomToolDiagnosticPhase,
|
|
388
|
+
message: str,
|
|
389
|
+
location: str | None = None,
|
|
390
|
+
path: str | None = None,
|
|
391
|
+
evidence: Mapping[str, Any] | None = None,
|
|
392
|
+
) -> None:
|
|
393
|
+
self.diagnostics.append(
|
|
394
|
+
custom_tool_diagnostic(
|
|
395
|
+
code,
|
|
396
|
+
phase=phase,
|
|
397
|
+
message=message,
|
|
398
|
+
location=location,
|
|
399
|
+
path=path,
|
|
400
|
+
evidence=evidence,
|
|
401
|
+
)
|
|
402
|
+
)
|
|
403
|
+
|
|
404
|
+
def _rejected(self) -> CustomToolCompilationResult:
|
|
405
|
+
return _rejected(tuple(self.diagnostics))
|
|
406
|
+
|
|
407
|
+
|
|
408
|
+
def _validate_contract(
|
|
409
|
+
model: type[_T],
|
|
410
|
+
value: _T | Mapping[str, Any],
|
|
411
|
+
*,
|
|
412
|
+
phase: CustomToolDiagnosticPhase,
|
|
413
|
+
) -> tuple[_T | None, CustomToolDiagnostic | None]:
|
|
414
|
+
if isinstance(value, model):
|
|
415
|
+
return value, None
|
|
416
|
+
try:
|
|
417
|
+
return model.model_validate(value), None
|
|
418
|
+
except ValidationError as exc:
|
|
419
|
+
schema_diagnostic = _schema_subset_validation_diagnostic(
|
|
420
|
+
exc,
|
|
421
|
+
model_name=model.__name__,
|
|
422
|
+
phase=phase,
|
|
423
|
+
)
|
|
424
|
+
if schema_diagnostic is not None:
|
|
425
|
+
return None, schema_diagnostic
|
|
426
|
+
return (
|
|
427
|
+
None,
|
|
428
|
+
malformed_input_diagnostic(
|
|
429
|
+
phase=phase,
|
|
430
|
+
model_name=model.__name__,
|
|
431
|
+
path=_validation_pointer(exc),
|
|
432
|
+
missing_field=_missing_field(exc),
|
|
433
|
+
code=_validation_code(exc),
|
|
434
|
+
),
|
|
435
|
+
)
|
|
436
|
+
except Exception:
|
|
437
|
+
return (
|
|
438
|
+
None,
|
|
439
|
+
malformed_input_diagnostic(
|
|
440
|
+
phase=phase,
|
|
441
|
+
model_name=model.__name__,
|
|
442
|
+
code=CustomToolDiagnosticCode.SOURCE_INVALID,
|
|
443
|
+
),
|
|
444
|
+
)
|
|
445
|
+
|
|
446
|
+
|
|
447
|
+
def _raw_source_hazards(
|
|
448
|
+
value: Any, *, phase: CustomToolDiagnosticPhase
|
|
449
|
+
) -> tuple[CustomToolDiagnostic, ...]:
|
|
450
|
+
if isinstance(value, BaseModel):
|
|
451
|
+
try:
|
|
452
|
+
value = value.model_dump(mode="json")
|
|
453
|
+
except Exception:
|
|
454
|
+
return (_hazard_diagnostic("/", phase, "runtime_object"),)
|
|
455
|
+
diagnostics: list[CustomToolDiagnostic] = []
|
|
456
|
+
_scan_raw_value(
|
|
457
|
+
value,
|
|
458
|
+
path="",
|
|
459
|
+
phase=phase,
|
|
460
|
+
diagnostics=diagnostics,
|
|
461
|
+
active_container_ids=set(),
|
|
462
|
+
)
|
|
463
|
+
return tuple(diagnostics)
|
|
464
|
+
|
|
465
|
+
|
|
466
|
+
def _scan_raw_value(
|
|
467
|
+
value: Any,
|
|
468
|
+
*,
|
|
469
|
+
path: str,
|
|
470
|
+
phase: CustomToolDiagnosticPhase,
|
|
471
|
+
diagnostics: list[CustomToolDiagnostic],
|
|
472
|
+
active_container_ids: set[int],
|
|
473
|
+
) -> None:
|
|
474
|
+
if isinstance(value, Mapping):
|
|
475
|
+
container_id = id(value)
|
|
476
|
+
if container_id in active_container_ids:
|
|
477
|
+
diagnostics.append(
|
|
478
|
+
_hazard_diagnostic(path or "/", phase, "recursive_reference")
|
|
479
|
+
)
|
|
480
|
+
return
|
|
481
|
+
active_container_ids.add(container_id)
|
|
482
|
+
try:
|
|
483
|
+
for key, item in value.items():
|
|
484
|
+
key_path = _join_pointer(path, str(key))
|
|
485
|
+
if not isinstance(key, str):
|
|
486
|
+
diagnostics.append(_hazard_diagnostic(key_path, phase, "non_json"))
|
|
487
|
+
continue
|
|
488
|
+
_scan_raw_value(
|
|
489
|
+
item,
|
|
490
|
+
path=key_path,
|
|
491
|
+
phase=phase,
|
|
492
|
+
diagnostics=diagnostics,
|
|
493
|
+
active_container_ids=active_container_ids,
|
|
494
|
+
)
|
|
495
|
+
finally:
|
|
496
|
+
active_container_ids.remove(container_id)
|
|
497
|
+
return
|
|
498
|
+
if isinstance(value, list | tuple):
|
|
499
|
+
container_id = id(value)
|
|
500
|
+
if container_id in active_container_ids:
|
|
501
|
+
diagnostics.append(
|
|
502
|
+
_hazard_diagnostic(path or "/", phase, "recursive_reference")
|
|
503
|
+
)
|
|
504
|
+
return
|
|
505
|
+
active_container_ids.add(container_id)
|
|
506
|
+
try:
|
|
507
|
+
for index, item in enumerate(value):
|
|
508
|
+
_scan_raw_value(
|
|
509
|
+
item,
|
|
510
|
+
path=_join_pointer(path, str(index)),
|
|
511
|
+
phase=phase,
|
|
512
|
+
diagnostics=diagnostics,
|
|
513
|
+
active_container_ids=active_container_ids,
|
|
514
|
+
)
|
|
515
|
+
finally:
|
|
516
|
+
active_container_ids.remove(container_id)
|
|
517
|
+
return
|
|
518
|
+
if isinstance(value, str):
|
|
519
|
+
hazard = _hazard_kind(value, field_name=path.rsplit("/", 1)[-1])
|
|
520
|
+
if hazard is not None:
|
|
521
|
+
diagnostics.append(_hazard_diagnostic(path or "/", phase, hazard))
|
|
522
|
+
return
|
|
523
|
+
if value is None or isinstance(value, bool | int | float):
|
|
524
|
+
return
|
|
525
|
+
diagnostics.append(_hazard_diagnostic(path or "/", phase, "runtime_object"))
|
|
526
|
+
|
|
527
|
+
|
|
528
|
+
def _hazard_kind(value: str, *, field_name: str) -> str | None:
|
|
529
|
+
stripped = value.strip()
|
|
530
|
+
if detect_secret_candidate(
|
|
531
|
+
field_path=f"/{field_name}",
|
|
532
|
+
field_name=field_name,
|
|
533
|
+
value=stripped,
|
|
534
|
+
policy=RedactionPolicy(),
|
|
535
|
+
):
|
|
536
|
+
return "secret_material"
|
|
537
|
+
if field_name == "runtime_kind" and stripped != "contract_only":
|
|
538
|
+
if stripped.lower() in _EXECUTABLE_RUNTIME_KINDS:
|
|
539
|
+
return "runtime_kind"
|
|
540
|
+
if _LIVE_URL_RE.search(stripped):
|
|
541
|
+
return "live_endpoint_url"
|
|
542
|
+
if _PARENT_TRAVERSAL_RE.search(stripped):
|
|
543
|
+
return "parent_traversal"
|
|
544
|
+
if _SHELL_COMMAND_RE.search(stripped):
|
|
545
|
+
return "shell_command"
|
|
546
|
+
if _WINDOWS_ABSOLUTE_PATH_RE.search(stripped) or _ABSOLUTE_PATH_RE.search(stripped):
|
|
547
|
+
return "absolute_path"
|
|
548
|
+
if _SCRIPT_BODY_RE.search(stripped):
|
|
549
|
+
return "script_body"
|
|
550
|
+
if _TEMPLATE_INTERPOLATION_RE.search(stripped):
|
|
551
|
+
return "template_interpolation"
|
|
552
|
+
if field_name == "description" and _INSTRUCTION_LIKE_RE.search(stripped):
|
|
553
|
+
return "instruction_like"
|
|
554
|
+
return None
|
|
555
|
+
|
|
556
|
+
|
|
557
|
+
def _hazard_diagnostic(
|
|
558
|
+
path: str, phase: CustomToolDiagnosticPhase, hazard: str
|
|
559
|
+
) -> CustomToolDiagnostic:
|
|
560
|
+
code = _hazard_code(hazard)
|
|
561
|
+
return custom_tool_diagnostic(
|
|
562
|
+
code,
|
|
563
|
+
phase=phase,
|
|
564
|
+
path=path or "/",
|
|
565
|
+
message="Custom tool source contains unsupported or hazardous material.",
|
|
566
|
+
evidence={"hazard": hazard},
|
|
567
|
+
)
|
|
568
|
+
|
|
569
|
+
|
|
570
|
+
def _hazard_code(hazard: str) -> CustomToolDiagnosticCode:
|
|
571
|
+
if hazard == "secret_material":
|
|
572
|
+
return CustomToolDiagnosticCode.SECRET_MATERIAL
|
|
573
|
+
if hazard == "runtime_kind":
|
|
574
|
+
return CustomToolDiagnosticCode.RUNTIME_KIND_UNSUPPORTED
|
|
575
|
+
if hazard in {
|
|
576
|
+
"absolute_path",
|
|
577
|
+
"live_endpoint_url",
|
|
578
|
+
"parent_traversal",
|
|
579
|
+
"script_body",
|
|
580
|
+
"shell_command",
|
|
581
|
+
"template_interpolation",
|
|
582
|
+
}:
|
|
583
|
+
return CustomToolDiagnosticCode.EXECUTABLE_MATERIAL
|
|
584
|
+
if hazard == "instruction_like":
|
|
585
|
+
return CustomToolDiagnosticCode.DESCRIPTION_UNSAFE
|
|
586
|
+
return CustomToolDiagnosticCode.SOURCE_MALFORMED
|
|
587
|
+
|
|
588
|
+
|
|
589
|
+
def _schema_subset_validation_diagnostic(
|
|
590
|
+
exc: ValidationError,
|
|
591
|
+
*,
|
|
592
|
+
model_name: str,
|
|
593
|
+
phase: CustomToolDiagnosticPhase,
|
|
594
|
+
) -> CustomToolDiagnostic | None:
|
|
595
|
+
for error in exc.errors():
|
|
596
|
+
ctx = error.get("ctx")
|
|
597
|
+
if not isinstance(ctx, Mapping):
|
|
598
|
+
continue
|
|
599
|
+
schema_error = ctx.get("error")
|
|
600
|
+
if not isinstance(schema_error, SchemaSubsetError):
|
|
601
|
+
continue
|
|
602
|
+
loc = error.get("loc")
|
|
603
|
+
path = _validation_pointer_from_loc(loc)
|
|
604
|
+
code = (
|
|
605
|
+
CustomToolDiagnosticCode.OUTPUT_SCHEMA_UNSUPPORTED
|
|
606
|
+
if _validation_loc_has_field(loc, "output_schema")
|
|
607
|
+
else CustomToolDiagnosticCode.INPUT_SCHEMA_UNSUPPORTED
|
|
608
|
+
)
|
|
609
|
+
return custom_tool_diagnostic(
|
|
610
|
+
code,
|
|
611
|
+
phase=phase,
|
|
612
|
+
path=path,
|
|
613
|
+
message="Custom-tool schema is outside the accepted JSON Schema subset.",
|
|
614
|
+
evidence={
|
|
615
|
+
"model": model_name,
|
|
616
|
+
"error_type": type(schema_error).__name__,
|
|
617
|
+
"schema_error": str(schema_error),
|
|
618
|
+
},
|
|
619
|
+
)
|
|
620
|
+
return None
|
|
621
|
+
|
|
622
|
+
|
|
623
|
+
def _validation_code(exc: ValidationError) -> CustomToolDiagnosticCode:
|
|
624
|
+
errors = exc.errors()
|
|
625
|
+
if not errors:
|
|
626
|
+
return CustomToolDiagnosticCode.SOURCE_INVALID
|
|
627
|
+
first = errors[0]
|
|
628
|
+
loc = tuple(str(part) for part in first.get("loc", ()))
|
|
629
|
+
message = str(first.get("msg", "")).lower()
|
|
630
|
+
error_text = str(errors).lower()
|
|
631
|
+
if "secret material" in error_text:
|
|
632
|
+
return CustomToolDiagnosticCode.SECRET_MATERIAL
|
|
633
|
+
if "runtime_kind" in loc:
|
|
634
|
+
return CustomToolDiagnosticCode.RUNTIME_KIND_UNSUPPORTED
|
|
635
|
+
if "produced_artifact_ids" in loc:
|
|
636
|
+
return CustomToolDiagnosticCode.ARTIFACT_POLICY_INVALID
|
|
637
|
+
if (
|
|
638
|
+
"required_capabilities" in loc
|
|
639
|
+
or "require capabilities" in error_text
|
|
640
|
+
or "requires explicit capabilities" in error_text
|
|
641
|
+
):
|
|
642
|
+
return CustomToolDiagnosticCode.CAPABILITY_MISSING
|
|
643
|
+
if "forbidden approval policy" in error_text:
|
|
644
|
+
return CustomToolDiagnosticCode.FORBIDDEN_TOOL_COMPILED
|
|
645
|
+
if "approval_policy" in loc or "side-effecting custom tools" in error_text:
|
|
646
|
+
return CustomToolDiagnosticCode.APPROVAL_POLICY_INVALID
|
|
647
|
+
if "timeout_policy" in loc:
|
|
648
|
+
return CustomToolDiagnosticCode.TIMEOUT_POLICY_INVALID
|
|
649
|
+
if "output_policy" in loc:
|
|
650
|
+
return CustomToolDiagnosticCode.OUTPUT_POLICY_INVALID
|
|
651
|
+
if "input schema" in message or "input_schema" in loc:
|
|
652
|
+
return CustomToolDiagnosticCode.INPUT_SCHEMA_UNSUPPORTED
|
|
653
|
+
if "output schema" in message or "output_schema" in loc:
|
|
654
|
+
return CustomToolDiagnosticCode.OUTPUT_SCHEMA_UNSUPPORTED
|
|
655
|
+
if "custom tool identities" in error_text:
|
|
656
|
+
return CustomToolDiagnosticCode.DUPLICATE_TOOL
|
|
657
|
+
if "custom tool model_tool_name" in error_text:
|
|
658
|
+
return CustomToolDiagnosticCode.DUPLICATE_MODEL_TOOL_NAME
|
|
659
|
+
if "custom tool implementation_id" in error_text:
|
|
660
|
+
return CustomToolDiagnosticCode.DUPLICATE_IMPLEMENTATION_ID
|
|
661
|
+
return CustomToolDiagnosticCode.SOURCE_INVALID
|
|
662
|
+
|
|
663
|
+
|
|
664
|
+
def _missing_field(exc: ValidationError) -> str | None:
|
|
665
|
+
errors = exc.errors()
|
|
666
|
+
if not errors or errors[0].get("type") != "missing":
|
|
667
|
+
return None
|
|
668
|
+
loc = errors[0].get("loc")
|
|
669
|
+
if not isinstance(loc, tuple | list) or not loc:
|
|
670
|
+
return None
|
|
671
|
+
return str(loc[-1])
|
|
672
|
+
|
|
673
|
+
|
|
674
|
+
def _validation_pointer(exc: ValidationError) -> str:
|
|
675
|
+
errors = exc.errors()
|
|
676
|
+
if not errors:
|
|
677
|
+
return "/"
|
|
678
|
+
return _validation_pointer_from_loc(errors[0].get("loc"))
|
|
679
|
+
|
|
680
|
+
|
|
681
|
+
def _validation_pointer_from_loc(loc: Any) -> str:
|
|
682
|
+
if not isinstance(loc, tuple | list) or not loc:
|
|
683
|
+
return "/"
|
|
684
|
+
parts = [str(part).replace("~", "~0").replace("/", "~1") for part in loc]
|
|
685
|
+
return "/" + "/".join(parts)
|
|
686
|
+
|
|
687
|
+
|
|
688
|
+
def _validation_loc_has_field(loc: Any, field_name: str) -> bool:
|
|
689
|
+
if not isinstance(loc, tuple | list):
|
|
690
|
+
return False
|
|
691
|
+
return any(str(part) == field_name for part in loc)
|
|
692
|
+
|
|
693
|
+
|
|
694
|
+
def _join_pointer(prefix: str, raw_part: str) -> str:
|
|
695
|
+
part = raw_part.replace("~", "~0").replace("/", "~1")
|
|
696
|
+
return f"{prefix}/{part}" if prefix else f"/{part}"
|
|
697
|
+
|
|
698
|
+
|
|
699
|
+
def _rejected(
|
|
700
|
+
diagnostics: tuple[CustomToolDiagnostic, ...],
|
|
701
|
+
) -> CustomToolCompilationResult:
|
|
702
|
+
return CustomToolCompilationResult(
|
|
703
|
+
accepted=False,
|
|
704
|
+
diagnostics=tuple(sorted(diagnostics, key=custom_tool_diagnostic_sort_key)),
|
|
705
|
+
)
|
|
706
|
+
|
|
707
|
+
|
|
708
|
+
def _sort_lowered(
|
|
709
|
+
lowered: list[tuple[ToolDescriptor, CustomToolCompilationRecord]],
|
|
710
|
+
) -> tuple[tuple[ToolDescriptor, CustomToolCompilationRecord], ...]:
|
|
711
|
+
return tuple(
|
|
712
|
+
sorted(
|
|
713
|
+
lowered,
|
|
714
|
+
key=lambda item: (
|
|
715
|
+
item[1].package_id,
|
|
716
|
+
item[0].tool_id,
|
|
717
|
+
item[0].tool_version,
|
|
718
|
+
item[0].model_tool_name,
|
|
719
|
+
item[0].implementation_id,
|
|
720
|
+
item[0].descriptor_sha256,
|
|
721
|
+
item[1].compilation_record_sha256,
|
|
722
|
+
),
|
|
723
|
+
)
|
|
724
|
+
)
|