millforge 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- millforge/__init__.py +1174 -0
- millforge/_forge/LICENSE +21 -0
- millforge/_forge/PROVENANCE.json +295 -0
- millforge/_forge/UPDATE_POLICY.md +24 -0
- millforge/_forge/__init__.py +14 -0
- millforge/_forge/adapter.py +2232 -0
- millforge/_forge/base_runner.py +121 -0
- millforge/_forge/clients/__init__.py +10 -0
- millforge/_forge/clients/base.py +200 -0
- millforge/_forge/context/__init__.py +23 -0
- millforge/_forge/context/manager.py +178 -0
- millforge/_forge/context/strategies.py +335 -0
- millforge/_forge/core/__init__.py +16 -0
- millforge/_forge/core/inference.py +433 -0
- millforge/_forge/core/messages.py +119 -0
- millforge/_forge/core/runner.py +479 -0
- millforge/_forge/core/steps.py +108 -0
- millforge/_forge/core/workflow.py +400 -0
- millforge/_forge/errors.py +222 -0
- millforge/_forge/guardrails/__init__.py +21 -0
- millforge/_forge/guardrails/error_tracker.py +71 -0
- millforge/_forge/guardrails/guardrails.py +194 -0
- millforge/_forge/guardrails/nudge.py +47 -0
- millforge/_forge/guardrails/response_validator.py +119 -0
- millforge/_forge/guardrails/step_enforcer.py +183 -0
- millforge/_forge/prompts/__init__.py +16 -0
- millforge/_forge/prompts/nudges.py +95 -0
- millforge/_forge/prompts/templates.py +285 -0
- millforge/_version.py +3 -0
- millforge/artifacts.py +570 -0
- millforge/base/__init__.py +97 -0
- millforge/base/composition.py +402 -0
- millforge/base/context.py +285 -0
- millforge/base/harness.py +138 -0
- millforge/base/identity.py +465 -0
- millforge/base/options.py +34 -0
- millforge/base/platform.py +17 -0
- millforge/base/prompt.py +317 -0
- millforge/base/runner.py +546 -0
- millforge/compiled_plan.py +970 -0
- millforge/compiler/__init__.py +231 -0
- millforge/compiler/artifact_validation.py +257 -0
- millforge/compiler/canonicalization.py +169 -0
- millforge/compiler/capabilities.py +66 -0
- millforge/compiler/catalogs.py +500 -0
- millforge/compiler/diagnostics.py +491 -0
- millforge/compiler/graph.py +678 -0
- millforge/compiler/lowering.py +198 -0
- millforge/compiler/output.py +692 -0
- millforge/compiler/parsing.py +1424 -0
- millforge/compiler/requests.py +1180 -0
- millforge/compiler/schema_validation.py +272 -0
- millforge/compiler/semantic.py +490 -0
- millforge/compiler/service.py +448 -0
- millforge/compiler/source.py +375 -0
- millforge/compiler/validators.py +184 -0
- millforge/connectors/__init__.py +95 -0
- millforge/connectors/admission.py +801 -0
- millforge/connectors/broker.py +202 -0
- millforge/connectors/contracts.py +1159 -0
- millforge/connectors/diagnostics.py +189 -0
- millforge/connectors/fake.py +66 -0
- millforge/connectors/runtime.py +236 -0
- millforge/contracts.py +2860 -0
- millforge/custom_tools/__init__.py +67 -0
- millforge/custom_tools/compiler.py +724 -0
- millforge/custom_tools/contracts.py +1093 -0
- millforge/custom_tools/diagnostics.py +205 -0
- millforge/eval_artifacts.py +952 -0
- millforge/eval_boundary.py +2435 -0
- millforge/eval_fixtures/__init__.py +1 -0
- millforge/eval_fixtures/default_pack/__init__.py +1 -0
- millforge/eval_fixtures/default_pack/fixtures/fixture.08a.bug_diagnosis.traceback.v1.json +52 -0
- millforge/eval_fixtures/default_pack/fixtures/fixture.08a.direct_edit.import_sort.v1.json +52 -0
- millforge/eval_fixtures/default_pack/fixtures/fixture.08a.evidence_discipline.no_source_change.v1.json +51 -0
- millforge/eval_fixtures/default_pack/fixtures/fixture.08a.false_closure.visible_green.v1.json +52 -0
- millforge/eval_fixtures/default_pack/fixtures/fixture.08a.multi_file.api_contract.v1.json +54 -0
- millforge/eval_fixtures/default_pack/fixtures/fixture.08a.recovery.malformed_artifact.v1.json +54 -0
- millforge/eval_fixtures/default_pack/manifest.json +12 -0
- millforge/eval_modes.py +1282 -0
- millforge/eval_presets.py +1398 -0
- millforge/eval_reports.py +2517 -0
- millforge/eval_suite.py +2429 -0
- millforge/eval_trials.py +2632 -0
- millforge/eval_workflow.py +794 -0
- millforge/exceptions.py +122 -0
- millforge/model_backend.py +2098 -0
- millforge/protocols.py +340 -0
- millforge/py.typed +0 -0
- millforge/runtime.py +1791 -0
- millforge/testing/__init__.py +1089 -0
- millforge/tools/__init__.py +83 -0
- millforge/tools/builtin_runtime.py +1339 -0
- millforge/tools/builtins.py +773 -0
- millforge/tools/execution.py +1545 -0
- millforge/tools/path_policy.py +155 -0
- millforge/tools/pi_compat/PI_LICENSE +21 -0
- millforge/tools/pi_compat/PROVENANCE.json +55 -0
- millforge/tools/pi_compat/UPDATE_POLICY.md +36 -0
- millforge/tools/pi_compat/__init__.py +34 -0
- millforge/tools/pi_compat/contracts.py +49 -0
- millforge/tools/pi_compat/editing.py +390 -0
- millforge/tools/pi_compat/mutations.py +57 -0
- millforge/tools/pi_compat/operations.py +401 -0
- millforge/tools/pi_compat/paths.py +155 -0
- millforge/tools/pi_compat/process.py +1375 -0
- millforge/tools/pi_compat/search.py +738 -0
- millforge/tools/pi_compat/truncation.py +267 -0
- millforge/tools/pi_compat_catalog.py +396 -0
- millforge/tools/pi_compat_runtime.py +460 -0
- millforge/tools/registry.py +553 -0
- millforge/tools/results.py +533 -0
- millforge-0.1.0.dist-info/METADATA +844 -0
- millforge-0.1.0.dist-info/RECORD +116 -0
- millforge-0.1.0.dist-info/WHEEL +4 -0
- millforge-0.1.0.dist-info/licenses/LICENSE +201 -0
|
@@ -0,0 +1,1545 @@
|
|
|
1
|
+
"""Compiled-plan scoped tool binding resolution and execution."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import inspect
|
|
6
|
+
from collections.abc import Awaitable, Callable, Mapping
|
|
7
|
+
from enum import Enum
|
|
8
|
+
from typing import Any, Literal
|
|
9
|
+
|
|
10
|
+
from millforge import (
|
|
11
|
+
ToolBindingRef,
|
|
12
|
+
SideEffectCertainty,
|
|
13
|
+
SideEffectClass,
|
|
14
|
+
ToolExecutionStatus,
|
|
15
|
+
ToolTraceDecision,
|
|
16
|
+
canonical_json_serialize,
|
|
17
|
+
)
|
|
18
|
+
from millforge.compiled_plan import CompiledHarnessNode, CompiledHarnessPlan
|
|
19
|
+
from millforge.contracts import (
|
|
20
|
+
SideEffectRecord,
|
|
21
|
+
ToolExecutionContext,
|
|
22
|
+
ToolExecutionResult,
|
|
23
|
+
ValidatedToolCall,
|
|
24
|
+
)
|
|
25
|
+
from millforge.compiler.catalogs import ToolCatalogSnapshot
|
|
26
|
+
from millforge.tools.registry import ToolOutputPolicy
|
|
27
|
+
from millforge.tools.results import (
|
|
28
|
+
MAX_MODEL_SUMMARY_UTF8,
|
|
29
|
+
ToolExecutionErrorCode,
|
|
30
|
+
canonical_sha256,
|
|
31
|
+
make_denial_result,
|
|
32
|
+
make_tool_result,
|
|
33
|
+
make_trace_record,
|
|
34
|
+
redact_tool_value,
|
|
35
|
+
sanitize_tool_execution_result,
|
|
36
|
+
validate_json_object_schema,
|
|
37
|
+
)
|
|
38
|
+
|
|
39
|
+
RuntimeToolImplementation = Callable[
|
|
40
|
+
[ValidatedToolCall, ToolExecutionContext],
|
|
41
|
+
ToolExecutionResult | Awaitable[ToolExecutionResult],
|
|
42
|
+
]
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
class ToolBindingDenialCode(str, Enum):
|
|
46
|
+
"""Stable binding-denial categories."""
|
|
47
|
+
|
|
48
|
+
BINDING_MISMATCH = "binding_mismatch"
|
|
49
|
+
CONFLICT = "conflict"
|
|
50
|
+
INVALID_ARGUMENTS = "invalid_arguments"
|
|
51
|
+
NOT_FOUND = "not_found"
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
class RuntimeToolRegistry:
|
|
55
|
+
"""Explicit in-process registry of source-owned runtime implementations."""
|
|
56
|
+
|
|
57
|
+
def __init__(self) -> None:
|
|
58
|
+
self._implementations: dict[str, RuntimeToolImplementation] = {}
|
|
59
|
+
|
|
60
|
+
def register(
|
|
61
|
+
self,
|
|
62
|
+
implementation_id: str,
|
|
63
|
+
implementation: RuntimeToolImplementation,
|
|
64
|
+
) -> None:
|
|
65
|
+
if not implementation_id.strip():
|
|
66
|
+
raise ValueError("implementation_id must be non-empty")
|
|
67
|
+
if not callable(implementation):
|
|
68
|
+
raise TypeError("implementation must be callable")
|
|
69
|
+
if implementation_id in self._implementations:
|
|
70
|
+
raise ValueError(f"duplicate implementation_id {implementation_id!r}")
|
|
71
|
+
self._implementations[implementation_id] = implementation
|
|
72
|
+
|
|
73
|
+
def resolve(self, implementation_id: str) -> RuntimeToolImplementation | None:
|
|
74
|
+
return self._implementations.get(implementation_id)
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
class CompiledToolBindingExecutor:
|
|
78
|
+
"""Tool executor admitted by exact compiled-plan and descriptor-snapshot bindings."""
|
|
79
|
+
|
|
80
|
+
def __init__(
|
|
81
|
+
self,
|
|
82
|
+
*,
|
|
83
|
+
plan: CompiledHarnessPlan,
|
|
84
|
+
descriptor_snapshot: ToolCatalogSnapshot,
|
|
85
|
+
runtime_registry: RuntimeToolRegistry,
|
|
86
|
+
connector_admission_snapshot: Any | None = None,
|
|
87
|
+
connector_broker: Any | None = None,
|
|
88
|
+
) -> None:
|
|
89
|
+
self._plan = plan
|
|
90
|
+
self._nodes_by_id = {node.node_id: node for node in plan.nodes}
|
|
91
|
+
self._node_by_model_name: dict[str, CompiledHarnessNode] = {}
|
|
92
|
+
self._terminal_result_map = dict(plan.terminal_result_map)
|
|
93
|
+
self._conflicting_model_names = _duplicate_values(
|
|
94
|
+
node.model_tool_name for node in plan.nodes
|
|
95
|
+
)
|
|
96
|
+
for node in plan.nodes:
|
|
97
|
+
if node.model_tool_name not in self._conflicting_model_names:
|
|
98
|
+
self._node_by_model_name[node.model_tool_name] = node
|
|
99
|
+
_validate_connector_admissions(
|
|
100
|
+
plan=plan,
|
|
101
|
+
connector_admission_snapshot=connector_admission_snapshot,
|
|
102
|
+
connector_broker=connector_broker,
|
|
103
|
+
)
|
|
104
|
+
self._node_defects = {
|
|
105
|
+
node.node_id: _node_binding_defect(
|
|
106
|
+
node=node,
|
|
107
|
+
descriptor_snapshot=descriptor_snapshot,
|
|
108
|
+
runtime_registry=runtime_registry,
|
|
109
|
+
)
|
|
110
|
+
for node in plan.nodes
|
|
111
|
+
}
|
|
112
|
+
self._descriptor_snapshot = descriptor_snapshot
|
|
113
|
+
self._connector_admission_snapshot = connector_admission_snapshot
|
|
114
|
+
self._connector_broker = connector_broker
|
|
115
|
+
self._runtime_registry = runtime_registry
|
|
116
|
+
self._trace_records: list[Any] = []
|
|
117
|
+
self._next_sequence = 1
|
|
118
|
+
|
|
119
|
+
def fork_for_invocation(self) -> CompiledToolBindingExecutor:
|
|
120
|
+
"""Create an executor with identical bindings and fresh trace state."""
|
|
121
|
+
return CompiledToolBindingExecutor(
|
|
122
|
+
plan=self._plan,
|
|
123
|
+
descriptor_snapshot=self._descriptor_snapshot,
|
|
124
|
+
runtime_registry=self._runtime_registry,
|
|
125
|
+
connector_admission_snapshot=self._connector_admission_snapshot,
|
|
126
|
+
connector_broker=self._connector_broker,
|
|
127
|
+
)
|
|
128
|
+
|
|
129
|
+
def supports_tool(self, name: str) -> bool:
|
|
130
|
+
node = self._node_by_model_name.get(name)
|
|
131
|
+
return node is not None and self._node_defects[node.node_id] is None
|
|
132
|
+
|
|
133
|
+
@property
|
|
134
|
+
def trace_records(self) -> tuple[Any, ...]:
|
|
135
|
+
"""Return emitted trace records in execution order."""
|
|
136
|
+
return tuple(self._trace_records)
|
|
137
|
+
|
|
138
|
+
def validate_model_tool_call(
|
|
139
|
+
self,
|
|
140
|
+
*,
|
|
141
|
+
model_tool_name: str,
|
|
142
|
+
call_id: str,
|
|
143
|
+
arguments: Mapping[str, Any],
|
|
144
|
+
) -> ValidatedToolCall | ToolExecutionResult:
|
|
145
|
+
"""Resolve a model-visible tool name into an exact compiled binding."""
|
|
146
|
+
input_sha256 = canonical_sha256(arguments)
|
|
147
|
+
if model_tool_name in self._conflicting_model_names:
|
|
148
|
+
return _denial_result(
|
|
149
|
+
call_id=call_id,
|
|
150
|
+
input_sha256=input_sha256,
|
|
151
|
+
code=ToolBindingDenialCode.CONFLICT,
|
|
152
|
+
summary="model-visible tool name is ambiguous",
|
|
153
|
+
evidence={"model_tool_name": model_tool_name},
|
|
154
|
+
)
|
|
155
|
+
node = self._node_by_model_name.get(model_tool_name)
|
|
156
|
+
if node is None:
|
|
157
|
+
return _denial_result(
|
|
158
|
+
call_id=call_id,
|
|
159
|
+
input_sha256=input_sha256,
|
|
160
|
+
code=ToolBindingDenialCode.NOT_FOUND,
|
|
161
|
+
summary="model-visible tool name is not compiled",
|
|
162
|
+
evidence={"model_tool_name": model_tool_name},
|
|
163
|
+
)
|
|
164
|
+
defect = self._node_defects[node.node_id]
|
|
165
|
+
if defect is not None:
|
|
166
|
+
code, summary, evidence, hard_failure = defect
|
|
167
|
+
return _denial_result(
|
|
168
|
+
call_id=call_id,
|
|
169
|
+
input_sha256=input_sha256,
|
|
170
|
+
code=code,
|
|
171
|
+
summary=summary,
|
|
172
|
+
evidence={"node_id": node.node_id, **evidence},
|
|
173
|
+
node=node,
|
|
174
|
+
status=(ToolExecutionStatus.HARD_FAILURE if hard_failure else None),
|
|
175
|
+
)
|
|
176
|
+
input_error = validate_json_object_schema(arguments, node.input_schema)
|
|
177
|
+
if input_error is not None:
|
|
178
|
+
return _denial_result(
|
|
179
|
+
call_id=call_id,
|
|
180
|
+
input_sha256=input_sha256,
|
|
181
|
+
code=ToolBindingDenialCode.INVALID_ARGUMENTS,
|
|
182
|
+
summary="tool arguments failed descriptor input schema validation",
|
|
183
|
+
evidence={"schema_error": input_error},
|
|
184
|
+
node=node,
|
|
185
|
+
)
|
|
186
|
+
return ValidatedToolCall(
|
|
187
|
+
call_id=call_id,
|
|
188
|
+
node_id=node.node_id,
|
|
189
|
+
binding=node.binding,
|
|
190
|
+
arguments=dict(arguments),
|
|
191
|
+
)
|
|
192
|
+
|
|
193
|
+
async def execute_model_tool(
|
|
194
|
+
self,
|
|
195
|
+
*,
|
|
196
|
+
model_tool_name: str,
|
|
197
|
+
call_id: str,
|
|
198
|
+
arguments: Mapping[str, Any],
|
|
199
|
+
context: ToolExecutionContext,
|
|
200
|
+
prerequisite_results: Mapping[str, ToolExecutionResult] | None = None,
|
|
201
|
+
model_turn: int = 0,
|
|
202
|
+
session_id: str | None = None,
|
|
203
|
+
) -> ToolExecutionResult:
|
|
204
|
+
resolved = self.validate_model_tool_call(
|
|
205
|
+
model_tool_name=model_tool_name,
|
|
206
|
+
call_id=call_id,
|
|
207
|
+
arguments=arguments,
|
|
208
|
+
)
|
|
209
|
+
if isinstance(resolved, ToolExecutionResult):
|
|
210
|
+
node = self._node_by_model_name.get(model_tool_name)
|
|
211
|
+
if node is not None:
|
|
212
|
+
connector_audit = (
|
|
213
|
+
self._connector_pre_entry_audit(
|
|
214
|
+
node=node,
|
|
215
|
+
context=context,
|
|
216
|
+
)
|
|
217
|
+
if _is_connector_tool_id(node.binding.tool_id)
|
|
218
|
+
else None
|
|
219
|
+
)
|
|
220
|
+
self._record_trace(
|
|
221
|
+
node=node,
|
|
222
|
+
call_id=call_id,
|
|
223
|
+
model_tool_name=model_tool_name,
|
|
224
|
+
input_sha256=resolved.input_sha256,
|
|
225
|
+
result=resolved,
|
|
226
|
+
context=context,
|
|
227
|
+
prerequisite_decisions={},
|
|
228
|
+
capability_decisions={},
|
|
229
|
+
connector_audit=connector_audit,
|
|
230
|
+
model_turn=model_turn,
|
|
231
|
+
session_id=session_id,
|
|
232
|
+
)
|
|
233
|
+
else:
|
|
234
|
+
binding_resolution_status: Literal["ambiguous", "uncompiled"] = (
|
|
235
|
+
"ambiguous"
|
|
236
|
+
if model_tool_name in self._conflicting_model_names
|
|
237
|
+
else "uncompiled"
|
|
238
|
+
)
|
|
239
|
+
self._record_trace(
|
|
240
|
+
node=None,
|
|
241
|
+
call_id=call_id,
|
|
242
|
+
model_tool_name=model_tool_name,
|
|
243
|
+
input_sha256=resolved.input_sha256,
|
|
244
|
+
result=resolved,
|
|
245
|
+
context=context,
|
|
246
|
+
prerequisite_decisions={},
|
|
247
|
+
capability_decisions={},
|
|
248
|
+
connector_audit=None,
|
|
249
|
+
model_turn=model_turn,
|
|
250
|
+
session_id=session_id,
|
|
251
|
+
binding_resolution_status=binding_resolution_status,
|
|
252
|
+
)
|
|
253
|
+
return resolved
|
|
254
|
+
return await self.execute(
|
|
255
|
+
resolved,
|
|
256
|
+
context,
|
|
257
|
+
prerequisite_results=prerequisite_results,
|
|
258
|
+
model_turn=model_turn,
|
|
259
|
+
model_tool_name=model_tool_name,
|
|
260
|
+
session_id=session_id,
|
|
261
|
+
)
|
|
262
|
+
|
|
263
|
+
async def execute(
|
|
264
|
+
self,
|
|
265
|
+
call: ValidatedToolCall,
|
|
266
|
+
context: ToolExecutionContext,
|
|
267
|
+
prerequisite_results: Mapping[str, ToolExecutionResult] | None = None,
|
|
268
|
+
model_turn: int = 0,
|
|
269
|
+
model_tool_name: str | None = None,
|
|
270
|
+
session_id: str | None = None,
|
|
271
|
+
) -> ToolExecutionResult:
|
|
272
|
+
input_sha256 = canonical_sha256(call.arguments)
|
|
273
|
+
node = self._nodes_by_id.get(call.node_id)
|
|
274
|
+
if node is None:
|
|
275
|
+
return _denial_result(
|
|
276
|
+
call_id=call.call_id,
|
|
277
|
+
input_sha256=input_sha256,
|
|
278
|
+
code=ToolBindingDenialCode.NOT_FOUND,
|
|
279
|
+
summary="compiled node is not admitted",
|
|
280
|
+
evidence={"node_id": call.node_id},
|
|
281
|
+
)
|
|
282
|
+
mismatch = _call_binding_mismatch(call, node)
|
|
283
|
+
if mismatch is not None:
|
|
284
|
+
result = _denial_result(
|
|
285
|
+
call_id=call.call_id,
|
|
286
|
+
input_sha256=input_sha256,
|
|
287
|
+
code=ToolBindingDenialCode.BINDING_MISMATCH,
|
|
288
|
+
summary="validated call binding does not match compiled node",
|
|
289
|
+
evidence={"node_id": node.node_id, **mismatch},
|
|
290
|
+
node=node,
|
|
291
|
+
)
|
|
292
|
+
self._record_trace(
|
|
293
|
+
node=node,
|
|
294
|
+
call_id=call.call_id,
|
|
295
|
+
model_tool_name=model_tool_name or node.model_tool_name,
|
|
296
|
+
input_sha256=input_sha256,
|
|
297
|
+
result=result,
|
|
298
|
+
context=context,
|
|
299
|
+
prerequisite_decisions={},
|
|
300
|
+
capability_decisions={},
|
|
301
|
+
connector_audit=None,
|
|
302
|
+
model_turn=model_turn,
|
|
303
|
+
session_id=session_id,
|
|
304
|
+
)
|
|
305
|
+
return result
|
|
306
|
+
defect = self._node_defects[node.node_id]
|
|
307
|
+
if defect is not None:
|
|
308
|
+
code, summary, evidence, hard_failure = defect
|
|
309
|
+
result = _denial_result(
|
|
310
|
+
call_id=call.call_id,
|
|
311
|
+
input_sha256=input_sha256,
|
|
312
|
+
code=code,
|
|
313
|
+
summary=summary,
|
|
314
|
+
evidence={"node_id": node.node_id, **evidence},
|
|
315
|
+
node=node,
|
|
316
|
+
status=(ToolExecutionStatus.HARD_FAILURE if hard_failure else None),
|
|
317
|
+
)
|
|
318
|
+
self._record_trace(
|
|
319
|
+
node=node,
|
|
320
|
+
call_id=call.call_id,
|
|
321
|
+
model_tool_name=model_tool_name or node.model_tool_name,
|
|
322
|
+
input_sha256=input_sha256,
|
|
323
|
+
result=result,
|
|
324
|
+
context=context,
|
|
325
|
+
prerequisite_decisions={},
|
|
326
|
+
capability_decisions={},
|
|
327
|
+
connector_audit=None,
|
|
328
|
+
model_turn=model_turn,
|
|
329
|
+
session_id=session_id,
|
|
330
|
+
)
|
|
331
|
+
return result
|
|
332
|
+
preflight = self._preflight_result(
|
|
333
|
+
node=node,
|
|
334
|
+
call=call,
|
|
335
|
+
context=context,
|
|
336
|
+
prerequisite_results=prerequisite_results or {},
|
|
337
|
+
input_sha256=input_sha256,
|
|
338
|
+
)
|
|
339
|
+
if preflight is not None:
|
|
340
|
+
result, prerequisite_decisions, capability_decisions = preflight
|
|
341
|
+
connector_audit = self._connector_pre_entry_audit(
|
|
342
|
+
node=node,
|
|
343
|
+
context=context,
|
|
344
|
+
)
|
|
345
|
+
self._record_trace(
|
|
346
|
+
node=node,
|
|
347
|
+
call_id=call.call_id,
|
|
348
|
+
model_tool_name=model_tool_name or node.model_tool_name,
|
|
349
|
+
input_sha256=input_sha256,
|
|
350
|
+
result=result,
|
|
351
|
+
context=context,
|
|
352
|
+
prerequisite_decisions=prerequisite_decisions,
|
|
353
|
+
capability_decisions=capability_decisions,
|
|
354
|
+
connector_audit=connector_audit,
|
|
355
|
+
model_turn=model_turn,
|
|
356
|
+
session_id=session_id,
|
|
357
|
+
)
|
|
358
|
+
return result
|
|
359
|
+
prerequisite_decisions = {
|
|
360
|
+
prereq.node_id: "allowed" for prereq in node.prerequisites
|
|
361
|
+
}
|
|
362
|
+
capability_decisions = {
|
|
363
|
+
capability: "allowed" for capability in node.required_capabilities
|
|
364
|
+
}
|
|
365
|
+
if _is_connector_tool_id(node.binding.tool_id):
|
|
366
|
+
result, connector_audit = await self._execute_connector(
|
|
367
|
+
node=node,
|
|
368
|
+
call=call,
|
|
369
|
+
context=context,
|
|
370
|
+
input_sha256=input_sha256,
|
|
371
|
+
)
|
|
372
|
+
self._record_trace(
|
|
373
|
+
node=node,
|
|
374
|
+
call_id=call.call_id,
|
|
375
|
+
model_tool_name=model_tool_name or node.model_tool_name,
|
|
376
|
+
input_sha256=input_sha256,
|
|
377
|
+
result=result,
|
|
378
|
+
context=context,
|
|
379
|
+
prerequisite_decisions=prerequisite_decisions,
|
|
380
|
+
capability_decisions=capability_decisions,
|
|
381
|
+
connector_audit=connector_audit,
|
|
382
|
+
model_turn=model_turn,
|
|
383
|
+
session_id=session_id,
|
|
384
|
+
)
|
|
385
|
+
return result
|
|
386
|
+
implementation = self._runtime_registry.resolve(node.binding.implementation_id)
|
|
387
|
+
if implementation is None:
|
|
388
|
+
return _denial_result(
|
|
389
|
+
call_id=call.call_id,
|
|
390
|
+
input_sha256=input_sha256,
|
|
391
|
+
code=ToolBindingDenialCode.NOT_FOUND,
|
|
392
|
+
summary="runtime implementation is not registered",
|
|
393
|
+
evidence={
|
|
394
|
+
"node_id": node.node_id,
|
|
395
|
+
"implementation_id": node.binding.implementation_id,
|
|
396
|
+
},
|
|
397
|
+
node=node,
|
|
398
|
+
)
|
|
399
|
+
try:
|
|
400
|
+
pending = implementation(call, context)
|
|
401
|
+
result = await pending if inspect.isawaitable(pending) else pending
|
|
402
|
+
except Exception as exc:
|
|
403
|
+
result = make_tool_result(
|
|
404
|
+
call_id=call.call_id,
|
|
405
|
+
status=ToolExecutionStatus.HARD_FAILURE,
|
|
406
|
+
code=ToolExecutionErrorCode.IMPLEMENTATION_ERROR,
|
|
407
|
+
summary="tool implementation raised an exception",
|
|
408
|
+
structured_data={
|
|
409
|
+
"category": ToolExecutionErrorCode.IMPLEMENTATION_ERROR.value,
|
|
410
|
+
"error_type": type(exc).__name__,
|
|
411
|
+
"error": exc,
|
|
412
|
+
},
|
|
413
|
+
side_effect_class=node.side_effect_class,
|
|
414
|
+
idempotency=node.idempotency,
|
|
415
|
+
side_effect_certainty=SideEffectCertainty.COMPLETION_UNKNOWN,
|
|
416
|
+
side_effect_record=SideEffectRecord(
|
|
417
|
+
certainty=SideEffectCertainty.COMPLETION_UNKNOWN,
|
|
418
|
+
detail_code=ToolExecutionErrorCode.AMBIGUOUS_SIDE_EFFECT.value,
|
|
419
|
+
summary="implementation failure left completion unknown",
|
|
420
|
+
retry_allowed=False,
|
|
421
|
+
),
|
|
422
|
+
input_sha256=input_sha256,
|
|
423
|
+
retryable=False,
|
|
424
|
+
)
|
|
425
|
+
else:
|
|
426
|
+
result = self._validate_implementation_result(
|
|
427
|
+
node=node,
|
|
428
|
+
call=call,
|
|
429
|
+
result=result,
|
|
430
|
+
input_sha256=input_sha256,
|
|
431
|
+
)
|
|
432
|
+
self._record_trace(
|
|
433
|
+
node=node,
|
|
434
|
+
call_id=call.call_id,
|
|
435
|
+
model_tool_name=model_tool_name or node.model_tool_name,
|
|
436
|
+
input_sha256=input_sha256,
|
|
437
|
+
result=result,
|
|
438
|
+
context=context,
|
|
439
|
+
prerequisite_decisions=prerequisite_decisions,
|
|
440
|
+
capability_decisions=capability_decisions,
|
|
441
|
+
connector_audit=None,
|
|
442
|
+
model_turn=model_turn,
|
|
443
|
+
session_id=session_id,
|
|
444
|
+
)
|
|
445
|
+
return result
|
|
446
|
+
|
|
447
|
+
async def _execute_connector(
|
|
448
|
+
self,
|
|
449
|
+
*,
|
|
450
|
+
node: CompiledHarnessNode,
|
|
451
|
+
call: ValidatedToolCall,
|
|
452
|
+
context: ToolExecutionContext,
|
|
453
|
+
input_sha256: str,
|
|
454
|
+
) -> tuple[ToolExecutionResult, dict[str, Any] | None]:
|
|
455
|
+
from millforge.connectors.broker import (
|
|
456
|
+
ConnectorBrokerOutcome,
|
|
457
|
+
ConnectorInvocationRequest,
|
|
458
|
+
connector_idempotency_key,
|
|
459
|
+
)
|
|
460
|
+
|
|
461
|
+
assert self._connector_admission_snapshot is not None
|
|
462
|
+
admission = self._connector_admission_snapshot.require(
|
|
463
|
+
node.binding.tool_id,
|
|
464
|
+
node.binding.tool_version,
|
|
465
|
+
node.binding.descriptor_sha256,
|
|
466
|
+
)
|
|
467
|
+
connector_audit = _connector_approval_audit_fields(
|
|
468
|
+
admission=admission,
|
|
469
|
+
node=node,
|
|
470
|
+
context=context,
|
|
471
|
+
broker_attempted=False,
|
|
472
|
+
)
|
|
473
|
+
approval_decision = connector_audit["approval_decision"]
|
|
474
|
+
if approval_decision == "forbidden":
|
|
475
|
+
return make_denial_result(
|
|
476
|
+
call_id=call.call_id,
|
|
477
|
+
code=ToolExecutionErrorCode.POLICY_DENIED,
|
|
478
|
+
summary="connector approval policy forbids runtime invocation",
|
|
479
|
+
evidence=connector_audit,
|
|
480
|
+
side_effect_class=node.side_effect_class,
|
|
481
|
+
idempotency=node.idempotency,
|
|
482
|
+
input_sha256=input_sha256,
|
|
483
|
+
), connector_audit
|
|
484
|
+
if approval_decision == "pending":
|
|
485
|
+
return make_denial_result(
|
|
486
|
+
call_id=call.call_id,
|
|
487
|
+
code=ToolExecutionErrorCode.POLICY_DENIED,
|
|
488
|
+
summary="connector approval is pending operator out-of-band review",
|
|
489
|
+
evidence=connector_audit,
|
|
490
|
+
side_effect_class=node.side_effect_class,
|
|
491
|
+
idempotency=node.idempotency,
|
|
492
|
+
input_sha256=input_sha256,
|
|
493
|
+
), connector_audit
|
|
494
|
+
if approval_decision != "approved":
|
|
495
|
+
return make_denial_result(
|
|
496
|
+
call_id=call.call_id,
|
|
497
|
+
code=ToolExecutionErrorCode.POLICY_DENIED,
|
|
498
|
+
summary="connector approval policy is not satisfied",
|
|
499
|
+
evidence=connector_audit,
|
|
500
|
+
side_effect_class=node.side_effect_class,
|
|
501
|
+
idempotency=node.idempotency,
|
|
502
|
+
input_sha256=input_sha256,
|
|
503
|
+
), connector_audit
|
|
504
|
+
broker = self._connector_broker
|
|
505
|
+
if broker is None or not broker.has_provider_tool(
|
|
506
|
+
admission.connector_id,
|
|
507
|
+
admission.provider_tool_name,
|
|
508
|
+
):
|
|
509
|
+
return make_denial_result(
|
|
510
|
+
call_id=call.call_id,
|
|
511
|
+
code=ToolExecutionErrorCode.NOT_FOUND,
|
|
512
|
+
summary="connector provider tool is not available from broker",
|
|
513
|
+
evidence={
|
|
514
|
+
"connector_id": admission.connector_id,
|
|
515
|
+
"provider_tool_name": admission.provider_tool_name,
|
|
516
|
+
},
|
|
517
|
+
side_effect_class=node.side_effect_class,
|
|
518
|
+
idempotency=node.idempotency,
|
|
519
|
+
input_sha256=input_sha256,
|
|
520
|
+
), connector_audit
|
|
521
|
+
drift = _connector_pre_entry_drift(
|
|
522
|
+
admission=admission,
|
|
523
|
+
node=node,
|
|
524
|
+
broker=broker,
|
|
525
|
+
)
|
|
526
|
+
if drift is not None:
|
|
527
|
+
field, expected, actual = drift
|
|
528
|
+
return make_denial_result(
|
|
529
|
+
call_id=call.call_id,
|
|
530
|
+
code=ToolExecutionErrorCode.BINDING_MISMATCH,
|
|
531
|
+
summary="connector provider evidence drifted before broker entry",
|
|
532
|
+
evidence={
|
|
533
|
+
"connector_id": admission.connector_id,
|
|
534
|
+
"provider_tool_name": admission.provider_tool_name,
|
|
535
|
+
"binding_field": field,
|
|
536
|
+
"expected": expected,
|
|
537
|
+
"actual": actual,
|
|
538
|
+
},
|
|
539
|
+
side_effect_class=node.side_effect_class,
|
|
540
|
+
idempotency=node.idempotency,
|
|
541
|
+
input_sha256=input_sha256,
|
|
542
|
+
), {**connector_audit, "drift_decision": "failed"}
|
|
543
|
+
connector_audit = {
|
|
544
|
+
**connector_audit,
|
|
545
|
+
"broker_attempted": True,
|
|
546
|
+
"drift_decision": "passed",
|
|
547
|
+
}
|
|
548
|
+
request = ConnectorInvocationRequest.from_runtime(
|
|
549
|
+
connector_id=admission.connector_id,
|
|
550
|
+
provider_tool_name=admission.provider_tool_name,
|
|
551
|
+
tool_id=node.binding.tool_id,
|
|
552
|
+
tool_version=node.binding.tool_version,
|
|
553
|
+
descriptor_sha256=node.binding.descriptor_sha256,
|
|
554
|
+
connector_identity_sha256=admission.connector_identity_sha256,
|
|
555
|
+
discovery_snapshot_sha256=admission.discovery_snapshot_sha256,
|
|
556
|
+
raw_tool_sha256=admission.raw_tool_sha256,
|
|
557
|
+
arguments=call.arguments,
|
|
558
|
+
request_id=context.request_id,
|
|
559
|
+
run_id=context.run_id,
|
|
560
|
+
stage_plane=context.stage.plane,
|
|
561
|
+
stage_kind_id=context.stage.stage_kind_id,
|
|
562
|
+
stage_node_id=context.stage.node_id,
|
|
563
|
+
timeout_seconds=context.timeout.timeout_seconds,
|
|
564
|
+
deadline_remaining_seconds=context.deadline.remaining(
|
|
565
|
+
lambda: context.current_monotonic
|
|
566
|
+
),
|
|
567
|
+
cancellation_requested=context.cancellation_requested,
|
|
568
|
+
cancellation_id=context.cancellation.cancellation_id,
|
|
569
|
+
idempotency_key=connector_idempotency_key(
|
|
570
|
+
idempotency=node.idempotency,
|
|
571
|
+
call_id=call.call_id,
|
|
572
|
+
),
|
|
573
|
+
)
|
|
574
|
+
try:
|
|
575
|
+
pending = broker.invoke(request)
|
|
576
|
+
outcome = await pending if inspect.isawaitable(pending) else pending
|
|
577
|
+
except Exception as exc:
|
|
578
|
+
return make_tool_result(
|
|
579
|
+
call_id=call.call_id,
|
|
580
|
+
status=ToolExecutionStatus.HARD_FAILURE,
|
|
581
|
+
code=ToolExecutionErrorCode.IMPLEMENTATION_ERROR,
|
|
582
|
+
summary="connector broker raised an exception",
|
|
583
|
+
structured_data={
|
|
584
|
+
"category": ToolExecutionErrorCode.IMPLEMENTATION_ERROR.value,
|
|
585
|
+
"error_type": type(exc).__name__,
|
|
586
|
+
},
|
|
587
|
+
side_effect_class=node.side_effect_class,
|
|
588
|
+
idempotency=node.idempotency,
|
|
589
|
+
side_effect_certainty=SideEffectCertainty.COMPLETION_UNKNOWN,
|
|
590
|
+
side_effect_record=SideEffectRecord(
|
|
591
|
+
certainty=SideEffectCertainty.COMPLETION_UNKNOWN,
|
|
592
|
+
detail_code=ToolExecutionErrorCode.AMBIGUOUS_SIDE_EFFECT.value,
|
|
593
|
+
summary="broker failure left connector completion unknown",
|
|
594
|
+
retry_allowed=False,
|
|
595
|
+
),
|
|
596
|
+
input_sha256=input_sha256,
|
|
597
|
+
retryable=False,
|
|
598
|
+
), connector_audit
|
|
599
|
+
if not isinstance(outcome, ConnectorBrokerOutcome):
|
|
600
|
+
return make_tool_result(
|
|
601
|
+
call_id=call.call_id,
|
|
602
|
+
status=ToolExecutionStatus.HARD_FAILURE,
|
|
603
|
+
code=ToolExecutionErrorCode.IMPLEMENTATION_ERROR,
|
|
604
|
+
summary="connector broker returned an invalid outcome",
|
|
605
|
+
structured_data={
|
|
606
|
+
"category": ToolExecutionErrorCode.IMPLEMENTATION_ERROR.value,
|
|
607
|
+
"error_type": type(outcome).__name__,
|
|
608
|
+
},
|
|
609
|
+
side_effect_class=node.side_effect_class,
|
|
610
|
+
idempotency=node.idempotency,
|
|
611
|
+
side_effect_certainty=SideEffectCertainty.COMPLETION_UNKNOWN,
|
|
612
|
+
input_sha256=input_sha256,
|
|
613
|
+
retryable=False,
|
|
614
|
+
), connector_audit
|
|
615
|
+
result = _connector_result_from_broker_outcome(
|
|
616
|
+
call_id=call.call_id,
|
|
617
|
+
node=node,
|
|
618
|
+
outcome=outcome,
|
|
619
|
+
input_sha256=input_sha256,
|
|
620
|
+
idempotency_key_policy=admission.idempotency_key_policy,
|
|
621
|
+
idempotency_key=request.idempotency_key,
|
|
622
|
+
)
|
|
623
|
+
return (
|
|
624
|
+
self._validate_implementation_result(
|
|
625
|
+
node=node,
|
|
626
|
+
call=call,
|
|
627
|
+
result=result,
|
|
628
|
+
input_sha256=input_sha256,
|
|
629
|
+
),
|
|
630
|
+
connector_audit,
|
|
631
|
+
)
|
|
632
|
+
|
|
633
|
+
def _preflight_result(
|
|
634
|
+
self,
|
|
635
|
+
*,
|
|
636
|
+
node: CompiledHarnessNode,
|
|
637
|
+
call: ValidatedToolCall,
|
|
638
|
+
context: ToolExecutionContext,
|
|
639
|
+
prerequisite_results: Mapping[str, ToolExecutionResult],
|
|
640
|
+
input_sha256: str,
|
|
641
|
+
) -> (
|
|
642
|
+
tuple[
|
|
643
|
+
ToolExecutionResult,
|
|
644
|
+
dict[str, str],
|
|
645
|
+
dict[str, str],
|
|
646
|
+
]
|
|
647
|
+
| None
|
|
648
|
+
):
|
|
649
|
+
prerequisite_decisions: dict[str, str] = {}
|
|
650
|
+
for prereq in node.prerequisites:
|
|
651
|
+
prereq_result = prerequisite_results.get(prereq.node_id)
|
|
652
|
+
if (
|
|
653
|
+
prereq_result is None
|
|
654
|
+
or prereq_result.status is not ToolExecutionStatus.SUCCESS
|
|
655
|
+
):
|
|
656
|
+
prerequisite_decisions[prereq.node_id] = "denied"
|
|
657
|
+
return (
|
|
658
|
+
make_denial_result(
|
|
659
|
+
call_id=call.call_id,
|
|
660
|
+
code=ToolExecutionErrorCode.PREREQUISITE_DENIED,
|
|
661
|
+
summary="required prerequisite has not completed successfully",
|
|
662
|
+
evidence={"prerequisite_node_id": prereq.node_id},
|
|
663
|
+
side_effect_class=node.side_effect_class,
|
|
664
|
+
idempotency=node.idempotency,
|
|
665
|
+
input_sha256=input_sha256,
|
|
666
|
+
),
|
|
667
|
+
prerequisite_decisions,
|
|
668
|
+
{},
|
|
669
|
+
)
|
|
670
|
+
prerequisite_decisions[prereq.node_id] = "allowed"
|
|
671
|
+
granted = {grant.capability_id for grant in context.capability_envelope.grants}
|
|
672
|
+
capability_decisions: dict[str, str] = {}
|
|
673
|
+
for capability in node.required_capabilities:
|
|
674
|
+
if capability not in granted:
|
|
675
|
+
capability_decisions[capability] = "denied"
|
|
676
|
+
return (
|
|
677
|
+
make_denial_result(
|
|
678
|
+
call_id=call.call_id,
|
|
679
|
+
code=ToolExecutionErrorCode.CAPABILITY_DENIED,
|
|
680
|
+
summary="required capability is absent from request envelope",
|
|
681
|
+
evidence={"capability_id": capability},
|
|
682
|
+
side_effect_class=node.side_effect_class,
|
|
683
|
+
idempotency=node.idempotency,
|
|
684
|
+
input_sha256=input_sha256,
|
|
685
|
+
),
|
|
686
|
+
prerequisite_decisions,
|
|
687
|
+
capability_decisions,
|
|
688
|
+
)
|
|
689
|
+
capability_decisions[capability] = "allowed"
|
|
690
|
+
policy_result = _builtin_pre_entry_policy_result(call, context, input_sha256)
|
|
691
|
+
if policy_result is not None:
|
|
692
|
+
return policy_result, prerequisite_decisions, capability_decisions
|
|
693
|
+
if node.terminal_result is not None:
|
|
694
|
+
mapped_terminal_result = self._terminal_result_map.get(node.node_id)
|
|
695
|
+
if mapped_terminal_result != node.terminal_result:
|
|
696
|
+
return (
|
|
697
|
+
make_denial_result(
|
|
698
|
+
call_id=call.call_id,
|
|
699
|
+
code=ToolExecutionErrorCode.TERMINAL_INTENT_INVALID,
|
|
700
|
+
summary="compiled terminal result map does not match terminal node",
|
|
701
|
+
evidence={
|
|
702
|
+
"node_id": node.node_id,
|
|
703
|
+
"expected": node.terminal_result,
|
|
704
|
+
"actual": str(mapped_terminal_result),
|
|
705
|
+
},
|
|
706
|
+
side_effect_class=node.side_effect_class,
|
|
707
|
+
idempotency=node.idempotency,
|
|
708
|
+
input_sha256=input_sha256,
|
|
709
|
+
),
|
|
710
|
+
prerequisite_decisions,
|
|
711
|
+
capability_decisions,
|
|
712
|
+
)
|
|
713
|
+
terminal_result = call.arguments.get("terminal_result")
|
|
714
|
+
if terminal_result != node.terminal_result:
|
|
715
|
+
return (
|
|
716
|
+
make_denial_result(
|
|
717
|
+
call_id=call.call_id,
|
|
718
|
+
code=ToolExecutionErrorCode.TERMINAL_INTENT_INVALID,
|
|
719
|
+
summary="terminal intent does not match compiled terminal result",
|
|
720
|
+
evidence={
|
|
721
|
+
"expected": node.terminal_result,
|
|
722
|
+
"actual": str(terminal_result),
|
|
723
|
+
},
|
|
724
|
+
side_effect_class=node.side_effect_class,
|
|
725
|
+
idempotency=node.idempotency,
|
|
726
|
+
input_sha256=input_sha256,
|
|
727
|
+
),
|
|
728
|
+
prerequisite_decisions,
|
|
729
|
+
capability_decisions,
|
|
730
|
+
)
|
|
731
|
+
if context.deadline.remaining(lambda: context.current_monotonic) <= 0:
|
|
732
|
+
return (
|
|
733
|
+
make_denial_result(
|
|
734
|
+
call_id=call.call_id,
|
|
735
|
+
code=ToolExecutionErrorCode.TIMEOUT,
|
|
736
|
+
summary="tool deadline expired before implementation entry",
|
|
737
|
+
evidence={"deadline": "expired"},
|
|
738
|
+
side_effect_class=node.side_effect_class,
|
|
739
|
+
idempotency=node.idempotency,
|
|
740
|
+
input_sha256=input_sha256,
|
|
741
|
+
),
|
|
742
|
+
prerequisite_decisions,
|
|
743
|
+
capability_decisions,
|
|
744
|
+
)
|
|
745
|
+
if context.cancellation_requested:
|
|
746
|
+
return (
|
|
747
|
+
make_denial_result(
|
|
748
|
+
call_id=call.call_id,
|
|
749
|
+
code=ToolExecutionErrorCode.CANCELLED,
|
|
750
|
+
summary="tool call was cancelled before implementation entry",
|
|
751
|
+
evidence={"cancellation_id": context.cancellation.cancellation_id},
|
|
752
|
+
side_effect_class=node.side_effect_class,
|
|
753
|
+
idempotency=node.idempotency,
|
|
754
|
+
input_sha256=input_sha256,
|
|
755
|
+
),
|
|
756
|
+
prerequisite_decisions,
|
|
757
|
+
capability_decisions,
|
|
758
|
+
)
|
|
759
|
+
return None
|
|
760
|
+
|
|
761
|
+
def _connector_pre_entry_audit(
|
|
762
|
+
self,
|
|
763
|
+
*,
|
|
764
|
+
node: CompiledHarnessNode,
|
|
765
|
+
context: ToolExecutionContext,
|
|
766
|
+
) -> dict[str, Any] | None:
|
|
767
|
+
if not _is_connector_tool_id(node.binding.tool_id):
|
|
768
|
+
return None
|
|
769
|
+
assert self._connector_admission_snapshot is not None
|
|
770
|
+
admission = self._connector_admission_snapshot.require(
|
|
771
|
+
node.binding.tool_id,
|
|
772
|
+
node.binding.tool_version,
|
|
773
|
+
node.binding.descriptor_sha256,
|
|
774
|
+
)
|
|
775
|
+
return _connector_approval_audit_fields(
|
|
776
|
+
admission=admission,
|
|
777
|
+
node=node,
|
|
778
|
+
context=context,
|
|
779
|
+
broker_attempted=False,
|
|
780
|
+
)
|
|
781
|
+
|
|
782
|
+
def _validate_implementation_result(
|
|
783
|
+
self,
|
|
784
|
+
*,
|
|
785
|
+
node: CompiledHarnessNode,
|
|
786
|
+
call: ValidatedToolCall,
|
|
787
|
+
result: ToolExecutionResult,
|
|
788
|
+
input_sha256: str,
|
|
789
|
+
) -> ToolExecutionResult:
|
|
790
|
+
lookup = self._descriptor_snapshot.resolve_exact(
|
|
791
|
+
node.binding.tool_id,
|
|
792
|
+
node.binding.tool_version,
|
|
793
|
+
)
|
|
794
|
+
output_schema = lookup.entry.output_schema if lookup.entry is not None else {}
|
|
795
|
+
output_error = validate_json_object_schema(
|
|
796
|
+
result.structured_data
|
|
797
|
+
if isinstance(result.structured_data, Mapping)
|
|
798
|
+
else {},
|
|
799
|
+
output_schema,
|
|
800
|
+
)
|
|
801
|
+
if output_error is not None:
|
|
802
|
+
return make_tool_result(
|
|
803
|
+
call_id=call.call_id,
|
|
804
|
+
status=ToolExecutionStatus.HARD_FAILURE,
|
|
805
|
+
code=ToolExecutionErrorCode.OUTPUT_VALIDATION_FAILED,
|
|
806
|
+
summary="tool output failed descriptor output schema validation",
|
|
807
|
+
structured_data={
|
|
808
|
+
"category": ToolExecutionErrorCode.OUTPUT_VALIDATION_FAILED.value,
|
|
809
|
+
"schema_error": output_error,
|
|
810
|
+
},
|
|
811
|
+
side_effect_class=node.side_effect_class,
|
|
812
|
+
idempotency=node.idempotency,
|
|
813
|
+
side_effect_certainty=result.side_effect_certainty,
|
|
814
|
+
side_effect_record=result.side_effect_record,
|
|
815
|
+
input_sha256=input_sha256,
|
|
816
|
+
retryable=False,
|
|
817
|
+
)
|
|
818
|
+
output_policy = self._output_policy_for_node(node)
|
|
819
|
+
return sanitize_tool_execution_result(
|
|
820
|
+
result,
|
|
821
|
+
output_policy=output_policy,
|
|
822
|
+
input_sha256=input_sha256,
|
|
823
|
+
)
|
|
824
|
+
|
|
825
|
+
def _record_trace(
|
|
826
|
+
self,
|
|
827
|
+
*,
|
|
828
|
+
node: CompiledHarnessNode | None,
|
|
829
|
+
call_id: str,
|
|
830
|
+
model_tool_name: str,
|
|
831
|
+
input_sha256: str,
|
|
832
|
+
result: ToolExecutionResult,
|
|
833
|
+
context: ToolExecutionContext,
|
|
834
|
+
prerequisite_decisions: Mapping[str, str],
|
|
835
|
+
capability_decisions: Mapping[str, str],
|
|
836
|
+
connector_audit: Mapping[str, Any] | None,
|
|
837
|
+
model_turn: int,
|
|
838
|
+
session_id: str | None,
|
|
839
|
+
binding_resolution_status: Literal[
|
|
840
|
+
"resolved", "ambiguous", "uncompiled"
|
|
841
|
+
] = "resolved",
|
|
842
|
+
) -> None:
|
|
843
|
+
if connector_audit is not None:
|
|
844
|
+
connector_audit = _connector_trace_audit_fields(
|
|
845
|
+
audit=connector_audit,
|
|
846
|
+
input_sha256=input_sha256,
|
|
847
|
+
result=result,
|
|
848
|
+
)
|
|
849
|
+
trace_node_id = (
|
|
850
|
+
node.node_id
|
|
851
|
+
if node is not None
|
|
852
|
+
else f"binding_resolution::{binding_resolution_status}"
|
|
853
|
+
)
|
|
854
|
+
trace = make_trace_record(
|
|
855
|
+
sequence=self._next_sequence,
|
|
856
|
+
request_id=context.request_id,
|
|
857
|
+
run_id=context.run_id,
|
|
858
|
+
session_id=session_id or context.run_id,
|
|
859
|
+
stage=context.stage,
|
|
860
|
+
node_id=trace_node_id,
|
|
861
|
+
model_turn=model_turn,
|
|
862
|
+
tool_call_id=call_id,
|
|
863
|
+
model_tool_name=model_tool_name,
|
|
864
|
+
binding=(
|
|
865
|
+
node.binding
|
|
866
|
+
if node is not None
|
|
867
|
+
else _binding_resolution_trace_binding(
|
|
868
|
+
model_tool_name=model_tool_name,
|
|
869
|
+
binding_resolution_status=binding_resolution_status,
|
|
870
|
+
)
|
|
871
|
+
),
|
|
872
|
+
binding_resolution_status=binding_resolution_status,
|
|
873
|
+
input_sha256=input_sha256,
|
|
874
|
+
prerequisite_decisions={
|
|
875
|
+
key: ToolTraceDecision(value)
|
|
876
|
+
for key, value in prerequisite_decisions.items()
|
|
877
|
+
},
|
|
878
|
+
capability_decisions={
|
|
879
|
+
key: ToolTraceDecision(value)
|
|
880
|
+
for key, value in capability_decisions.items()
|
|
881
|
+
},
|
|
882
|
+
result=result,
|
|
883
|
+
connector_audit=connector_audit,
|
|
884
|
+
summary_max_utf8=(
|
|
885
|
+
self._summary_max_utf8_for_node(node)
|
|
886
|
+
if node is not None
|
|
887
|
+
else MAX_MODEL_SUMMARY_UTF8
|
|
888
|
+
),
|
|
889
|
+
)
|
|
890
|
+
self._trace_records.append(trace)
|
|
891
|
+
self._next_sequence += 1
|
|
892
|
+
|
|
893
|
+
def _summary_max_utf8_for_node(self, node: CompiledHarnessNode) -> int:
|
|
894
|
+
output_policy = self._output_policy_for_node(node)
|
|
895
|
+
if output_policy is None:
|
|
896
|
+
return MAX_MODEL_SUMMARY_UTF8
|
|
897
|
+
return output_policy.max_summary_utf8
|
|
898
|
+
|
|
899
|
+
def _output_policy_for_node(
|
|
900
|
+
self, node: CompiledHarnessNode
|
|
901
|
+
) -> ToolOutputPolicy | None:
|
|
902
|
+
lookup = self._descriptor_snapshot.resolve_exact(
|
|
903
|
+
node.binding.tool_id,
|
|
904
|
+
node.binding.tool_version,
|
|
905
|
+
)
|
|
906
|
+
if lookup.entry is None or lookup.entry.output_policy is None:
|
|
907
|
+
return None
|
|
908
|
+
if not isinstance(lookup.entry.output_policy, ToolOutputPolicy):
|
|
909
|
+
raise ValueError(
|
|
910
|
+
"descriptor snapshot output_policy must be ToolOutputPolicy"
|
|
911
|
+
)
|
|
912
|
+
return lookup.entry.output_policy
|
|
913
|
+
|
|
914
|
+
|
|
915
|
+
def _builtin_pre_entry_policy_result(
|
|
916
|
+
call: ValidatedToolCall,
|
|
917
|
+
context: ToolExecutionContext,
|
|
918
|
+
input_sha256: str,
|
|
919
|
+
) -> ToolExecutionResult | None:
|
|
920
|
+
"""Lazily consult the built-in pre-entry policy hook.
|
|
921
|
+
|
|
922
|
+
The lazy import avoids a module cycle with ``builtin_runtime`` while keeping
|
|
923
|
+
the executor-owned policy gate in the runtime boundary.
|
|
924
|
+
"""
|
|
925
|
+
if not call.binding.tool_id.startswith("builtin."):
|
|
926
|
+
return None
|
|
927
|
+
from millforge.tools.builtin_runtime import validate_builtin_pre_entry_policy
|
|
928
|
+
|
|
929
|
+
return validate_builtin_pre_entry_policy(
|
|
930
|
+
call,
|
|
931
|
+
context,
|
|
932
|
+
input_sha256=input_sha256,
|
|
933
|
+
)
|
|
934
|
+
|
|
935
|
+
|
|
936
|
+
def create_tool_executor(
|
|
937
|
+
*,
|
|
938
|
+
plan: CompiledHarnessPlan,
|
|
939
|
+
descriptor_snapshot: ToolCatalogSnapshot,
|
|
940
|
+
runtime_registry: RuntimeToolRegistry,
|
|
941
|
+
connector_admission_snapshot: Any | None = None,
|
|
942
|
+
connector_broker: Any | None = None,
|
|
943
|
+
) -> CompiledToolBindingExecutor:
|
|
944
|
+
"""Create a compiled-plan scoped executor."""
|
|
945
|
+
return CompiledToolBindingExecutor(
|
|
946
|
+
plan=plan,
|
|
947
|
+
descriptor_snapshot=descriptor_snapshot,
|
|
948
|
+
runtime_registry=runtime_registry,
|
|
949
|
+
connector_admission_snapshot=connector_admission_snapshot,
|
|
950
|
+
connector_broker=connector_broker,
|
|
951
|
+
)
|
|
952
|
+
|
|
953
|
+
|
|
954
|
+
def _validate_connector_admissions(
|
|
955
|
+
*,
|
|
956
|
+
plan: CompiledHarnessPlan,
|
|
957
|
+
connector_admission_snapshot: Any | None,
|
|
958
|
+
connector_broker: Any | None,
|
|
959
|
+
) -> None:
|
|
960
|
+
connector_nodes = [
|
|
961
|
+
node for node in plan.nodes if _is_connector_tool_id(node.binding.tool_id)
|
|
962
|
+
]
|
|
963
|
+
if not connector_nodes:
|
|
964
|
+
return
|
|
965
|
+
if connector_admission_snapshot is None:
|
|
966
|
+
from millforge.connectors.runtime import ConnectorAdmissionSnapshotError
|
|
967
|
+
|
|
968
|
+
raise ConnectorAdmissionSnapshotError(
|
|
969
|
+
"compiled connector descriptors require connector admission snapshot"
|
|
970
|
+
)
|
|
971
|
+
if connector_broker is None:
|
|
972
|
+
from millforge.connectors.runtime import ConnectorAdmissionSnapshotError
|
|
973
|
+
|
|
974
|
+
raise ConnectorAdmissionSnapshotError(
|
|
975
|
+
"compiled connector descriptors require connector broker"
|
|
976
|
+
)
|
|
977
|
+
for node in connector_nodes:
|
|
978
|
+
admission = connector_admission_snapshot.require(
|
|
979
|
+
node.binding.tool_id,
|
|
980
|
+
node.binding.tool_version,
|
|
981
|
+
node.binding.descriptor_sha256,
|
|
982
|
+
)
|
|
983
|
+
mismatch = _connector_admission_node_mismatch(admission, node)
|
|
984
|
+
if mismatch is not None:
|
|
985
|
+
from millforge.connectors.runtime import ConnectorAdmissionSnapshotError
|
|
986
|
+
|
|
987
|
+
raise ConnectorAdmissionSnapshotError(
|
|
988
|
+
"compiled connector descriptor is inconsistent with admission binding: "
|
|
989
|
+
f"{mismatch}"
|
|
990
|
+
)
|
|
991
|
+
|
|
992
|
+
|
|
993
|
+
def _is_connector_tool_id(tool_id: str) -> bool:
|
|
994
|
+
return tool_id.startswith("connector.")
|
|
995
|
+
|
|
996
|
+
|
|
997
|
+
def _connector_pre_entry_drift(
|
|
998
|
+
*,
|
|
999
|
+
admission: Any,
|
|
1000
|
+
node: CompiledHarnessNode,
|
|
1001
|
+
broker: Any,
|
|
1002
|
+
) -> tuple[str, str, str] | None:
|
|
1003
|
+
admitted_capabilities = tuple(admission.required_capabilities)
|
|
1004
|
+
compiled_capabilities = tuple(node.required_capabilities)
|
|
1005
|
+
if compiled_capabilities != admitted_capabilities:
|
|
1006
|
+
return (
|
|
1007
|
+
"required_capabilities",
|
|
1008
|
+
_evidence_text(admitted_capabilities),
|
|
1009
|
+
_evidence_text(compiled_capabilities),
|
|
1010
|
+
)
|
|
1011
|
+
evidence_factory = getattr(broker, "provider_tool_evidence", None)
|
|
1012
|
+
if not callable(evidence_factory):
|
|
1013
|
+
return ("provider_tool_evidence", "present", "missing")
|
|
1014
|
+
evidence = evidence_factory(admission.connector_id, admission.provider_tool_name)
|
|
1015
|
+
if evidence is None:
|
|
1016
|
+
return ("provider_tool_evidence", "present", "missing")
|
|
1017
|
+
expected_values = {
|
|
1018
|
+
"connector_id": admission.connector_id,
|
|
1019
|
+
"provider_tool_name": admission.provider_tool_name,
|
|
1020
|
+
"connector_identity_sha256": admission.connector_identity_sha256,
|
|
1021
|
+
"discovery_snapshot_sha256": admission.discovery_snapshot_sha256,
|
|
1022
|
+
"raw_tool_sha256": admission.raw_tool_sha256,
|
|
1023
|
+
"input_schema_sha256": admission.input_schema_sha256,
|
|
1024
|
+
"output_schema_sha256": admission.output_schema_sha256,
|
|
1025
|
+
"provider_description_sha256": admission.provider_description_sha256,
|
|
1026
|
+
}
|
|
1027
|
+
for field, expected in expected_values.items():
|
|
1028
|
+
if expected is None:
|
|
1029
|
+
continue
|
|
1030
|
+
actual = getattr(evidence, field, None)
|
|
1031
|
+
if actual != expected:
|
|
1032
|
+
return (field, _evidence_text(expected), _evidence_text(actual))
|
|
1033
|
+
return None
|
|
1034
|
+
|
|
1035
|
+
|
|
1036
|
+
def _connector_admission_node_mismatch(
|
|
1037
|
+
admission: Any, node: CompiledHarnessNode
|
|
1038
|
+
) -> str | None:
|
|
1039
|
+
expected = {
|
|
1040
|
+
"side_effect_class": admission.side_effect_class,
|
|
1041
|
+
"idempotency": admission.idempotency,
|
|
1042
|
+
}
|
|
1043
|
+
actual = {
|
|
1044
|
+
"side_effect_class": node.side_effect_class,
|
|
1045
|
+
"idempotency": node.idempotency,
|
|
1046
|
+
}
|
|
1047
|
+
for field, expected_value in expected.items():
|
|
1048
|
+
if not _projection_values_equal(actual[field], expected_value):
|
|
1049
|
+
return field
|
|
1050
|
+
return None
|
|
1051
|
+
|
|
1052
|
+
|
|
1053
|
+
def _connector_approval_identity_kind(grant: Any | None) -> str:
|
|
1054
|
+
if grant is None:
|
|
1055
|
+
return "none"
|
|
1056
|
+
has_approval_id = getattr(grant, "approval_id", None) is not None
|
|
1057
|
+
has_nonce = getattr(grant, "nonce", None) is not None
|
|
1058
|
+
if has_approval_id and has_nonce:
|
|
1059
|
+
return "approval_id_and_nonce"
|
|
1060
|
+
if has_approval_id:
|
|
1061
|
+
return "approval_id"
|
|
1062
|
+
if has_nonce:
|
|
1063
|
+
return "nonce"
|
|
1064
|
+
return "none"
|
|
1065
|
+
|
|
1066
|
+
|
|
1067
|
+
def _connector_approval_evidence(
|
|
1068
|
+
grants: tuple[Any, ...],
|
|
1069
|
+
representative: Any | None = None,
|
|
1070
|
+
) -> dict[str, Any]:
|
|
1071
|
+
representative_grant = (
|
|
1072
|
+
representative
|
|
1073
|
+
if representative is not None
|
|
1074
|
+
else (grants[0] if grants else None)
|
|
1075
|
+
)
|
|
1076
|
+
return {
|
|
1077
|
+
"grant_count": len(grants),
|
|
1078
|
+
"approval_id_present": any(
|
|
1079
|
+
getattr(grant, "approval_id", None) is not None for grant in grants
|
|
1080
|
+
),
|
|
1081
|
+
"nonce_present": any(
|
|
1082
|
+
getattr(grant, "nonce", None) is not None for grant in grants
|
|
1083
|
+
),
|
|
1084
|
+
"identity_kind": _connector_approval_identity_kind(representative_grant),
|
|
1085
|
+
}
|
|
1086
|
+
|
|
1087
|
+
|
|
1088
|
+
def _connector_explicit_approval_assessment(
|
|
1089
|
+
*,
|
|
1090
|
+
admission: Any,
|
|
1091
|
+
node: CompiledHarnessNode,
|
|
1092
|
+
context: ToolExecutionContext,
|
|
1093
|
+
) -> tuple[str, dict[str, Any]]:
|
|
1094
|
+
grants = context.connector_approval_grants
|
|
1095
|
+
if not grants:
|
|
1096
|
+
return "missing", _connector_approval_evidence(grants)
|
|
1097
|
+
expired_scope_seen = False
|
|
1098
|
+
wrong_stage_seen = False
|
|
1099
|
+
wrong_run_seen = False
|
|
1100
|
+
wrong_scope_seen = False
|
|
1101
|
+
for grant in grants:
|
|
1102
|
+
scope_matches = (
|
|
1103
|
+
grant.connector_id == admission.connector_id
|
|
1104
|
+
and grant.provider_tool_name == admission.provider_tool_name
|
|
1105
|
+
and grant.tool_id == node.binding.tool_id
|
|
1106
|
+
and grant.tool_version == node.binding.tool_version
|
|
1107
|
+
and grant.descriptor_sha256 == node.binding.descriptor_sha256
|
|
1108
|
+
and grant.request_id == context.request_id
|
|
1109
|
+
and grant.run_id == context.run_id
|
|
1110
|
+
and _stage_identity_matches(grant.stage, context.stage)
|
|
1111
|
+
and grant.approval_policy == admission.approval_policy.value
|
|
1112
|
+
)
|
|
1113
|
+
if scope_matches:
|
|
1114
|
+
if grant.expires_at_monotonic <= context.current_monotonic:
|
|
1115
|
+
expired_scope_seen = True
|
|
1116
|
+
continue
|
|
1117
|
+
return "approved", _connector_approval_evidence(grants, grant)
|
|
1118
|
+
wrong_stage_seen = wrong_stage_seen or (
|
|
1119
|
+
grant.connector_id == admission.connector_id
|
|
1120
|
+
and grant.provider_tool_name == admission.provider_tool_name
|
|
1121
|
+
and grant.tool_id == node.binding.tool_id
|
|
1122
|
+
and grant.tool_version == node.binding.tool_version
|
|
1123
|
+
and grant.descriptor_sha256 == node.binding.descriptor_sha256
|
|
1124
|
+
and grant.request_id == context.request_id
|
|
1125
|
+
and grant.run_id == context.run_id
|
|
1126
|
+
and not _stage_identity_matches(grant.stage, context.stage)
|
|
1127
|
+
)
|
|
1128
|
+
wrong_run_seen = wrong_run_seen or (
|
|
1129
|
+
grant.connector_id == admission.connector_id
|
|
1130
|
+
and grant.provider_tool_name == admission.provider_tool_name
|
|
1131
|
+
and grant.tool_id == node.binding.tool_id
|
|
1132
|
+
and grant.tool_version == node.binding.tool_version
|
|
1133
|
+
and grant.descriptor_sha256 == node.binding.descriptor_sha256
|
|
1134
|
+
and grant.request_id == context.request_id
|
|
1135
|
+
and grant.run_id != context.run_id
|
|
1136
|
+
)
|
|
1137
|
+
wrong_scope_seen = wrong_scope_seen or (
|
|
1138
|
+
grant.approval_policy == admission.approval_policy.value
|
|
1139
|
+
)
|
|
1140
|
+
if expired_scope_seen:
|
|
1141
|
+
return "expired_or_stale", _connector_approval_evidence(grants)
|
|
1142
|
+
if wrong_stage_seen:
|
|
1143
|
+
return "wrong_stage", _connector_approval_evidence(grants)
|
|
1144
|
+
if wrong_run_seen:
|
|
1145
|
+
return "wrong_run", _connector_approval_evidence(grants)
|
|
1146
|
+
if wrong_scope_seen:
|
|
1147
|
+
return "wrong_scope", _connector_approval_evidence(grants)
|
|
1148
|
+
return "missing", _connector_approval_evidence(grants)
|
|
1149
|
+
|
|
1150
|
+
|
|
1151
|
+
def _connector_approval_audit_fields(
|
|
1152
|
+
*,
|
|
1153
|
+
admission: Any,
|
|
1154
|
+
node: CompiledHarnessNode,
|
|
1155
|
+
context: ToolExecutionContext,
|
|
1156
|
+
broker_attempted: bool,
|
|
1157
|
+
) -> dict[str, Any]:
|
|
1158
|
+
from millforge.connectors import ConnectorApprovalPolicy
|
|
1159
|
+
|
|
1160
|
+
base = {
|
|
1161
|
+
"connector_id": admission.connector_id,
|
|
1162
|
+
"provider_tool_name": admission.provider_tool_name,
|
|
1163
|
+
"connector_tool_id": node.binding.tool_id,
|
|
1164
|
+
"connector_tool_version": node.binding.tool_version,
|
|
1165
|
+
"connector_descriptor_sha256": node.binding.descriptor_sha256,
|
|
1166
|
+
"connector_identity_sha256": admission.connector_identity_sha256,
|
|
1167
|
+
"discovery_snapshot_sha256": admission.discovery_snapshot_sha256,
|
|
1168
|
+
"approval_policy": admission.approval_policy.value,
|
|
1169
|
+
"broker_attempted": broker_attempted,
|
|
1170
|
+
"drift_decision": "not_reached",
|
|
1171
|
+
}
|
|
1172
|
+
if admission.approval_policy is ConnectorApprovalPolicy.NONE:
|
|
1173
|
+
decision = "approved"
|
|
1174
|
+
evidence = _connector_approval_evidence(context.connector_approval_grants)
|
|
1175
|
+
elif admission.approval_policy is ConnectorApprovalPolicy.FORBIDDEN:
|
|
1176
|
+
decision = "forbidden"
|
|
1177
|
+
evidence = _connector_approval_evidence(context.connector_approval_grants)
|
|
1178
|
+
elif admission.approval_policy is ConnectorApprovalPolicy.OPERATOR_OUT_OF_BAND:
|
|
1179
|
+
decision = "pending"
|
|
1180
|
+
evidence = _connector_approval_evidence(context.connector_approval_grants)
|
|
1181
|
+
elif admission.approval_policy is ConnectorApprovalPolicy.MILLRACE_EXPLICIT:
|
|
1182
|
+
decision, evidence = _connector_explicit_approval_assessment(
|
|
1183
|
+
admission=admission,
|
|
1184
|
+
node=node,
|
|
1185
|
+
context=context,
|
|
1186
|
+
)
|
|
1187
|
+
else:
|
|
1188
|
+
decision = "missing"
|
|
1189
|
+
evidence = _connector_approval_evidence(context.connector_approval_grants)
|
|
1190
|
+
return {
|
|
1191
|
+
**base,
|
|
1192
|
+
"approval_decision": decision,
|
|
1193
|
+
"approval_evidence": evidence,
|
|
1194
|
+
}
|
|
1195
|
+
|
|
1196
|
+
|
|
1197
|
+
def _connector_trace_audit_fields(
|
|
1198
|
+
*,
|
|
1199
|
+
audit: Mapping[str, Any],
|
|
1200
|
+
input_sha256: str,
|
|
1201
|
+
result: ToolExecutionResult,
|
|
1202
|
+
) -> dict[str, Any]:
|
|
1203
|
+
result_evidence = {}
|
|
1204
|
+
if isinstance(result.structured_data, Mapping):
|
|
1205
|
+
raw_evidence = result.structured_data.get("evidence")
|
|
1206
|
+
if isinstance(raw_evidence, Mapping):
|
|
1207
|
+
result_evidence = {
|
|
1208
|
+
key: raw_evidence[key]
|
|
1209
|
+
for key in ("binding_field", "expected", "actual")
|
|
1210
|
+
if key in raw_evidence
|
|
1211
|
+
}
|
|
1212
|
+
redacted_evidence = redact_tool_value(
|
|
1213
|
+
{
|
|
1214
|
+
"approval_decision": audit.get("approval_decision"),
|
|
1215
|
+
"approval_evidence": audit.get("approval_evidence", {}),
|
|
1216
|
+
"broker_attempted": audit.get("broker_attempted"),
|
|
1217
|
+
"drift_decision": audit.get("drift_decision"),
|
|
1218
|
+
**result_evidence,
|
|
1219
|
+
"error_code": result.error_code,
|
|
1220
|
+
"execution_status": result.status.value,
|
|
1221
|
+
"side_effect_certainty": result.side_effect_certainty.value,
|
|
1222
|
+
"summary": result.summary,
|
|
1223
|
+
}
|
|
1224
|
+
)
|
|
1225
|
+
return {
|
|
1226
|
+
**audit,
|
|
1227
|
+
"request_sha256": input_sha256,
|
|
1228
|
+
"response_sha256": canonical_sha256(redact_tool_value(result.structured_data)),
|
|
1229
|
+
"retry_decision": "retry_allowed" if result.retryable else "retry_denied",
|
|
1230
|
+
"redacted_evidence": redacted_evidence,
|
|
1231
|
+
}
|
|
1232
|
+
|
|
1233
|
+
|
|
1234
|
+
def _connector_result_from_broker_outcome(
|
|
1235
|
+
*,
|
|
1236
|
+
call_id: str,
|
|
1237
|
+
node: CompiledHarnessNode,
|
|
1238
|
+
outcome: Any,
|
|
1239
|
+
input_sha256: str,
|
|
1240
|
+
idempotency_key_policy: str | None,
|
|
1241
|
+
idempotency_key: str | None,
|
|
1242
|
+
) -> ToolExecutionResult:
|
|
1243
|
+
status = outcome.status
|
|
1244
|
+
code = _connector_error_code_for_outcome(outcome)
|
|
1245
|
+
certainty = outcome.side_effect_certainty
|
|
1246
|
+
retryable = _connector_retryable_for_outcome(
|
|
1247
|
+
status=status,
|
|
1248
|
+
certainty=certainty,
|
|
1249
|
+
idempotency=node.idempotency,
|
|
1250
|
+
requested_retryable=outcome.retryable,
|
|
1251
|
+
idempotency_key_policy=idempotency_key_policy,
|
|
1252
|
+
idempotency_key=idempotency_key,
|
|
1253
|
+
)
|
|
1254
|
+
side_effect_record = _connector_side_effect_record(
|
|
1255
|
+
status=status,
|
|
1256
|
+
code=code,
|
|
1257
|
+
certainty=certainty,
|
|
1258
|
+
summary=outcome.summary,
|
|
1259
|
+
retryable=retryable,
|
|
1260
|
+
)
|
|
1261
|
+
return make_tool_result(
|
|
1262
|
+
call_id=call_id,
|
|
1263
|
+
status=status,
|
|
1264
|
+
code=code,
|
|
1265
|
+
summary=outcome.summary,
|
|
1266
|
+
structured_data=outcome.structured_data,
|
|
1267
|
+
side_effect_class=node.side_effect_class,
|
|
1268
|
+
idempotency=node.idempotency,
|
|
1269
|
+
side_effect_certainty=certainty,
|
|
1270
|
+
side_effect_record=side_effect_record,
|
|
1271
|
+
input_sha256=input_sha256,
|
|
1272
|
+
retryable=retryable,
|
|
1273
|
+
)
|
|
1274
|
+
|
|
1275
|
+
|
|
1276
|
+
def _connector_error_code_for_outcome(outcome: Any) -> ToolExecutionErrorCode | None:
|
|
1277
|
+
if outcome.status is ToolExecutionStatus.SUCCESS:
|
|
1278
|
+
return None
|
|
1279
|
+
if outcome.error_code is not None:
|
|
1280
|
+
return outcome.error_code
|
|
1281
|
+
if outcome.status is ToolExecutionStatus.TIMED_OUT:
|
|
1282
|
+
return ToolExecutionErrorCode.TIMEOUT
|
|
1283
|
+
if outcome.status is ToolExecutionStatus.CANCELLED:
|
|
1284
|
+
return ToolExecutionErrorCode.CANCELLED
|
|
1285
|
+
if outcome.status is ToolExecutionStatus.AMBIGUOUS:
|
|
1286
|
+
return ToolExecutionErrorCode.AMBIGUOUS_SIDE_EFFECT
|
|
1287
|
+
return ToolExecutionErrorCode.IMPLEMENTATION_ERROR
|
|
1288
|
+
|
|
1289
|
+
|
|
1290
|
+
def _connector_retryable_for_outcome(
|
|
1291
|
+
*,
|
|
1292
|
+
status: ToolExecutionStatus,
|
|
1293
|
+
certainty: SideEffectCertainty,
|
|
1294
|
+
idempotency: Any,
|
|
1295
|
+
requested_retryable: bool,
|
|
1296
|
+
idempotency_key_policy: str | None,
|
|
1297
|
+
idempotency_key: str | None,
|
|
1298
|
+
) -> bool:
|
|
1299
|
+
if status is ToolExecutionStatus.SUCCESS:
|
|
1300
|
+
return False
|
|
1301
|
+
if not requested_retryable:
|
|
1302
|
+
return False
|
|
1303
|
+
from millforge import IdempotencyClass
|
|
1304
|
+
|
|
1305
|
+
has_safe_idempotency_key = (
|
|
1306
|
+
idempotency is IdempotencyClass.IDEMPOTENT_WITH_KEY
|
|
1307
|
+
and idempotency_key_policy == "call_id"
|
|
1308
|
+
and idempotency_key is not None
|
|
1309
|
+
)
|
|
1310
|
+
if certainty is SideEffectCertainty.CONFIRMED_COMPLETE:
|
|
1311
|
+
return has_safe_idempotency_key
|
|
1312
|
+
if idempotency is IdempotencyClass.IDEMPOTENT:
|
|
1313
|
+
return True
|
|
1314
|
+
return has_safe_idempotency_key
|
|
1315
|
+
|
|
1316
|
+
|
|
1317
|
+
def _connector_side_effect_record(
|
|
1318
|
+
*,
|
|
1319
|
+
status: ToolExecutionStatus,
|
|
1320
|
+
code: ToolExecutionErrorCode | None,
|
|
1321
|
+
certainty: SideEffectCertainty,
|
|
1322
|
+
summary: str,
|
|
1323
|
+
retryable: bool,
|
|
1324
|
+
) -> SideEffectRecord | None:
|
|
1325
|
+
if status is ToolExecutionStatus.SUCCESS:
|
|
1326
|
+
return None
|
|
1327
|
+
detail_code = (
|
|
1328
|
+
code.value
|
|
1329
|
+
if code is not None
|
|
1330
|
+
else ToolExecutionErrorCode.IMPLEMENTATION_ERROR.value
|
|
1331
|
+
)
|
|
1332
|
+
return SideEffectRecord(
|
|
1333
|
+
certainty=certainty,
|
|
1334
|
+
detail_code=detail_code,
|
|
1335
|
+
summary=f"connector broker outcome: {summary}",
|
|
1336
|
+
retry_allowed=retryable,
|
|
1337
|
+
)
|
|
1338
|
+
|
|
1339
|
+
|
|
1340
|
+
def _connector_explicit_approval_denial(
|
|
1341
|
+
*,
|
|
1342
|
+
admission: Any,
|
|
1343
|
+
node: CompiledHarnessNode,
|
|
1344
|
+
context: ToolExecutionContext,
|
|
1345
|
+
) -> dict[str, Any] | None:
|
|
1346
|
+
audit = _connector_approval_audit_fields(
|
|
1347
|
+
admission=admission,
|
|
1348
|
+
node=node,
|
|
1349
|
+
context=context,
|
|
1350
|
+
broker_attempted=False,
|
|
1351
|
+
)
|
|
1352
|
+
if audit["approval_decision"] == "approved":
|
|
1353
|
+
return None
|
|
1354
|
+
return audit
|
|
1355
|
+
|
|
1356
|
+
|
|
1357
|
+
def _stage_identity_matches(left: Any, right: Any) -> bool:
|
|
1358
|
+
return (
|
|
1359
|
+
getattr(left, "plane", None) == getattr(right, "plane", None)
|
|
1360
|
+
and getattr(left, "stage_kind_id", None)
|
|
1361
|
+
== getattr(right, "stage_kind_id", None)
|
|
1362
|
+
and getattr(left, "node_id", None) == getattr(right, "node_id", None)
|
|
1363
|
+
)
|
|
1364
|
+
|
|
1365
|
+
|
|
1366
|
+
def _node_binding_defect(
|
|
1367
|
+
*,
|
|
1368
|
+
node: CompiledHarnessNode,
|
|
1369
|
+
descriptor_snapshot: ToolCatalogSnapshot,
|
|
1370
|
+
runtime_registry: RuntimeToolRegistry,
|
|
1371
|
+
) -> tuple[ToolBindingDenialCode, str, dict[str, str], bool] | None:
|
|
1372
|
+
lookup = descriptor_snapshot.resolve_exact(
|
|
1373
|
+
node.binding.tool_id,
|
|
1374
|
+
node.binding.tool_version,
|
|
1375
|
+
)
|
|
1376
|
+
if lookup.entry is None:
|
|
1377
|
+
return (
|
|
1378
|
+
ToolBindingDenialCode.NOT_FOUND,
|
|
1379
|
+
"compiled binding is absent from descriptor snapshot",
|
|
1380
|
+
{
|
|
1381
|
+
"binding_field": "tool_id/tool_version",
|
|
1382
|
+
"tool_id": node.binding.tool_id,
|
|
1383
|
+
"tool_version": str(node.binding.tool_version),
|
|
1384
|
+
},
|
|
1385
|
+
False,
|
|
1386
|
+
)
|
|
1387
|
+
entry = lookup.entry
|
|
1388
|
+
if entry.output_policy is not None and not isinstance(
|
|
1389
|
+
entry.output_policy, ToolOutputPolicy
|
|
1390
|
+
):
|
|
1391
|
+
return (
|
|
1392
|
+
ToolBindingDenialCode.BINDING_MISMATCH,
|
|
1393
|
+
"descriptor snapshot output policy is malformed",
|
|
1394
|
+
{"binding_field": "output_policy"},
|
|
1395
|
+
True,
|
|
1396
|
+
)
|
|
1397
|
+
expected: dict[str, Any] = {
|
|
1398
|
+
"descriptor_sha256": entry.descriptor_sha256,
|
|
1399
|
+
"implementation_id": entry.implementation_id,
|
|
1400
|
+
"model_tool_name": entry.model_tool_name,
|
|
1401
|
+
"input_schema": entry.input_schema,
|
|
1402
|
+
"required_capabilities": tuple(entry.required_capabilities),
|
|
1403
|
+
"produced_artifact_ids": tuple(entry.produced_artifact_ids),
|
|
1404
|
+
"side_effect_class": entry.side_effect_class,
|
|
1405
|
+
"idempotency": entry.idempotency,
|
|
1406
|
+
}
|
|
1407
|
+
actual: dict[str, Any] = {
|
|
1408
|
+
"descriptor_sha256": node.binding.descriptor_sha256,
|
|
1409
|
+
"implementation_id": node.binding.implementation_id,
|
|
1410
|
+
"model_tool_name": node.model_tool_name,
|
|
1411
|
+
"input_schema": node.input_schema,
|
|
1412
|
+
"required_capabilities": tuple(node.required_capabilities),
|
|
1413
|
+
"produced_artifact_ids": tuple(node.produced_artifact_ids),
|
|
1414
|
+
"side_effect_class": node.side_effect_class,
|
|
1415
|
+
"idempotency": node.idempotency,
|
|
1416
|
+
}
|
|
1417
|
+
connector_tool = _is_connector_tool_id(node.binding.tool_id)
|
|
1418
|
+
for field, expected_value in expected.items():
|
|
1419
|
+
if connector_tool and field == "required_capabilities":
|
|
1420
|
+
continue
|
|
1421
|
+
if not _projection_values_equal(actual[field], expected_value):
|
|
1422
|
+
return (
|
|
1423
|
+
ToolBindingDenialCode.BINDING_MISMATCH,
|
|
1424
|
+
"compiled node projection does not match descriptor snapshot",
|
|
1425
|
+
{
|
|
1426
|
+
"binding_field": field,
|
|
1427
|
+
"expected": _evidence_text(expected_value),
|
|
1428
|
+
"actual": _evidence_text(actual[field]),
|
|
1429
|
+
},
|
|
1430
|
+
False,
|
|
1431
|
+
)
|
|
1432
|
+
if connector_tool:
|
|
1433
|
+
return None
|
|
1434
|
+
if runtime_registry.resolve(node.binding.implementation_id) is None:
|
|
1435
|
+
return (
|
|
1436
|
+
ToolBindingDenialCode.NOT_FOUND,
|
|
1437
|
+
"runtime implementation is not registered",
|
|
1438
|
+
{
|
|
1439
|
+
"binding_field": "implementation_id",
|
|
1440
|
+
"implementation_id": node.binding.implementation_id,
|
|
1441
|
+
},
|
|
1442
|
+
True,
|
|
1443
|
+
)
|
|
1444
|
+
return None
|
|
1445
|
+
|
|
1446
|
+
|
|
1447
|
+
def _call_binding_mismatch(
|
|
1448
|
+
call: ValidatedToolCall,
|
|
1449
|
+
node: CompiledHarnessNode,
|
|
1450
|
+
) -> dict[str, str] | None:
|
|
1451
|
+
for field in (
|
|
1452
|
+
"tool_id",
|
|
1453
|
+
"tool_version",
|
|
1454
|
+
"descriptor_sha256",
|
|
1455
|
+
"implementation_id",
|
|
1456
|
+
):
|
|
1457
|
+
actual = getattr(call.binding, field)
|
|
1458
|
+
expected = getattr(node.binding, field)
|
|
1459
|
+
if actual != expected:
|
|
1460
|
+
return {
|
|
1461
|
+
"binding_field": field,
|
|
1462
|
+
"expected": _evidence_text(expected),
|
|
1463
|
+
"actual": _evidence_text(actual),
|
|
1464
|
+
}
|
|
1465
|
+
return None
|
|
1466
|
+
|
|
1467
|
+
|
|
1468
|
+
def _denial_result(
|
|
1469
|
+
*,
|
|
1470
|
+
call_id: str,
|
|
1471
|
+
input_sha256: str,
|
|
1472
|
+
code: ToolBindingDenialCode,
|
|
1473
|
+
summary: str,
|
|
1474
|
+
evidence: Mapping[str, str],
|
|
1475
|
+
node: CompiledHarnessNode | None = None,
|
|
1476
|
+
status: ToolExecutionStatus | None = None,
|
|
1477
|
+
) -> ToolExecutionResult:
|
|
1478
|
+
result_code = ToolExecutionErrorCode(code.value)
|
|
1479
|
+
return make_denial_result(
|
|
1480
|
+
call_id=call_id,
|
|
1481
|
+
code=result_code,
|
|
1482
|
+
summary=summary,
|
|
1483
|
+
side_effect_class=node.side_effect_class
|
|
1484
|
+
if node is not None
|
|
1485
|
+
else SideEffectClass.READ_ONLY,
|
|
1486
|
+
idempotency=node.idempotency if node is not None else _default_idempotency(),
|
|
1487
|
+
input_sha256=input_sha256,
|
|
1488
|
+
evidence={key: _evidence_text(value) for key, value in evidence.items()},
|
|
1489
|
+
status=status,
|
|
1490
|
+
)
|
|
1491
|
+
|
|
1492
|
+
|
|
1493
|
+
def _default_idempotency() -> Any:
|
|
1494
|
+
from millforge import IdempotencyClass
|
|
1495
|
+
|
|
1496
|
+
return IdempotencyClass.IDEMPOTENT
|
|
1497
|
+
|
|
1498
|
+
|
|
1499
|
+
def _binding_resolution_trace_binding(
|
|
1500
|
+
*,
|
|
1501
|
+
model_tool_name: str,
|
|
1502
|
+
binding_resolution_status: str,
|
|
1503
|
+
) -> ToolBindingRef:
|
|
1504
|
+
return ToolBindingRef(
|
|
1505
|
+
tool_id=model_tool_name,
|
|
1506
|
+
tool_version=1,
|
|
1507
|
+
descriptor_sha256="0" * 64,
|
|
1508
|
+
implementation_id=f"binding-resolution::{binding_resolution_status}",
|
|
1509
|
+
)
|
|
1510
|
+
|
|
1511
|
+
|
|
1512
|
+
def _duplicate_values(values: Any) -> frozenset[str]:
|
|
1513
|
+
seen: set[str] = set()
|
|
1514
|
+
duplicates: set[str] = set()
|
|
1515
|
+
for value in values:
|
|
1516
|
+
if value in seen:
|
|
1517
|
+
duplicates.add(value)
|
|
1518
|
+
seen.add(value)
|
|
1519
|
+
return frozenset(duplicates)
|
|
1520
|
+
|
|
1521
|
+
|
|
1522
|
+
def _projection_values_equal(left: Any, right: Any) -> bool:
|
|
1523
|
+
if isinstance(left, Mapping) or isinstance(right, Mapping):
|
|
1524
|
+
return canonical_json_serialize(_json_value(left)) == canonical_json_serialize(
|
|
1525
|
+
_json_value(right)
|
|
1526
|
+
)
|
|
1527
|
+
return left == right
|
|
1528
|
+
|
|
1529
|
+
|
|
1530
|
+
def _evidence_text(value: Any) -> str:
|
|
1531
|
+
value = _json_value(value)
|
|
1532
|
+
text = (
|
|
1533
|
+
canonical_json_serialize(value).strip()
|
|
1534
|
+
if isinstance(value, Mapping)
|
|
1535
|
+
else str(value)
|
|
1536
|
+
)
|
|
1537
|
+
return text[:512]
|
|
1538
|
+
|
|
1539
|
+
|
|
1540
|
+
def _json_value(value: Any) -> Any:
|
|
1541
|
+
if isinstance(value, Mapping):
|
|
1542
|
+
return {key: _json_value(item) for key, item in value.items()}
|
|
1543
|
+
if isinstance(value, tuple):
|
|
1544
|
+
return [_json_value(item) for item in value]
|
|
1545
|
+
return value
|