millforge 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- millforge/__init__.py +1174 -0
- millforge/_forge/LICENSE +21 -0
- millforge/_forge/PROVENANCE.json +295 -0
- millforge/_forge/UPDATE_POLICY.md +24 -0
- millforge/_forge/__init__.py +14 -0
- millforge/_forge/adapter.py +2232 -0
- millforge/_forge/base_runner.py +121 -0
- millforge/_forge/clients/__init__.py +10 -0
- millforge/_forge/clients/base.py +200 -0
- millforge/_forge/context/__init__.py +23 -0
- millforge/_forge/context/manager.py +178 -0
- millforge/_forge/context/strategies.py +335 -0
- millforge/_forge/core/__init__.py +16 -0
- millforge/_forge/core/inference.py +433 -0
- millforge/_forge/core/messages.py +119 -0
- millforge/_forge/core/runner.py +479 -0
- millforge/_forge/core/steps.py +108 -0
- millforge/_forge/core/workflow.py +400 -0
- millforge/_forge/errors.py +222 -0
- millforge/_forge/guardrails/__init__.py +21 -0
- millforge/_forge/guardrails/error_tracker.py +71 -0
- millforge/_forge/guardrails/guardrails.py +194 -0
- millforge/_forge/guardrails/nudge.py +47 -0
- millforge/_forge/guardrails/response_validator.py +119 -0
- millforge/_forge/guardrails/step_enforcer.py +183 -0
- millforge/_forge/prompts/__init__.py +16 -0
- millforge/_forge/prompts/nudges.py +95 -0
- millforge/_forge/prompts/templates.py +285 -0
- millforge/_version.py +3 -0
- millforge/artifacts.py +570 -0
- millforge/base/__init__.py +97 -0
- millforge/base/composition.py +402 -0
- millforge/base/context.py +285 -0
- millforge/base/harness.py +138 -0
- millforge/base/identity.py +465 -0
- millforge/base/options.py +34 -0
- millforge/base/platform.py +17 -0
- millforge/base/prompt.py +317 -0
- millforge/base/runner.py +546 -0
- millforge/compiled_plan.py +970 -0
- millforge/compiler/__init__.py +231 -0
- millforge/compiler/artifact_validation.py +257 -0
- millforge/compiler/canonicalization.py +169 -0
- millforge/compiler/capabilities.py +66 -0
- millforge/compiler/catalogs.py +500 -0
- millforge/compiler/diagnostics.py +491 -0
- millforge/compiler/graph.py +678 -0
- millforge/compiler/lowering.py +198 -0
- millforge/compiler/output.py +692 -0
- millforge/compiler/parsing.py +1424 -0
- millforge/compiler/requests.py +1180 -0
- millforge/compiler/schema_validation.py +272 -0
- millforge/compiler/semantic.py +490 -0
- millforge/compiler/service.py +448 -0
- millforge/compiler/source.py +375 -0
- millforge/compiler/validators.py +184 -0
- millforge/connectors/__init__.py +95 -0
- millforge/connectors/admission.py +801 -0
- millforge/connectors/broker.py +202 -0
- millforge/connectors/contracts.py +1159 -0
- millforge/connectors/diagnostics.py +189 -0
- millforge/connectors/fake.py +66 -0
- millforge/connectors/runtime.py +236 -0
- millforge/contracts.py +2860 -0
- millforge/custom_tools/__init__.py +67 -0
- millforge/custom_tools/compiler.py +724 -0
- millforge/custom_tools/contracts.py +1093 -0
- millforge/custom_tools/diagnostics.py +205 -0
- millforge/eval_artifacts.py +952 -0
- millforge/eval_boundary.py +2435 -0
- millforge/eval_fixtures/__init__.py +1 -0
- millforge/eval_fixtures/default_pack/__init__.py +1 -0
- millforge/eval_fixtures/default_pack/fixtures/fixture.08a.bug_diagnosis.traceback.v1.json +52 -0
- millforge/eval_fixtures/default_pack/fixtures/fixture.08a.direct_edit.import_sort.v1.json +52 -0
- millforge/eval_fixtures/default_pack/fixtures/fixture.08a.evidence_discipline.no_source_change.v1.json +51 -0
- millforge/eval_fixtures/default_pack/fixtures/fixture.08a.false_closure.visible_green.v1.json +52 -0
- millforge/eval_fixtures/default_pack/fixtures/fixture.08a.multi_file.api_contract.v1.json +54 -0
- millforge/eval_fixtures/default_pack/fixtures/fixture.08a.recovery.malformed_artifact.v1.json +54 -0
- millforge/eval_fixtures/default_pack/manifest.json +12 -0
- millforge/eval_modes.py +1282 -0
- millforge/eval_presets.py +1398 -0
- millforge/eval_reports.py +2517 -0
- millforge/eval_suite.py +2429 -0
- millforge/eval_trials.py +2632 -0
- millforge/eval_workflow.py +794 -0
- millforge/exceptions.py +122 -0
- millforge/model_backend.py +2098 -0
- millforge/protocols.py +340 -0
- millforge/py.typed +0 -0
- millforge/runtime.py +1791 -0
- millforge/testing/__init__.py +1089 -0
- millforge/tools/__init__.py +83 -0
- millforge/tools/builtin_runtime.py +1339 -0
- millforge/tools/builtins.py +773 -0
- millforge/tools/execution.py +1545 -0
- millforge/tools/path_policy.py +155 -0
- millforge/tools/pi_compat/PI_LICENSE +21 -0
- millforge/tools/pi_compat/PROVENANCE.json +55 -0
- millforge/tools/pi_compat/UPDATE_POLICY.md +36 -0
- millforge/tools/pi_compat/__init__.py +34 -0
- millforge/tools/pi_compat/contracts.py +49 -0
- millforge/tools/pi_compat/editing.py +390 -0
- millforge/tools/pi_compat/mutations.py +57 -0
- millforge/tools/pi_compat/operations.py +401 -0
- millforge/tools/pi_compat/paths.py +155 -0
- millforge/tools/pi_compat/process.py +1375 -0
- millforge/tools/pi_compat/search.py +738 -0
- millforge/tools/pi_compat/truncation.py +267 -0
- millforge/tools/pi_compat_catalog.py +396 -0
- millforge/tools/pi_compat_runtime.py +460 -0
- millforge/tools/registry.py +553 -0
- millforge/tools/results.py +533 -0
- millforge-0.1.0.dist-info/METADATA +844 -0
- millforge-0.1.0.dist-info/RECORD +116 -0
- millforge-0.1.0.dist-info/WHEEL +4 -0
- millforge-0.1.0.dist-info/licenses/LICENSE +201 -0
|
@@ -0,0 +1,533 @@
|
|
|
1
|
+
"""Deterministic tool execution result, validation, and trace helpers."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import hashlib
|
|
6
|
+
import math
|
|
7
|
+
import re
|
|
8
|
+
from collections.abc import Mapping
|
|
9
|
+
from enum import Enum
|
|
10
|
+
from typing import Any, Literal
|
|
11
|
+
|
|
12
|
+
from millforge import (
|
|
13
|
+
ArtifactRef,
|
|
14
|
+
IdempotencyClass,
|
|
15
|
+
SideEffectCertainty,
|
|
16
|
+
SideEffectClass,
|
|
17
|
+
SideEffectRecord,
|
|
18
|
+
TimingMetadata,
|
|
19
|
+
ToolBindingRef,
|
|
20
|
+
ToolExecutionResult,
|
|
21
|
+
ToolExecutionStatus,
|
|
22
|
+
ToolTraceDecision,
|
|
23
|
+
ToolTraceDecisionRecord,
|
|
24
|
+
ToolTraceIdempotency,
|
|
25
|
+
ToolTraceRecord,
|
|
26
|
+
ToolTraceSideEffectClass,
|
|
27
|
+
canonical_json_serialize,
|
|
28
|
+
redact_diagnostic_text,
|
|
29
|
+
redact_diagnostic_value,
|
|
30
|
+
RedactionPolicy,
|
|
31
|
+
)
|
|
32
|
+
from millforge.tools.registry import ToolOutputPolicy
|
|
33
|
+
|
|
34
|
+
MAX_MODEL_SUMMARY_UTF8 = 8192
|
|
35
|
+
_HOST_PATH_RE = re.compile(r"(?<![\w.-])(?:/[^\s:;,]+)+")
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
class ToolExecutionErrorCode(str, Enum):
|
|
39
|
+
"""Stable tool execution result categories."""
|
|
40
|
+
|
|
41
|
+
INVALID_ARGUMENTS = "invalid_arguments"
|
|
42
|
+
CAPABILITY_DENIED = "capability_denied"
|
|
43
|
+
POLICY_DENIED = "policy_denied"
|
|
44
|
+
PREREQUISITE_DENIED = "prerequisite_denied"
|
|
45
|
+
NOT_FOUND = "not_found"
|
|
46
|
+
CONFLICT = "conflict"
|
|
47
|
+
PERMISSION_DENIED = "permission_denied"
|
|
48
|
+
IO_ERROR = "io_error"
|
|
49
|
+
PROCESS_EXIT_NONZERO = "process_exit_nonzero"
|
|
50
|
+
PROCESS_LAUNCH_ERROR = "process_launch_error"
|
|
51
|
+
TIMEOUT = "timeout"
|
|
52
|
+
CANCELLED = "cancelled"
|
|
53
|
+
IMPLEMENTATION_ERROR = "implementation_error"
|
|
54
|
+
AMBIGUOUS_SIDE_EFFECT = "ambiguous_side_effect"
|
|
55
|
+
OUTPUT_VALIDATION_FAILED = "output_validation_failed"
|
|
56
|
+
TERMINAL_INTENT_INVALID = "terminal_intent_invalid"
|
|
57
|
+
BINDING_MISMATCH = "binding_mismatch"
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
MODEL_CORRECTABLE_CODES = frozenset(
|
|
61
|
+
{
|
|
62
|
+
ToolExecutionErrorCode.INVALID_ARGUMENTS,
|
|
63
|
+
ToolExecutionErrorCode.CAPABILITY_DENIED,
|
|
64
|
+
ToolExecutionErrorCode.POLICY_DENIED,
|
|
65
|
+
ToolExecutionErrorCode.PREREQUISITE_DENIED,
|
|
66
|
+
ToolExecutionErrorCode.NOT_FOUND,
|
|
67
|
+
ToolExecutionErrorCode.CONFLICT,
|
|
68
|
+
ToolExecutionErrorCode.TIMEOUT,
|
|
69
|
+
ToolExecutionErrorCode.CANCELLED,
|
|
70
|
+
ToolExecutionErrorCode.TERMINAL_INTENT_INVALID,
|
|
71
|
+
}
|
|
72
|
+
)
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
def canonical_sha256(value: Any) -> str:
|
|
76
|
+
"""Hash a JSON-compatible value in the project canonical format."""
|
|
77
|
+
return hashlib.sha256(canonical_json_serialize(value).encode("utf-8")).hexdigest()
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
def redact_tool_value(value: Any, *, policy: RedactionPolicy | None = None) -> Any:
|
|
81
|
+
"""Redact and bound a value before trace persistence or model return."""
|
|
82
|
+
redacted = redact_diagnostic_value(value, policy=policy)
|
|
83
|
+
return _redact_host_paths(redacted)
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
def bounded_summary(
|
|
87
|
+
value: Any,
|
|
88
|
+
*,
|
|
89
|
+
max_utf8: int = MAX_MODEL_SUMMARY_UTF8,
|
|
90
|
+
policy: RedactionPolicy | None = None,
|
|
91
|
+
) -> str:
|
|
92
|
+
"""Return a redacted non-empty summary bounded by UTF-8 byte length."""
|
|
93
|
+
if isinstance(value, str):
|
|
94
|
+
text = _redact_host_paths(redact_diagnostic_text(value, policy=policy))
|
|
95
|
+
else:
|
|
96
|
+
text = canonical_json_serialize(redact_tool_value(value, policy=policy)).strip()
|
|
97
|
+
if len(text.encode("utf-8")) > max_utf8:
|
|
98
|
+
raw = text.encode("utf-8")[:max_utf8]
|
|
99
|
+
text = raw.decode("utf-8", errors="ignore") + "[truncated]"
|
|
100
|
+
return text or "[empty]"
|
|
101
|
+
|
|
102
|
+
|
|
103
|
+
def output_hash(value: Any, *, policy: RedactionPolicy | None = None) -> str:
|
|
104
|
+
"""Hash the safe redacted output value."""
|
|
105
|
+
return canonical_sha256(redact_tool_value(value, policy=policy))
|
|
106
|
+
|
|
107
|
+
|
|
108
|
+
def validate_json_object_schema(
|
|
109
|
+
value: Mapping[str, Any],
|
|
110
|
+
schema: Mapping[str, Any],
|
|
111
|
+
) -> str | None:
|
|
112
|
+
"""Validate the descriptor schema subset used by built-in tools."""
|
|
113
|
+
return _validate_schema_value(value, schema, path="$")
|
|
114
|
+
|
|
115
|
+
|
|
116
|
+
def make_tool_result(
|
|
117
|
+
*,
|
|
118
|
+
call_id: str,
|
|
119
|
+
status: ToolExecutionStatus,
|
|
120
|
+
code: ToolExecutionErrorCode | None,
|
|
121
|
+
summary: str,
|
|
122
|
+
structured_data: Any,
|
|
123
|
+
side_effect_class: SideEffectClass,
|
|
124
|
+
idempotency: IdempotencyClass,
|
|
125
|
+
side_effect_certainty: SideEffectCertainty,
|
|
126
|
+
input_sha256: str,
|
|
127
|
+
retryable: bool = False,
|
|
128
|
+
artifact_refs: tuple[ArtifactRef, ...] = (),
|
|
129
|
+
output_sha256: str | None = None,
|
|
130
|
+
timing: TimingMetadata | None = None,
|
|
131
|
+
side_effect_record: SideEffectRecord | None = None,
|
|
132
|
+
output_policy: ToolOutputPolicy | None = None,
|
|
133
|
+
) -> ToolExecutionResult:
|
|
134
|
+
"""Build a bounded ``ToolExecutionResult`` for model-visible return."""
|
|
135
|
+
summary_limit = (
|
|
136
|
+
output_policy.max_summary_utf8
|
|
137
|
+
if output_policy is not None
|
|
138
|
+
else MAX_MODEL_SUMMARY_UTF8
|
|
139
|
+
)
|
|
140
|
+
safe_data = _model_visible_value(structured_data, output_policy=output_policy)
|
|
141
|
+
safe_summary = _model_visible_summary(
|
|
142
|
+
summary,
|
|
143
|
+
max_utf8=summary_limit,
|
|
144
|
+
output_policy=output_policy,
|
|
145
|
+
)
|
|
146
|
+
safe_side_effect_record = (
|
|
147
|
+
None
|
|
148
|
+
if side_effect_record is None
|
|
149
|
+
else side_effect_record.model_copy(
|
|
150
|
+
update={
|
|
151
|
+
"summary": _model_visible_summary(
|
|
152
|
+
side_effect_record.summary,
|
|
153
|
+
max_utf8=summary_limit,
|
|
154
|
+
output_policy=output_policy,
|
|
155
|
+
)
|
|
156
|
+
}
|
|
157
|
+
)
|
|
158
|
+
)
|
|
159
|
+
if code is None and output_sha256 is None:
|
|
160
|
+
output_sha256 = canonical_sha256(safe_data)
|
|
161
|
+
return ToolExecutionResult(
|
|
162
|
+
call_id=call_id,
|
|
163
|
+
status=status,
|
|
164
|
+
summary=safe_summary,
|
|
165
|
+
structured_data=safe_data,
|
|
166
|
+
artifact_refs=artifact_refs,
|
|
167
|
+
error_code=None if code is None else code.value,
|
|
168
|
+
retryable=retryable,
|
|
169
|
+
side_effect_class=side_effect_class,
|
|
170
|
+
idempotency=idempotency,
|
|
171
|
+
side_effect_certainty=side_effect_certainty,
|
|
172
|
+
side_effect_record=safe_side_effect_record,
|
|
173
|
+
input_sha256=input_sha256,
|
|
174
|
+
output_sha256=output_sha256,
|
|
175
|
+
timing=timing or zero_timing(),
|
|
176
|
+
)
|
|
177
|
+
|
|
178
|
+
|
|
179
|
+
def sanitize_tool_execution_result(
|
|
180
|
+
result: ToolExecutionResult,
|
|
181
|
+
*,
|
|
182
|
+
output_policy: ToolOutputPolicy | None = None,
|
|
183
|
+
input_sha256: str | None = None,
|
|
184
|
+
) -> ToolExecutionResult:
|
|
185
|
+
"""Sanitize an implementation-produced result for model-visible return."""
|
|
186
|
+
summary_limit = (
|
|
187
|
+
output_policy.max_summary_utf8
|
|
188
|
+
if output_policy is not None
|
|
189
|
+
else MAX_MODEL_SUMMARY_UTF8
|
|
190
|
+
)
|
|
191
|
+
safe_data = _model_visible_value(
|
|
192
|
+
result.structured_data, output_policy=output_policy
|
|
193
|
+
)
|
|
194
|
+
safe_summary = _model_visible_summary(
|
|
195
|
+
result.summary,
|
|
196
|
+
max_utf8=summary_limit,
|
|
197
|
+
output_policy=output_policy,
|
|
198
|
+
)
|
|
199
|
+
safe_side_effect_record = (
|
|
200
|
+
None
|
|
201
|
+
if result.side_effect_record is None
|
|
202
|
+
else result.side_effect_record.model_copy(
|
|
203
|
+
update={
|
|
204
|
+
"summary": _model_visible_summary(
|
|
205
|
+
result.side_effect_record.summary,
|
|
206
|
+
max_utf8=summary_limit,
|
|
207
|
+
output_policy=output_policy,
|
|
208
|
+
)
|
|
209
|
+
}
|
|
210
|
+
)
|
|
211
|
+
)
|
|
212
|
+
output_sha256 = result.output_sha256
|
|
213
|
+
if result.status is ToolExecutionStatus.SUCCESS or output_sha256 is not None:
|
|
214
|
+
output_sha256 = canonical_sha256(safe_data)
|
|
215
|
+
update = {
|
|
216
|
+
"summary": safe_summary,
|
|
217
|
+
"structured_data": safe_data,
|
|
218
|
+
"side_effect_record": safe_side_effect_record,
|
|
219
|
+
"output_sha256": output_sha256,
|
|
220
|
+
}
|
|
221
|
+
if input_sha256 is not None:
|
|
222
|
+
update["input_sha256"] = input_sha256
|
|
223
|
+
return result.model_copy(update=update)
|
|
224
|
+
|
|
225
|
+
|
|
226
|
+
def make_denial_result(
|
|
227
|
+
*,
|
|
228
|
+
call_id: str,
|
|
229
|
+
code: ToolExecutionErrorCode,
|
|
230
|
+
summary: str,
|
|
231
|
+
evidence: Mapping[str, Any],
|
|
232
|
+
side_effect_class: SideEffectClass,
|
|
233
|
+
idempotency: IdempotencyClass,
|
|
234
|
+
input_sha256: str,
|
|
235
|
+
status: ToolExecutionStatus | None = None,
|
|
236
|
+
) -> ToolExecutionResult:
|
|
237
|
+
"""Build a deterministic pre-entry denial result."""
|
|
238
|
+
if status is None:
|
|
239
|
+
hard = code not in MODEL_CORRECTABLE_CODES
|
|
240
|
+
status = (
|
|
241
|
+
ToolExecutionStatus.HARD_FAILURE
|
|
242
|
+
if hard
|
|
243
|
+
else ToolExecutionStatus.NOT_EXECUTED
|
|
244
|
+
)
|
|
245
|
+
return make_tool_result(
|
|
246
|
+
call_id=call_id,
|
|
247
|
+
status=status,
|
|
248
|
+
code=code,
|
|
249
|
+
summary=summary,
|
|
250
|
+
structured_data={"category": code.value, "evidence": dict(evidence)},
|
|
251
|
+
side_effect_class=side_effect_class,
|
|
252
|
+
idempotency=idempotency,
|
|
253
|
+
side_effect_certainty=SideEffectCertainty.NOT_ATTEMPTED,
|
|
254
|
+
input_sha256=input_sha256,
|
|
255
|
+
output_sha256=None,
|
|
256
|
+
)
|
|
257
|
+
|
|
258
|
+
|
|
259
|
+
def make_trace_record(
|
|
260
|
+
*,
|
|
261
|
+
sequence: int,
|
|
262
|
+
request_id: str,
|
|
263
|
+
run_id: str,
|
|
264
|
+
session_id: str,
|
|
265
|
+
stage: Any,
|
|
266
|
+
node_id: str,
|
|
267
|
+
model_turn: int,
|
|
268
|
+
tool_call_id: str,
|
|
269
|
+
model_tool_name: str,
|
|
270
|
+
binding: ToolBindingRef,
|
|
271
|
+
binding_resolution_status: Literal[
|
|
272
|
+
"resolved", "ambiguous", "uncompiled"
|
|
273
|
+
] = "resolved",
|
|
274
|
+
input_sha256: str,
|
|
275
|
+
prerequisite_decisions: Mapping[str, ToolTraceDecision],
|
|
276
|
+
capability_decisions: Mapping[str, ToolTraceDecision],
|
|
277
|
+
result: ToolExecutionResult,
|
|
278
|
+
connector_audit: Mapping[str, Any] | None = None,
|
|
279
|
+
summary_max_utf8: int = MAX_MODEL_SUMMARY_UTF8,
|
|
280
|
+
occurred_at: str = "1970-01-01T00:00:00+00:00",
|
|
281
|
+
monotonic_offset_ms: float = 0.0,
|
|
282
|
+
) -> ToolTraceRecord:
|
|
283
|
+
"""Build and validate a redacted trace record for an attempted tool call."""
|
|
284
|
+
record = result.side_effect_record
|
|
285
|
+
summary_policy = RedactionPolicy(
|
|
286
|
+
max_string_length=max(summary_max_utf8, 1),
|
|
287
|
+
max_total_bytes=max(summary_max_utf8, 1),
|
|
288
|
+
)
|
|
289
|
+
connector_fields = dict(connector_audit or {})
|
|
290
|
+
return ToolTraceRecord(
|
|
291
|
+
schema_version="1.0",
|
|
292
|
+
sequence=sequence,
|
|
293
|
+
occurred_at=occurred_at,
|
|
294
|
+
monotonic_offset_ms=monotonic_offset_ms,
|
|
295
|
+
request_id=request_id,
|
|
296
|
+
run_id=run_id,
|
|
297
|
+
session_id=session_id,
|
|
298
|
+
stage=stage,
|
|
299
|
+
node_id=node_id,
|
|
300
|
+
model_turn=model_turn,
|
|
301
|
+
tool_call_id=tool_call_id,
|
|
302
|
+
model_tool_name=model_tool_name,
|
|
303
|
+
binding=binding,
|
|
304
|
+
binding_resolution_status=binding_resolution_status,
|
|
305
|
+
input_sha256=input_sha256,
|
|
306
|
+
prerequisite_decisions=tuple(
|
|
307
|
+
ToolTraceDecisionRecord(key=key, decision=decision)
|
|
308
|
+
for key, decision in sorted(prerequisite_decisions.items())
|
|
309
|
+
),
|
|
310
|
+
capability_decisions=tuple(
|
|
311
|
+
ToolTraceDecisionRecord(key=key, decision=decision)
|
|
312
|
+
for key, decision in sorted(capability_decisions.items())
|
|
313
|
+
),
|
|
314
|
+
execution_status=result.status,
|
|
315
|
+
retryable=result.retryable,
|
|
316
|
+
side_effect_class=ToolTraceSideEffectClass(result.side_effect_class.value),
|
|
317
|
+
idempotency=ToolTraceIdempotency(result.idempotency.value),
|
|
318
|
+
side_effect_certainty=result.side_effect_certainty,
|
|
319
|
+
**connector_fields,
|
|
320
|
+
side_effect_detail_code=None if record is None else record.detail_code,
|
|
321
|
+
side_effect_detail_summary=None
|
|
322
|
+
if record is None
|
|
323
|
+
else bounded_summary(
|
|
324
|
+
record.summary,
|
|
325
|
+
max_utf8=summary_max_utf8,
|
|
326
|
+
policy=summary_policy,
|
|
327
|
+
),
|
|
328
|
+
side_effect_retry_allowed=None if record is None else record.retry_allowed,
|
|
329
|
+
output_sha256=result.output_sha256,
|
|
330
|
+
duration_ms=result.duration_ms,
|
|
331
|
+
summary=bounded_summary(
|
|
332
|
+
result.summary,
|
|
333
|
+
max_utf8=summary_max_utf8,
|
|
334
|
+
policy=summary_policy,
|
|
335
|
+
),
|
|
336
|
+
)
|
|
337
|
+
|
|
338
|
+
|
|
339
|
+
def zero_timing() -> TimingMetadata:
|
|
340
|
+
"""Return deterministic timing metadata for synchronous tests."""
|
|
341
|
+
return TimingMetadata(
|
|
342
|
+
started_at="1970-01-01T00:00:00+00:00",
|
|
343
|
+
completed_at="1970-01-01T00:00:00+00:00",
|
|
344
|
+
duration_ms=0.0,
|
|
345
|
+
)
|
|
346
|
+
|
|
347
|
+
|
|
348
|
+
def _output_redaction_policy(
|
|
349
|
+
output_policy: ToolOutputPolicy | None,
|
|
350
|
+
) -> RedactionPolicy | None:
|
|
351
|
+
if output_policy is None:
|
|
352
|
+
return None
|
|
353
|
+
return RedactionPolicy(
|
|
354
|
+
max_string_length=max(
|
|
355
|
+
output_policy.max_output_bytes, output_policy.max_summary_utf8
|
|
356
|
+
),
|
|
357
|
+
max_total_bytes=output_policy.max_output_bytes,
|
|
358
|
+
)
|
|
359
|
+
|
|
360
|
+
|
|
361
|
+
def _model_visible_value(value: Any, *, output_policy: ToolOutputPolicy | None) -> Any:
|
|
362
|
+
if output_policy is not None and not output_policy.redact_secrets:
|
|
363
|
+
return _bound_unredacted_value(
|
|
364
|
+
value, max_total_bytes=output_policy.max_output_bytes
|
|
365
|
+
)
|
|
366
|
+
return redact_tool_value(value, policy=_output_redaction_policy(output_policy))
|
|
367
|
+
|
|
368
|
+
|
|
369
|
+
def _model_visible_summary(
|
|
370
|
+
value: Any,
|
|
371
|
+
*,
|
|
372
|
+
max_utf8: int,
|
|
373
|
+
output_policy: ToolOutputPolicy | None,
|
|
374
|
+
) -> str:
|
|
375
|
+
if output_policy is not None and not output_policy.redact_secrets:
|
|
376
|
+
if isinstance(value, str):
|
|
377
|
+
return _bound_unredacted_text(value, max_utf8=max_utf8) or "[empty]"
|
|
378
|
+
return (
|
|
379
|
+
_bound_unredacted_text(
|
|
380
|
+
canonical_json_serialize(
|
|
381
|
+
_bound_unredacted_value(value, max_total_bytes=max_utf8)
|
|
382
|
+
).strip(),
|
|
383
|
+
max_utf8=max_utf8,
|
|
384
|
+
)
|
|
385
|
+
or "[empty]"
|
|
386
|
+
)
|
|
387
|
+
return bounded_summary(
|
|
388
|
+
value,
|
|
389
|
+
max_utf8=max_utf8,
|
|
390
|
+
policy=_summary_redaction_policy(max_utf8),
|
|
391
|
+
)
|
|
392
|
+
|
|
393
|
+
|
|
394
|
+
def _bound_unredacted_value(value: Any, *, max_total_bytes: int) -> Any:
|
|
395
|
+
"""Bound JSON-compatible model output without redacting its content."""
|
|
396
|
+
bounded = _copy_json_value(value)
|
|
397
|
+
if isinstance(bounded, str):
|
|
398
|
+
return _bound_unredacted_text(bounded, max_utf8=max_total_bytes)
|
|
399
|
+
while len(canonical_json_serialize(bounded).encode("utf-8")) > max_total_bytes:
|
|
400
|
+
path, text = _longest_string_value(bounded)
|
|
401
|
+
if path is None:
|
|
402
|
+
break
|
|
403
|
+
current_size = len(canonical_json_serialize(bounded).encode("utf-8"))
|
|
404
|
+
target_size = max(
|
|
405
|
+
0, len(text.encode("utf-8")) - (current_size - max_total_bytes)
|
|
406
|
+
)
|
|
407
|
+
replacement = _bound_unredacted_text(text, max_utf8=target_size)
|
|
408
|
+
if replacement == text:
|
|
409
|
+
break
|
|
410
|
+
_replace_json_value(bounded, path, replacement)
|
|
411
|
+
return bounded
|
|
412
|
+
|
|
413
|
+
|
|
414
|
+
def _copy_json_value(value: Any) -> Any:
|
|
415
|
+
if isinstance(value, Mapping):
|
|
416
|
+
return {str(key): _copy_json_value(item) for key, item in value.items()}
|
|
417
|
+
if isinstance(value, list | tuple):
|
|
418
|
+
return [_copy_json_value(item) for item in value]
|
|
419
|
+
return value
|
|
420
|
+
|
|
421
|
+
|
|
422
|
+
def _longest_string_value(
|
|
423
|
+
value: Any,
|
|
424
|
+
path: tuple[str | int, ...] = (),
|
|
425
|
+
) -> tuple[tuple[str | int, ...] | None, str]:
|
|
426
|
+
if isinstance(value, str):
|
|
427
|
+
return path, value
|
|
428
|
+
candidates: list[tuple[tuple[str | int, ...] | None, str]] = []
|
|
429
|
+
if isinstance(value, Mapping):
|
|
430
|
+
candidates.extend(
|
|
431
|
+
_longest_string_value(item, (*path, str(key)))
|
|
432
|
+
for key, item in value.items()
|
|
433
|
+
)
|
|
434
|
+
elif isinstance(value, list):
|
|
435
|
+
candidates.extend(
|
|
436
|
+
_longest_string_value(item, (*path, index))
|
|
437
|
+
for index, item in enumerate(value)
|
|
438
|
+
)
|
|
439
|
+
candidates = [candidate for candidate in candidates if candidate[0] is not None]
|
|
440
|
+
if not candidates:
|
|
441
|
+
return None, ""
|
|
442
|
+
return max(candidates, key=lambda candidate: len(candidate[1].encode("utf-8")))
|
|
443
|
+
|
|
444
|
+
|
|
445
|
+
def _replace_json_value(
|
|
446
|
+
value: Any,
|
|
447
|
+
path: tuple[str | int, ...],
|
|
448
|
+
replacement: str,
|
|
449
|
+
) -> None:
|
|
450
|
+
target = value
|
|
451
|
+
for segment in path[:-1]:
|
|
452
|
+
target = target[segment]
|
|
453
|
+
target[path[-1]] = replacement
|
|
454
|
+
|
|
455
|
+
|
|
456
|
+
def _bound_unredacted_text(value: str, *, max_utf8: int) -> str:
|
|
457
|
+
encoded = value.encode("utf-8")
|
|
458
|
+
if len(encoded) <= max_utf8:
|
|
459
|
+
return value
|
|
460
|
+
marker = "[truncated]"
|
|
461
|
+
if max_utf8 <= len(marker):
|
|
462
|
+
return marker[:max_utf8]
|
|
463
|
+
prefix = encoded[: max_utf8 - len(marker)].decode("utf-8", errors="ignore")
|
|
464
|
+
return f"{prefix}{marker}"
|
|
465
|
+
|
|
466
|
+
|
|
467
|
+
def _summary_redaction_policy(summary_limit: int) -> RedactionPolicy:
|
|
468
|
+
return RedactionPolicy(
|
|
469
|
+
max_string_length=summary_limit,
|
|
470
|
+
max_total_bytes=summary_limit,
|
|
471
|
+
)
|
|
472
|
+
|
|
473
|
+
|
|
474
|
+
def _validate_schema_value(
|
|
475
|
+
value: Any, schema: Mapping[str, Any], *, path: str
|
|
476
|
+
) -> str | None:
|
|
477
|
+
expected_type = schema.get("type")
|
|
478
|
+
if expected_type == "object":
|
|
479
|
+
if not isinstance(value, Mapping):
|
|
480
|
+
return f"{path} must be object"
|
|
481
|
+
properties = schema.get("properties", {})
|
|
482
|
+
required = schema.get("required", [])
|
|
483
|
+
for key in required:
|
|
484
|
+
if key not in value:
|
|
485
|
+
return f"{path}.{key} is required"
|
|
486
|
+
if schema.get("additionalProperties") is False:
|
|
487
|
+
extra = set(value) - set(properties)
|
|
488
|
+
if extra:
|
|
489
|
+
return f"{path}.{sorted(extra)[0]} is not allowed"
|
|
490
|
+
for key, item in value.items():
|
|
491
|
+
child_schema = properties.get(key)
|
|
492
|
+
if child_schema is None:
|
|
493
|
+
continue
|
|
494
|
+
error = _validate_schema_value(item, child_schema, path=f"{path}.{key}")
|
|
495
|
+
if error is not None:
|
|
496
|
+
return error
|
|
497
|
+
elif expected_type == "array":
|
|
498
|
+
if not isinstance(value, list):
|
|
499
|
+
return f"{path} must be array"
|
|
500
|
+
item_schema = schema.get("items", {})
|
|
501
|
+
for index, item in enumerate(value):
|
|
502
|
+
error = _validate_schema_value(item, item_schema, path=f"{path}[{index}]")
|
|
503
|
+
if error is not None:
|
|
504
|
+
return error
|
|
505
|
+
elif expected_type == "string":
|
|
506
|
+
if not isinstance(value, str):
|
|
507
|
+
return f"{path} must be string"
|
|
508
|
+
elif expected_type == "integer":
|
|
509
|
+
if not isinstance(value, int) or isinstance(value, bool):
|
|
510
|
+
return f"{path} must be integer"
|
|
511
|
+
elif expected_type == "number":
|
|
512
|
+
if (
|
|
513
|
+
isinstance(value, bool)
|
|
514
|
+
or not isinstance(value, int | float)
|
|
515
|
+
or not math.isfinite(value)
|
|
516
|
+
):
|
|
517
|
+
return f"{path} must be number"
|
|
518
|
+
elif expected_type == "boolean":
|
|
519
|
+
if not isinstance(value, bool):
|
|
520
|
+
return f"{path} must be boolean"
|
|
521
|
+
if "enum" in schema and value not in schema["enum"]:
|
|
522
|
+
return f"{path} must be one of {schema['enum']!r}"
|
|
523
|
+
return None
|
|
524
|
+
|
|
525
|
+
|
|
526
|
+
def _redact_host_paths(value: Any) -> Any:
|
|
527
|
+
if isinstance(value, str):
|
|
528
|
+
return _HOST_PATH_RE.sub("[path]", value)
|
|
529
|
+
if isinstance(value, Mapping):
|
|
530
|
+
return {str(key): _redact_host_paths(item) for key, item in value.items()}
|
|
531
|
+
if isinstance(value, list):
|
|
532
|
+
return [_redact_host_paths(item) for item in value]
|
|
533
|
+
return value
|