millforge 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- millforge/__init__.py +1174 -0
- millforge/_forge/LICENSE +21 -0
- millforge/_forge/PROVENANCE.json +295 -0
- millforge/_forge/UPDATE_POLICY.md +24 -0
- millforge/_forge/__init__.py +14 -0
- millforge/_forge/adapter.py +2232 -0
- millforge/_forge/base_runner.py +121 -0
- millforge/_forge/clients/__init__.py +10 -0
- millforge/_forge/clients/base.py +200 -0
- millforge/_forge/context/__init__.py +23 -0
- millforge/_forge/context/manager.py +178 -0
- millforge/_forge/context/strategies.py +335 -0
- millforge/_forge/core/__init__.py +16 -0
- millforge/_forge/core/inference.py +433 -0
- millforge/_forge/core/messages.py +119 -0
- millforge/_forge/core/runner.py +479 -0
- millforge/_forge/core/steps.py +108 -0
- millforge/_forge/core/workflow.py +400 -0
- millforge/_forge/errors.py +222 -0
- millforge/_forge/guardrails/__init__.py +21 -0
- millforge/_forge/guardrails/error_tracker.py +71 -0
- millforge/_forge/guardrails/guardrails.py +194 -0
- millforge/_forge/guardrails/nudge.py +47 -0
- millforge/_forge/guardrails/response_validator.py +119 -0
- millforge/_forge/guardrails/step_enforcer.py +183 -0
- millforge/_forge/prompts/__init__.py +16 -0
- millforge/_forge/prompts/nudges.py +95 -0
- millforge/_forge/prompts/templates.py +285 -0
- millforge/_version.py +3 -0
- millforge/artifacts.py +570 -0
- millforge/base/__init__.py +97 -0
- millforge/base/composition.py +402 -0
- millforge/base/context.py +285 -0
- millforge/base/harness.py +138 -0
- millforge/base/identity.py +465 -0
- millforge/base/options.py +34 -0
- millforge/base/platform.py +17 -0
- millforge/base/prompt.py +317 -0
- millforge/base/runner.py +546 -0
- millforge/compiled_plan.py +970 -0
- millforge/compiler/__init__.py +231 -0
- millforge/compiler/artifact_validation.py +257 -0
- millforge/compiler/canonicalization.py +169 -0
- millforge/compiler/capabilities.py +66 -0
- millforge/compiler/catalogs.py +500 -0
- millforge/compiler/diagnostics.py +491 -0
- millforge/compiler/graph.py +678 -0
- millforge/compiler/lowering.py +198 -0
- millforge/compiler/output.py +692 -0
- millforge/compiler/parsing.py +1424 -0
- millforge/compiler/requests.py +1180 -0
- millforge/compiler/schema_validation.py +272 -0
- millforge/compiler/semantic.py +490 -0
- millforge/compiler/service.py +448 -0
- millforge/compiler/source.py +375 -0
- millforge/compiler/validators.py +184 -0
- millforge/connectors/__init__.py +95 -0
- millforge/connectors/admission.py +801 -0
- millforge/connectors/broker.py +202 -0
- millforge/connectors/contracts.py +1159 -0
- millforge/connectors/diagnostics.py +189 -0
- millforge/connectors/fake.py +66 -0
- millforge/connectors/runtime.py +236 -0
- millforge/contracts.py +2860 -0
- millforge/custom_tools/__init__.py +67 -0
- millforge/custom_tools/compiler.py +724 -0
- millforge/custom_tools/contracts.py +1093 -0
- millforge/custom_tools/diagnostics.py +205 -0
- millforge/eval_artifacts.py +952 -0
- millforge/eval_boundary.py +2435 -0
- millforge/eval_fixtures/__init__.py +1 -0
- millforge/eval_fixtures/default_pack/__init__.py +1 -0
- millforge/eval_fixtures/default_pack/fixtures/fixture.08a.bug_diagnosis.traceback.v1.json +52 -0
- millforge/eval_fixtures/default_pack/fixtures/fixture.08a.direct_edit.import_sort.v1.json +52 -0
- millforge/eval_fixtures/default_pack/fixtures/fixture.08a.evidence_discipline.no_source_change.v1.json +51 -0
- millforge/eval_fixtures/default_pack/fixtures/fixture.08a.false_closure.visible_green.v1.json +52 -0
- millforge/eval_fixtures/default_pack/fixtures/fixture.08a.multi_file.api_contract.v1.json +54 -0
- millforge/eval_fixtures/default_pack/fixtures/fixture.08a.recovery.malformed_artifact.v1.json +54 -0
- millforge/eval_fixtures/default_pack/manifest.json +12 -0
- millforge/eval_modes.py +1282 -0
- millforge/eval_presets.py +1398 -0
- millforge/eval_reports.py +2517 -0
- millforge/eval_suite.py +2429 -0
- millforge/eval_trials.py +2632 -0
- millforge/eval_workflow.py +794 -0
- millforge/exceptions.py +122 -0
- millforge/model_backend.py +2098 -0
- millforge/protocols.py +340 -0
- millforge/py.typed +0 -0
- millforge/runtime.py +1791 -0
- millforge/testing/__init__.py +1089 -0
- millforge/tools/__init__.py +83 -0
- millforge/tools/builtin_runtime.py +1339 -0
- millforge/tools/builtins.py +773 -0
- millforge/tools/execution.py +1545 -0
- millforge/tools/path_policy.py +155 -0
- millforge/tools/pi_compat/PI_LICENSE +21 -0
- millforge/tools/pi_compat/PROVENANCE.json +55 -0
- millforge/tools/pi_compat/UPDATE_POLICY.md +36 -0
- millforge/tools/pi_compat/__init__.py +34 -0
- millforge/tools/pi_compat/contracts.py +49 -0
- millforge/tools/pi_compat/editing.py +390 -0
- millforge/tools/pi_compat/mutations.py +57 -0
- millforge/tools/pi_compat/operations.py +401 -0
- millforge/tools/pi_compat/paths.py +155 -0
- millforge/tools/pi_compat/process.py +1375 -0
- millforge/tools/pi_compat/search.py +738 -0
- millforge/tools/pi_compat/truncation.py +267 -0
- millforge/tools/pi_compat_catalog.py +396 -0
- millforge/tools/pi_compat_runtime.py +460 -0
- millforge/tools/registry.py +553 -0
- millforge/tools/results.py +533 -0
- millforge-0.1.0.dist-info/METADATA +844 -0
- millforge-0.1.0.dist-info/RECORD +116 -0
- millforge-0.1.0.dist-info/WHEEL +4 -0
- millforge-0.1.0.dist-info/licenses/LICENSE +201 -0
|
@@ -0,0 +1,2232 @@
|
|
|
1
|
+
"""Private translation from compiled Millforge plans to Forge objects."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import asyncio
|
|
6
|
+
import hashlib
|
|
7
|
+
import json
|
|
8
|
+
from collections.abc import AsyncIterator, Callable, Mapping
|
|
9
|
+
from dataclasses import dataclass
|
|
10
|
+
from pathlib import Path
|
|
11
|
+
from typing import Any, Literal, NoReturn, Protocol
|
|
12
|
+
|
|
13
|
+
from millforge._forge.clients.base import StreamChunk, TokenUsage as ForgeTokenUsage
|
|
14
|
+
from millforge._forge.context.manager import ContextManager
|
|
15
|
+
from millforge._forge.context.strategies import TieredCompact
|
|
16
|
+
from millforge._forge.core.messages import (
|
|
17
|
+
Message,
|
|
18
|
+
MessageMeta,
|
|
19
|
+
MessageRole,
|
|
20
|
+
MessageType,
|
|
21
|
+
)
|
|
22
|
+
from millforge._forge.core.runner import WorkflowRunner
|
|
23
|
+
from millforge._forge.core.workflow import (
|
|
24
|
+
TextResponse,
|
|
25
|
+
ToolCall,
|
|
26
|
+
ToolDef,
|
|
27
|
+
ToolSpec,
|
|
28
|
+
Workflow,
|
|
29
|
+
)
|
|
30
|
+
from millforge._forge.errors import (
|
|
31
|
+
ContextBudgetExceeded,
|
|
32
|
+
ForgeError,
|
|
33
|
+
MaxIterationsError,
|
|
34
|
+
NonRetryableToolError,
|
|
35
|
+
PrerequisiteError,
|
|
36
|
+
StepEnforcementError,
|
|
37
|
+
ToolCallError,
|
|
38
|
+
ToolExecutionError,
|
|
39
|
+
ToolResolutionError,
|
|
40
|
+
WorkflowCancelledError,
|
|
41
|
+
)
|
|
42
|
+
from millforge.compiled_plan import (
|
|
43
|
+
ArgumentMatch,
|
|
44
|
+
CompiledContextPolicy,
|
|
45
|
+
CompiledHarnessNode,
|
|
46
|
+
CompiledHarnessPlan,
|
|
47
|
+
CompiledPrerequisite,
|
|
48
|
+
DiagnosticField,
|
|
49
|
+
IdempotencyClass,
|
|
50
|
+
SessionEvent,
|
|
51
|
+
SessionEventType,
|
|
52
|
+
SideEffectCertainty,
|
|
53
|
+
SideEffectClass,
|
|
54
|
+
ToolExecutionStatus,
|
|
55
|
+
ToolTraceDecision,
|
|
56
|
+
ToolTraceDecisionRecord,
|
|
57
|
+
ToolTraceIdempotency,
|
|
58
|
+
ToolTraceRecord,
|
|
59
|
+
ToolTraceSideEffectClass,
|
|
60
|
+
canonical_json_serialize,
|
|
61
|
+
verify_compiled_plan_sha256,
|
|
62
|
+
)
|
|
63
|
+
from millforge.contracts import (
|
|
64
|
+
ArtifactRef,
|
|
65
|
+
AssistantMessage,
|
|
66
|
+
CancellationRef,
|
|
67
|
+
Deadline,
|
|
68
|
+
DiagnosticMetadata,
|
|
69
|
+
GuardedSessionResult,
|
|
70
|
+
GuardedSessionStatus,
|
|
71
|
+
GuardedSessionRequest,
|
|
72
|
+
HarnessExecutionRequest,
|
|
73
|
+
InvalidToolArguments,
|
|
74
|
+
ModelCompletionRequest,
|
|
75
|
+
ModelCompletionResponse,
|
|
76
|
+
ModelMessage,
|
|
77
|
+
ModelToolDefinition,
|
|
78
|
+
ModelToolCall,
|
|
79
|
+
ParsedToolArguments,
|
|
80
|
+
SamplingRequest,
|
|
81
|
+
SelectedOutput,
|
|
82
|
+
SelectedOutputRequirement,
|
|
83
|
+
SystemMessage,
|
|
84
|
+
TerminalIntent,
|
|
85
|
+
ToolExecutionContext,
|
|
86
|
+
ToolExecutionResult,
|
|
87
|
+
ToolResultMessage,
|
|
88
|
+
TerminalSelectedOutputRequirement,
|
|
89
|
+
TimingMetadata,
|
|
90
|
+
TokenUsage,
|
|
91
|
+
UserMessage,
|
|
92
|
+
UsageMetadata,
|
|
93
|
+
ValidatedToolCall,
|
|
94
|
+
_selected_output_requirements_by_terminal_result,
|
|
95
|
+
admit_selected_output,
|
|
96
|
+
)
|
|
97
|
+
from millforge.exceptions import (
|
|
98
|
+
BackendTranslationError,
|
|
99
|
+
DeadlineExceededError,
|
|
100
|
+
ModelTransportError,
|
|
101
|
+
OperationCancelledError,
|
|
102
|
+
ToolInvokeError,
|
|
103
|
+
)
|
|
104
|
+
from millforge.model_backend import (
|
|
105
|
+
ModelProviderError,
|
|
106
|
+
ModelRequestDeadlineExceededError,
|
|
107
|
+
ProviderErrorCategory,
|
|
108
|
+
)
|
|
109
|
+
from millforge.protocols import (
|
|
110
|
+
CancellationResolver,
|
|
111
|
+
CompiledHarnessLoader,
|
|
112
|
+
ModelClient,
|
|
113
|
+
RuntimeClock,
|
|
114
|
+
ToolExecutor,
|
|
115
|
+
)
|
|
116
|
+
from millforge.tools.results import bounded_summary, canonical_sha256
|
|
117
|
+
|
|
118
|
+
|
|
119
|
+
class ModelClientLike(Protocol):
|
|
120
|
+
async def complete(
|
|
121
|
+
self, request: ModelCompletionRequest
|
|
122
|
+
) -> ModelCompletionResponse: ...
|
|
123
|
+
|
|
124
|
+
|
|
125
|
+
class ToolExecutorLike(Protocol):
|
|
126
|
+
async def execute(
|
|
127
|
+
self, call: ValidatedToolCall, context: ToolExecutionContext
|
|
128
|
+
) -> ToolExecutionResult: ...
|
|
129
|
+
|
|
130
|
+
def supports_tool(self, name: str) -> bool: ...
|
|
131
|
+
|
|
132
|
+
|
|
133
|
+
class CancellationTokenLike(Protocol):
|
|
134
|
+
@property
|
|
135
|
+
def cancellation_id(self) -> str: ...
|
|
136
|
+
|
|
137
|
+
def is_cancelled(self) -> bool: ...
|
|
138
|
+
|
|
139
|
+
async def wait(self) -> None: ...
|
|
140
|
+
|
|
141
|
+
@property
|
|
142
|
+
def reason(self) -> str | None: ...
|
|
143
|
+
|
|
144
|
+
|
|
145
|
+
class CancellationResolverLike(Protocol):
|
|
146
|
+
def resolve(self, ref: Any) -> CancellationTokenLike: ...
|
|
147
|
+
|
|
148
|
+
|
|
149
|
+
class RuntimeClockLike(Protocol):
|
|
150
|
+
def utc_now(self) -> Any: ...
|
|
151
|
+
|
|
152
|
+
def monotonic(self) -> float: ...
|
|
153
|
+
|
|
154
|
+
|
|
155
|
+
class ForgeBindingRejectedError(ValueError):
|
|
156
|
+
"""Compiled semantics cannot be represented by the private Forge subset."""
|
|
157
|
+
|
|
158
|
+
code = "binding_rejected"
|
|
159
|
+
|
|
160
|
+
|
|
161
|
+
class ForgeBridgeError(ValueError):
|
|
162
|
+
"""Bridge-owned validation rejected a private Forge interaction."""
|
|
163
|
+
|
|
164
|
+
code = "bridge_rejected"
|
|
165
|
+
|
|
166
|
+
|
|
167
|
+
class _WorkflowOperationCancelledError(
|
|
168
|
+
OperationCancelledError,
|
|
169
|
+
NonRetryableToolError,
|
|
170
|
+
):
|
|
171
|
+
pass
|
|
172
|
+
|
|
173
|
+
|
|
174
|
+
class _WorkflowDeadlineExceededError(
|
|
175
|
+
DeadlineExceededError,
|
|
176
|
+
NonRetryableToolError,
|
|
177
|
+
):
|
|
178
|
+
pass
|
|
179
|
+
|
|
180
|
+
|
|
181
|
+
DiagnosticCategory = Literal[
|
|
182
|
+
"binding",
|
|
183
|
+
"compiled_harness",
|
|
184
|
+
"backend",
|
|
185
|
+
"model",
|
|
186
|
+
"tool",
|
|
187
|
+
"budget",
|
|
188
|
+
"timeout",
|
|
189
|
+
"cancellation",
|
|
190
|
+
"artifact",
|
|
191
|
+
"internal",
|
|
192
|
+
]
|
|
193
|
+
|
|
194
|
+
|
|
195
|
+
@dataclass(frozen=True)
|
|
196
|
+
class TerminalCandidate:
|
|
197
|
+
"""Internal terminal candidate before it is accepted as ``TerminalIntent``."""
|
|
198
|
+
|
|
199
|
+
call_id: str
|
|
200
|
+
node_id: str
|
|
201
|
+
tool_name: str
|
|
202
|
+
terminal_result: str
|
|
203
|
+
summary: str
|
|
204
|
+
artifact_refs: tuple[ArtifactRef, ...]
|
|
205
|
+
selected_output: SelectedOutput | None = None
|
|
206
|
+
selected_output_schema_sha256: str | None = None
|
|
207
|
+
|
|
208
|
+
|
|
209
|
+
@dataclass(frozen=True)
|
|
210
|
+
class ForgeRunnerOptions:
|
|
211
|
+
"""Budget values passed to ``WorkflowRunner``."""
|
|
212
|
+
|
|
213
|
+
max_iterations: int
|
|
214
|
+
max_retries_per_step: int
|
|
215
|
+
max_tool_errors: int
|
|
216
|
+
max_premature_attempts: int
|
|
217
|
+
max_prereq_violations: int
|
|
218
|
+
|
|
219
|
+
|
|
220
|
+
@dataclass(frozen=True)
|
|
221
|
+
class ForgeWorkflowInput:
|
|
222
|
+
"""Private Forge workflow plus Millforge-owned translation metadata."""
|
|
223
|
+
|
|
224
|
+
workflow: Workflow
|
|
225
|
+
runner_options: ForgeRunnerOptions
|
|
226
|
+
binding_by_tool: dict[str, str]
|
|
227
|
+
node_id_by_tool: dict[str, str]
|
|
228
|
+
terminal_result_by_tool: dict[str, str]
|
|
229
|
+
cancellation_id: str | None = None
|
|
230
|
+
|
|
231
|
+
|
|
232
|
+
class ForgeGuardrailBackend:
|
|
233
|
+
"""GuardrailBackend implementation backed by the private Forge runner subset."""
|
|
234
|
+
|
|
235
|
+
def __init__(
|
|
236
|
+
self,
|
|
237
|
+
*,
|
|
238
|
+
model_client: ModelClient,
|
|
239
|
+
tool_executor: ToolExecutor,
|
|
240
|
+
plan_loader: CompiledHarnessLoader,
|
|
241
|
+
context_factory: ForgeContextFactory,
|
|
242
|
+
clock: RuntimeClock,
|
|
243
|
+
cancellation_resolver: CancellationResolver,
|
|
244
|
+
) -> None:
|
|
245
|
+
self._model_client = model_client
|
|
246
|
+
self._tool_executor = tool_executor
|
|
247
|
+
self._plan_loader = plan_loader
|
|
248
|
+
self._context_factory = context_factory
|
|
249
|
+
self._clock = clock
|
|
250
|
+
self._cancellation_resolver = cancellation_resolver
|
|
251
|
+
self._requests: list[GuardedSessionRequest] = []
|
|
252
|
+
|
|
253
|
+
@property
|
|
254
|
+
def requests(self) -> tuple[GuardedSessionRequest, ...]:
|
|
255
|
+
return tuple(self._requests)
|
|
256
|
+
|
|
257
|
+
@property
|
|
258
|
+
def call_count(self) -> int:
|
|
259
|
+
return len(self._requests)
|
|
260
|
+
|
|
261
|
+
async def run_session(self, request: GuardedSessionRequest) -> GuardedSessionResult:
|
|
262
|
+
"""Run one guarded Forge workflow session."""
|
|
263
|
+
self._requests.append(request)
|
|
264
|
+
started_at = self._clock.utc_now()
|
|
265
|
+
event_translator = ForgeEventTranslator(
|
|
266
|
+
session_request=request,
|
|
267
|
+
clock=self._clock,
|
|
268
|
+
)
|
|
269
|
+
tool_bridge: ForgeToolBridge | None = None
|
|
270
|
+
model_bridge: ForgeModelBridge | None = None
|
|
271
|
+
cancellation_watcher: asyncio.Task[None] | None = None
|
|
272
|
+
|
|
273
|
+
try:
|
|
274
|
+
self._validate_request(request)
|
|
275
|
+
token = self._cancellation_resolver.resolve(
|
|
276
|
+
request.execution_request.cancellation
|
|
277
|
+
)
|
|
278
|
+
self._check_deadline(request)
|
|
279
|
+
self._check_cancelled(token)
|
|
280
|
+
self._check_remaining_deadline(request)
|
|
281
|
+
cancel_event = asyncio.Event()
|
|
282
|
+
cancellation_watcher = asyncio.create_task(
|
|
283
|
+
_watch_cancellation(token, cancel_event),
|
|
284
|
+
name=f"millforge-cancellation-watcher:{request.session_id}",
|
|
285
|
+
)
|
|
286
|
+
|
|
287
|
+
plan = await self._load_verified_plan(request)
|
|
288
|
+
self._recheck_plan_against_request(plan, request)
|
|
289
|
+
self._check_cancelled(token)
|
|
290
|
+
event_translator.emit(SessionEventType.SESSION_STARTED)
|
|
291
|
+
|
|
292
|
+
model_bridge = ForgeModelBridge(
|
|
293
|
+
model_client=self._model_client,
|
|
294
|
+
model=request.execution_request.model_profile.profile_id,
|
|
295
|
+
event_translator=event_translator,
|
|
296
|
+
selected_output_requirements=(
|
|
297
|
+
request.execution_request.selected_output_requirements
|
|
298
|
+
),
|
|
299
|
+
terminal_result_by_tool={
|
|
300
|
+
node.model_tool_name: node.terminal_result
|
|
301
|
+
for node in plan.nodes
|
|
302
|
+
if node.terminal_result is not None
|
|
303
|
+
},
|
|
304
|
+
)
|
|
305
|
+
tool_bridge = ForgeToolBridge(
|
|
306
|
+
plan=plan,
|
|
307
|
+
session_request=request,
|
|
308
|
+
executor=self._tool_executor,
|
|
309
|
+
cancellation_resolver=self._cancellation_resolver,
|
|
310
|
+
clock=self._clock,
|
|
311
|
+
)
|
|
312
|
+
workflow_input = ForgeWorkflowFactory(
|
|
313
|
+
{
|
|
314
|
+
node.binding.implementation_id: tool_bridge.make_callable(
|
|
315
|
+
node.model_tool_name
|
|
316
|
+
)
|
|
317
|
+
for node in plan.nodes
|
|
318
|
+
},
|
|
319
|
+
cancellation_id=token.cancellation_id,
|
|
320
|
+
).build(plan)
|
|
321
|
+
event_translator.workflow_constructed(
|
|
322
|
+
tool_count=len(workflow_input.workflow.tools)
|
|
323
|
+
)
|
|
324
|
+
context_manager = self._context_factory.build(plan.context_policy)
|
|
325
|
+
runner = WorkflowRunner(
|
|
326
|
+
model_bridge,
|
|
327
|
+
context_manager,
|
|
328
|
+
max_iterations=workflow_input.runner_options.max_iterations,
|
|
329
|
+
max_retries_per_step=workflow_input.runner_options.max_retries_per_step,
|
|
330
|
+
max_tool_errors=workflow_input.runner_options.max_tool_errors,
|
|
331
|
+
max_premature_attempts=workflow_input.runner_options.max_premature_attempts,
|
|
332
|
+
max_prereq_violations=workflow_input.runner_options.max_prereq_violations,
|
|
333
|
+
stream=False,
|
|
334
|
+
on_message=_runner_message_observer(event_translator),
|
|
335
|
+
tool_call_invoker=lambda call: tool_bridge.invoke(
|
|
336
|
+
call.tool,
|
|
337
|
+
call.args,
|
|
338
|
+
call_id=call.call_id,
|
|
339
|
+
),
|
|
340
|
+
)
|
|
341
|
+
initial_messages = ForgeSessionInputBuilder().build(plan, request)
|
|
342
|
+
await runner.run(
|
|
343
|
+
workflow_input.workflow,
|
|
344
|
+
user_message="",
|
|
345
|
+
initial_messages=initial_messages,
|
|
346
|
+
cancel_event=cancel_event,
|
|
347
|
+
)
|
|
348
|
+
self._check_cancelled(token)
|
|
349
|
+
if tool_bridge.terminal_intent is None:
|
|
350
|
+
raise ForgeBridgeError("Workflow completed without terminal intent")
|
|
351
|
+
status = (
|
|
352
|
+
GuardedSessionStatus.REJECTED
|
|
353
|
+
if tool_bridge.terminal_intent.disposition in {"blocked", "rejected"}
|
|
354
|
+
else GuardedSessionStatus.TERMINAL
|
|
355
|
+
)
|
|
356
|
+
return self._result(
|
|
357
|
+
request,
|
|
358
|
+
status=status,
|
|
359
|
+
started_at=started_at,
|
|
360
|
+
terminal_intent=tool_bridge.terminal_intent,
|
|
361
|
+
artifact_refs=tool_bridge.terminal_intent.artifact_refs,
|
|
362
|
+
usage=_usage_from_bridges(model_bridge, tool_bridge),
|
|
363
|
+
events=_ordered_events(event_translator.events, tool_bridge.events),
|
|
364
|
+
tool_trace=tool_bridge.tool_trace,
|
|
365
|
+
)
|
|
366
|
+
except OperationCancelledError:
|
|
367
|
+
return self._failure_result(
|
|
368
|
+
request,
|
|
369
|
+
status=GuardedSessionStatus.CANCELLED,
|
|
370
|
+
code="workflow_cancelled",
|
|
371
|
+
category="cancellation",
|
|
372
|
+
message="Workflow cancelled",
|
|
373
|
+
started_at=started_at,
|
|
374
|
+
events=_ordered_events(
|
|
375
|
+
event_translator.events,
|
|
376
|
+
tool_bridge.events if tool_bridge is not None else (),
|
|
377
|
+
_single_event(event_translator, SessionEventType.CANCELLED),
|
|
378
|
+
),
|
|
379
|
+
tool_trace=tool_bridge.tool_trace if tool_bridge is not None else (),
|
|
380
|
+
usage=_usage_from_optional_bridges(model_bridge, tool_bridge),
|
|
381
|
+
)
|
|
382
|
+
except DeadlineExceededError:
|
|
383
|
+
return self._failure_result(
|
|
384
|
+
request,
|
|
385
|
+
status=GuardedSessionStatus.TIMED_OUT,
|
|
386
|
+
code="deadline_expired",
|
|
387
|
+
category="timeout",
|
|
388
|
+
message="Workflow deadline expired",
|
|
389
|
+
started_at=started_at,
|
|
390
|
+
events=_ordered_events(
|
|
391
|
+
event_translator.events,
|
|
392
|
+
tool_bridge.events if tool_bridge is not None else (),
|
|
393
|
+
_single_event(event_translator, SessionEventType.TIMED_OUT),
|
|
394
|
+
),
|
|
395
|
+
tool_trace=tool_bridge.tool_trace if tool_bridge is not None else (),
|
|
396
|
+
usage=_usage_from_optional_bridges(model_bridge, tool_bridge),
|
|
397
|
+
)
|
|
398
|
+
except (ForgeBindingRejectedError, ForgeBridgeError) as exc:
|
|
399
|
+
return self._failure_result(
|
|
400
|
+
request,
|
|
401
|
+
status=GuardedSessionStatus.BACKEND_FAILED,
|
|
402
|
+
code=getattr(exc, "code", "bridge_rejected"),
|
|
403
|
+
category="backend",
|
|
404
|
+
message="Forge backend rejected the compiled workflow",
|
|
405
|
+
started_at=started_at,
|
|
406
|
+
events=_ordered_events(event_translator.events),
|
|
407
|
+
usage=_usage_from_optional_bridges(model_bridge, tool_bridge),
|
|
408
|
+
)
|
|
409
|
+
except ToolCallError:
|
|
410
|
+
event_translator.correction_issued(code="tool_arg_validation")
|
|
411
|
+
return self._failure_result(
|
|
412
|
+
request,
|
|
413
|
+
status=GuardedSessionStatus.MODEL_FAILED,
|
|
414
|
+
code="malformed_tool_call",
|
|
415
|
+
category="model",
|
|
416
|
+
message="Model did not produce a valid tool call",
|
|
417
|
+
started_at=started_at,
|
|
418
|
+
events=_ordered_events(event_translator.events),
|
|
419
|
+
usage=_usage_from_optional_bridges(model_bridge, tool_bridge),
|
|
420
|
+
)
|
|
421
|
+
except StepEnforcementError as exc:
|
|
422
|
+
event_translator.correction_issued(code="step")
|
|
423
|
+
event_translator.premature_terminal_rejected(
|
|
424
|
+
node_id=workflow_input.node_id_by_tool.get(exc.terminal_tool)
|
|
425
|
+
)
|
|
426
|
+
return self._failure_result(
|
|
427
|
+
request,
|
|
428
|
+
status=GuardedSessionStatus.INVALID_TERMINAL,
|
|
429
|
+
code="workflow_order_rejected",
|
|
430
|
+
category="backend",
|
|
431
|
+
message="Model exhausted ordered workflow corrections",
|
|
432
|
+
started_at=started_at,
|
|
433
|
+
events=_ordered_events(
|
|
434
|
+
event_translator.events,
|
|
435
|
+
tool_bridge.events if tool_bridge is not None else (),
|
|
436
|
+
),
|
|
437
|
+
tool_trace=tool_bridge.tool_trace if tool_bridge is not None else (),
|
|
438
|
+
usage=_usage_from_optional_bridges(model_bridge, tool_bridge),
|
|
439
|
+
)
|
|
440
|
+
except PrerequisiteError:
|
|
441
|
+
event_translator.budget_exhausted(code="prerequisite_budget_exhausted")
|
|
442
|
+
return self._failure_result(
|
|
443
|
+
request,
|
|
444
|
+
status=GuardedSessionStatus.PREREQUISITE_BUDGET_EXHAUSTED,
|
|
445
|
+
code="prerequisite_budget_exhausted",
|
|
446
|
+
category="backend",
|
|
447
|
+
message="Model exhausted prerequisite correction budget",
|
|
448
|
+
started_at=started_at,
|
|
449
|
+
events=_ordered_events(
|
|
450
|
+
event_translator.events,
|
|
451
|
+
tool_bridge.events if tool_bridge is not None else (),
|
|
452
|
+
),
|
|
453
|
+
tool_trace=tool_bridge.tool_trace if tool_bridge is not None else (),
|
|
454
|
+
usage=_usage_from_optional_bridges(model_bridge, tool_bridge),
|
|
455
|
+
)
|
|
456
|
+
except (MaxIterationsError, ContextBudgetExceeded):
|
|
457
|
+
event_translator.budget_exhausted(code="workflow_budget_exhausted")
|
|
458
|
+
return self._failure_result(
|
|
459
|
+
request,
|
|
460
|
+
status=GuardedSessionStatus.BUDGET_EXHAUSTED,
|
|
461
|
+
code="workflow_budget_exhausted",
|
|
462
|
+
category="backend",
|
|
463
|
+
message="Workflow budget exhausted",
|
|
464
|
+
started_at=started_at,
|
|
465
|
+
events=_ordered_events(
|
|
466
|
+
event_translator.events,
|
|
467
|
+
tool_bridge.events if tool_bridge is not None else (),
|
|
468
|
+
),
|
|
469
|
+
tool_trace=tool_bridge.tool_trace if tool_bridge is not None else (),
|
|
470
|
+
usage=_usage_from_optional_bridges(model_bridge, tool_bridge),
|
|
471
|
+
)
|
|
472
|
+
except WorkflowCancelledError:
|
|
473
|
+
return self._failure_result(
|
|
474
|
+
request,
|
|
475
|
+
status=GuardedSessionStatus.CANCELLED,
|
|
476
|
+
code="workflow_cancelled",
|
|
477
|
+
category="cancellation",
|
|
478
|
+
message="Workflow cancelled",
|
|
479
|
+
started_at=started_at,
|
|
480
|
+
events=_ordered_events(
|
|
481
|
+
event_translator.events,
|
|
482
|
+
tool_bridge.events if tool_bridge is not None else (),
|
|
483
|
+
_single_event(event_translator, SessionEventType.CANCELLED),
|
|
484
|
+
),
|
|
485
|
+
tool_trace=tool_bridge.tool_trace if tool_bridge is not None else (),
|
|
486
|
+
usage=_usage_from_optional_bridges(model_bridge, tool_bridge),
|
|
487
|
+
)
|
|
488
|
+
except (ToolExecutionError, ToolResolutionError, NonRetryableToolError):
|
|
489
|
+
return self._failure_result(
|
|
490
|
+
request,
|
|
491
|
+
status=GuardedSessionStatus.TOOL_FAILED,
|
|
492
|
+
code="tool_execution_failed",
|
|
493
|
+
category="tool",
|
|
494
|
+
message="Tool execution failed",
|
|
495
|
+
started_at=started_at,
|
|
496
|
+
events=_ordered_events(
|
|
497
|
+
event_translator.events,
|
|
498
|
+
tool_bridge.events if tool_bridge is not None else (),
|
|
499
|
+
),
|
|
500
|
+
tool_trace=tool_bridge.tool_trace if tool_bridge is not None else (),
|
|
501
|
+
usage=_usage_from_optional_bridges(model_bridge, tool_bridge),
|
|
502
|
+
)
|
|
503
|
+
except (BackendTranslationError, ToolInvokeError):
|
|
504
|
+
return self._failure_result(
|
|
505
|
+
request,
|
|
506
|
+
status=GuardedSessionStatus.BACKEND_FAILED,
|
|
507
|
+
code="backend_translation_failed",
|
|
508
|
+
category="backend",
|
|
509
|
+
message="Forge backend translation failed",
|
|
510
|
+
started_at=started_at,
|
|
511
|
+
events=_ordered_events(event_translator.events),
|
|
512
|
+
usage=_usage_from_optional_bridges(model_bridge, tool_bridge),
|
|
513
|
+
)
|
|
514
|
+
except ModelTransportError as exc:
|
|
515
|
+
if (
|
|
516
|
+
isinstance(exc, ModelProviderError)
|
|
517
|
+
and exc.category is ProviderErrorCategory.CANCELLED
|
|
518
|
+
):
|
|
519
|
+
return self._failure_result(
|
|
520
|
+
request,
|
|
521
|
+
status=GuardedSessionStatus.CANCELLED,
|
|
522
|
+
code="workflow_cancelled",
|
|
523
|
+
category="cancellation",
|
|
524
|
+
message="Workflow cancelled",
|
|
525
|
+
started_at=started_at,
|
|
526
|
+
events=_ordered_events(
|
|
527
|
+
event_translator.events,
|
|
528
|
+
tool_bridge.events if tool_bridge is not None else (),
|
|
529
|
+
_single_event(event_translator, SessionEventType.CANCELLED),
|
|
530
|
+
),
|
|
531
|
+
tool_trace=tool_bridge.tool_trace
|
|
532
|
+
if tool_bridge is not None
|
|
533
|
+
else (),
|
|
534
|
+
usage=_usage_from_optional_bridges(model_bridge, tool_bridge),
|
|
535
|
+
)
|
|
536
|
+
if isinstance(exc, ModelRequestDeadlineExceededError):
|
|
537
|
+
return self._failure_result(
|
|
538
|
+
request,
|
|
539
|
+
status=GuardedSessionStatus.TIMED_OUT,
|
|
540
|
+
code="deadline_expired",
|
|
541
|
+
category="timeout",
|
|
542
|
+
message="Workflow deadline expired",
|
|
543
|
+
started_at=started_at,
|
|
544
|
+
events=_ordered_events(
|
|
545
|
+
event_translator.events,
|
|
546
|
+
tool_bridge.events if tool_bridge is not None else (),
|
|
547
|
+
_single_event(event_translator, SessionEventType.TIMED_OUT),
|
|
548
|
+
),
|
|
549
|
+
tool_trace=tool_bridge.tool_trace
|
|
550
|
+
if tool_bridge is not None
|
|
551
|
+
else (),
|
|
552
|
+
usage=_usage_from_optional_bridges(model_bridge, tool_bridge),
|
|
553
|
+
)
|
|
554
|
+
return self._failure_result(
|
|
555
|
+
request,
|
|
556
|
+
status=GuardedSessionStatus.MODEL_FAILED,
|
|
557
|
+
code="model_transport_failed",
|
|
558
|
+
category="model",
|
|
559
|
+
message="Model transport failed",
|
|
560
|
+
started_at=started_at,
|
|
561
|
+
events=_ordered_events(event_translator.events),
|
|
562
|
+
usage=_usage_from_optional_bridges(model_bridge, tool_bridge),
|
|
563
|
+
)
|
|
564
|
+
except ForgeError:
|
|
565
|
+
return self._failure_result(
|
|
566
|
+
request,
|
|
567
|
+
status=GuardedSessionStatus.BACKEND_FAILED,
|
|
568
|
+
code="forge_backend_failed",
|
|
569
|
+
category="backend",
|
|
570
|
+
message="Forge backend failed",
|
|
571
|
+
started_at=started_at,
|
|
572
|
+
events=_ordered_events(event_translator.events),
|
|
573
|
+
usage=_usage_from_optional_bridges(model_bridge, tool_bridge),
|
|
574
|
+
)
|
|
575
|
+
except Exception as exc:
|
|
576
|
+
category: DiagnosticCategory
|
|
577
|
+
if model_bridge is not None and (
|
|
578
|
+
model_bridge.in_model_call or model_bridge.model_exception_seen
|
|
579
|
+
):
|
|
580
|
+
status = GuardedSessionStatus.MODEL_FAILED
|
|
581
|
+
code = "model_transport_failed"
|
|
582
|
+
category = "model"
|
|
583
|
+
message = "Model transport failed"
|
|
584
|
+
else:
|
|
585
|
+
status = GuardedSessionStatus.BACKEND_FAILED
|
|
586
|
+
code = "unknown_forge_exception"
|
|
587
|
+
category = "backend"
|
|
588
|
+
message = "Forge backend failed"
|
|
589
|
+
return self._failure_result(
|
|
590
|
+
request,
|
|
591
|
+
status=status,
|
|
592
|
+
code=code,
|
|
593
|
+
category=category,
|
|
594
|
+
message=message,
|
|
595
|
+
started_at=started_at,
|
|
596
|
+
events=_ordered_events(
|
|
597
|
+
event_translator.events,
|
|
598
|
+
tool_bridge.events if tool_bridge is not None else (),
|
|
599
|
+
),
|
|
600
|
+
tool_trace=tool_bridge.tool_trace if tool_bridge is not None else (),
|
|
601
|
+
usage=_usage_from_optional_bridges(model_bridge, tool_bridge),
|
|
602
|
+
cause=exc,
|
|
603
|
+
)
|
|
604
|
+
finally:
|
|
605
|
+
await _cancel_and_await_watcher(cancellation_watcher)
|
|
606
|
+
|
|
607
|
+
def _validate_request(self, request: GuardedSessionRequest) -> None:
|
|
608
|
+
if not request.session_id.strip():
|
|
609
|
+
raise BackendTranslationError("guarded session_id must be non-empty")
|
|
610
|
+
execution = request.execution_request
|
|
611
|
+
if not execution.request_id.strip() or not execution.run_id.strip():
|
|
612
|
+
raise BackendTranslationError("execution identity must be non-empty")
|
|
613
|
+
|
|
614
|
+
def _check_deadline(self, request: GuardedSessionRequest) -> None:
|
|
615
|
+
now = self._clock.monotonic()
|
|
616
|
+
if now >= request.deadline.effective_deadline_monotonic:
|
|
617
|
+
raise DeadlineExceededError("guarded session deadline expired")
|
|
618
|
+
|
|
619
|
+
def _check_remaining_deadline(self, request: GuardedSessionRequest) -> None:
|
|
620
|
+
if request.deadline.remaining(self._clock) <= 0:
|
|
621
|
+
raise DeadlineExceededError("guarded session has no remaining deadline")
|
|
622
|
+
|
|
623
|
+
def _check_cancelled(self, token: CancellationTokenLike) -> None:
|
|
624
|
+
if token.is_cancelled():
|
|
625
|
+
raise OperationCancelledError(token.reason or "workflow cancelled")
|
|
626
|
+
|
|
627
|
+
async def _load_verified_plan(
|
|
628
|
+
self, request: GuardedSessionRequest
|
|
629
|
+
) -> CompiledHarnessPlan:
|
|
630
|
+
execution = request.execution_request
|
|
631
|
+
plan = await self._plan_loader.load(execution.compiled_harness)
|
|
632
|
+
verified, computed_hash, warnings, restored = verify_compiled_plan_sha256(
|
|
633
|
+
canonical_json_serialize(plan.model_dump(mode="json")),
|
|
634
|
+
expected_compiled_hash=execution.compiled_harness.expected_hash.digest,
|
|
635
|
+
expected_harness_id=execution.compiled_harness.identity.harness_id,
|
|
636
|
+
expected_harness_version=(
|
|
637
|
+
execution.compiled_harness.identity.harness_version
|
|
638
|
+
),
|
|
639
|
+
)
|
|
640
|
+
if not verified or restored is None:
|
|
641
|
+
if computed_hash != plan.compiled_sha256:
|
|
642
|
+
raise ForgeBindingRejectedError("Compiled plan canonical hash mismatch")
|
|
643
|
+
if computed_hash != execution.compiled_harness.expected_hash.digest:
|
|
644
|
+
raise ForgeBindingRejectedError("Compiled plan expected hash mismatch")
|
|
645
|
+
raise ForgeBindingRejectedError("; ".join(warnings))
|
|
646
|
+
return restored
|
|
647
|
+
|
|
648
|
+
def _recheck_plan_against_request(
|
|
649
|
+
self,
|
|
650
|
+
plan: CompiledHarnessPlan,
|
|
651
|
+
request: GuardedSessionRequest,
|
|
652
|
+
) -> None:
|
|
653
|
+
execution = request.execution_request
|
|
654
|
+
ref = execution.compiled_harness
|
|
655
|
+
if plan.harness_id != ref.identity.harness_id:
|
|
656
|
+
raise ForgeBindingRejectedError("Compiled plan harness_id mismatch")
|
|
657
|
+
if plan.harness_version != ref.identity.harness_version:
|
|
658
|
+
raise ForgeBindingRejectedError("Compiled plan harness_version mismatch")
|
|
659
|
+
if execution.stage.plane != "execution":
|
|
660
|
+
raise ForgeBindingRejectedError("Guarded session stage is not execution")
|
|
661
|
+
if execution.stage.stage_kind_id not in plan.stage_kind_ids:
|
|
662
|
+
raise ForgeBindingRejectedError("Compiled plan does not support stage kind")
|
|
663
|
+
if execution.model_profile.profile_id != plan.model_profile.profile_id:
|
|
664
|
+
raise ForgeBindingRejectedError("Compiled plan model profile mismatch")
|
|
665
|
+
granted = {
|
|
666
|
+
grant.capability_id for grant in execution.capability_envelope.grants
|
|
667
|
+
}
|
|
668
|
+
missing = sorted(set(plan.required_capabilities) - granted)
|
|
669
|
+
if missing:
|
|
670
|
+
raise ForgeBindingRejectedError("Compiled plan capability grants missing")
|
|
671
|
+
|
|
672
|
+
def _result(
|
|
673
|
+
self,
|
|
674
|
+
request: GuardedSessionRequest,
|
|
675
|
+
*,
|
|
676
|
+
status: GuardedSessionStatus,
|
|
677
|
+
started_at: Any,
|
|
678
|
+
terminal_intent: TerminalIntent | None = None,
|
|
679
|
+
artifact_refs: tuple[ArtifactRef, ...] = (),
|
|
680
|
+
usage: UsageMetadata | None = None,
|
|
681
|
+
diagnostic: DiagnosticMetadata | None = None,
|
|
682
|
+
events: tuple[SessionEvent, ...] = (),
|
|
683
|
+
tool_trace: tuple[ToolTraceRecord, ...] = (),
|
|
684
|
+
) -> GuardedSessionResult:
|
|
685
|
+
return GuardedSessionResult(
|
|
686
|
+
session_id=request.session_id,
|
|
687
|
+
status=status,
|
|
688
|
+
terminal_intent=terminal_intent,
|
|
689
|
+
artifact_refs=artifact_refs,
|
|
690
|
+
usage=usage,
|
|
691
|
+
timing=_timing_metadata(started_at, self._clock.utc_now()),
|
|
692
|
+
diagnostic=diagnostic,
|
|
693
|
+
events=events,
|
|
694
|
+
tool_trace=tool_trace,
|
|
695
|
+
)
|
|
696
|
+
|
|
697
|
+
def _failure_result(
|
|
698
|
+
self,
|
|
699
|
+
request: GuardedSessionRequest,
|
|
700
|
+
*,
|
|
701
|
+
status: GuardedSessionStatus,
|
|
702
|
+
code: str,
|
|
703
|
+
category: DiagnosticCategory,
|
|
704
|
+
message: str,
|
|
705
|
+
started_at: Any,
|
|
706
|
+
events: tuple[SessionEvent, ...],
|
|
707
|
+
tool_trace: tuple[ToolTraceRecord, ...] = (),
|
|
708
|
+
usage: UsageMetadata | None = None,
|
|
709
|
+
cause: Exception | None = None,
|
|
710
|
+
) -> GuardedSessionResult:
|
|
711
|
+
_ = cause
|
|
712
|
+
diagnostic = DiagnosticMetadata(
|
|
713
|
+
error_code=code,
|
|
714
|
+
category=category,
|
|
715
|
+
message=message,
|
|
716
|
+
retryable=status
|
|
717
|
+
in {
|
|
718
|
+
GuardedSessionStatus.BACKEND_FAILED,
|
|
719
|
+
GuardedSessionStatus.MODEL_FAILED,
|
|
720
|
+
GuardedSessionStatus.TOOL_FAILED,
|
|
721
|
+
},
|
|
722
|
+
origin=code,
|
|
723
|
+
fields=(),
|
|
724
|
+
)
|
|
725
|
+
return self._result(
|
|
726
|
+
request,
|
|
727
|
+
status=status,
|
|
728
|
+
started_at=started_at,
|
|
729
|
+
diagnostic=diagnostic,
|
|
730
|
+
usage=usage,
|
|
731
|
+
events=events,
|
|
732
|
+
tool_trace=tool_trace,
|
|
733
|
+
)
|
|
734
|
+
|
|
735
|
+
|
|
736
|
+
class ForgeEventTranslator:
|
|
737
|
+
"""Bridge-owned typed event collector for private Forge activity."""
|
|
738
|
+
|
|
739
|
+
def __init__(
|
|
740
|
+
self,
|
|
741
|
+
*,
|
|
742
|
+
session_request: GuardedSessionRequest,
|
|
743
|
+
clock: RuntimeClockLike,
|
|
744
|
+
) -> None:
|
|
745
|
+
self._session_request = session_request
|
|
746
|
+
self._clock = clock
|
|
747
|
+
self._events: list[SessionEvent] = []
|
|
748
|
+
|
|
749
|
+
@property
|
|
750
|
+
def events(self) -> tuple[SessionEvent, ...]:
|
|
751
|
+
return tuple(self._events)
|
|
752
|
+
|
|
753
|
+
def workflow_constructed(self, *, tool_count: int) -> SessionEvent:
|
|
754
|
+
return self.emit(
|
|
755
|
+
SessionEventType.WORKFLOW_CONSTRUCTED,
|
|
756
|
+
fields={"tool_count": tool_count},
|
|
757
|
+
)
|
|
758
|
+
|
|
759
|
+
def correction_issued(self, *, code: str) -> SessionEvent:
|
|
760
|
+
return self.emit(SessionEventType.CORRECTION_ISSUED, code=code)
|
|
761
|
+
|
|
762
|
+
def premature_terminal_rejected(
|
|
763
|
+
self, *, node_id: str | None = None
|
|
764
|
+
) -> SessionEvent:
|
|
765
|
+
return self.emit(
|
|
766
|
+
SessionEventType.PREMATURE_TERMINAL_REJECTED,
|
|
767
|
+
node_id=node_id,
|
|
768
|
+
)
|
|
769
|
+
|
|
770
|
+
def context_compacted(self, *, kept_messages: int) -> SessionEvent:
|
|
771
|
+
return self.emit(
|
|
772
|
+
SessionEventType.CONTEXT_COMPACTED,
|
|
773
|
+
fields={"kept_messages": kept_messages},
|
|
774
|
+
)
|
|
775
|
+
|
|
776
|
+
def budget_exhausted(self, *, code: str) -> SessionEvent:
|
|
777
|
+
return self.emit(SessionEventType.BUDGET_EXHAUSTED, code=code)
|
|
778
|
+
|
|
779
|
+
def sanitized_metadata(
|
|
780
|
+
self, **fields: str | int | float | bool | None
|
|
781
|
+
) -> tuple[DiagnosticField, ...]:
|
|
782
|
+
return _diagnostic_fields(fields)
|
|
783
|
+
|
|
784
|
+
def emit(
|
|
785
|
+
self,
|
|
786
|
+
event_type: SessionEventType,
|
|
787
|
+
*,
|
|
788
|
+
node_id: str | None = None,
|
|
789
|
+
model_turn: int | None = None,
|
|
790
|
+
tool_call_id: str | None = None,
|
|
791
|
+
code: str | None = None,
|
|
792
|
+
fields: Mapping[str, str | int | float | bool | None] | None = None,
|
|
793
|
+
) -> SessionEvent:
|
|
794
|
+
request = self._session_request.execution_request
|
|
795
|
+
event = SessionEvent(
|
|
796
|
+
schema_version="1.0",
|
|
797
|
+
sequence=len(self._events) + 1,
|
|
798
|
+
occurred_at=self._clock.utc_now().isoformat(),
|
|
799
|
+
monotonic_offset_ms=self._clock.monotonic() * 1000,
|
|
800
|
+
event_type=event_type,
|
|
801
|
+
request_id=request.request_id,
|
|
802
|
+
run_id=request.run_id,
|
|
803
|
+
session_id=self._session_request.session_id,
|
|
804
|
+
stage=request.stage,
|
|
805
|
+
node_id=node_id,
|
|
806
|
+
model_turn=model_turn,
|
|
807
|
+
tool_call_id=tool_call_id,
|
|
808
|
+
code=code,
|
|
809
|
+
fields=_diagnostic_fields(fields or {}),
|
|
810
|
+
)
|
|
811
|
+
self._events.append(event)
|
|
812
|
+
return event
|
|
813
|
+
|
|
814
|
+
|
|
815
|
+
def _runner_message_observer(
|
|
816
|
+
event_translator: ForgeEventTranslator,
|
|
817
|
+
) -> Callable[[Message], None]:
|
|
818
|
+
def observe(message: Message) -> None:
|
|
819
|
+
metadata_type = message.metadata.type
|
|
820
|
+
if metadata_type == MessageType.STEP_NUDGE:
|
|
821
|
+
event_translator.correction_issued(code="step")
|
|
822
|
+
event_translator.premature_terminal_rejected()
|
|
823
|
+
elif metadata_type == MessageType.PREREQUISITE_NUDGE:
|
|
824
|
+
event_translator.correction_issued(code="prerequisite")
|
|
825
|
+
elif metadata_type == MessageType.RETRY_NUDGE:
|
|
826
|
+
code = (
|
|
827
|
+
"tool_arg_validation" if message.role == MessageRole.TOOL else "retry"
|
|
828
|
+
)
|
|
829
|
+
event_translator.correction_issued(code=code)
|
|
830
|
+
|
|
831
|
+
return observe
|
|
832
|
+
|
|
833
|
+
|
|
834
|
+
class ForgeWorkflowFactory:
|
|
835
|
+
"""Translate a compiled harness plan into private Forge workflow objects."""
|
|
836
|
+
|
|
837
|
+
def __init__(
|
|
838
|
+
self,
|
|
839
|
+
bindings: Mapping[str, Callable[..., Any]],
|
|
840
|
+
*,
|
|
841
|
+
cancellation_id: str | None = None,
|
|
842
|
+
) -> None:
|
|
843
|
+
self._bindings = dict(bindings)
|
|
844
|
+
self._cancellation_id = cancellation_id
|
|
845
|
+
|
|
846
|
+
def build(self, plan: CompiledHarnessPlan) -> ForgeWorkflowInput:
|
|
847
|
+
_validate_plan_identities(plan)
|
|
848
|
+
tool_name_by_node_id = {
|
|
849
|
+
node.node_id: node.model_tool_name for node in plan.nodes
|
|
850
|
+
}
|
|
851
|
+
tools: dict[str, ToolDef] = {}
|
|
852
|
+
binding_by_tool: dict[str, str] = {}
|
|
853
|
+
node_id_by_tool: dict[str, str] = {}
|
|
854
|
+
terminal_result_by_tool: dict[str, str] = {}
|
|
855
|
+
|
|
856
|
+
for node in plan.nodes:
|
|
857
|
+
implementation = self._bindings.get(node.binding.implementation_id)
|
|
858
|
+
if implementation is None:
|
|
859
|
+
_reject(
|
|
860
|
+
"Missing tool binding implementation "
|
|
861
|
+
f"{node.binding.implementation_id!r} for node {node.node_id!r}"
|
|
862
|
+
)
|
|
863
|
+
spec = _tool_spec_from_node(node)
|
|
864
|
+
tools[node.model_tool_name] = ToolDef(
|
|
865
|
+
spec=spec,
|
|
866
|
+
callable=implementation,
|
|
867
|
+
prerequisites=[
|
|
868
|
+
_translate_prerequisite(prereq, tool_name_by_node_id)
|
|
869
|
+
for prereq in node.prerequisites
|
|
870
|
+
],
|
|
871
|
+
)
|
|
872
|
+
binding_by_tool[node.model_tool_name] = node.binding.implementation_id
|
|
873
|
+
node_id_by_tool[node.model_tool_name] = node.node_id
|
|
874
|
+
if node.terminal_result is not None:
|
|
875
|
+
terminal_result_by_tool[node.model_tool_name] = node.terminal_result
|
|
876
|
+
|
|
877
|
+
terminal_tools = [
|
|
878
|
+
node.model_tool_name
|
|
879
|
+
for node in plan.nodes
|
|
880
|
+
if node.terminal_result is not None
|
|
881
|
+
]
|
|
882
|
+
if not terminal_tools:
|
|
883
|
+
_reject("Compiled plan has no terminal nodes")
|
|
884
|
+
|
|
885
|
+
workflow = Workflow(
|
|
886
|
+
name=plan.harness_id,
|
|
887
|
+
description=f"Compiled Millforge harness {plan.harness_id}",
|
|
888
|
+
tools=tools,
|
|
889
|
+
required_steps=[
|
|
890
|
+
node.model_tool_name
|
|
891
|
+
for node in plan.nodes
|
|
892
|
+
if node.required and node.terminal_result is None
|
|
893
|
+
],
|
|
894
|
+
terminal_tool=terminal_tools,
|
|
895
|
+
system_prompt_template=plan.prompt_policy.system_instructions,
|
|
896
|
+
)
|
|
897
|
+
return ForgeWorkflowInput(
|
|
898
|
+
workflow=workflow,
|
|
899
|
+
runner_options=ForgeRunnerOptions(
|
|
900
|
+
max_iterations=plan.budgets.max_iterations,
|
|
901
|
+
max_retries_per_step=plan.budgets.max_validation_retries,
|
|
902
|
+
max_tool_errors=plan.budgets.max_tool_errors,
|
|
903
|
+
max_premature_attempts=plan.budgets.max_premature_terminal_attempts,
|
|
904
|
+
max_prereq_violations=plan.budgets.max_prerequisite_violations,
|
|
905
|
+
),
|
|
906
|
+
binding_by_tool=binding_by_tool,
|
|
907
|
+
node_id_by_tool=node_id_by_tool,
|
|
908
|
+
terminal_result_by_tool=terminal_result_by_tool,
|
|
909
|
+
cancellation_id=self._cancellation_id,
|
|
910
|
+
)
|
|
911
|
+
|
|
912
|
+
|
|
913
|
+
class ForgeSessionInputBuilder:
|
|
914
|
+
"""Build deterministic private Forge initial messages."""
|
|
915
|
+
|
|
916
|
+
def build(
|
|
917
|
+
self,
|
|
918
|
+
plan: CompiledHarnessPlan,
|
|
919
|
+
session_request: GuardedSessionRequest,
|
|
920
|
+
) -> list[Message]:
|
|
921
|
+
request = session_request.execution_request
|
|
922
|
+
content = request.task.instruction
|
|
923
|
+
if plan.prompt_policy.include_request_context:
|
|
924
|
+
context = json.dumps(
|
|
925
|
+
_request_context_payload(request),
|
|
926
|
+
sort_keys=True,
|
|
927
|
+
separators=(",", ":"),
|
|
928
|
+
allow_nan=False,
|
|
929
|
+
)
|
|
930
|
+
content = f"{content}{_REQUEST_CONTEXT_SEPARATOR}{context}"
|
|
931
|
+
return [
|
|
932
|
+
Message(
|
|
933
|
+
MessageRole.SYSTEM,
|
|
934
|
+
plan.prompt_policy.system_instructions,
|
|
935
|
+
MessageMeta(MessageType.SYSTEM_PROMPT),
|
|
936
|
+
),
|
|
937
|
+
Message(
|
|
938
|
+
MessageRole.USER,
|
|
939
|
+
content,
|
|
940
|
+
MessageMeta(MessageType.USER_INPUT),
|
|
941
|
+
),
|
|
942
|
+
]
|
|
943
|
+
|
|
944
|
+
|
|
945
|
+
class ForgeContextFactory:
|
|
946
|
+
"""Construct private Forge context management from compiled policy only."""
|
|
947
|
+
|
|
948
|
+
def build(self, policy: CompiledContextPolicy) -> ContextManager:
|
|
949
|
+
if policy.strategy_id != "forge.tiered.v1":
|
|
950
|
+
_reject(f"Unsupported context strategy {policy.strategy_id!r}")
|
|
951
|
+
return ContextManager(
|
|
952
|
+
strategy=TieredCompact(
|
|
953
|
+
keep_recent=policy.keep_recent_iterations,
|
|
954
|
+
phase_thresholds=policy.phase_thresholds,
|
|
955
|
+
),
|
|
956
|
+
budget_tokens=policy.budget_tokens,
|
|
957
|
+
on_compact=_context_compaction_callback,
|
|
958
|
+
context_thresholds=None,
|
|
959
|
+
on_context_threshold=None,
|
|
960
|
+
)
|
|
961
|
+
|
|
962
|
+
|
|
963
|
+
class ForgeModelBridge:
|
|
964
|
+
"""Private Forge ``LLMClient`` backed by a public ``ModelClient``."""
|
|
965
|
+
|
|
966
|
+
api_format = "openai"
|
|
967
|
+
|
|
968
|
+
def __init__(
|
|
969
|
+
self,
|
|
970
|
+
*,
|
|
971
|
+
model_client: ModelClientLike,
|
|
972
|
+
model: str,
|
|
973
|
+
event_translator: ForgeEventTranslator | None = None,
|
|
974
|
+
selected_output_requirements: tuple[
|
|
975
|
+
TerminalSelectedOutputRequirement, ...
|
|
976
|
+
] = (),
|
|
977
|
+
terminal_result_by_tool: Mapping[str, str] | None = None,
|
|
978
|
+
) -> None:
|
|
979
|
+
self._model_client = model_client
|
|
980
|
+
self._event_translator = event_translator
|
|
981
|
+
self._selected_output_by_terminal_result = (
|
|
982
|
+
_selected_output_requirements_by_terminal_result(
|
|
983
|
+
selected_output_requirements
|
|
984
|
+
)
|
|
985
|
+
)
|
|
986
|
+
self._terminal_result_by_tool = dict(terminal_result_by_tool or {})
|
|
987
|
+
self.model = model
|
|
988
|
+
self.last_usage: dict[int, ForgeTokenUsage] = {}
|
|
989
|
+
self._model_calls = 0
|
|
990
|
+
self._input_tokens = 0
|
|
991
|
+
self._output_tokens = 0
|
|
992
|
+
self._provider_reported = False
|
|
993
|
+
self._in_model_call = False
|
|
994
|
+
self._model_exception_seen = False
|
|
995
|
+
self._model_turn = 0
|
|
996
|
+
|
|
997
|
+
@property
|
|
998
|
+
def events(self) -> tuple[SessionEvent, ...]:
|
|
999
|
+
if self._event_translator is None:
|
|
1000
|
+
return ()
|
|
1001
|
+
return self._event_translator.events
|
|
1002
|
+
|
|
1003
|
+
@property
|
|
1004
|
+
def model_calls(self) -> int:
|
|
1005
|
+
return self._model_calls
|
|
1006
|
+
|
|
1007
|
+
@property
|
|
1008
|
+
def token_usage(self) -> TokenUsage | None:
|
|
1009
|
+
if self._input_tokens == 0 and self._output_tokens == 0:
|
|
1010
|
+
return None
|
|
1011
|
+
return TokenUsage(
|
|
1012
|
+
input_tokens=self._input_tokens,
|
|
1013
|
+
output_tokens=self._output_tokens,
|
|
1014
|
+
total_tokens=self._input_tokens + self._output_tokens,
|
|
1015
|
+
provider_reported=self._provider_reported,
|
|
1016
|
+
)
|
|
1017
|
+
|
|
1018
|
+
@property
|
|
1019
|
+
def in_model_call(self) -> bool:
|
|
1020
|
+
return self._in_model_call
|
|
1021
|
+
|
|
1022
|
+
@property
|
|
1023
|
+
def model_exception_seen(self) -> bool:
|
|
1024
|
+
return self._model_exception_seen
|
|
1025
|
+
|
|
1026
|
+
async def send(
|
|
1027
|
+
self,
|
|
1028
|
+
messages: list[dict[str, str]],
|
|
1029
|
+
tools: list[ToolSpec] | None = None,
|
|
1030
|
+
sampling: dict[str, Any] | None = None,
|
|
1031
|
+
passthrough: dict[str, Any] | None = None,
|
|
1032
|
+
inbound_anthropic_body: dict[str, Any] | None = None,
|
|
1033
|
+
raw_openai_tools: list[dict[str, Any]] | None = None,
|
|
1034
|
+
) -> list[ToolCall] | TextResponse:
|
|
1035
|
+
if sampling:
|
|
1036
|
+
raise ForgeBridgeError(
|
|
1037
|
+
"ForgeModelBridge does not accept sampling overrides"
|
|
1038
|
+
)
|
|
1039
|
+
if passthrough or inbound_anthropic_body or raw_openai_tools:
|
|
1040
|
+
raise ForgeBridgeError(
|
|
1041
|
+
"ForgeModelBridge does not accept provider passthrough"
|
|
1042
|
+
)
|
|
1043
|
+
model_turn = self._model_turn
|
|
1044
|
+
self._model_turn += 1
|
|
1045
|
+
request = ModelCompletionRequest(
|
|
1046
|
+
request_id=self._request_id(model_turn),
|
|
1047
|
+
run_id=self._run_id(),
|
|
1048
|
+
model_profile_id=self.model,
|
|
1049
|
+
messages=tuple(
|
|
1050
|
+
_model_message_from_private(message) for message in messages
|
|
1051
|
+
),
|
|
1052
|
+
tools=tuple(
|
|
1053
|
+
_tool_definition_from_spec(
|
|
1054
|
+
spec,
|
|
1055
|
+
selected_output=(
|
|
1056
|
+
self._selected_output_by_terminal_result.get(
|
|
1057
|
+
self._terminal_result_by_tool.get(spec.name, "")
|
|
1058
|
+
)
|
|
1059
|
+
),
|
|
1060
|
+
)
|
|
1061
|
+
for spec in tools or ()
|
|
1062
|
+
),
|
|
1063
|
+
sampling_overrides=SamplingRequest(),
|
|
1064
|
+
maximum_output_tokens_override=None,
|
|
1065
|
+
request_options={"parallel_tool_calls": False},
|
|
1066
|
+
deadline=self._deadline(),
|
|
1067
|
+
cancellation=self._cancellation(),
|
|
1068
|
+
secret_refs=(
|
|
1069
|
+
self._event_translator._session_request.execution_request.secret_refs
|
|
1070
|
+
if self._event_translator is not None
|
|
1071
|
+
else ()
|
|
1072
|
+
),
|
|
1073
|
+
)
|
|
1074
|
+
if self._event_translator is not None:
|
|
1075
|
+
self._event_translator.emit(
|
|
1076
|
+
SessionEventType.MODEL_REQUEST_STARTED,
|
|
1077
|
+
model_turn=model_turn,
|
|
1078
|
+
fields={
|
|
1079
|
+
"message_count": len(request.messages),
|
|
1080
|
+
"tool_count": len(request.tools),
|
|
1081
|
+
**_prompt_event_fields(request.messages),
|
|
1082
|
+
},
|
|
1083
|
+
)
|
|
1084
|
+
try:
|
|
1085
|
+
self._in_model_call = True
|
|
1086
|
+
response = await self._model_client.complete(request)
|
|
1087
|
+
except Exception as exc:
|
|
1088
|
+
self._model_exception_seen = True
|
|
1089
|
+
if self._event_translator is not None:
|
|
1090
|
+
self._event_translator.emit(
|
|
1091
|
+
SessionEventType.MODEL_REQUEST_FAILED,
|
|
1092
|
+
model_turn=model_turn,
|
|
1093
|
+
code=type(exc).__name__,
|
|
1094
|
+
)
|
|
1095
|
+
raise
|
|
1096
|
+
finally:
|
|
1097
|
+
self._in_model_call = False
|
|
1098
|
+
if self._event_translator is not None:
|
|
1099
|
+
self._event_translator.emit(
|
|
1100
|
+
SessionEventType.MODEL_REQUEST_COMPLETED,
|
|
1101
|
+
model_turn=model_turn,
|
|
1102
|
+
fields={"tool_call_count": len(response.tool_calls)},
|
|
1103
|
+
)
|
|
1104
|
+
self._store_usage(response.usage)
|
|
1105
|
+
if not response.tool_calls:
|
|
1106
|
+
return TextResponse(response.content or "")
|
|
1107
|
+
calls: list[ToolCall] = []
|
|
1108
|
+
for call in response.tool_calls:
|
|
1109
|
+
args = _parsed_model_tool_args(call)
|
|
1110
|
+
calls.append(
|
|
1111
|
+
ToolCall(
|
|
1112
|
+
tool=call.name,
|
|
1113
|
+
args=args,
|
|
1114
|
+
reasoning=response.content,
|
|
1115
|
+
call_id=call.call_id,
|
|
1116
|
+
reasoning_content=response.message.reasoning_content,
|
|
1117
|
+
)
|
|
1118
|
+
)
|
|
1119
|
+
return calls
|
|
1120
|
+
|
|
1121
|
+
async def send_stream(
|
|
1122
|
+
self,
|
|
1123
|
+
messages: list[dict[str, str]],
|
|
1124
|
+
tools: list[ToolSpec] | None = None,
|
|
1125
|
+
sampling: dict[str, Any] | None = None,
|
|
1126
|
+
passthrough: dict[str, Any] | None = None,
|
|
1127
|
+
inbound_anthropic_body: dict[str, Any] | None = None,
|
|
1128
|
+
raw_openai_tools: list[dict[str, Any]] | None = None,
|
|
1129
|
+
) -> AsyncIterator[StreamChunk]:
|
|
1130
|
+
_ = (
|
|
1131
|
+
messages,
|
|
1132
|
+
tools,
|
|
1133
|
+
sampling,
|
|
1134
|
+
passthrough,
|
|
1135
|
+
inbound_anthropic_body,
|
|
1136
|
+
raw_openai_tools,
|
|
1137
|
+
)
|
|
1138
|
+
raise NotImplementedError("ForgeModelBridge explicitly rejects streaming")
|
|
1139
|
+
if False:
|
|
1140
|
+
yield StreamChunk(type="final") # pragma: no cover
|
|
1141
|
+
|
|
1142
|
+
async def get_context_length(self) -> int | None:
|
|
1143
|
+
return None
|
|
1144
|
+
|
|
1145
|
+
async def aclose(self) -> None:
|
|
1146
|
+
return None
|
|
1147
|
+
|
|
1148
|
+
def _request_id(self, model_turn: int) -> str:
|
|
1149
|
+
if self._event_translator is None:
|
|
1150
|
+
return f"model-request-{model_turn}"
|
|
1151
|
+
return self._event_translator._session_request.execution_request.request_id
|
|
1152
|
+
|
|
1153
|
+
def _run_id(self) -> str:
|
|
1154
|
+
if self._event_translator is None:
|
|
1155
|
+
return "model-run"
|
|
1156
|
+
return self._event_translator._session_request.execution_request.run_id
|
|
1157
|
+
|
|
1158
|
+
def _deadline(self) -> Deadline:
|
|
1159
|
+
if self._event_translator is not None:
|
|
1160
|
+
return self._event_translator._session_request.deadline
|
|
1161
|
+
return Deadline(
|
|
1162
|
+
started_monotonic=0.0,
|
|
1163
|
+
outer_deadline_monotonic=1.0,
|
|
1164
|
+
effective_deadline_monotonic=1.0,
|
|
1165
|
+
source="request",
|
|
1166
|
+
)
|
|
1167
|
+
|
|
1168
|
+
def _cancellation(self) -> CancellationRef:
|
|
1169
|
+
if self._event_translator is not None:
|
|
1170
|
+
return (
|
|
1171
|
+
self._event_translator._session_request.execution_request.cancellation
|
|
1172
|
+
)
|
|
1173
|
+
return CancellationRef(cancellation_id="model-cancel")
|
|
1174
|
+
|
|
1175
|
+
def _store_usage(self, usage: TokenUsage | None) -> None:
|
|
1176
|
+
self._model_calls += 1
|
|
1177
|
+
if usage is None:
|
|
1178
|
+
self.last_usage = {}
|
|
1179
|
+
return
|
|
1180
|
+
token_usage = usage
|
|
1181
|
+
self._input_tokens += usage.input_tokens
|
|
1182
|
+
self._output_tokens += usage.output_tokens
|
|
1183
|
+
self._provider_reported = (
|
|
1184
|
+
self._provider_reported or token_usage.provider_reported
|
|
1185
|
+
)
|
|
1186
|
+
self.last_usage = {
|
|
1187
|
+
0: ForgeTokenUsage(
|
|
1188
|
+
prompt_tokens=token_usage.input_tokens,
|
|
1189
|
+
completion_tokens=token_usage.output_tokens,
|
|
1190
|
+
total_tokens=token_usage.total_tokens,
|
|
1191
|
+
)
|
|
1192
|
+
}
|
|
1193
|
+
|
|
1194
|
+
|
|
1195
|
+
class ForgeToolBridge:
|
|
1196
|
+
"""Private Forge tool callables backed by a public ``ToolExecutor``."""
|
|
1197
|
+
|
|
1198
|
+
def __init__(
|
|
1199
|
+
self,
|
|
1200
|
+
*,
|
|
1201
|
+
plan: CompiledHarnessPlan,
|
|
1202
|
+
session_request: GuardedSessionRequest,
|
|
1203
|
+
executor: ToolExecutorLike,
|
|
1204
|
+
cancellation_resolver: CancellationResolverLike,
|
|
1205
|
+
clock: RuntimeClockLike,
|
|
1206
|
+
) -> None:
|
|
1207
|
+
self._plan = plan
|
|
1208
|
+
self._session_request = session_request
|
|
1209
|
+
self._selected_output_by_terminal_result = (
|
|
1210
|
+
_selected_output_requirements_by_terminal_result(
|
|
1211
|
+
session_request.execution_request.selected_output_requirements
|
|
1212
|
+
)
|
|
1213
|
+
)
|
|
1214
|
+
self._executor = executor
|
|
1215
|
+
self._cancellation_resolver = cancellation_resolver
|
|
1216
|
+
self._clock = clock
|
|
1217
|
+
self._nodes_by_tool = {node.model_tool_name: node for node in plan.nodes}
|
|
1218
|
+
self._owned_successes: dict[
|
|
1219
|
+
str, tuple[ToolExecutionResult, dict[str, Any]]
|
|
1220
|
+
] = {}
|
|
1221
|
+
self._owned_artifacts: dict[str, ArtifactRef] = {}
|
|
1222
|
+
self._events: list[SessionEvent] = []
|
|
1223
|
+
self._tool_trace: list[ToolTraceRecord] = []
|
|
1224
|
+
self._terminal_candidate: TerminalCandidate | None = None
|
|
1225
|
+
self._terminal_intent: TerminalIntent | None = None
|
|
1226
|
+
self._sequence = 0
|
|
1227
|
+
self._bridge_call_counter = 0
|
|
1228
|
+
|
|
1229
|
+
@property
|
|
1230
|
+
def events(self) -> tuple[SessionEvent, ...]:
|
|
1231
|
+
return tuple(self._events)
|
|
1232
|
+
|
|
1233
|
+
@property
|
|
1234
|
+
def tool_trace(self) -> tuple[ToolTraceRecord, ...]:
|
|
1235
|
+
return tuple(self._tool_trace)
|
|
1236
|
+
|
|
1237
|
+
@property
|
|
1238
|
+
def terminal_candidate(self) -> TerminalCandidate | None:
|
|
1239
|
+
return self._terminal_candidate
|
|
1240
|
+
|
|
1241
|
+
@property
|
|
1242
|
+
def terminal_intent(self) -> TerminalIntent | None:
|
|
1243
|
+
return self._terminal_intent
|
|
1244
|
+
|
|
1245
|
+
def make_callable(self, tool_name: str) -> Callable[..., Any]:
|
|
1246
|
+
if tool_name not in self._nodes_by_tool:
|
|
1247
|
+
raise ForgeBindingRejectedError(
|
|
1248
|
+
f"Cannot build callable for tool outside compiled plan: {tool_name!r}"
|
|
1249
|
+
)
|
|
1250
|
+
|
|
1251
|
+
async def _callable(**kwargs: Any) -> str:
|
|
1252
|
+
return await self.invoke(tool_name, kwargs)
|
|
1253
|
+
|
|
1254
|
+
return _callable
|
|
1255
|
+
|
|
1256
|
+
async def invoke(
|
|
1257
|
+
self,
|
|
1258
|
+
tool_name: str,
|
|
1259
|
+
arguments: Mapping[str, Any],
|
|
1260
|
+
*,
|
|
1261
|
+
call_id: str | None = None,
|
|
1262
|
+
) -> str:
|
|
1263
|
+
node = self._nodes_by_tool.get(tool_name)
|
|
1264
|
+
if node is None:
|
|
1265
|
+
raise NonRetryableToolError(f"Tool {tool_name!r} is outside compiled plan")
|
|
1266
|
+
args = dict(arguments)
|
|
1267
|
+
call_id = self._resolve_call_id(call_id)
|
|
1268
|
+
input_sha256 = _sha256_json(args)
|
|
1269
|
+
self._append_event(
|
|
1270
|
+
SessionEventType.TOOL_STARTED,
|
|
1271
|
+
node_id=node.node_id,
|
|
1272
|
+
tool_call_id=call_id,
|
|
1273
|
+
)
|
|
1274
|
+
prereq_decisions = self._prerequisite_decisions(node, args)
|
|
1275
|
+
capability_decisions = self._capability_decisions(node)
|
|
1276
|
+
if _has_denial((*prereq_decisions, *capability_decisions)):
|
|
1277
|
+
self._append_trace(
|
|
1278
|
+
node=node,
|
|
1279
|
+
call_id=call_id,
|
|
1280
|
+
input_sha256=input_sha256,
|
|
1281
|
+
prerequisite_decisions=prereq_decisions,
|
|
1282
|
+
capability_decisions=capability_decisions,
|
|
1283
|
+
status=ToolExecutionStatus.NOT_EXECUTED,
|
|
1284
|
+
retryable=False,
|
|
1285
|
+
side_effect_certainty=SideEffectCertainty.NOT_ATTEMPTED,
|
|
1286
|
+
summary="tool rejected before execution",
|
|
1287
|
+
)
|
|
1288
|
+
self._append_event(
|
|
1289
|
+
SessionEventType.PREREQUISITE_REJECTED,
|
|
1290
|
+
node_id=node.node_id,
|
|
1291
|
+
tool_call_id=call_id,
|
|
1292
|
+
)
|
|
1293
|
+
raise ToolResolutionError(
|
|
1294
|
+
"Tool prerequisites or capabilities are not satisfied",
|
|
1295
|
+
tool_name=tool_name,
|
|
1296
|
+
)
|
|
1297
|
+
|
|
1298
|
+
token = self._cancellation_resolver.resolve(
|
|
1299
|
+
self._session_request.execution_request.cancellation
|
|
1300
|
+
)
|
|
1301
|
+
if token.is_cancelled():
|
|
1302
|
+
self._append_trace(
|
|
1303
|
+
node=node,
|
|
1304
|
+
call_id=call_id,
|
|
1305
|
+
input_sha256=input_sha256,
|
|
1306
|
+
prerequisite_decisions=prereq_decisions,
|
|
1307
|
+
capability_decisions=capability_decisions,
|
|
1308
|
+
status=ToolExecutionStatus.CANCELLED,
|
|
1309
|
+
retryable=False,
|
|
1310
|
+
side_effect_certainty=SideEffectCertainty.NOT_ATTEMPTED,
|
|
1311
|
+
summary=token.reason or "tool call cancelled",
|
|
1312
|
+
)
|
|
1313
|
+
self._append_event(
|
|
1314
|
+
SessionEventType.CANCELLED,
|
|
1315
|
+
node_id=node.node_id,
|
|
1316
|
+
tool_call_id=call_id,
|
|
1317
|
+
)
|
|
1318
|
+
raise _WorkflowOperationCancelledError(
|
|
1319
|
+
token.reason or "tool call cancelled"
|
|
1320
|
+
)
|
|
1321
|
+
if self._session_request.deadline.remaining(self._clock) <= 0:
|
|
1322
|
+
self._append_trace(
|
|
1323
|
+
node=node,
|
|
1324
|
+
call_id=call_id,
|
|
1325
|
+
input_sha256=input_sha256,
|
|
1326
|
+
prerequisite_decisions=prereq_decisions,
|
|
1327
|
+
capability_decisions=capability_decisions,
|
|
1328
|
+
status=ToolExecutionStatus.TIMED_OUT,
|
|
1329
|
+
retryable=False,
|
|
1330
|
+
side_effect_certainty=SideEffectCertainty.NOT_ATTEMPTED,
|
|
1331
|
+
summary="tool call deadline expired",
|
|
1332
|
+
)
|
|
1333
|
+
self._append_event(
|
|
1334
|
+
SessionEventType.TIMED_OUT,
|
|
1335
|
+
node_id=node.node_id,
|
|
1336
|
+
tool_call_id=call_id,
|
|
1337
|
+
)
|
|
1338
|
+
raise _WorkflowDeadlineExceededError("tool call deadline expired")
|
|
1339
|
+
|
|
1340
|
+
admitted_selected_output: SelectedOutput | None = None
|
|
1341
|
+
execution_args = args
|
|
1342
|
+
selected_output_requirement = (
|
|
1343
|
+
self._selected_output_by_terminal_result.get(node.terminal_result)
|
|
1344
|
+
if node.terminal_result is not None
|
|
1345
|
+
else None
|
|
1346
|
+
)
|
|
1347
|
+
if node.terminal_result is not None:
|
|
1348
|
+
try:
|
|
1349
|
+
if selected_output_requirement is None:
|
|
1350
|
+
if "candidate" in args:
|
|
1351
|
+
raise ValueError(
|
|
1352
|
+
"selected output candidate is not allowed for this terminal result"
|
|
1353
|
+
)
|
|
1354
|
+
else:
|
|
1355
|
+
admitted_selected_output = admit_selected_output(
|
|
1356
|
+
selected_output_requirement,
|
|
1357
|
+
present="candidate" in args,
|
|
1358
|
+
value=args.get("candidate"),
|
|
1359
|
+
)
|
|
1360
|
+
except ValueError as exc:
|
|
1361
|
+
self._append_trace(
|
|
1362
|
+
node=node,
|
|
1363
|
+
call_id=call_id,
|
|
1364
|
+
input_sha256=input_sha256,
|
|
1365
|
+
prerequisite_decisions=prereq_decisions,
|
|
1366
|
+
capability_decisions=capability_decisions,
|
|
1367
|
+
status=ToolExecutionStatus.NOT_EXECUTED,
|
|
1368
|
+
retryable=True,
|
|
1369
|
+
side_effect_certainty=SideEffectCertainty.NOT_ATTEMPTED,
|
|
1370
|
+
summary="selected output candidate rejected",
|
|
1371
|
+
)
|
|
1372
|
+
self._append_event(
|
|
1373
|
+
SessionEventType.TERMINAL_INTENT_REJECTED,
|
|
1374
|
+
node_id=node.node_id,
|
|
1375
|
+
tool_call_id=call_id,
|
|
1376
|
+
code="selected_output_invalid",
|
|
1377
|
+
)
|
|
1378
|
+
self._append_event(
|
|
1379
|
+
SessionEventType.CORRECTION_ISSUED,
|
|
1380
|
+
node_id=node.node_id,
|
|
1381
|
+
tool_call_id=call_id,
|
|
1382
|
+
code="selected_output_invalid",
|
|
1383
|
+
)
|
|
1384
|
+
raise ToolResolutionError(
|
|
1385
|
+
str(exc),
|
|
1386
|
+
tool_name=tool_name,
|
|
1387
|
+
) from None
|
|
1388
|
+
if selected_output_requirement is not None:
|
|
1389
|
+
execution_args = dict(args)
|
|
1390
|
+
execution_args.pop("candidate", None)
|
|
1391
|
+
|
|
1392
|
+
execution_input_sha256 = _sha256_json(execution_args)
|
|
1393
|
+
validated_call = ValidatedToolCall(
|
|
1394
|
+
call_id=call_id,
|
|
1395
|
+
node_id=node.node_id,
|
|
1396
|
+
binding=node.binding,
|
|
1397
|
+
arguments=execution_args,
|
|
1398
|
+
)
|
|
1399
|
+
try:
|
|
1400
|
+
context = self._execution_context()
|
|
1401
|
+
prerequisite_results = {
|
|
1402
|
+
node_id: result
|
|
1403
|
+
for node_id, (result, _args) in self._owned_successes.items()
|
|
1404
|
+
}
|
|
1405
|
+
execute_model_tool = getattr(self._executor, "execute_model_tool", None)
|
|
1406
|
+
if callable(execute_model_tool):
|
|
1407
|
+
result = await execute_model_tool(
|
|
1408
|
+
model_tool_name=tool_name,
|
|
1409
|
+
call_id=call_id,
|
|
1410
|
+
arguments=execution_args,
|
|
1411
|
+
context=context,
|
|
1412
|
+
prerequisite_results=prerequisite_results,
|
|
1413
|
+
session_id=self._session_request.session_id,
|
|
1414
|
+
)
|
|
1415
|
+
else:
|
|
1416
|
+
result = await self._executor.execute(validated_call, context)
|
|
1417
|
+
except Exception as exc:
|
|
1418
|
+
self._append_trace(
|
|
1419
|
+
node=node,
|
|
1420
|
+
call_id=call_id,
|
|
1421
|
+
input_sha256=input_sha256,
|
|
1422
|
+
prerequisite_decisions=prereq_decisions,
|
|
1423
|
+
capability_decisions=capability_decisions,
|
|
1424
|
+
status=ToolExecutionStatus.HARD_FAILURE,
|
|
1425
|
+
retryable=False,
|
|
1426
|
+
side_effect_certainty=SideEffectCertainty.COMPLETION_UNKNOWN,
|
|
1427
|
+
side_effect_detail_code="implementation_completion_unknown",
|
|
1428
|
+
side_effect_detail_summary=(
|
|
1429
|
+
f"{node.model_tool_name} side-effect completion is unknown "
|
|
1430
|
+
"after implementation exception"
|
|
1431
|
+
),
|
|
1432
|
+
side_effect_retry_allowed=False,
|
|
1433
|
+
summary=f"tool implementation defect: {type(exc).__name__}",
|
|
1434
|
+
)
|
|
1435
|
+
self._append_event(
|
|
1436
|
+
SessionEventType.TOOL_FAILED,
|
|
1437
|
+
node_id=node.node_id,
|
|
1438
|
+
tool_call_id=call_id,
|
|
1439
|
+
code="implementation_defect",
|
|
1440
|
+
)
|
|
1441
|
+
raise
|
|
1442
|
+
|
|
1443
|
+
result_defect = _tool_result_boundary_defect(
|
|
1444
|
+
result=result,
|
|
1445
|
+
expected_call=validated_call,
|
|
1446
|
+
expected_input_sha256=execution_input_sha256,
|
|
1447
|
+
)
|
|
1448
|
+
if result_defect is not None:
|
|
1449
|
+
self._append_trace(
|
|
1450
|
+
node=node,
|
|
1451
|
+
call_id=call_id,
|
|
1452
|
+
input_sha256=input_sha256,
|
|
1453
|
+
prerequisite_decisions=prereq_decisions,
|
|
1454
|
+
capability_decisions=capability_decisions,
|
|
1455
|
+
status=ToolExecutionStatus.HARD_FAILURE,
|
|
1456
|
+
retryable=False,
|
|
1457
|
+
side_effect_certainty=SideEffectCertainty.COMPLETION_UNKNOWN,
|
|
1458
|
+
side_effect_detail_code="binding_completion_unknown",
|
|
1459
|
+
side_effect_detail_summary=(
|
|
1460
|
+
f"{node.model_tool_name} side-effect completion is unknown "
|
|
1461
|
+
"after result boundary defect"
|
|
1462
|
+
),
|
|
1463
|
+
side_effect_retry_allowed=False,
|
|
1464
|
+
summary=f"tool binding defect: {result_defect}",
|
|
1465
|
+
duration_ms=result.duration_ms,
|
|
1466
|
+
)
|
|
1467
|
+
self._append_event(
|
|
1468
|
+
SessionEventType.TOOL_FAILED,
|
|
1469
|
+
node_id=node.node_id,
|
|
1470
|
+
tool_call_id=call_id,
|
|
1471
|
+
code="binding_defect",
|
|
1472
|
+
)
|
|
1473
|
+
raise NonRetryableToolError(f"Tool {result_defect}")
|
|
1474
|
+
try:
|
|
1475
|
+
output_sha256 = _validate_output_hash(result)
|
|
1476
|
+
except NonRetryableToolError:
|
|
1477
|
+
self._append_trace(
|
|
1478
|
+
node=node,
|
|
1479
|
+
call_id=call_id,
|
|
1480
|
+
input_sha256=input_sha256,
|
|
1481
|
+
prerequisite_decisions=prereq_decisions,
|
|
1482
|
+
capability_decisions=capability_decisions,
|
|
1483
|
+
status=ToolExecutionStatus.HARD_FAILURE,
|
|
1484
|
+
retryable=False,
|
|
1485
|
+
side_effect_certainty=SideEffectCertainty.COMPLETION_UNKNOWN,
|
|
1486
|
+
side_effect_detail_code="output_hash_completion_unknown",
|
|
1487
|
+
side_effect_detail_summary=(
|
|
1488
|
+
f"{node.model_tool_name} side-effect completion is unknown "
|
|
1489
|
+
"after output hash mismatch"
|
|
1490
|
+
),
|
|
1491
|
+
side_effect_retry_allowed=False,
|
|
1492
|
+
output_sha256=result.output_sha256,
|
|
1493
|
+
duration_ms=result.duration_ms,
|
|
1494
|
+
summary="tool binding defect: output_sha256 mismatch",
|
|
1495
|
+
)
|
|
1496
|
+
self._append_event(
|
|
1497
|
+
SessionEventType.TOOL_FAILED,
|
|
1498
|
+
node_id=node.node_id,
|
|
1499
|
+
tool_call_id=call_id,
|
|
1500
|
+
code="output_hash_mismatch",
|
|
1501
|
+
)
|
|
1502
|
+
raise
|
|
1503
|
+
status = result.status
|
|
1504
|
+
self._append_trace(
|
|
1505
|
+
node=node,
|
|
1506
|
+
call_id=call_id,
|
|
1507
|
+
input_sha256=input_sha256,
|
|
1508
|
+
prerequisite_decisions=prereq_decisions,
|
|
1509
|
+
capability_decisions=capability_decisions,
|
|
1510
|
+
status=status,
|
|
1511
|
+
retryable=result.retryable,
|
|
1512
|
+
side_effect_certainty=result.side_effect_certainty,
|
|
1513
|
+
side_effect_class=result.side_effect_class,
|
|
1514
|
+
idempotency=result.idempotency,
|
|
1515
|
+
side_effect_detail_code=result.side_effect_record.detail_code
|
|
1516
|
+
if result.side_effect_record is not None
|
|
1517
|
+
else None,
|
|
1518
|
+
side_effect_detail_summary=result.side_effect_record.summary
|
|
1519
|
+
if result.side_effect_record is not None
|
|
1520
|
+
else None,
|
|
1521
|
+
side_effect_retry_allowed=result.side_effect_record.retry_allowed
|
|
1522
|
+
if result.side_effect_record is not None
|
|
1523
|
+
else None,
|
|
1524
|
+
output_sha256=output_sha256,
|
|
1525
|
+
duration_ms=result.duration_ms,
|
|
1526
|
+
summary=result.summary,
|
|
1527
|
+
)
|
|
1528
|
+
if result.status is ToolExecutionStatus.CANCELLED:
|
|
1529
|
+
self._append_event(
|
|
1530
|
+
SessionEventType.CANCELLED,
|
|
1531
|
+
node_id=node.node_id,
|
|
1532
|
+
tool_call_id=call_id,
|
|
1533
|
+
code=result.error_code,
|
|
1534
|
+
)
|
|
1535
|
+
raise _WorkflowOperationCancelledError(_failure_message(result))
|
|
1536
|
+
if result.status is ToolExecutionStatus.TIMED_OUT:
|
|
1537
|
+
self._append_event(
|
|
1538
|
+
SessionEventType.TIMED_OUT,
|
|
1539
|
+
node_id=node.node_id,
|
|
1540
|
+
tool_call_id=call_id,
|
|
1541
|
+
code=result.error_code,
|
|
1542
|
+
)
|
|
1543
|
+
raise _WorkflowDeadlineExceededError(_failure_message(result))
|
|
1544
|
+
if result.status != ToolExecutionStatus.SUCCESS:
|
|
1545
|
+
self._append_event(
|
|
1546
|
+
SessionEventType.TOOL_FAILED,
|
|
1547
|
+
node_id=node.node_id,
|
|
1548
|
+
tool_call_id=call_id,
|
|
1549
|
+
code=result.error_code,
|
|
1550
|
+
)
|
|
1551
|
+
message = _failure_message(result)
|
|
1552
|
+
if result.status is ToolExecutionStatus.SOFT_FAILURE or result.retryable:
|
|
1553
|
+
raise ToolResolutionError(
|
|
1554
|
+
message,
|
|
1555
|
+
tool_name=tool_name,
|
|
1556
|
+
)
|
|
1557
|
+
raise NonRetryableToolError(message)
|
|
1558
|
+
|
|
1559
|
+
self._owned_successes[node.node_id] = (result, args)
|
|
1560
|
+
for artifact_ref in result.artifact_refs:
|
|
1561
|
+
self._owned_artifacts[artifact_ref.artifact_id] = artifact_ref
|
|
1562
|
+
self._append_event(
|
|
1563
|
+
SessionEventType.TOOL_COMPLETED,
|
|
1564
|
+
node_id=node.node_id,
|
|
1565
|
+
tool_call_id=call_id,
|
|
1566
|
+
)
|
|
1567
|
+
if node.terminal_result is not None:
|
|
1568
|
+
self._accept_terminal_candidate(
|
|
1569
|
+
node,
|
|
1570
|
+
call_id,
|
|
1571
|
+
result,
|
|
1572
|
+
selected_output=admitted_selected_output,
|
|
1573
|
+
)
|
|
1574
|
+
return _model_visible_tool_content(result)
|
|
1575
|
+
|
|
1576
|
+
def _resolve_call_id(self, call_id: str | None) -> str:
|
|
1577
|
+
if call_id is not None:
|
|
1578
|
+
return call_id
|
|
1579
|
+
call_id = f"bridge_call_{self._bridge_call_counter:09d}"
|
|
1580
|
+
self._bridge_call_counter += 1
|
|
1581
|
+
return call_id
|
|
1582
|
+
|
|
1583
|
+
def _execution_context(self) -> ToolExecutionContext:
|
|
1584
|
+
request = self._session_request.execution_request
|
|
1585
|
+
token = self._cancellation_resolver.resolve(request.cancellation)
|
|
1586
|
+
trusted = self._session_request.tool_execution_context
|
|
1587
|
+
if trusted is not None:
|
|
1588
|
+
return trusted.model_copy(
|
|
1589
|
+
update={
|
|
1590
|
+
"deadline": self._session_request.deadline,
|
|
1591
|
+
"cancellation_requested": token.is_cancelled(),
|
|
1592
|
+
"current_monotonic": self._clock.monotonic(),
|
|
1593
|
+
}
|
|
1594
|
+
)
|
|
1595
|
+
return ToolExecutionContext(
|
|
1596
|
+
request_id=request.request_id,
|
|
1597
|
+
run_id=request.run_id,
|
|
1598
|
+
stage=request.stage,
|
|
1599
|
+
run_directory=request.run_directory,
|
|
1600
|
+
capability_envelope=request.capability_envelope,
|
|
1601
|
+
timeout=request.timeout,
|
|
1602
|
+
cancellation=request.cancellation,
|
|
1603
|
+
workspace_root=Path.cwd(),
|
|
1604
|
+
artifact_root=request.run_directory.path / "millforge",
|
|
1605
|
+
compiled_artifact_policy=self._plan.artifact_policy,
|
|
1606
|
+
input_artifacts=request.input_artifacts,
|
|
1607
|
+
work_item_id=request.work_item_id,
|
|
1608
|
+
deadline=self._session_request.deadline,
|
|
1609
|
+
cancellation_requested=token.is_cancelled(),
|
|
1610
|
+
current_monotonic=self._clock.monotonic(),
|
|
1611
|
+
)
|
|
1612
|
+
|
|
1613
|
+
def _prerequisite_decisions(
|
|
1614
|
+
self, node: CompiledHarnessNode, args: Mapping[str, Any]
|
|
1615
|
+
) -> tuple[ToolTraceDecisionRecord, ...]:
|
|
1616
|
+
decisions: list[ToolTraceDecisionRecord] = []
|
|
1617
|
+
for prereq in node.prerequisites:
|
|
1618
|
+
prior = self._owned_successes.get(prereq.node_id)
|
|
1619
|
+
satisfied = prior is not None
|
|
1620
|
+
prior_args = prior[1] if prior else {}
|
|
1621
|
+
for match in prereq.argument_matches:
|
|
1622
|
+
satisfied = satisfied and prior_args.get(
|
|
1623
|
+
match.prerequisite_argument
|
|
1624
|
+
) == args.get(match.current_argument)
|
|
1625
|
+
decisions.append(
|
|
1626
|
+
ToolTraceDecisionRecord(
|
|
1627
|
+
key=prereq.node_id,
|
|
1628
|
+
decision=ToolTraceDecision.ALLOWED
|
|
1629
|
+
if satisfied
|
|
1630
|
+
else ToolTraceDecision.DENIED,
|
|
1631
|
+
)
|
|
1632
|
+
)
|
|
1633
|
+
return tuple(decisions)
|
|
1634
|
+
|
|
1635
|
+
def _capability_decisions(
|
|
1636
|
+
self, node: CompiledHarnessNode
|
|
1637
|
+
) -> tuple[ToolTraceDecisionRecord, ...]:
|
|
1638
|
+
granted = {
|
|
1639
|
+
grant.capability_id
|
|
1640
|
+
for grant in self._session_request.execution_request.capability_envelope.grants
|
|
1641
|
+
}
|
|
1642
|
+
decisions = [
|
|
1643
|
+
ToolTraceDecisionRecord(
|
|
1644
|
+
key=capability,
|
|
1645
|
+
decision=ToolTraceDecision.ALLOWED
|
|
1646
|
+
if capability in granted
|
|
1647
|
+
else ToolTraceDecision.DENIED,
|
|
1648
|
+
)
|
|
1649
|
+
for capability in node.required_capabilities
|
|
1650
|
+
]
|
|
1651
|
+
if not self._executor.supports_tool(node.model_tool_name):
|
|
1652
|
+
decisions.append(
|
|
1653
|
+
ToolTraceDecisionRecord(
|
|
1654
|
+
key=f"tool:{node.model_tool_name}",
|
|
1655
|
+
decision=ToolTraceDecision.DENIED,
|
|
1656
|
+
)
|
|
1657
|
+
)
|
|
1658
|
+
return tuple(decisions)
|
|
1659
|
+
|
|
1660
|
+
def _accept_terminal_candidate(
|
|
1661
|
+
self,
|
|
1662
|
+
node: CompiledHarnessNode,
|
|
1663
|
+
call_id: str,
|
|
1664
|
+
result: ToolExecutionResult,
|
|
1665
|
+
*,
|
|
1666
|
+
selected_output: SelectedOutput | None,
|
|
1667
|
+
) -> None:
|
|
1668
|
+
assert node.terminal_result is not None
|
|
1669
|
+
candidate = TerminalCandidate(
|
|
1670
|
+
call_id=call_id,
|
|
1671
|
+
node_id=node.node_id,
|
|
1672
|
+
tool_name=node.model_tool_name,
|
|
1673
|
+
terminal_result=node.terminal_result,
|
|
1674
|
+
summary=result.summary,
|
|
1675
|
+
artifact_refs=tuple(self._owned_artifacts.values()),
|
|
1676
|
+
selected_output=selected_output,
|
|
1677
|
+
selected_output_schema_sha256=(
|
|
1678
|
+
self._selected_output_by_terminal_result[
|
|
1679
|
+
node.terminal_result
|
|
1680
|
+
].schema_sha256
|
|
1681
|
+
if selected_output is not None
|
|
1682
|
+
else None
|
|
1683
|
+
),
|
|
1684
|
+
)
|
|
1685
|
+
self._terminal_candidate = candidate
|
|
1686
|
+
missing_required = [
|
|
1687
|
+
item.node_id
|
|
1688
|
+
for item in self._plan.nodes
|
|
1689
|
+
if item.required and item.node_id not in self._owned_successes
|
|
1690
|
+
]
|
|
1691
|
+
required_artifacts = {
|
|
1692
|
+
artifact_id
|
|
1693
|
+
for requirement in self._plan.artifact_policy.required_by_terminal
|
|
1694
|
+
if requirement.terminal_result == node.terminal_result
|
|
1695
|
+
for artifact_id in requirement.artifact_ids
|
|
1696
|
+
}
|
|
1697
|
+
missing_artifacts = sorted(required_artifacts - set(self._owned_artifacts))
|
|
1698
|
+
mapping_valid = (
|
|
1699
|
+
self._plan.terminal_result_map.get(node.node_id) == node.terminal_result
|
|
1700
|
+
)
|
|
1701
|
+
if missing_required or missing_artifacts or not mapping_valid:
|
|
1702
|
+
self._append_event(
|
|
1703
|
+
SessionEventType.TERMINAL_INTENT_REJECTED,
|
|
1704
|
+
node_id=node.node_id,
|
|
1705
|
+
tool_call_id=call_id,
|
|
1706
|
+
)
|
|
1707
|
+
raise NonRetryableToolError(
|
|
1708
|
+
"Terminal candidate failed owned-history validation"
|
|
1709
|
+
)
|
|
1710
|
+
request = self._session_request.execution_request
|
|
1711
|
+
disposition = _terminal_disposition(
|
|
1712
|
+
node.terminal_result,
|
|
1713
|
+
legacy_blocked_suffix=not node.binding.tool_id.startswith(
|
|
1714
|
+
"builtin.pi_compat.terminal."
|
|
1715
|
+
),
|
|
1716
|
+
)
|
|
1717
|
+
self._terminal_intent = TerminalIntent(
|
|
1718
|
+
request_id=request.request_id,
|
|
1719
|
+
run_id=request.run_id,
|
|
1720
|
+
stage=request.stage,
|
|
1721
|
+
terminal_node_id=node.node_id,
|
|
1722
|
+
terminal_result=node.terminal_result,
|
|
1723
|
+
disposition=disposition,
|
|
1724
|
+
summary=candidate.summary,
|
|
1725
|
+
artifact_refs=candidate.artifact_refs,
|
|
1726
|
+
selected_output=candidate.selected_output,
|
|
1727
|
+
selected_output_schema_sha256=candidate.selected_output_schema_sha256,
|
|
1728
|
+
)
|
|
1729
|
+
self._append_event(
|
|
1730
|
+
SessionEventType.TERMINAL_INTENT_ACCEPTED,
|
|
1731
|
+
node_id=node.node_id,
|
|
1732
|
+
tool_call_id=call_id,
|
|
1733
|
+
)
|
|
1734
|
+
|
|
1735
|
+
def _append_event(
|
|
1736
|
+
self,
|
|
1737
|
+
event_type: SessionEventType,
|
|
1738
|
+
*,
|
|
1739
|
+
node_id: str | None = None,
|
|
1740
|
+
tool_call_id: str | None = None,
|
|
1741
|
+
code: str | None = None,
|
|
1742
|
+
) -> None:
|
|
1743
|
+
self._sequence += 1
|
|
1744
|
+
request = self._session_request.execution_request
|
|
1745
|
+
self._events.append(
|
|
1746
|
+
SessionEvent(
|
|
1747
|
+
schema_version="1.0",
|
|
1748
|
+
sequence=self._sequence,
|
|
1749
|
+
occurred_at=self._clock.utc_now().isoformat(),
|
|
1750
|
+
monotonic_offset_ms=self._clock.monotonic() * 1000,
|
|
1751
|
+
event_type=event_type,
|
|
1752
|
+
request_id=request.request_id,
|
|
1753
|
+
run_id=request.run_id,
|
|
1754
|
+
session_id=self._session_request.session_id,
|
|
1755
|
+
stage=request.stage,
|
|
1756
|
+
node_id=node_id,
|
|
1757
|
+
model_turn=None,
|
|
1758
|
+
tool_call_id=tool_call_id,
|
|
1759
|
+
code=code,
|
|
1760
|
+
fields=(),
|
|
1761
|
+
)
|
|
1762
|
+
)
|
|
1763
|
+
|
|
1764
|
+
def _append_trace(
|
|
1765
|
+
self,
|
|
1766
|
+
*,
|
|
1767
|
+
node: CompiledHarnessNode,
|
|
1768
|
+
call_id: str,
|
|
1769
|
+
input_sha256: str,
|
|
1770
|
+
prerequisite_decisions: tuple[ToolTraceDecisionRecord, ...],
|
|
1771
|
+
capability_decisions: tuple[ToolTraceDecisionRecord, ...],
|
|
1772
|
+
status: ToolExecutionStatus,
|
|
1773
|
+
retryable: bool,
|
|
1774
|
+
side_effect_certainty: SideEffectCertainty,
|
|
1775
|
+
summary: str,
|
|
1776
|
+
side_effect_class: SideEffectClass | None = None,
|
|
1777
|
+
idempotency: IdempotencyClass | None = None,
|
|
1778
|
+
side_effect_detail_code: str | None = None,
|
|
1779
|
+
side_effect_detail_summary: str | None = None,
|
|
1780
|
+
side_effect_retry_allowed: bool | None = None,
|
|
1781
|
+
output_sha256: str | None = None,
|
|
1782
|
+
duration_ms: float = 0.0,
|
|
1783
|
+
) -> None:
|
|
1784
|
+
request = self._session_request.execution_request
|
|
1785
|
+
self._tool_trace.append(
|
|
1786
|
+
ToolTraceRecord(
|
|
1787
|
+
schema_version="1.0",
|
|
1788
|
+
sequence=len(self._tool_trace) + 1,
|
|
1789
|
+
occurred_at=self._clock.utc_now().isoformat(),
|
|
1790
|
+
monotonic_offset_ms=self._clock.monotonic() * 1000,
|
|
1791
|
+
request_id=request.request_id,
|
|
1792
|
+
run_id=request.run_id,
|
|
1793
|
+
session_id=self._session_request.session_id,
|
|
1794
|
+
stage=request.stage,
|
|
1795
|
+
node_id=node.node_id,
|
|
1796
|
+
model_turn=0,
|
|
1797
|
+
tool_call_id=call_id,
|
|
1798
|
+
model_tool_name=node.model_tool_name,
|
|
1799
|
+
binding=node.binding,
|
|
1800
|
+
input_sha256=input_sha256,
|
|
1801
|
+
prerequisite_decisions=prerequisite_decisions,
|
|
1802
|
+
capability_decisions=capability_decisions,
|
|
1803
|
+
execution_status=status,
|
|
1804
|
+
retryable=retryable,
|
|
1805
|
+
side_effect_class=_trace_side_effect(
|
|
1806
|
+
side_effect_class or node.side_effect_class
|
|
1807
|
+
),
|
|
1808
|
+
idempotency=_trace_idempotency(idempotency or node.idempotency),
|
|
1809
|
+
side_effect_certainty=side_effect_certainty,
|
|
1810
|
+
side_effect_detail_code=side_effect_detail_code,
|
|
1811
|
+
side_effect_detail_summary=None
|
|
1812
|
+
if side_effect_detail_summary is None
|
|
1813
|
+
else bounded_summary(side_effect_detail_summary, max_utf8=2048),
|
|
1814
|
+
side_effect_retry_allowed=side_effect_retry_allowed,
|
|
1815
|
+
output_sha256=output_sha256,
|
|
1816
|
+
duration_ms=duration_ms,
|
|
1817
|
+
summary=bounded_summary(summary, max_utf8=2048),
|
|
1818
|
+
)
|
|
1819
|
+
)
|
|
1820
|
+
|
|
1821
|
+
|
|
1822
|
+
async def _watch_cancellation(
|
|
1823
|
+
token: CancellationTokenLike,
|
|
1824
|
+
cancel_event: asyncio.Event,
|
|
1825
|
+
) -> None:
|
|
1826
|
+
await token.wait()
|
|
1827
|
+
cancel_event.set()
|
|
1828
|
+
|
|
1829
|
+
|
|
1830
|
+
async def _cancel_and_await_watcher(
|
|
1831
|
+
watcher: asyncio.Task[None] | None,
|
|
1832
|
+
) -> None:
|
|
1833
|
+
if watcher is None:
|
|
1834
|
+
return
|
|
1835
|
+
if not watcher.done():
|
|
1836
|
+
watcher.cancel()
|
|
1837
|
+
await asyncio.gather(watcher, return_exceptions=True)
|
|
1838
|
+
|
|
1839
|
+
|
|
1840
|
+
def _reject(message: str) -> NoReturn:
|
|
1841
|
+
raise ForgeBindingRejectedError(message)
|
|
1842
|
+
|
|
1843
|
+
|
|
1844
|
+
def _tool_spec_from_node(node: CompiledHarnessNode) -> ToolSpec:
|
|
1845
|
+
try:
|
|
1846
|
+
return ToolSpec.from_json_schema(
|
|
1847
|
+
node.model_tool_name,
|
|
1848
|
+
node.description,
|
|
1849
|
+
node.input_schema,
|
|
1850
|
+
)
|
|
1851
|
+
except ValueError as exc:
|
|
1852
|
+
_reject(f"Unsupported input schema for node {node.node_id!r}: {exc}")
|
|
1853
|
+
|
|
1854
|
+
|
|
1855
|
+
def _translate_prerequisite(
|
|
1856
|
+
prereq: CompiledPrerequisite,
|
|
1857
|
+
tool_name_by_node_id: Mapping[str, str],
|
|
1858
|
+
) -> str | dict[str, str]:
|
|
1859
|
+
prereq_tool = tool_name_by_node_id.get(prereq.node_id)
|
|
1860
|
+
if prereq_tool is None:
|
|
1861
|
+
_reject(f"Prerequisite references unknown node {prereq.node_id!r}")
|
|
1862
|
+
if not prereq.argument_matches:
|
|
1863
|
+
return prereq_tool
|
|
1864
|
+
match = _single_supported_argument_match(prereq.argument_matches)
|
|
1865
|
+
if match.prerequisite_argument == match.current_argument:
|
|
1866
|
+
return {"tool": prereq_tool, "match_arg": match.current_argument}
|
|
1867
|
+
return {
|
|
1868
|
+
"tool": prereq_tool,
|
|
1869
|
+
"prerequisite_arg": match.prerequisite_argument,
|
|
1870
|
+
"current_arg": match.current_argument,
|
|
1871
|
+
}
|
|
1872
|
+
|
|
1873
|
+
|
|
1874
|
+
def _single_supported_argument_match(
|
|
1875
|
+
matches: tuple[ArgumentMatch, ...],
|
|
1876
|
+
) -> ArgumentMatch:
|
|
1877
|
+
if len(matches) != 1:
|
|
1878
|
+
_reject("Forge prerequisites support at most one argument match")
|
|
1879
|
+
return matches[0]
|
|
1880
|
+
|
|
1881
|
+
|
|
1882
|
+
def _validate_plan_identities(plan: CompiledHarnessPlan) -> None:
|
|
1883
|
+
_unique_or_reject((node.node_id for node in plan.nodes), "node_id")
|
|
1884
|
+
_unique_or_reject((node.model_tool_name for node in plan.nodes), "model_tool_name")
|
|
1885
|
+
_unique_or_reject(
|
|
1886
|
+
(f"{node.binding.tool_id}:{node.binding.tool_version}" for node in plan.nodes),
|
|
1887
|
+
"binding identity",
|
|
1888
|
+
)
|
|
1889
|
+
node_ids = {node.node_id for node in plan.nodes}
|
|
1890
|
+
terminal_node_ids = {
|
|
1891
|
+
node.node_id for node in plan.nodes if node.terminal_result is not None
|
|
1892
|
+
}
|
|
1893
|
+
if set(plan.terminal_result_map) != terminal_node_ids:
|
|
1894
|
+
_reject("terminal_result_map must exactly name terminal nodes")
|
|
1895
|
+
for node in plan.nodes:
|
|
1896
|
+
if not node.node_id or not node.model_tool_name:
|
|
1897
|
+
_reject("Compiled node identities must be non-empty")
|
|
1898
|
+
for prereq in node.prerequisites:
|
|
1899
|
+
if prereq.node_id not in node_ids:
|
|
1900
|
+
_reject(f"Prerequisite references unknown node {prereq.node_id!r}")
|
|
1901
|
+
|
|
1902
|
+
|
|
1903
|
+
def _unique_or_reject(values: Any, label: str) -> None:
|
|
1904
|
+
items = tuple(values)
|
|
1905
|
+
if len(set(items)) != len(items):
|
|
1906
|
+
_reject(f"Duplicate {label} values are unsupported")
|
|
1907
|
+
|
|
1908
|
+
|
|
1909
|
+
_REQUEST_CONTEXT_SEPARATOR = "\n\n--- millforge request context ---\n"
|
|
1910
|
+
|
|
1911
|
+
|
|
1912
|
+
def _request_context_payload(request: HarnessExecutionRequest) -> dict[str, Any]:
|
|
1913
|
+
payload: dict[str, Any] = {
|
|
1914
|
+
"input_artifacts": [
|
|
1915
|
+
{
|
|
1916
|
+
"artifact_id": artifact.artifact_id,
|
|
1917
|
+
"content_type": artifact.content_type,
|
|
1918
|
+
"path": _path_text(artifact.path),
|
|
1919
|
+
}
|
|
1920
|
+
for artifact in request.input_artifacts
|
|
1921
|
+
],
|
|
1922
|
+
"request_id": request.request_id,
|
|
1923
|
+
"run_id": request.run_id,
|
|
1924
|
+
"stage": {
|
|
1925
|
+
"node_id": request.stage.node_id,
|
|
1926
|
+
"plane": request.stage.plane,
|
|
1927
|
+
"stage_kind_id": request.stage.stage_kind_id,
|
|
1928
|
+
},
|
|
1929
|
+
"work_item_id": request.work_item_id,
|
|
1930
|
+
}
|
|
1931
|
+
if request.selected_output_requirements:
|
|
1932
|
+
payload["selected_output_requirements"] = [
|
|
1933
|
+
{
|
|
1934
|
+
"candidate_transport": "terminal_tool_argument",
|
|
1935
|
+
"required": item.selected_output.required,
|
|
1936
|
+
"schema_sha256": item.selected_output.schema_sha256,
|
|
1937
|
+
"terminal_result": item.terminal_result,
|
|
1938
|
+
}
|
|
1939
|
+
for item in request.selected_output_requirements
|
|
1940
|
+
]
|
|
1941
|
+
return payload
|
|
1942
|
+
|
|
1943
|
+
|
|
1944
|
+
def _context_compaction_callback(_event: Any) -> None:
|
|
1945
|
+
return None
|
|
1946
|
+
|
|
1947
|
+
|
|
1948
|
+
def _path_text(path: Path) -> str:
|
|
1949
|
+
return path.as_posix()
|
|
1950
|
+
|
|
1951
|
+
|
|
1952
|
+
def _model_message_from_private(message: Mapping[str, Any]) -> ModelMessage:
|
|
1953
|
+
role = message.get("role")
|
|
1954
|
+
content = message.get("content")
|
|
1955
|
+
if not isinstance(role, str):
|
|
1956
|
+
raise ForgeBridgeError("private message role is missing")
|
|
1957
|
+
tool_calls: list[ModelToolCall] = []
|
|
1958
|
+
for index, raw_call in enumerate(message.get("tool_calls") or ()):
|
|
1959
|
+
function = raw_call.get("function", {}) if isinstance(raw_call, dict) else {}
|
|
1960
|
+
args = function.get("arguments", {})
|
|
1961
|
+
call_id = raw_call.get("id") or f"private_call_{index:09d}"
|
|
1962
|
+
tool_calls.append(
|
|
1963
|
+
ModelToolCall(
|
|
1964
|
+
call_id=str(call_id),
|
|
1965
|
+
name=str(function.get("name", "")),
|
|
1966
|
+
arguments=_tool_arguments_from_private(args),
|
|
1967
|
+
)
|
|
1968
|
+
)
|
|
1969
|
+
if role == "system":
|
|
1970
|
+
return SystemMessage(content=str(content or ""))
|
|
1971
|
+
if role == "user":
|
|
1972
|
+
return UserMessage(content=str(content or ""))
|
|
1973
|
+
if role == "assistant":
|
|
1974
|
+
assistant_content = (
|
|
1975
|
+
None if content is None or (content == "" and tool_calls) else str(content)
|
|
1976
|
+
)
|
|
1977
|
+
return AssistantMessage(
|
|
1978
|
+
content=assistant_content,
|
|
1979
|
+
tool_calls=tuple(tool_calls),
|
|
1980
|
+
reasoning_content=(
|
|
1981
|
+
str(message["reasoning_content"])
|
|
1982
|
+
if message.get("reasoning_content") is not None
|
|
1983
|
+
else None
|
|
1984
|
+
),
|
|
1985
|
+
)
|
|
1986
|
+
if role == "tool":
|
|
1987
|
+
return ToolResultMessage(
|
|
1988
|
+
tool_call_id=str(message.get("tool_call_id") or ""),
|
|
1989
|
+
tool_name=str(message.get("tool_name") or message.get("name") or ""),
|
|
1990
|
+
content=str(content or ""),
|
|
1991
|
+
)
|
|
1992
|
+
raise ForgeBridgeError(f"Unsupported private message role {role!r}")
|
|
1993
|
+
|
|
1994
|
+
|
|
1995
|
+
def _prompt_event_fields(
|
|
1996
|
+
messages: tuple[ModelMessage, ...],
|
|
1997
|
+
) -> dict[str, str | int]:
|
|
1998
|
+
fields: dict[str, str | int] = {}
|
|
1999
|
+
for index, message in enumerate(messages):
|
|
2000
|
+
content = message.content or ""
|
|
2001
|
+
encoded = content.encode("utf-8")
|
|
2002
|
+
prefix = f"prompt_{index}"
|
|
2003
|
+
fields[f"{prefix}_role"] = message.role
|
|
2004
|
+
fields[f"{prefix}_byte_size"] = len(encoded)
|
|
2005
|
+
fields[f"{prefix}_sha256"] = hashlib.sha256(encoded).hexdigest()
|
|
2006
|
+
return fields
|
|
2007
|
+
|
|
2008
|
+
|
|
2009
|
+
def _tool_arguments_from_private(
|
|
2010
|
+
value: Any,
|
|
2011
|
+
) -> ParsedToolArguments | InvalidToolArguments:
|
|
2012
|
+
if isinstance(value, dict):
|
|
2013
|
+
return ParsedToolArguments(value=value)
|
|
2014
|
+
if isinstance(value, str):
|
|
2015
|
+
try:
|
|
2016
|
+
parsed = json.loads(value)
|
|
2017
|
+
except json.JSONDecodeError:
|
|
2018
|
+
return InvalidToolArguments(
|
|
2019
|
+
raw=value,
|
|
2020
|
+
error_code="invalid_json",
|
|
2021
|
+
)
|
|
2022
|
+
if isinstance(parsed, dict):
|
|
2023
|
+
return ParsedToolArguments(value=parsed)
|
|
2024
|
+
return InvalidToolArguments(
|
|
2025
|
+
raw=value,
|
|
2026
|
+
error_code="invalid_arguments",
|
|
2027
|
+
)
|
|
2028
|
+
|
|
2029
|
+
|
|
2030
|
+
def _tool_definition_from_spec(
|
|
2031
|
+
spec: ToolSpec,
|
|
2032
|
+
*,
|
|
2033
|
+
selected_output: SelectedOutputRequirement | None = None,
|
|
2034
|
+
) -> ModelToolDefinition:
|
|
2035
|
+
input_schema = spec.get_json_schema()
|
|
2036
|
+
if selected_output is not None:
|
|
2037
|
+
input_schema = _selected_output_terminal_tool_schema(
|
|
2038
|
+
input_schema,
|
|
2039
|
+
selected_output,
|
|
2040
|
+
)
|
|
2041
|
+
return ModelToolDefinition(
|
|
2042
|
+
name=spec.name,
|
|
2043
|
+
description=spec.description,
|
|
2044
|
+
input_schema=input_schema,
|
|
2045
|
+
)
|
|
2046
|
+
|
|
2047
|
+
|
|
2048
|
+
def _selected_output_terminal_tool_schema(
|
|
2049
|
+
input_schema: Mapping[str, Any],
|
|
2050
|
+
selected_output: SelectedOutputRequirement,
|
|
2051
|
+
) -> dict[str, Any]:
|
|
2052
|
+
"""Derive one model-visible terminal schema without mutating its descriptor."""
|
|
2053
|
+
|
|
2054
|
+
derived = json.loads(
|
|
2055
|
+
json.dumps(input_schema, sort_keys=True, separators=(",", ":"))
|
|
2056
|
+
)
|
|
2057
|
+
properties = derived.get("properties")
|
|
2058
|
+
required = derived.get("required")
|
|
2059
|
+
if not isinstance(properties, dict) or not isinstance(required, list):
|
|
2060
|
+
raise ForgeBindingRejectedError(
|
|
2061
|
+
"Terminal tool schema is not a closed object schema"
|
|
2062
|
+
)
|
|
2063
|
+
if "candidate" in properties:
|
|
2064
|
+
raise ForgeBindingRejectedError(
|
|
2065
|
+
"Terminal tool schema already reserves the selected output candidate field"
|
|
2066
|
+
)
|
|
2067
|
+
properties["candidate"] = json.loads(selected_output.canonical_schema_bytes)
|
|
2068
|
+
if selected_output.required:
|
|
2069
|
+
required.append("candidate")
|
|
2070
|
+
derived["required"] = sorted(required)
|
|
2071
|
+
return derived
|
|
2072
|
+
|
|
2073
|
+
|
|
2074
|
+
def _parsed_model_tool_args(call: ModelToolCall) -> Any:
|
|
2075
|
+
if isinstance(call.arguments, ParsedToolArguments):
|
|
2076
|
+
return dict(call.arguments.value)
|
|
2077
|
+
return call.arguments.raw
|
|
2078
|
+
|
|
2079
|
+
|
|
2080
|
+
def _sha256_json(value: Any) -> str:
|
|
2081
|
+
return hashlib.sha256(canonical_json_serialize(value).encode("utf-8")).hexdigest()
|
|
2082
|
+
|
|
2083
|
+
|
|
2084
|
+
def _validate_output_hash(result: ToolExecutionResult) -> str | None:
|
|
2085
|
+
computed = canonical_sha256(result.structured_data)
|
|
2086
|
+
if result.output_sha256 is not None and result.output_sha256 != computed:
|
|
2087
|
+
raise NonRetryableToolError(
|
|
2088
|
+
"Tool result output_sha256 did not match safe output"
|
|
2089
|
+
)
|
|
2090
|
+
return result.output_sha256 or computed
|
|
2091
|
+
|
|
2092
|
+
|
|
2093
|
+
def _tool_result_boundary_defect(
|
|
2094
|
+
*,
|
|
2095
|
+
result: ToolExecutionResult,
|
|
2096
|
+
expected_call: ValidatedToolCall,
|
|
2097
|
+
expected_input_sha256: str,
|
|
2098
|
+
) -> str | None:
|
|
2099
|
+
if result.call_id != expected_call.call_id:
|
|
2100
|
+
return "result call_id mismatch"
|
|
2101
|
+
if result.input_sha256 != expected_input_sha256:
|
|
2102
|
+
return "result input_sha256 mismatch"
|
|
2103
|
+
return None
|
|
2104
|
+
|
|
2105
|
+
|
|
2106
|
+
def _safe_output_payload(result: ToolExecutionResult) -> dict[str, Any]:
|
|
2107
|
+
return {
|
|
2108
|
+
"summary": result.summary,
|
|
2109
|
+
"structured_data": result.structured_data,
|
|
2110
|
+
"artifact_refs": [ref.model_dump(mode="json") for ref in result.artifact_refs],
|
|
2111
|
+
}
|
|
2112
|
+
|
|
2113
|
+
|
|
2114
|
+
def _trace_side_effect(value: SideEffectClass) -> ToolTraceSideEffectClass:
|
|
2115
|
+
return ToolTraceSideEffectClass(value.value)
|
|
2116
|
+
|
|
2117
|
+
|
|
2118
|
+
def _trace_idempotency(value: IdempotencyClass) -> ToolTraceIdempotency:
|
|
2119
|
+
return ToolTraceIdempotency(value.value)
|
|
2120
|
+
|
|
2121
|
+
|
|
2122
|
+
def _has_denial(records: tuple[ToolTraceDecisionRecord, ...]) -> bool:
|
|
2123
|
+
return any(record.decision == ToolTraceDecision.DENIED for record in records)
|
|
2124
|
+
|
|
2125
|
+
|
|
2126
|
+
def _terminal_disposition(
|
|
2127
|
+
terminal_result: str,
|
|
2128
|
+
*,
|
|
2129
|
+
legacy_blocked_suffix: bool = False,
|
|
2130
|
+
) -> Literal["success", "blocked", "rejected"]:
|
|
2131
|
+
if terminal_result == "COMPLETE":
|
|
2132
|
+
return "success"
|
|
2133
|
+
if terminal_result == "BLOCKED":
|
|
2134
|
+
return "blocked"
|
|
2135
|
+
if terminal_result == "REJECTED":
|
|
2136
|
+
return "rejected"
|
|
2137
|
+
return (
|
|
2138
|
+
"blocked"
|
|
2139
|
+
if legacy_blocked_suffix and terminal_result.endswith("_BLOCKED")
|
|
2140
|
+
else "success"
|
|
2141
|
+
)
|
|
2142
|
+
|
|
2143
|
+
|
|
2144
|
+
def _model_visible_tool_content(result: ToolExecutionResult) -> str:
|
|
2145
|
+
if result.status == ToolExecutionStatus.SUCCESS:
|
|
2146
|
+
return result.summary
|
|
2147
|
+
return _failure_message(result)
|
|
2148
|
+
|
|
2149
|
+
|
|
2150
|
+
def _failure_message(result: ToolExecutionResult) -> str:
|
|
2151
|
+
code = result.error_code or "tool_failed"
|
|
2152
|
+
return f"[{code}] {result.summary}"
|
|
2153
|
+
|
|
2154
|
+
|
|
2155
|
+
def _diagnostic_fields(
|
|
2156
|
+
fields: Mapping[str, str | int | float | bool | None],
|
|
2157
|
+
) -> tuple[DiagnosticField, ...]:
|
|
2158
|
+
return tuple(
|
|
2159
|
+
DiagnosticField(
|
|
2160
|
+
key=str(key),
|
|
2161
|
+
value=value[:256] if isinstance(value, str) else value,
|
|
2162
|
+
)
|
|
2163
|
+
for key, value in fields.items()
|
|
2164
|
+
)
|
|
2165
|
+
|
|
2166
|
+
|
|
2167
|
+
def _timing_metadata(started_at: Any, completed_at: Any) -> TimingMetadata:
|
|
2168
|
+
return TimingMetadata(
|
|
2169
|
+
started_at=started_at.strftime("%Y-%m-%dT%H:%M:%SZ"),
|
|
2170
|
+
completed_at=completed_at.strftime("%Y-%m-%dT%H:%M:%SZ"),
|
|
2171
|
+
duration_ms=max(0.0, (completed_at - started_at).total_seconds() * 1000.0),
|
|
2172
|
+
)
|
|
2173
|
+
|
|
2174
|
+
|
|
2175
|
+
def _single_event(
|
|
2176
|
+
translator: ForgeEventTranslator,
|
|
2177
|
+
event_type: SessionEventType,
|
|
2178
|
+
*,
|
|
2179
|
+
code: str | None = None,
|
|
2180
|
+
) -> tuple[SessionEvent, ...]:
|
|
2181
|
+
return (translator.emit(event_type, code=code),)
|
|
2182
|
+
|
|
2183
|
+
|
|
2184
|
+
def _ordered_events(
|
|
2185
|
+
*groups: tuple[SessionEvent, ...],
|
|
2186
|
+
) -> tuple[SessionEvent, ...]:
|
|
2187
|
+
events = [event for group in groups for event in group]
|
|
2188
|
+
return tuple(
|
|
2189
|
+
event.model_copy(update={"sequence": sequence})
|
|
2190
|
+
for sequence, event in enumerate(events, start=1)
|
|
2191
|
+
)
|
|
2192
|
+
|
|
2193
|
+
|
|
2194
|
+
def _usage_from_bridges(
|
|
2195
|
+
model_bridge: ForgeModelBridge,
|
|
2196
|
+
tool_bridge: ForgeToolBridge,
|
|
2197
|
+
) -> UsageMetadata:
|
|
2198
|
+
executed_tool_calls = sum(
|
|
2199
|
+
1
|
|
2200
|
+
for trace in tool_bridge.tool_trace
|
|
2201
|
+
if trace.execution_status != ToolExecutionStatus.NOT_EXECUTED
|
|
2202
|
+
)
|
|
2203
|
+
return UsageMetadata(
|
|
2204
|
+
model_calls=model_bridge.model_calls,
|
|
2205
|
+
tool_calls=executed_tool_calls,
|
|
2206
|
+
token_usage=model_bridge.token_usage,
|
|
2207
|
+
)
|
|
2208
|
+
|
|
2209
|
+
|
|
2210
|
+
def _usage_from_optional_bridges(
|
|
2211
|
+
model_bridge: ForgeModelBridge | None,
|
|
2212
|
+
tool_bridge: ForgeToolBridge | None,
|
|
2213
|
+
) -> UsageMetadata | None:
|
|
2214
|
+
if model_bridge is None or tool_bridge is None:
|
|
2215
|
+
return None
|
|
2216
|
+
return _usage_from_bridges(model_bridge, tool_bridge)
|
|
2217
|
+
|
|
2218
|
+
|
|
2219
|
+
__all__ = [
|
|
2220
|
+
"ForgeGuardrailBackend",
|
|
2221
|
+
"ForgeBindingRejectedError",
|
|
2222
|
+
"ForgeBridgeError",
|
|
2223
|
+
"ForgeContextFactory",
|
|
2224
|
+
"ForgeEventTranslator",
|
|
2225
|
+
"ForgeModelBridge",
|
|
2226
|
+
"ForgeRunnerOptions",
|
|
2227
|
+
"ForgeSessionInputBuilder",
|
|
2228
|
+
"ForgeToolBridge",
|
|
2229
|
+
"ForgeWorkflowFactory",
|
|
2230
|
+
"ForgeWorkflowInput",
|
|
2231
|
+
"TerminalCandidate",
|
|
2232
|
+
]
|