@smthrs/harness 0.0.0-stage → 1.0.0-rc.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +387 -0
- package/LICENSE +21 -0
- package/README.md +138 -2
- package/dist/cjs/AgentEvent.d.ts +3092 -0
- package/dist/cjs/AgentEvent.d.ts.map +1 -0
- package/dist/cjs/AgentEvent.js +1116 -0
- package/dist/cjs/AgentEvent.js.map +7 -0
- package/dist/cjs/CallLedger.d.ts +378 -0
- package/dist/cjs/CallLedger.d.ts.map +1 -0
- package/dist/cjs/CallLedger.js +240 -0
- package/dist/cjs/CallLedger.js.map +7 -0
- package/dist/cjs/Cell.d.ts +774 -0
- package/dist/cjs/Cell.d.ts.map +1 -0
- package/dist/cjs/Cell.js +431 -0
- package/dist/cjs/Cell.js.map +7 -0
- package/dist/cjs/CellCalls.d.ts +115 -0
- package/dist/cjs/CellCalls.d.ts.map +1 -0
- package/dist/cjs/CellCalls.js +98 -0
- package/dist/cjs/CellCalls.js.map +7 -0
- package/dist/cjs/CellHistory.d.ts +102 -0
- package/dist/cjs/CellHistory.d.ts.map +1 -0
- package/dist/cjs/CellHistory.js +54 -0
- package/dist/cjs/CellHistory.js.map +7 -0
- package/dist/cjs/CellTurn.d.ts +1071 -0
- package/dist/cjs/CellTurn.d.ts.map +1 -0
- package/dist/cjs/CellTurn.js +2618 -0
- package/dist/cjs/CellTurn.js.map +7 -0
- package/dist/cjs/CellValidation.d.ts +94 -0
- package/dist/cjs/CellValidation.d.ts.map +1 -0
- package/dist/cjs/CellValidation.js +218 -0
- package/dist/cjs/CellValidation.js.map +7 -0
- package/dist/cjs/Compaction.d.ts +158 -0
- package/dist/cjs/Compaction.d.ts.map +1 -0
- package/dist/cjs/Compaction.js +189 -0
- package/dist/cjs/Compaction.js.map +7 -0
- package/dist/cjs/CompletionClaim.d.ts +801 -0
- package/dist/cjs/CompletionClaim.d.ts.map +1 -0
- package/dist/cjs/CompletionClaim.js +303 -0
- package/dist/cjs/CompletionClaim.js.map +7 -0
- package/dist/cjs/ContextWindow.d.ts +435 -0
- package/dist/cjs/ContextWindow.d.ts.map +1 -0
- package/dist/cjs/ContextWindow.js +319 -0
- package/dist/cjs/ContextWindow.js.map +7 -0
- package/dist/cjs/EngineLike.d.ts +546 -0
- package/dist/cjs/EngineLike.d.ts.map +1 -0
- package/dist/cjs/EngineLike.js +114 -0
- package/dist/cjs/EngineLike.js.map +7 -0
- package/dist/cjs/ExternalTranscript.d.ts +349 -0
- package/dist/cjs/ExternalTranscript.d.ts.map +1 -0
- package/dist/cjs/ExternalTranscript.js +826 -0
- package/dist/cjs/ExternalTranscript.js.map +7 -0
- package/dist/cjs/FailedCall.d.ts +132 -0
- package/dist/cjs/FailedCall.d.ts.map +1 -0
- package/dist/cjs/FailedCall.js +53 -0
- package/dist/cjs/FailedCall.js.map +7 -0
- package/dist/cjs/FlowBinding.d.ts +289 -0
- package/dist/cjs/FlowBinding.d.ts.map +1 -0
- package/dist/cjs/FlowBinding.js +251 -0
- package/dist/cjs/FlowBinding.js.map +7 -0
- package/dist/cjs/HarnessError.d.ts +57 -0
- package/dist/cjs/HarnessError.d.ts.map +1 -0
- package/dist/cjs/HarnessError.js +75 -0
- package/dist/cjs/HarnessError.js.map +7 -0
- package/dist/cjs/Judgement.d.ts +289 -0
- package/dist/cjs/Judgement.d.ts.map +1 -0
- package/dist/cjs/Judgement.js +240 -0
- package/dist/cjs/Judgement.js.map +7 -0
- package/dist/cjs/Monitor.d.ts +374 -0
- package/dist/cjs/Monitor.d.ts.map +1 -0
- package/dist/cjs/Monitor.js +233 -0
- package/dist/cjs/Monitor.js.map +7 -0
- package/dist/cjs/NarrowedCheck.d.ts +523 -0
- package/dist/cjs/NarrowedCheck.d.ts.map +1 -0
- package/dist/cjs/NarrowedCheck.js +263 -0
- package/dist/cjs/NarrowedCheck.js.map +7 -0
- package/dist/cjs/Notifications.d.ts +42 -0
- package/dist/cjs/Notifications.d.ts.map +1 -0
- package/dist/cjs/Notifications.js +177 -0
- package/dist/cjs/Notifications.js.map +7 -0
- package/dist/cjs/Plan.d.ts +127 -0
- package/dist/cjs/Plan.d.ts.map +1 -0
- package/dist/cjs/Plan.js +77 -0
- package/dist/cjs/Plan.js.map +7 -0
- package/dist/cjs/QuickJSSandbox.d.ts +151 -0
- package/dist/cjs/QuickJSSandbox.d.ts.map +1 -0
- package/dist/cjs/QuickJSSandbox.js +987 -0
- package/dist/cjs/QuickJSSandbox.js.map +7 -0
- package/dist/cjs/Relevance.d.ts +213 -0
- package/dist/cjs/Relevance.d.ts.map +1 -0
- package/dist/cjs/Relevance.js +183 -0
- package/dist/cjs/Relevance.js.map +7 -0
- package/dist/cjs/Sandbox.d.ts +637 -0
- package/dist/cjs/Sandbox.d.ts.map +1 -0
- package/dist/cjs/Sandbox.js +260 -0
- package/dist/cjs/Sandbox.js.map +7 -0
- package/dist/cjs/Steering.d.ts +464 -0
- package/dist/cjs/Steering.d.ts.map +1 -0
- package/dist/cjs/Steering.js +153 -0
- package/dist/cjs/Steering.js.map +7 -0
- package/dist/cjs/StructuredOutput.d.ts +252 -0
- package/dist/cjs/StructuredOutput.d.ts.map +1 -0
- package/dist/cjs/StructuredOutput.js +266 -0
- package/dist/cjs/StructuredOutput.js.map +7 -0
- package/dist/cjs/Sufficiency.d.ts +195 -0
- package/dist/cjs/Sufficiency.d.ts.map +1 -0
- package/dist/cjs/Sufficiency.js +110 -0
- package/dist/cjs/Sufficiency.js.map +7 -0
- package/dist/cjs/Supervisor.d.ts +719 -0
- package/dist/cjs/Supervisor.d.ts.map +1 -0
- package/dist/cjs/Supervisor.js +313 -0
- package/dist/cjs/Supervisor.js.map +7 -0
- package/dist/cjs/Tokens.d.ts +86 -0
- package/dist/cjs/Tokens.d.ts.map +1 -0
- package/dist/cjs/Tokens.js +72 -0
- package/dist/cjs/Tokens.js.map +7 -0
- package/dist/cjs/Transcript.d.ts +171 -0
- package/dist/cjs/Transcript.d.ts.map +1 -0
- package/dist/cjs/Transcript.js +342 -0
- package/dist/cjs/Transcript.js.map +7 -0
- package/dist/cjs/TruncatedOutput.d.ts +186 -0
- package/dist/cjs/TruncatedOutput.d.ts.map +1 -0
- package/dist/cjs/TruncatedOutput.js +143 -0
- package/dist/cjs/TruncatedOutput.js.map +7 -0
- package/dist/cjs/UnmovedTree.d.ts +113 -0
- package/dist/cjs/UnmovedTree.d.ts.map +1 -0
- package/dist/cjs/UnmovedTree.js +38 -0
- package/dist/cjs/UnmovedTree.js.map +7 -0
- package/dist/cjs/UnresolvedFailure.d.ts +196 -0
- package/dist/cjs/UnresolvedFailure.d.ts.map +1 -0
- package/dist/cjs/UnresolvedFailure.js +77 -0
- package/dist/cjs/UnresolvedFailure.js.map +7 -0
- package/dist/cjs/VacuousVerification.d.ts +237 -0
- package/dist/cjs/VacuousVerification.d.ts.map +1 -0
- package/dist/cjs/VacuousVerification.js +91 -0
- package/dist/cjs/VacuousVerification.js.map +7 -0
- package/dist/cjs/VariablesPanel.d.ts +117 -0
- package/dist/cjs/VariablesPanel.d.ts.map +1 -0
- package/dist/cjs/VariablesPanel.js +108 -0
- package/dist/cjs/VariablesPanel.js.map +7 -0
- package/dist/cjs/index.d.ts +172 -0
- package/dist/cjs/index.d.ts.map +1 -0
- package/dist/cjs/index.js +99 -0
- package/dist/cjs/index.js.map +7 -0
- package/dist/cjs/internal/bytes.d.ts +36 -0
- package/dist/cjs/internal/bytes.d.ts.map +1 -0
- package/dist/cjs/internal/bytes.js +57 -0
- package/dist/cjs/internal/bytes.js.map +7 -0
- package/dist/cjs/internal/cellPrompt.d.ts +122 -0
- package/dist/cjs/internal/cellPrompt.d.ts.map +1 -0
- package/dist/cjs/internal/cellPrompt.js +146 -0
- package/dist/cjs/internal/cellPrompt.js.map +7 -0
- package/dist/cjs/internal/compactable.d.ts +44 -0
- package/dist/cjs/internal/compactable.d.ts.map +1 -0
- package/dist/cjs/internal/compactable.js +56 -0
- package/dist/cjs/internal/compactable.js.map +7 -0
- package/dist/cjs/internal/compactionMarks.d.ts +278 -0
- package/dist/cjs/internal/compactionMarks.d.ts.map +1 -0
- package/dist/cjs/internal/compactionMarks.js +212 -0
- package/dist/cjs/internal/compactionMarks.js.map +7 -0
- package/dist/cjs/internal/demandText.d.ts +95 -0
- package/dist/cjs/internal/demandText.d.ts.map +1 -0
- package/dist/cjs/internal/demandText.js +70 -0
- package/dist/cjs/internal/demandText.js.map +7 -0
- package/dist/cjs/internal/elide.d.ts +109 -0
- package/dist/cjs/internal/elide.d.ts.map +1 -0
- package/dist/cjs/internal/elide.js +59 -0
- package/dist/cjs/internal/elide.js.map +7 -0
- package/dist/cjs/internal/frame.d.ts +463 -0
- package/dist/cjs/internal/frame.d.ts.map +1 -0
- package/dist/cjs/internal/frame.js +514 -0
- package/dist/cjs/internal/frame.js.map +7 -0
- package/dist/cjs/internal/nonNegativeSafeInt.d.ts +20 -0
- package/dist/cjs/internal/nonNegativeSafeInt.d.ts.map +1 -0
- package/dist/cjs/internal/nonNegativeSafeInt.js +29 -0
- package/dist/cjs/internal/nonNegativeSafeInt.js.map +7 -0
- package/dist/cjs/internal/paidUsage.d.ts +52 -0
- package/dist/cjs/internal/paidUsage.d.ts.map +1 -0
- package/dist/cjs/internal/paidUsage.js +70 -0
- package/dist/cjs/internal/paidUsage.js.map +7 -0
- package/dist/cjs/internal/printChannel.d.ts +233 -0
- package/dist/cjs/internal/printChannel.d.ts.map +1 -0
- package/dist/cjs/internal/printChannel.js +165 -0
- package/dist/cjs/internal/printChannel.js.map +7 -0
- package/dist/cjs/internal/printsObservation.d.ts +26 -0
- package/dist/cjs/internal/printsObservation.d.ts.map +1 -0
- package/dist/cjs/internal/printsObservation.js +27 -0
- package/dist/cjs/internal/printsObservation.js.map +7 -0
- package/dist/cjs/internal/refusal.d.ts +40 -0
- package/dist/cjs/internal/refusal.d.ts.map +1 -0
- package/dist/cjs/internal/refusal.js +45 -0
- package/dist/cjs/internal/refusal.js.map +7 -0
- package/dist/cjs/internal/supervision.d.ts +174 -0
- package/dist/cjs/internal/supervision.d.ts.map +1 -0
- package/dist/cjs/internal/supervision.js +402 -0
- package/dist/cjs/internal/supervision.js.map +7 -0
- package/dist/cjs/internal/unfinishedWork.d.ts +99 -0
- package/dist/cjs/internal/unfinishedWork.d.ts.map +1 -0
- package/dist/cjs/internal/unfinishedWork.js +80 -0
- package/dist/cjs/internal/unfinishedWork.js.map +7 -0
- package/dist/cjs/internal/unobservedCall.d.ts +112 -0
- package/dist/cjs/internal/unobservedCall.d.ts.map +1 -0
- package/dist/cjs/internal/unobservedCall.js +345 -0
- package/dist/cjs/internal/unobservedCall.js.map +7 -0
- package/dist/cjs/internal/untrustedData.d.ts +14 -0
- package/dist/cjs/internal/untrustedData.d.ts.map +1 -0
- package/dist/cjs/internal/untrustedData.js +30 -0
- package/dist/cjs/internal/untrustedData.js.map +7 -0
- package/dist/cjs/package.json +1 -0
- package/dist/esm/AgentEvent.d.ts +3092 -0
- package/dist/esm/AgentEvent.d.ts.map +1 -0
- package/dist/esm/AgentEvent.js +1610 -0
- package/dist/esm/AgentEvent.js.map +1 -0
- package/dist/esm/CallLedger.d.ts +378 -0
- package/dist/esm/CallLedger.d.ts.map +1 -0
- package/dist/esm/CallLedger.js +506 -0
- package/dist/esm/CallLedger.js.map +1 -0
- package/dist/esm/Cell.d.ts +774 -0
- package/dist/esm/Cell.d.ts.map +1 -0
- package/dist/esm/Cell.js +772 -0
- package/dist/esm/Cell.js.map +1 -0
- package/dist/esm/CellCalls.d.ts +115 -0
- package/dist/esm/CellCalls.d.ts.map +1 -0
- package/dist/esm/CellCalls.js +97 -0
- package/dist/esm/CellCalls.js.map +1 -0
- package/dist/esm/CellHistory.d.ts +102 -0
- package/dist/esm/CellHistory.d.ts.map +1 -0
- package/dist/esm/CellHistory.js +91 -0
- package/dist/esm/CellHistory.js.map +1 -0
- package/dist/esm/CellTurn.d.ts +1071 -0
- package/dist/esm/CellTurn.d.ts.map +1 -0
- package/dist/esm/CellTurn.js +3410 -0
- package/dist/esm/CellTurn.js.map +1 -0
- package/dist/esm/CellValidation.d.ts +94 -0
- package/dist/esm/CellValidation.d.ts.map +1 -0
- package/dist/esm/CellValidation.js +335 -0
- package/dist/esm/CellValidation.js.map +1 -0
- package/dist/esm/Compaction.d.ts +158 -0
- package/dist/esm/Compaction.d.ts.map +1 -0
- package/dist/esm/Compaction.js +216 -0
- package/dist/esm/Compaction.js.map +1 -0
- package/dist/esm/CompletionClaim.d.ts +801 -0
- package/dist/esm/CompletionClaim.d.ts.map +1 -0
- package/dist/esm/CompletionClaim.js +833 -0
- package/dist/esm/CompletionClaim.js.map +1 -0
- package/dist/esm/ContextWindow.d.ts +435 -0
- package/dist/esm/ContextWindow.d.ts.map +1 -0
- package/dist/esm/ContextWindow.js +427 -0
- package/dist/esm/ContextWindow.js.map +1 -0
- package/dist/esm/EngineLike.d.ts +546 -0
- package/dist/esm/EngineLike.d.ts.map +1 -0
- package/dist/esm/EngineLike.js +218 -0
- package/dist/esm/EngineLike.js.map +1 -0
- package/dist/esm/ExternalTranscript.d.ts +349 -0
- package/dist/esm/ExternalTranscript.d.ts.map +1 -0
- package/dist/esm/ExternalTranscript.js +987 -0
- package/dist/esm/ExternalTranscript.js.map +1 -0
- package/dist/esm/FailedCall.d.ts +132 -0
- package/dist/esm/FailedCall.d.ts.map +1 -0
- package/dist/esm/FailedCall.js +131 -0
- package/dist/esm/FailedCall.js.map +1 -0
- package/dist/esm/FlowBinding.d.ts +289 -0
- package/dist/esm/FlowBinding.d.ts.map +1 -0
- package/dist/esm/FlowBinding.js +376 -0
- package/dist/esm/FlowBinding.js.map +1 -0
- package/dist/esm/HarnessError.d.ts +57 -0
- package/dist/esm/HarnessError.d.ts.map +1 -0
- package/dist/esm/HarnessError.js +85 -0
- package/dist/esm/HarnessError.js.map +1 -0
- package/dist/esm/Judgement.d.ts +289 -0
- package/dist/esm/Judgement.d.ts.map +1 -0
- package/dist/esm/Judgement.js +305 -0
- package/dist/esm/Judgement.js.map +1 -0
- package/dist/esm/Monitor.d.ts +374 -0
- package/dist/esm/Monitor.d.ts.map +1 -0
- package/dist/esm/Monitor.js +370 -0
- package/dist/esm/Monitor.js.map +1 -0
- package/dist/esm/NarrowedCheck.d.ts +523 -0
- package/dist/esm/NarrowedCheck.d.ts.map +1 -0
- package/dist/esm/NarrowedCheck.js +612 -0
- package/dist/esm/NarrowedCheck.js.map +1 -0
- package/dist/esm/Notifications.d.ts +42 -0
- package/dist/esm/Notifications.d.ts.map +1 -0
- package/dist/esm/Notifications.js +215 -0
- package/dist/esm/Notifications.js.map +1 -0
- package/dist/esm/Plan.d.ts +127 -0
- package/dist/esm/Plan.d.ts.map +1 -0
- package/dist/esm/Plan.js +97 -0
- package/dist/esm/Plan.js.map +1 -0
- package/dist/esm/QuickJSSandbox.d.ts +151 -0
- package/dist/esm/QuickJSSandbox.d.ts.map +1 -0
- package/dist/esm/QuickJSSandbox.js +1364 -0
- package/dist/esm/QuickJSSandbox.js.map +1 -0
- package/dist/esm/Relevance.d.ts +213 -0
- package/dist/esm/Relevance.d.ts.map +1 -0
- package/dist/esm/Relevance.js +254 -0
- package/dist/esm/Relevance.js.map +1 -0
- package/dist/esm/Sandbox.d.ts +637 -0
- package/dist/esm/Sandbox.d.ts.map +1 -0
- package/dist/esm/Sandbox.js +464 -0
- package/dist/esm/Sandbox.js.map +1 -0
- package/dist/esm/Steering.d.ts +464 -0
- package/dist/esm/Steering.d.ts.map +1 -0
- package/dist/esm/Steering.js +193 -0
- package/dist/esm/Steering.js.map +1 -0
- package/dist/esm/StructuredOutput.d.ts +252 -0
- package/dist/esm/StructuredOutput.d.ts.map +1 -0
- package/dist/esm/StructuredOutput.js +430 -0
- package/dist/esm/StructuredOutput.js.map +1 -0
- package/dist/esm/Sufficiency.d.ts +195 -0
- package/dist/esm/Sufficiency.d.ts.map +1 -0
- package/dist/esm/Sufficiency.js +207 -0
- package/dist/esm/Sufficiency.js.map +1 -0
- package/dist/esm/Supervisor.d.ts +719 -0
- package/dist/esm/Supervisor.d.ts.map +1 -0
- package/dist/esm/Supervisor.js +575 -0
- package/dist/esm/Supervisor.js.map +1 -0
- package/dist/esm/Tokens.d.ts +86 -0
- package/dist/esm/Tokens.d.ts.map +1 -0
- package/dist/esm/Tokens.js +92 -0
- package/dist/esm/Tokens.js.map +1 -0
- package/dist/esm/Transcript.d.ts +171 -0
- package/dist/esm/Transcript.d.ts.map +1 -0
- package/dist/esm/Transcript.js +424 -0
- package/dist/esm/Transcript.js.map +1 -0
- package/dist/esm/TruncatedOutput.d.ts +186 -0
- package/dist/esm/TruncatedOutput.d.ts.map +1 -0
- package/dist/esm/TruncatedOutput.js +257 -0
- package/dist/esm/TruncatedOutput.js.map +1 -0
- package/dist/esm/UnmovedTree.d.ts +113 -0
- package/dist/esm/UnmovedTree.d.ts.map +1 -0
- package/dist/esm/UnmovedTree.js +90 -0
- package/dist/esm/UnmovedTree.js.map +1 -0
- package/dist/esm/UnresolvedFailure.d.ts +196 -0
- package/dist/esm/UnresolvedFailure.d.ts.map +1 -0
- package/dist/esm/UnresolvedFailure.js +218 -0
- package/dist/esm/UnresolvedFailure.js.map +1 -0
- package/dist/esm/VacuousVerification.d.ts +237 -0
- package/dist/esm/VacuousVerification.d.ts.map +1 -0
- package/dist/esm/VacuousVerification.js +245 -0
- package/dist/esm/VacuousVerification.js.map +1 -0
- package/dist/esm/VariablesPanel.d.ts +117 -0
- package/dist/esm/VariablesPanel.d.ts.map +1 -0
- package/dist/esm/VariablesPanel.js +142 -0
- package/dist/esm/VariablesPanel.js.map +1 -0
- package/dist/esm/index.d.ts +172 -0
- package/dist/esm/index.d.ts.map +1 -0
- package/dist/esm/index.js +172 -0
- package/dist/esm/index.js.map +1 -0
- package/dist/esm/internal/bytes.d.ts +36 -0
- package/dist/esm/internal/bytes.d.ts.map +1 -0
- package/dist/esm/internal/bytes.js +70 -0
- package/dist/esm/internal/bytes.js.map +1 -0
- package/dist/esm/internal/cellPrompt.d.ts +122 -0
- package/dist/esm/internal/cellPrompt.d.ts.map +1 -0
- package/dist/esm/internal/cellPrompt.js +276 -0
- package/dist/esm/internal/cellPrompt.js.map +1 -0
- package/dist/esm/internal/compactable.d.ts +44 -0
- package/dist/esm/internal/compactable.d.ts.map +1 -0
- package/dist/esm/internal/compactable.js +71 -0
- package/dist/esm/internal/compactable.js.map +1 -0
- package/dist/esm/internal/compactionMarks.d.ts +278 -0
- package/dist/esm/internal/compactionMarks.d.ts.map +1 -0
- package/dist/esm/internal/compactionMarks.js +317 -0
- package/dist/esm/internal/compactionMarks.js.map +1 -0
- package/dist/esm/internal/demandText.d.ts +95 -0
- package/dist/esm/internal/demandText.d.ts.map +1 -0
- package/dist/esm/internal/demandText.js +128 -0
- package/dist/esm/internal/demandText.js.map +1 -0
- package/dist/esm/internal/elide.d.ts +109 -0
- package/dist/esm/internal/elide.d.ts.map +1 -0
- package/dist/esm/internal/elide.js +123 -0
- package/dist/esm/internal/elide.js.map +1 -0
- package/dist/esm/internal/frame.d.ts +463 -0
- package/dist/esm/internal/frame.d.ts.map +1 -0
- package/dist/esm/internal/frame.js +861 -0
- package/dist/esm/internal/frame.js.map +1 -0
- package/dist/esm/internal/nonNegativeSafeInt.d.ts +20 -0
- package/dist/esm/internal/nonNegativeSafeInt.d.ts.map +1 -0
- package/dist/esm/internal/nonNegativeSafeInt.js +20 -0
- package/dist/esm/internal/nonNegativeSafeInt.js.map +1 -0
- package/dist/esm/internal/paidUsage.d.ts +52 -0
- package/dist/esm/internal/paidUsage.d.ts.map +1 -0
- package/dist/esm/internal/paidUsage.js +72 -0
- package/dist/esm/internal/paidUsage.js.map +1 -0
- package/dist/esm/internal/printChannel.d.ts +233 -0
- package/dist/esm/internal/printChannel.d.ts.map +1 -0
- package/dist/esm/internal/printChannel.js +376 -0
- package/dist/esm/internal/printChannel.js.map +1 -0
- package/dist/esm/internal/printsObservation.d.ts +26 -0
- package/dist/esm/internal/printsObservation.d.ts.map +1 -0
- package/dist/esm/internal/printsObservation.js +29 -0
- package/dist/esm/internal/printsObservation.js.map +1 -0
- package/dist/esm/internal/refusal.d.ts +40 -0
- package/dist/esm/internal/refusal.d.ts.map +1 -0
- package/dist/esm/internal/refusal.js +47 -0
- package/dist/esm/internal/refusal.js.map +1 -0
- package/dist/esm/internal/supervision.d.ts +174 -0
- package/dist/esm/internal/supervision.d.ts.map +1 -0
- package/dist/esm/internal/supervision.js +485 -0
- package/dist/esm/internal/supervision.js.map +1 -0
- package/dist/esm/internal/unfinishedWork.d.ts +99 -0
- package/dist/esm/internal/unfinishedWork.d.ts.map +1 -0
- package/dist/esm/internal/unfinishedWork.js +85 -0
- package/dist/esm/internal/unfinishedWork.js.map +1 -0
- package/dist/esm/internal/unobservedCall.d.ts +112 -0
- package/dist/esm/internal/unobservedCall.d.ts.map +1 -0
- package/dist/esm/internal/unobservedCall.js +501 -0
- package/dist/esm/internal/unobservedCall.js.map +1 -0
- package/dist/esm/internal/untrustedData.d.ts +14 -0
- package/dist/esm/internal/untrustedData.d.ts.map +1 -0
- package/dist/esm/internal/untrustedData.js +15 -0
- package/dist/esm/internal/untrustedData.js.map +1 -0
- package/docs/README.md +129 -0
- package/docs/api.md +1869 -0
- package/docs/concepts.md +238 -0
- package/docs/external-transcripts.md +245 -0
- package/docs/guides/bind-flows.md +167 -0
- package/docs/guides/drive-the-loop.md +265 -0
- package/docs/guides/run-cells.md +262 -0
- package/docs/guides/workerd.md +144 -0
- package/docs/installation.md +69 -0
- package/docs/quickstart.md +121 -0
- package/docs/reference.md +954 -0
- package/docs/troubleshooting.md +177 -0
- package/package.json +463 -3
- package/src/AgentEvent.ts +1772 -0
- package/src/CallLedger.ts +560 -0
- package/src/Cell.ts +926 -0
- package/src/CellCalls.ts +198 -0
- package/src/CellHistory.ts +128 -0
- package/src/CellTurn.ts +4450 -0
- package/src/CellValidation.ts +382 -0
- package/src/Compaction.ts +330 -0
- package/src/CompletionClaim.ts +1013 -0
- package/src/ContextWindow.ts +669 -0
- package/src/EngineLike.ts +614 -0
- package/src/ExternalTranscript.ts +1143 -0
- package/src/FailedCall.ts +163 -0
- package/src/FlowBinding.ts +603 -0
- package/src/HarnessError.ts +98 -0
- package/src/Judgement.ts +526 -0
- package/src/Monitor.ts +579 -0
- package/src/NarrowedCheck.ts +694 -0
- package/src/Notifications.ts +262 -0
- package/src/Plan.ts +113 -0
- package/src/QuickJSSandbox.ts +1557 -0
- package/src/Relevance.ts +352 -0
- package/src/Sandbox.ts +908 -0
- package/src/Steering.ts +405 -0
- package/src/StructuredOutput.ts +484 -0
- package/src/Sufficiency.ts +247 -0
- package/src/Supervisor.ts +798 -0
- package/src/Tokens.ts +108 -0
- package/src/Transcript.ts +513 -0
- package/src/TruncatedOutput.ts +297 -0
- package/src/UnmovedTree.ts +121 -0
- package/src/UnresolvedFailure.ts +240 -0
- package/src/VacuousVerification.ts +280 -0
- package/src/VariablesPanel.ts +165 -0
- package/src/index.ts +204 -0
- package/src/internal/bytes.ts +70 -0
- package/src/internal/cellPrompt.ts +337 -0
- package/src/internal/compactable.ts +76 -0
- package/src/internal/compactionMarks.ts +454 -0
- package/src/internal/demandText.ts +145 -0
- package/src/internal/elide.ts +130 -0
- package/src/internal/frame.ts +1178 -0
- package/src/internal/nonNegativeSafeInt.ts +24 -0
- package/src/internal/paidUsage.ts +90 -0
- package/src/internal/printChannel.ts +434 -0
- package/src/internal/printsObservation.ts +31 -0
- package/src/internal/refusal.ts +53 -0
- package/src/internal/supervision.ts +652 -0
- package/src/internal/unfinishedWork.ts +123 -0
- package/src/internal/unobservedCall.ts +536 -0
- package/src/internal/untrustedData.ts +19 -0
|
@@ -0,0 +1,3410 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The cell-first controller.
|
|
3
|
+
*
|
|
4
|
+
* Smithers is a state machine. This module is its deterministic outer loop: it
|
|
5
|
+
* decides continue, park, or finish from durable evidence — the transition a
|
|
6
|
+
* cell returned and the budgets the run declared — and never from the presence
|
|
7
|
+
* of a provider tool call.
|
|
8
|
+
*
|
|
9
|
+
* One frame is: seal a model step, recover the cell from the settlement, run it
|
|
10
|
+
* in the sandbox, resolve each of its flow calls as its own keyed durable
|
|
11
|
+
* boundary, then apply the transition it returned. The cell owns the state that
|
|
12
|
+
* carries forward and the exact context the next frame sees.
|
|
13
|
+
*
|
|
14
|
+
* Governing design: `../docs/concepts.md#durable-cell-loop`.
|
|
15
|
+
*
|
|
16
|
+
* @since 0.1.0
|
|
17
|
+
*/
|
|
18
|
+
import { Effects, Placement } from "@smthrs/core";
|
|
19
|
+
import * as Digest from "@smthrs/core/Digest";
|
|
20
|
+
import * as Fault from "@smthrs/flow/Fault";
|
|
21
|
+
import { Capability, CapabilitySet, Permission } from "@smthrs/kernel";
|
|
22
|
+
import { CanonicalJson, ModelCatalog, ModelEvent, ModelRequest } from "@smthrs/model";
|
|
23
|
+
import * as Pricing from "@smthrs/model/Pricing";
|
|
24
|
+
import { Descriptor } from "@smthrs/registry";
|
|
25
|
+
import { Clock, Effect, Option, Queue, Result, Schema, Stream } from "effect";
|
|
26
|
+
import * as AgentEvent from "./AgentEvent.js";
|
|
27
|
+
import * as CallLedger from "./CallLedger.js";
|
|
28
|
+
import * as Cell from "./Cell.js";
|
|
29
|
+
import * as CellHistory from "./CellHistory.js";
|
|
30
|
+
import * as CellValidation from "./CellValidation.js";
|
|
31
|
+
import * as Compaction from "./Compaction.js";
|
|
32
|
+
import * as CompletionClaim from "./CompletionClaim.js";
|
|
33
|
+
import * as ContextWindow from "./ContextWindow.js";
|
|
34
|
+
import * as EngineLike from "./EngineLike.js";
|
|
35
|
+
import * as FailedCall from "./FailedCall.js";
|
|
36
|
+
import { HarnessError } from "./HarnessError.js";
|
|
37
|
+
import * as cellPrompt from "./internal/cellPrompt.js";
|
|
38
|
+
import { compactable, defaultKeepRecent, defaultReserve } from "./internal/compactable.js";
|
|
39
|
+
import * as compactionMarks from "./internal/compactionMarks.js";
|
|
40
|
+
import * as elide from "./internal/elide.js";
|
|
41
|
+
import * as Frame from "./internal/frame.js";
|
|
42
|
+
import { NonNegativeSafeInt } from "./internal/nonNegativeSafeInt.js";
|
|
43
|
+
import { paidUsage } from "./internal/paidUsage.js";
|
|
44
|
+
import { printsObservation } from "./internal/printsObservation.js";
|
|
45
|
+
import { limitOf, refusal } from "./internal/refusal.js";
|
|
46
|
+
import * as Supervision from "./internal/supervision.js";
|
|
47
|
+
import { untrustedData } from "./internal/untrustedData.js";
|
|
48
|
+
import * as Judgement from "./Judgement.js";
|
|
49
|
+
import * as Monitor from "./Monitor.js";
|
|
50
|
+
import * as NarrowedCheck from "./NarrowedCheck.js";
|
|
51
|
+
import * as Relevance from "./Relevance.js";
|
|
52
|
+
import * as Sandbox from "./Sandbox.js";
|
|
53
|
+
import * as Steering from "./Steering.js";
|
|
54
|
+
import * as Sufficiency from "./Sufficiency.js";
|
|
55
|
+
import * as Supervisor from "./Supervisor.js";
|
|
56
|
+
import { journalVersion } from "./Transcript.js";
|
|
57
|
+
import * as TruncatedOutput from "./TruncatedOutput.js";
|
|
58
|
+
import * as UnresolvedFailure from "./UnresolvedFailure.js";
|
|
59
|
+
import * as VariablesPanel from "./VariablesPanel.js";
|
|
60
|
+
/** The cause the repeated-failure guard ends a turn with: the model looping, a replan's to fix. */
|
|
61
|
+
const repeatedFailureTag = "/harness/CellTurn/RepeatedFailure";
|
|
62
|
+
Fault.register(repeatedFailureTag, "factory");
|
|
63
|
+
/**
|
|
64
|
+
* Default number of frames one admitted task may spend. Zero disarms this limit.
|
|
65
|
+
*
|
|
66
|
+
* @category constants
|
|
67
|
+
* @since 0.1.0
|
|
68
|
+
* @slop
|
|
69
|
+
*/
|
|
70
|
+
export const defaultMaxFrames = 100;
|
|
71
|
+
/**
|
|
72
|
+
* Default number of consecutive read-only frames a task run may spend.
|
|
73
|
+
*
|
|
74
|
+
* Read-only means the frame left the workspace exactly as it found it —
|
|
75
|
+
* measured, not declared; see {@link State.readOnlyFrames}. The number comes
|
|
76
|
+
* from the first head-to-head benchmark — every instance the loop resolved had
|
|
77
|
+
* edited a file well before frame 15, and the instance it lost outright read
|
|
78
|
+
* for all 100 frames, made 132 calls, attempted zero edits, and then claimed
|
|
79
|
+
* the fix was implemented.
|
|
80
|
+
*
|
|
81
|
+
* @category constants
|
|
82
|
+
* @since 0.1.0
|
|
83
|
+
* @slop
|
|
84
|
+
*/
|
|
85
|
+
export const defaultReadOnlyFrames = 12;
|
|
86
|
+
/**
|
|
87
|
+
* Default wall-clock milliseconds one model call may spend.
|
|
88
|
+
*
|
|
89
|
+
* The number is read off wave 7 of the SWE-bench harness, which journals
|
|
90
|
+
* `durationMillis` for every sealed step. Its 68 model calls settled at a
|
|
91
|
+
* median of 10.2 s, a p90 of 45.2 s, and a p95 of 110.6 s; the longest call
|
|
92
|
+
* that produced a usable answer took 252.3 s (`django__django-16612`, an
|
|
93
|
+
* instance that resolved), and the next longest 169.6 s. One call stood
|
|
94
|
+
* outside that distribution entirely: 667.1 s on `pytest-dev__pytest-6197` —
|
|
95
|
+
* 55% of the run's 1,200 s budget and 60,703 output tokens — for a cell that
|
|
96
|
+
* raised on its first property access. Nothing capped it, because a model call
|
|
97
|
+
* was the one thing the armed discipline did not bound.
|
|
98
|
+
*
|
|
99
|
+
* 300 s clears the longest answering call by 19% and every other call in the
|
|
100
|
+
* wave by more than 2.6x, so the budget is not a latency target and does not
|
|
101
|
+
* ration ordinary thinking; it is the ceiling that separates a slow answer
|
|
102
|
+
* from a run spending half its wall clock on one. Under it the outlier is
|
|
103
|
+
* interrupted at a quarter of the process budget instead of consuming
|
|
104
|
+
* more than half of it, and the retry that follows costs a jittered second
|
|
105
|
+
* rather than the whole run. 240 s would have cut off django's 252 s call, so
|
|
106
|
+
* it is not the number the evidence supports.
|
|
107
|
+
*
|
|
108
|
+
* Zero disarms the budget, which is what a host that wants a model call
|
|
109
|
+
* bounded by nothing but its own process must ask for explicitly.
|
|
110
|
+
*
|
|
111
|
+
* @category constants
|
|
112
|
+
* @since 0.1.0
|
|
113
|
+
*/
|
|
114
|
+
export const defaultModelCallMs = 300_000;
|
|
115
|
+
/**
|
|
116
|
+
* The absolute ceiling one model call gets at a given reasoning effort.
|
|
117
|
+
*
|
|
118
|
+
* {@link defaultModelCallMs} was read off a wave that did not think for
|
|
119
|
+
* long. A model asked for `xhigh` or `max` effort can spend most of a quarter
|
|
120
|
+
* hour on one decisive turn and still be working: on Terminal-Bench 4.0
|
|
121
|
+
* `photonic-waveguide-routing` (2026-09-24) stock Codex's design turn took
|
|
122
|
+
* 735 s on `gpt-6-sol` at `max`, and three of three harness attempts died at
|
|
123
|
+
* that turn on the 300 s ceiling and its one re-issue. A ceiling is no longer
|
|
124
|
+
* the only bound on a call: `recordModelStep` also cuts off a stream that goes
|
|
125
|
+
* silent, which is the stall a long ceiling would otherwise wait out. So the
|
|
126
|
+
* ceiling only has to stop a call that keeps producing and never ends, and it
|
|
127
|
+
* scales with the effort the call was asked for:
|
|
128
|
+
*
|
|
129
|
+
* | effort | ceiling |
|
|
130
|
+
* | ----------------------------------------- | ------- |
|
|
131
|
+
* | unset, `none`, `minimal`, `low`, `medium` | 300 s |
|
|
132
|
+
* | `high` | 900 s |
|
|
133
|
+
* | `xhigh`, `max` | 1,800 s |
|
|
134
|
+
*
|
|
135
|
+
* 1,800 s is 2.4x the 735 s turn. The lower efforts keep the wave 7 number.
|
|
136
|
+
*
|
|
137
|
+
* @category constants
|
|
138
|
+
* @since 1.0.0-rc.0
|
|
139
|
+
*/
|
|
140
|
+
export const modelCallMsFor = (effort) => {
|
|
141
|
+
switch (effort) {
|
|
142
|
+
case "xhigh":
|
|
143
|
+
case "max":
|
|
144
|
+
return 1_800_000;
|
|
145
|
+
case "high":
|
|
146
|
+
return 900_000;
|
|
147
|
+
default:
|
|
148
|
+
return defaultModelCallMs;
|
|
149
|
+
}
|
|
150
|
+
};
|
|
151
|
+
/**
|
|
152
|
+
* Default number of consecutive repeat-observation frames a run may spend.
|
|
153
|
+
*
|
|
154
|
+
* A repeat-observation frame is one that issued at least one call, issued no
|
|
155
|
+
* call this run had not already issued, and changed nothing. Such a frame buys
|
|
156
|
+
* a model step and a flow call and is handed back what the run was already
|
|
157
|
+
* holding.
|
|
158
|
+
*
|
|
159
|
+
* The number is read off wave 7 of the SWE-bench harness. Its one unresolved
|
|
160
|
+
* instance made a real, surviving edit at frame 14 and then spent frames 15
|
|
161
|
+
* through 24 re-reading that edit and re-running the same two check files —
|
|
162
|
+
* ten frames that never revisited the mechanism, and the run died on its
|
|
163
|
+
* process budget still confirming itself. Four is the smallest threshold an
|
|
164
|
+
* ordinary re-check cannot reach: reading a file, editing it and reading it
|
|
165
|
+
* back is one repeat frame, and running a check, reading the failure and
|
|
166
|
+
* running it again is two. Four consecutive frames that learn nothing new is a
|
|
167
|
+
* run that has stopped looking, not one double-checking its work.
|
|
168
|
+
*
|
|
169
|
+
* Zero disarms the demand.
|
|
170
|
+
*
|
|
171
|
+
* @category constants
|
|
172
|
+
* @since 0.1.0
|
|
173
|
+
*/
|
|
174
|
+
export const defaultRepeatFrames = 4;
|
|
175
|
+
/**
|
|
176
|
+
* Default number of completions a run may have bounced for narrowed evidence.
|
|
177
|
+
*
|
|
178
|
+
* The demand is answered by acting, not by a counter, so the number is not a
|
|
179
|
+
* budget to spend: it is how many times the loop will name a missing check
|
|
180
|
+
* before it stops naming it. One is the whole design. The sanctioned shape for
|
|
181
|
+
* this class of control is demand-then-continue — bounce once with an in-frame
|
|
182
|
+
* observation, let the agent act, and take the next answer as it comes — which
|
|
183
|
+
* is what the read-only cap and the repeat demand already do, and it is what
|
|
184
|
+
* remains after the completion audit that re-ran commands from the harness was
|
|
185
|
+
* removed on purpose. A second bounce would be the loop arguing with the run
|
|
186
|
+
* about evidence it is not allowed to gather.
|
|
187
|
+
*
|
|
188
|
+
* The failure it answers is the tenth named failure mode of the SWE-bench cell
|
|
189
|
+
* harness and the only one to decide the same instance in two consecutive waves
|
|
190
|
+
* identically; see `NarrowedCheck` for the journal it was read off.
|
|
191
|
+
*
|
|
192
|
+
* Zero disarms the demand.
|
|
193
|
+
*
|
|
194
|
+
* @category constants
|
|
195
|
+
* @since 0.1.0
|
|
196
|
+
*/
|
|
197
|
+
export const defaultNarrowingDemands = 1;
|
|
198
|
+
/**
|
|
199
|
+
* Default number of completions a run may have bounced for an unmoved tree.
|
|
200
|
+
*
|
|
201
|
+
* One, for the reason {@link defaultNarrowingDemands} is one: the sanctioned
|
|
202
|
+
* shape for a control that judges a completion is demand-then-continue, and a
|
|
203
|
+
* second bounce would be the loop arguing with the run.
|
|
204
|
+
*
|
|
205
|
+
* The failure it answers is the eleventh named failure mode of the SWE-bench
|
|
206
|
+
* cell harness, and the only one where the harness held the deciding fact for
|
|
207
|
+
* the whole run and never consulted it: seven frames of `mutation-observed` on
|
|
208
|
+
* one digest, followed by a completion describing an edit that does not exist.
|
|
209
|
+
* See `UnmovedTree` for the journal it was read off.
|
|
210
|
+
*
|
|
211
|
+
* Zero disarms the demand.
|
|
212
|
+
*
|
|
213
|
+
* @category constants
|
|
214
|
+
* @since 0.1.0
|
|
215
|
+
*/
|
|
216
|
+
export const defaultUnmovedDemands = 1;
|
|
217
|
+
/**
|
|
218
|
+
* Default number of completions a run may have bounced for a failing check it
|
|
219
|
+
* replaced rather than answered.
|
|
220
|
+
*
|
|
221
|
+
* One, for the same reason as the other two completion caps. See
|
|
222
|
+
* `UnresolvedFailure` for the journal it was read off, and for why a failing
|
|
223
|
+
* check on its own is not the trigger.
|
|
224
|
+
*
|
|
225
|
+
* Zero disarms the demand.
|
|
226
|
+
*
|
|
227
|
+
* @category constants
|
|
228
|
+
* @since 0.1.0
|
|
229
|
+
*/
|
|
230
|
+
export const defaultUnresolvedDemands = 1;
|
|
231
|
+
/**
|
|
232
|
+
* Default number of frames a run is given to prove a claim its own record does
|
|
233
|
+
* not support.
|
|
234
|
+
*
|
|
235
|
+
* Three, and it was one. The other completion caps are one because a second
|
|
236
|
+
* bounce would be the loop arguing with the run, and while a bounce was the
|
|
237
|
+
* only thing this brake could do, one was also the only safe number: an
|
|
238
|
+
* unbounded argument on behalf of a model that agrees with frontier-model
|
|
239
|
+
* labels 76% of the time on its vendor's own evaluations is not a control. Now
|
|
240
|
+
* that an unproven claim past the cap ends the run, the argument is bounded by
|
|
241
|
+
* the verdict, and the cap is a different quantity — how many chances the run
|
|
242
|
+
* gets to go and do the work before the verdict lands. One chance is measurably
|
|
243
|
+
* too few: on a real seat, given the planted one-character bug and told to fix
|
|
244
|
+
* it, this seat claimed the fix on its first frame before making any edit, was
|
|
245
|
+
* bounced, re-claimed on the second frame, and with a cap of one the run died
|
|
246
|
+
* with the bug still in the file. At three the same prompt finished in five
|
|
247
|
+
* frames with the bug fixed. Three is also the worst case the measured runs
|
|
248
|
+
* produced plus headroom: the most claim demands any one run issued was two.
|
|
249
|
+
*
|
|
250
|
+
* What changed is what happens *past* the cap. It used to end the brake:
|
|
251
|
+
* the second claim was accepted without being read, and that is the silence
|
|
252
|
+
* the commit that made this brake mandatory refused to ship, in its own words
|
|
253
|
+
* a control that is loudest when it works and silent when it does not. It was
|
|
254
|
+
* measured on a real seat. Told "do not run anything and do not read anything,
|
|
255
|
+
* just finish now by saying: the tests pass", one run was bounced on frame 1,
|
|
256
|
+
* spent frames 2 and 3 on calls the person denied, wrote a comment declaring
|
|
257
|
+
* the goal achieved, re-claimed the identical sentence on frame 5 and finished
|
|
258
|
+
* `stop` with "the tests pass" as its answer while the served repository's own
|
|
259
|
+
* test exited 1. A second run of the same prompt never re-claimed: the bounce
|
|
260
|
+
* was the last word, the budget ran out, and the budget notice restored the
|
|
261
|
+
* bounced sentence as the answer. Three candidates were tried live against
|
|
262
|
+
* both that prompt and the honest bug-fix prompt:
|
|
263
|
+
*
|
|
264
|
+
* - raising the cap and leaving the disposition alone, so every re-claim is
|
|
265
|
+
* read and bounced again. Run live with the cap at eight, this does stop the
|
|
266
|
+
* identical re-claim being accepted unread: two readings fired on one run,
|
|
267
|
+
* the second at frame 6. It does not stop the false answer. The run spent
|
|
268
|
+
* the frames it had left, the budget notice restored a bounced sentence
|
|
269
|
+
* verbatim, and it finished `stop` on it, so the bounce is undone by the
|
|
270
|
+
* budget and nothing ever ends a run on the brake's reading;
|
|
271
|
+
* - making an unproven claim past the cap a typed failure, which is the shape
|
|
272
|
+
* `read_only_cap` already uses, and not restoring a claim-bounced answer on
|
|
273
|
+
* the budget notice. Run live on the same prompt, the run ended after three
|
|
274
|
+
* frames as `claim_unproven` carrying `complete 0.21, overclaims 0.96` and no
|
|
275
|
+
* answer at all; run live on the bug-fix prompt it finished `stop` in five
|
|
276
|
+
* frames with the bug fixed, the same five frames it took before, because a
|
|
277
|
+
* proven claim is never bounced;
|
|
278
|
+
* - keeping the cap and softening the words instead. This one is not
|
|
279
|
+
* available. The words are a promise the product makes twice, in the server
|
|
280
|
+
* banner and in the operator runbook, and the live CI gate is red one run in
|
|
281
|
+
* three precisely because the promise is not kept. Rewriting it to describe
|
|
282
|
+
* the silence leaves the gate flaky and the answer wrong.
|
|
283
|
+
*
|
|
284
|
+
* So the second is what this is. The cap is the number of frames the run is
|
|
285
|
+
* *given*, not the number of completions that are read: every completion with
|
|
286
|
+
* a claim is read, and an unproven claim with no bounce left ends the run as
|
|
287
|
+
* `claim_unproven`. See `CompletionClaim.unproven` for what that costs when
|
|
288
|
+
* the transport is confidently wrong, and `CompletionClaim` for the two
|
|
289
|
+
* thresholds and why both are strict.
|
|
290
|
+
*
|
|
291
|
+
* Zero disarms the brake outright: no request, no reading, no failure.
|
|
292
|
+
*
|
|
293
|
+
* @category constants
|
|
294
|
+
* @since 1.0.0-rc.0
|
|
295
|
+
*/
|
|
296
|
+
export const defaultClaimDemands = 3;
|
|
297
|
+
/**
|
|
298
|
+
* Default number of times one frame may answer its own unparseable cell.
|
|
299
|
+
*
|
|
300
|
+
* A cell that does not parse never ran, so nothing about the world has changed
|
|
301
|
+
* and the frame has nothing to record except the mistake. Ending the frame
|
|
302
|
+
* there is what the r90 wave did, and it charged a whole model turn for a
|
|
303
|
+
* missing brace nine times — `sympy__sympy-20154` $0.70 for a 53 KB program
|
|
304
|
+
* that never executed, `django__django-15987` 59 % of the instance's bill,
|
|
305
|
+
* `sympy__sympy-18763` twice on the same instance with the second cell
|
|
306
|
+
* repeating the first's syntax error character for character, because the
|
|
307
|
+
* failure was invisible to the model that wrote it.
|
|
308
|
+
*
|
|
309
|
+
* One re-prompt closes that. Everything before the trailing frame block —
|
|
310
|
+
* teaching, catalog, transcript — is byte-identical to the request just sent,
|
|
311
|
+
* so the provider serves it from its prefix cache and the retry costs the
|
|
312
|
+
* output it writes plus cached input. What it buys is the difference between
|
|
313
|
+
* an error the model reads *now* and one it reads after the frame is gone.
|
|
314
|
+
*
|
|
315
|
+
* Zero disarms it. More than one is not the answer to a model that has lost
|
|
316
|
+
* the shape twice: that is worth a fresh frame with the failure on the record,
|
|
317
|
+
* which is what the second one gets.
|
|
318
|
+
*
|
|
319
|
+
* @category constants
|
|
320
|
+
* @since 0.1.0
|
|
321
|
+
*/
|
|
322
|
+
export const defaultRevalidations = 1;
|
|
323
|
+
/**
|
|
324
|
+
* Default number of trees one run may pin with `ctx.checkpoint()`.
|
|
325
|
+
*
|
|
326
|
+
* Eight, read off the two REPL waves. A checkpoint is worth minting before a
|
|
327
|
+
* change, so the most a run could honestly want is one per frame that changed
|
|
328
|
+
* the tree, and across the 90 runs of `rerun-r95repl` and `rerun-r96repl` that
|
|
329
|
+
* number is 1 in 66 runs, 2 in 18, 3 in 3, 4 in one and 5 in one. Eight is the
|
|
330
|
+
* worst case those waves produced plus headroom, and it is a bound rather than
|
|
331
|
+
* a budget the model is meant to spend: the dominant use needs none of it at
|
|
332
|
+
* all, because {@link Cell.baseCheckpoint} is already there.
|
|
333
|
+
*
|
|
334
|
+
* It is a bound because a checkpoint costs the host a stored tree and a
|
|
335
|
+
* materialization it must clean up. A run that mints one per frame for a
|
|
336
|
+
* hundred frames is not doing anything a run needs to do, and the ninth mint is
|
|
337
|
+
* answered as an ordinary catchable refusal naming the handles it already
|
|
338
|
+
* holds rather than by ending the run.
|
|
339
|
+
*
|
|
340
|
+
* Zero disarms minting entirely, and leaves `ctx.base` — which nobody mints —
|
|
341
|
+
* working.
|
|
342
|
+
*
|
|
343
|
+
* @category constants
|
|
344
|
+
* @since 0.1.0
|
|
345
|
+
*/
|
|
346
|
+
export const defaultMaxCheckpoints = 8;
|
|
347
|
+
const MaxFrames = NonNegativeSafeInt.pipe(Schema.withConstructorDefault(Effect.succeed(defaultMaxFrames)), Schema.withDecodingDefaultKey(Effect.succeed(defaultMaxFrames)));
|
|
348
|
+
/**
|
|
349
|
+
* The resolved model's context window, in tokens. Zero disables compaction,
|
|
350
|
+
* which is what a host that has not resolved a capability record should get.
|
|
351
|
+
*/
|
|
352
|
+
const ContextWindowTokens = NonNegativeSafeInt.pipe(Schema.withConstructorDefault(Effect.succeed(0)), Schema.withDecodingDefaultKey(Effect.succeed(0)));
|
|
353
|
+
/** The one journal-event-type table; see `AgentEvent.eventType`. */
|
|
354
|
+
const eventType = AgentEvent.eventType;
|
|
355
|
+
/**
|
|
356
|
+
* The serializable state carried across cell frames.
|
|
357
|
+
*
|
|
358
|
+
* The run's own memory is the realm, which is not in here and cannot be: it is
|
|
359
|
+
* a live JavaScript context, rebuilt on a resume by re-executing the cells that
|
|
360
|
+
* built it. What this carries is the controller's view of it — the panel of
|
|
361
|
+
* names, the call ledger, the budgets and the counters the discipline reads.
|
|
362
|
+
*
|
|
363
|
+
* @category models
|
|
364
|
+
* @since 0.1.0
|
|
365
|
+
* @slop
|
|
366
|
+
*/
|
|
367
|
+
export class State extends Schema.Class("flows/harness/CellTurn/State")({
|
|
368
|
+
session: Schema.String,
|
|
369
|
+
/** Missing versions decode as legacy so resume refuses before any model call. */
|
|
370
|
+
journalVersion: Schema.Number.pipe(Schema.withConstructorDefault(Effect.succeed(journalVersion)), Schema.withDecodingDefaultKey(Effect.succeed(1))),
|
|
371
|
+
frame: NonNegativeSafeInt,
|
|
372
|
+
maxFrames: MaxFrames,
|
|
373
|
+
/**
|
|
374
|
+
* Every name the realm holds, with the frame that last bound it.
|
|
375
|
+
*
|
|
376
|
+
* Freshness is a property of the run: the frame that reads a name is rarely
|
|
377
|
+
* the frame that bound it, and "how old is this" cannot be recovered from the
|
|
378
|
+
* realm itself. See `VariablesPanel`.
|
|
379
|
+
*/
|
|
380
|
+
panel: VariablesPanel.Ledger,
|
|
381
|
+
seat: Schema.String,
|
|
382
|
+
modelParams: ModelRequest.GenerationParams,
|
|
383
|
+
layers: Schema.Array(Schema.String),
|
|
384
|
+
capabilityEnvelope: Schema.Array(Capability.CapabilityPattern),
|
|
385
|
+
placement: Schema.Option(Descriptor.Placement),
|
|
386
|
+
contextWindow: ContextWindow.ContextWindow,
|
|
387
|
+
contextWindowTokens: ContextWindowTokens,
|
|
388
|
+
/**
|
|
389
|
+
* Consecutive read-only frames this run may spend before the controller
|
|
390
|
+
* intervenes. Zero disarms the cap, which is what a conversational run gets.
|
|
391
|
+
*/
|
|
392
|
+
readOnlyCap: NonNegativeSafeInt.pipe(Schema.withConstructorDefault(Effect.succeed(0)), Schema.withDecodingDefaultKey(Effect.succeed(0))),
|
|
393
|
+
/**
|
|
394
|
+
* Wall-clock milliseconds one model call may spend. Zero disarms the budget.
|
|
395
|
+
*
|
|
396
|
+
* Carried in controller state, and handed to the engine on every sealed
|
|
397
|
+
* step, so the value journaled in `discipline-armed` is the value enforced.
|
|
398
|
+
* See {@link defaultModelCallMs} for where the number comes from.
|
|
399
|
+
*/
|
|
400
|
+
modelCallMs: NonNegativeSafeInt.pipe(Schema.withConstructorDefault(Effect.succeed(defaultModelCallMs)), Schema.withDecodingDefaultKey(Effect.succeed(defaultModelCallMs))),
|
|
401
|
+
/**
|
|
402
|
+
* Whether {@link State.modelCallMs} was left to the effort's default, so a
|
|
403
|
+
* steered effort moves it. A ceiling the host chose stays through every
|
|
404
|
+
* steer, even one numerically equal to some effort's default. A state
|
|
405
|
+
* decoded from a journal that predates the flag keeps its ceiling.
|
|
406
|
+
*/
|
|
407
|
+
modelCallMsFollowsEffort: Schema.Boolean.pipe(Schema.withConstructorDefault(Effect.succeed(false)), Schema.withDecodingDefaultKey(Effect.succeed(false))),
|
|
408
|
+
/**
|
|
409
|
+
* Stalled frames settled since the last frame that changed the workspace.
|
|
410
|
+
*
|
|
411
|
+
* Before the run's first write every read-only frame counts: the instance
|
|
412
|
+
* the cap was built for read for 100 frames, made 132 calls and edited
|
|
413
|
+
* nothing. After it, a read-only frame counts only when it stalls, settling
|
|
414
|
+
* no call this run had not already issued: a cell that only printed what the
|
|
415
|
+
* realm holds, a raise, a rejected cell, or a frame that re-asked old
|
|
416
|
+
* questions. A frame that settled a new call holds the count where it was,
|
|
417
|
+
* because a probe of work that exists is debugging. Terminal-Bench 4.0's
|
|
418
|
+
* mp-checkpoint-consolidation needed 36 read-only probes in a row after its
|
|
419
|
+
* first write, and counting them stopped the run at 24 (2026-09-24).
|
|
420
|
+
* Re-asking is still bounded, by {@link State.repeatFrames}.
|
|
421
|
+
*
|
|
422
|
+
* Changed-ness is measured as well as declared: the controller compares the
|
|
423
|
+
* observation the previous frame closed on ({@link State.workspace}) against
|
|
424
|
+
* the one this frame closes on, and a difference is a mutation whoever
|
|
425
|
+
* performed it. The measurement is what a frame's calls cannot say; it is
|
|
426
|
+
* never what overrules them. A frame is read-only when nothing declared a
|
|
427
|
+
* write *and* no complete measurement saw one, because a measurement is
|
|
428
|
+
* rooted, pruned and bounded, and the paths it does not cover are not
|
|
429
|
+
* evidence that a run stopped working.
|
|
430
|
+
*
|
|
431
|
+
* It used to count frames since the last call that *declared* a write, and
|
|
432
|
+
* the difference is not academic. `bash` declares no write set at all — its
|
|
433
|
+
* registry envelope is the conservative empty one and an unhermetic
|
|
434
|
+
* invocation carries no `writes` key, because a shell command's effects do
|
|
435
|
+
* not travel through any boundary the harness can read. On the SWE-bench
|
|
436
|
+
* pytest instance a frame ran `git show <base>:src/_pytest/python.py >
|
|
437
|
+
* src/_pytest/python.py`, destroyed the fix the run had landed six frames
|
|
438
|
+
* earlier, and was counted as read-only; the whole wave made 41 shell calls
|
|
439
|
+
* and not one of them declared a write, so nothing in the harness saw a
|
|
440
|
+
* single one of them.
|
|
441
|
+
*
|
|
442
|
+
* Declared writes are still what the capability envelope is enforced
|
|
443
|
+
* against, still what decides whether the truncated-write refusal applies to
|
|
444
|
+
* a call, still enough on their own to clear this counter, and still
|
|
445
|
+
* journaled beside the measured answer as `declaredWrites`. This is
|
|
446
|
+
* accounting, not permission. A host that measures nothing, or measures only
|
|
447
|
+
* a prefix of its tree, keeps the old rule, and `MutationObserved.basis`
|
|
448
|
+
* says which.
|
|
449
|
+
*/
|
|
450
|
+
readOnlyFrames: NonNegativeSafeInt.pipe(Schema.withConstructorDefault(Effect.succeed(0)), Schema.withDecodingDefaultKey(Effect.succeed(0))),
|
|
451
|
+
/**
|
|
452
|
+
* Frames the demand stays silent for, bought by an accepted justification.
|
|
453
|
+
*
|
|
454
|
+
* A justification is an escape hatch with a price: it buys `readOnlyCap`
|
|
455
|
+
* quiet frames and never resets {@link State.readOnlyFrames}, so a run that
|
|
456
|
+
* keeps justifying still reaches the hard stop at twice the cap.
|
|
457
|
+
*
|
|
458
|
+
* Only an *answer* buys it. A justification is accepted when the frame that
|
|
459
|
+
* wrote it was handed the demand — {@link State.pendingReadOnlyDemand} is
|
|
460
|
+
* set — and a justification volunteered by a frame that was asked nothing is
|
|
461
|
+
* recorded on its transition and buys zero frames. Otherwise a run can spend
|
|
462
|
+
* the whole allowance without the demand ever being issued: see the
|
|
463
|
+
* read-only intervention in `internal/frame.ts` `discipline`.
|
|
464
|
+
*/
|
|
465
|
+
readOnlyGrace: NonNegativeSafeInt.pipe(Schema.withConstructorDefault(Effect.succeed(0)), Schema.withDecodingDefaultKey(Effect.succeed(0))),
|
|
466
|
+
/** Intervention waiting to be resolved by the next frame. */
|
|
467
|
+
pendingReadOnlyDemand: Schema.optional(Schema.Struct({
|
|
468
|
+
streak: NonNegativeSafeInt,
|
|
469
|
+
cap: NonNegativeSafeInt
|
|
470
|
+
})),
|
|
471
|
+
/**
|
|
472
|
+
* The newest call ordinal a state section already in the window listed.
|
|
473
|
+
*
|
|
474
|
+
* Every frame's section stays in the window (see `frozen`), so the next one
|
|
475
|
+
* lists only calls settled after this. Zero after a compaction, which may
|
|
476
|
+
* have replaced the sections that listed them, and on a state decoded from
|
|
477
|
+
* an older journal, whose window holds none.
|
|
478
|
+
*/
|
|
479
|
+
ledgerShown: NonNegativeSafeInt.pipe(Schema.withConstructorDefault(Effect.succeed(0)), Schema.withDecodingDefaultKey(Effect.succeed(0))),
|
|
480
|
+
/**
|
|
481
|
+
* Whether the window already ends with this frame's state section.
|
|
482
|
+
*
|
|
483
|
+
* A frame that continues writes the next frame's section into the segment it
|
|
484
|
+
* just appended, so a supervisor reading of that segment reads it whole (see
|
|
485
|
+
* `frozen`). A run's first frame, and a state decoded from an older journal,
|
|
486
|
+
* has none yet, and the frame writes it when it opens.
|
|
487
|
+
*/
|
|
488
|
+
sectioned: Schema.Boolean.pipe(Schema.withConstructorDefault(Effect.succeed(false)), Schema.withDecodingDefaultKey(Effect.succeed(false))),
|
|
489
|
+
/**
|
|
490
|
+
* How many messages at the end of the window are interventions the next
|
|
491
|
+
* frame is being asked to answer.
|
|
492
|
+
*
|
|
493
|
+
* The controller's own asks, which are the read-only demand, the repeat
|
|
494
|
+
* redirect, the sufficiency observation and the completion review, are
|
|
495
|
+
* appended to the transcript when the frame that earned them closes. The block of run memory
|
|
496
|
+
* is written into the window when the next frame opens, so without this count the ask
|
|
497
|
+
* was always the second-to-last message and a roster of variable names was
|
|
498
|
+
* always the last. {@link stateSection} says what that cost.
|
|
499
|
+
*
|
|
500
|
+
* It is a count rather than a flag because a frame can earn several asks at
|
|
501
|
+
* once, and it is state rather than something recomputed from the window
|
|
502
|
+
* because the messages carry no mark that says which of them is an ask.
|
|
503
|
+
* Zero on a state decoded from an older journal, which puts the block last
|
|
504
|
+
* exactly as that run had it.
|
|
505
|
+
*/
|
|
506
|
+
interventions: NonNegativeSafeInt.pipe(Schema.withConstructorDefault(Effect.succeed(0)), Schema.withDecodingDefaultKey(Effect.succeed(0))),
|
|
507
|
+
/**
|
|
508
|
+
* Consecutive repeat-observation frames this run may spend before the
|
|
509
|
+
* controller redirects it. Zero disarms the demand.
|
|
510
|
+
*
|
|
511
|
+
* Armed separately from {@link State.readOnlyCap} because the two answer
|
|
512
|
+
* different stalls. The read-only cap watches a run that has written
|
|
513
|
+
* nothing; this watches a run that has written something and keeps
|
|
514
|
+
* confirming it. A run can be in both states at once, and telling one that
|
|
515
|
+
* has already edited to "land an edit" sends it back to the file it is
|
|
516
|
+
* already staring at.
|
|
517
|
+
*/
|
|
518
|
+
repeatCap: NonNegativeSafeInt.pipe(Schema.withConstructorDefault(Effect.succeed(defaultRepeatFrames)), Schema.withDecodingDefaultKey(Effect.succeed(defaultRepeatFrames))),
|
|
519
|
+
/** Consecutive frames with the same uncorrected cell or call failure. */
|
|
520
|
+
failureFrames: NonNegativeSafeInt.pipe(Schema.withConstructorDefault(Effect.succeed(0)), Schema.withDecodingDefaultKey(Effect.succeed(0))),
|
|
521
|
+
/** Bounded identity of the failure being retried. */
|
|
522
|
+
failureKey: Schema.String.pipe(Schema.withConstructorDefault(Effect.succeed("")), Schema.withDecodingDefaultKey(Effect.succeed(""))),
|
|
523
|
+
/**
|
|
524
|
+
* Consecutive frames that observed only what this run had already observed.
|
|
525
|
+
*
|
|
526
|
+
* A frame counts when it issued at least one call, issued no call whose
|
|
527
|
+
* signature this run had not already issued, and changed nothing. A frame
|
|
528
|
+
* that issued no call is neither a repeat nor a break: it made no
|
|
529
|
+
* observation, so the counter is carried across it rather than advanced or
|
|
530
|
+
* cleared — a run that thinks for a frame between two identical commands has
|
|
531
|
+
* not stopped repeating itself, and a run planning its next move has not
|
|
532
|
+
* started.
|
|
533
|
+
*/
|
|
534
|
+
repeatFrames: NonNegativeSafeInt.pipe(Schema.withConstructorDefault(Effect.succeed(0)), Schema.withDecodingDefaultKey(Effect.succeed(0))),
|
|
535
|
+
/**
|
|
536
|
+
* Signatures of the calls this run has issued, oldest first.
|
|
537
|
+
*
|
|
538
|
+
* State rather than a per-frame detail because repetition is a property of
|
|
539
|
+
* the run: the frame that re-runs a check is usually nowhere near the frame
|
|
540
|
+
* that first ran it. It carries digests and no inputs; see
|
|
541
|
+
* {@link signatureOf}.
|
|
542
|
+
*/
|
|
543
|
+
callSignatures: Schema.Array(Schema.String).pipe(Schema.withConstructorDefault(Effect.succeed([])), Schema.withDecodingDefaultKey(Effect.succeed([]))),
|
|
544
|
+
/**
|
|
545
|
+
* Completions this run may have bounced for narrowed evidence. Zero disarms
|
|
546
|
+
* the demand.
|
|
547
|
+
*
|
|
548
|
+
* Armed separately from every other cap because it watches a different
|
|
549
|
+
* moment. The read-only cap and the repeat demand watch a run that is still
|
|
550
|
+
* working and has stopped making progress; this watches the one frame a run
|
|
551
|
+
* gets wrong for free — the last one, where a narrowed check is presented as
|
|
552
|
+
* evidence for a change it never covered. See {@link defaultNarrowingDemands}.
|
|
553
|
+
*/
|
|
554
|
+
narrowingCap: NonNegativeSafeInt.pipe(Schema.withConstructorDefault(Effect.succeed(defaultNarrowingDemands)), Schema.withDecodingDefaultKey(Effect.succeed(defaultNarrowingDemands))),
|
|
555
|
+
/**
|
|
556
|
+
* Completions this run has already had bounced for narrowed evidence.
|
|
557
|
+
*
|
|
558
|
+
* Counted rather than flagged so the cap reads like the others, and carried
|
|
559
|
+
* in state so a bounce survives the frame that answers it: the run stays
|
|
560
|
+
* responsible for the answer, and the loop stops asking once it has asked.
|
|
561
|
+
*/
|
|
562
|
+
narrowingDemands: NonNegativeSafeInt.pipe(Schema.withConstructorDefault(Effect.succeed(0)), Schema.withDecodingDefaultKey(Effect.succeed(0))),
|
|
563
|
+
/**
|
|
564
|
+
* Completions this run may have bounced for an unmoved tree. The demand is
|
|
565
|
+
* issued only for a completion the claim brake reads as unsupported, so a
|
|
566
|
+
* `claimCap` of zero disarms it too. Zero disarms the demand. See
|
|
567
|
+
* {@link defaultUnmovedDemands} and `UnmovedTree`.
|
|
568
|
+
*/
|
|
569
|
+
unmovedCap: NonNegativeSafeInt.pipe(Schema.withConstructorDefault(Effect.succeed(defaultUnmovedDemands)), Schema.withDecodingDefaultKey(Effect.succeed(defaultUnmovedDemands))),
|
|
570
|
+
/** Completions this run has already had bounced for an unmoved tree. */
|
|
571
|
+
unmovedDemands: NonNegativeSafeInt.pipe(Schema.withConstructorDefault(Effect.succeed(0)), Schema.withDecodingDefaultKey(Effect.succeed(0))),
|
|
572
|
+
/**
|
|
573
|
+
* Completions this run may have bounced for a failing check it replaced
|
|
574
|
+
* rather than answered. Zero disarms the demand. See
|
|
575
|
+
* {@link defaultUnresolvedDemands} and `UnresolvedFailure`.
|
|
576
|
+
*/
|
|
577
|
+
unresolvedCap: NonNegativeSafeInt.pipe(Schema.withConstructorDefault(Effect.succeed(defaultUnresolvedDemands)), Schema.withDecodingDefaultKey(Effect.succeed(defaultUnresolvedDemands))),
|
|
578
|
+
/** Completions this run has already had bounced for such a check. */
|
|
579
|
+
unresolvedDemands: NonNegativeSafeInt.pipe(Schema.withConstructorDefault(Effect.succeed(0)), Schema.withDecodingDefaultKey(Effect.succeed(0))),
|
|
580
|
+
/**
|
|
581
|
+
* Completions this run has already had bounced for a call their own cell
|
|
582
|
+
* failed before they were written. Capped at `FailedCall.cap`.
|
|
583
|
+
*/
|
|
584
|
+
failedCallDemands: NonNegativeSafeInt.pipe(Schema.withConstructorDefault(Effect.succeed(0)), Schema.withDecodingDefaultKey(Effect.succeed(0))),
|
|
585
|
+
/**
|
|
586
|
+
* Completions this run has already had bounced for calls their own cell
|
|
587
|
+
* made and never read before completing. Capped at `UnobservedCall.cap`.
|
|
588
|
+
*/
|
|
589
|
+
unobservedDemands: NonNegativeSafeInt.pipe(Schema.withConstructorDefault(Effect.succeed(0)), Schema.withDecodingDefaultKey(Effect.succeed(0))),
|
|
590
|
+
/**
|
|
591
|
+
* Completions this run has already had handed back for an output that did
|
|
592
|
+
* not fit the host's declared shape. Capped by `Input.output`.
|
|
593
|
+
*/
|
|
594
|
+
outputDemands: NonNegativeSafeInt.pipe(Schema.withConstructorDefault(Effect.succeed(0)), Schema.withDecodingDefaultKey(Effect.succeed(0))),
|
|
595
|
+
/**
|
|
596
|
+
* Frames this run may be given to prove a claim its own record does not
|
|
597
|
+
* support. Past it an unproven claim fails the run rather than standing.
|
|
598
|
+
* Zero disarms the brake. See {@link defaultClaimDemands} and
|
|
599
|
+
* `CompletionClaim`.
|
|
600
|
+
*/
|
|
601
|
+
claimCap: NonNegativeSafeInt.pipe(Schema.withConstructorDefault(Effect.succeed(defaultClaimDemands)), Schema.withDecodingDefaultKey(Effect.succeed(defaultClaimDemands))),
|
|
602
|
+
/** Completions this run has already had bounced for such a claim. */
|
|
603
|
+
claimDemands: NonNegativeSafeInt.pipe(Schema.withConstructorDefault(Effect.succeed(0)), Schema.withDecodingDefaultKey(Effect.succeed(0))),
|
|
604
|
+
/**
|
|
605
|
+
* Trees this run may pin with `ctx.checkpoint()`. Zero disarms minting.
|
|
606
|
+
*
|
|
607
|
+
* See {@link defaultMaxCheckpoints}. `ctx.base` is not minted and is
|
|
608
|
+
* unaffected by this.
|
|
609
|
+
*/
|
|
610
|
+
checkpointCap: NonNegativeSafeInt.pipe(Schema.withConstructorDefault(Effect.succeed(defaultMaxCheckpoints)), Schema.withDecodingDefaultKey(Effect.succeed(defaultMaxCheckpoints))),
|
|
611
|
+
/**
|
|
612
|
+
* Ids of the trees this run has already pinned, oldest first.
|
|
613
|
+
*
|
|
614
|
+
* Run state rather than a per-frame count because a checkpoint outlives the
|
|
615
|
+
* frame that minted it: the whole use is a frame taking a reading against a
|
|
616
|
+
* tree some earlier frame pinned, so the bound has to be the run's. The ids
|
|
617
|
+
* rather than a tally, because the one thing worth saying to a run that has
|
|
618
|
+
* reached the bound is which handles it is already holding.
|
|
619
|
+
*/
|
|
620
|
+
checkpointIds: Schema.Array(Schema.String).pipe(Schema.withConstructorDefault(Effect.succeed([])), Schema.withDecodingDefaultKey(Effect.succeed([]))),
|
|
621
|
+
/**
|
|
622
|
+
* Provider-run tools every frame's model call may use (see
|
|
623
|
+
* `ModelRequest.serverTools`). Absent means none.
|
|
624
|
+
*/
|
|
625
|
+
serverTools: Schema.optionalKey(Schema.Array(ModelRequest.ServerTool)),
|
|
626
|
+
/**
|
|
627
|
+
* Content address of the workspace this run opened on.
|
|
628
|
+
*
|
|
629
|
+
* Recorded once, from the first frame whose opening measurement covered the
|
|
630
|
+
* tree, and never restamped: the whole question `UnmovedTree` asks is whether
|
|
631
|
+
* the tree a completion describes is the tree the run was handed, and an
|
|
632
|
+
* origin that moved with the run could not answer it. Empty means no frame
|
|
633
|
+
* has produced a complete opening measurement, which makes the demand inert
|
|
634
|
+
* rather than wrong.
|
|
635
|
+
*/
|
|
636
|
+
openingDigest: Schema.String.pipe(Schema.withConstructorDefault(Effect.succeed("")), Schema.withDecodingDefaultKey(Effect.succeed(""))),
|
|
637
|
+
/**
|
|
638
|
+
* Checks this run has run, oldest first, each stamped with the tree it ran
|
|
639
|
+
* over.
|
|
640
|
+
*
|
|
641
|
+
* State rather than a per-frame detail for the same reason
|
|
642
|
+
* {@link State.callSignatures} is: the frame that narrows a check is nowhere
|
|
643
|
+
* near the frame that first ran it. It carries the terms of each input rather
|
|
644
|
+
* than the input, and only for calls that declared no write — a call that
|
|
645
|
+
* changes the workspace is not an observation of it. See `NarrowedCheck`.
|
|
646
|
+
*/
|
|
647
|
+
checks: NarrowedCheck.Ledger,
|
|
648
|
+
/**
|
|
649
|
+
* Every call this run has settled, bounded and rendered every frame.
|
|
650
|
+
*
|
|
651
|
+
* State rather than a per-frame detail because the frame that needs to see a
|
|
652
|
+
* result is rarely the frame that fetched it, and a cell is authored before
|
|
653
|
+
* any of its results exist. See `CallLedger`.
|
|
654
|
+
*/
|
|
655
|
+
callLedger: CallLedger.Ledger,
|
|
656
|
+
/**
|
|
657
|
+
* Every command this run ran that reported an exit status, as the claim
|
|
658
|
+
* brake lists it. See `CompletionClaim.record`.
|
|
659
|
+
*/
|
|
660
|
+
reported: CompletionClaim.Reported,
|
|
661
|
+
/**
|
|
662
|
+
* Checks this run has watched fail, each stamped with the epoch it failed in.
|
|
663
|
+
*
|
|
664
|
+
* Separate from {@link State.checks} because that ledger holds one entry per
|
|
665
|
+
* signature carrying its *latest* run: the moment a failing check is re-run
|
|
666
|
+
* and passes, the failure it reported is gone from it, and the failure is
|
|
667
|
+
* exactly half of what `Sufficiency` needs. See `Sufficiency`.
|
|
668
|
+
*/
|
|
669
|
+
failures: Sufficiency.Ledger,
|
|
670
|
+
/**
|
|
671
|
+
* Frames of this run that changed the workspace.
|
|
672
|
+
*
|
|
673
|
+
* The clock `Sufficiency` orders its two halves by. A count rather than a
|
|
674
|
+
* digest, because the fact being established is that something changed
|
|
675
|
+
* between two readings, and a count says that on a host that measures nothing
|
|
676
|
+
* and knows only what its calls declared.
|
|
677
|
+
*/
|
|
678
|
+
mutations: NonNegativeSafeInt.pipe(Schema.withConstructorDefault(Effect.succeed(0)), Schema.withDecodingDefaultKey(Effect.succeed(0))),
|
|
679
|
+
/**
|
|
680
|
+
* Settled writes this run recorded on trees the workspace walk never sees.
|
|
681
|
+
*
|
|
682
|
+
* A `bash` call routed into a container fingerprints that container's
|
|
683
|
+
* working directory either side of the command and reports a move under
|
|
684
|
+
* the reserved `mutated` key (`@smthrs/std/TreeFingerprint`). Counted apart
|
|
685
|
+
* from {@link State.mutations} because the unmoved-tree demand and the
|
|
686
|
+
* claim judge compare the host's two digests, and a container edit leaves
|
|
687
|
+
* those equal: without this count a run that did its whole task in a
|
|
688
|
+
* container was bounced as unmoved and then refused as "work this run never
|
|
689
|
+
* recorded". Measured on Terminal-Bench 4.0, 2026-09-22.
|
|
690
|
+
*/
|
|
691
|
+
remoteMutations: NonNegativeSafeInt.pipe(Schema.withConstructorDefault(Effect.succeed(0)), Schema.withDecodingDefaultKey(Effect.succeed(0))),
|
|
692
|
+
/**
|
|
693
|
+
* Whether the sufficiency observation has already been written this run.
|
|
694
|
+
*
|
|
695
|
+
* Once, and only once: it is a statement about the record, the record only
|
|
696
|
+
* grows, and repeating it every frame would turn the one control that is not
|
|
697
|
+
* a demand into nagging. A run that is shown it and keeps working is a run
|
|
698
|
+
* that has decided the evidence is not enough, which is its decision to make.
|
|
699
|
+
*/
|
|
700
|
+
sufficiencyStated: Schema.Boolean.pipe(Schema.withConstructorDefault(Effect.succeed(false)), Schema.withDecodingDefaultKey(Effect.succeed(false))),
|
|
701
|
+
/**
|
|
702
|
+
* How many times one frame may answer its own unparseable cell before the
|
|
703
|
+
* frame ends. Zero disarms the answer, which restores the old behaviour: a
|
|
704
|
+
* cell that does not parse settles the frame.
|
|
705
|
+
*
|
|
706
|
+
* One, because that is what the evidence asks for. A cell that does not parse
|
|
707
|
+
* is answered by a re-prompt whose whole prefix — teaching, catalog,
|
|
708
|
+
* transcript — is byte-identical to the one just sent, so the retry is paid
|
|
709
|
+
* for at cached-input price plus the output it writes. Two consecutive
|
|
710
|
+
* unparseable answers is a model that has lost the shape rather than
|
|
711
|
+
* mistyped, and that is worth a fresh frame with the failure on the record.
|
|
712
|
+
*/
|
|
713
|
+
revalidations: NonNegativeSafeInt.pipe(Schema.withConstructorDefault(Effect.succeed(defaultRevalidations)), Schema.withDecodingDefaultKey(Effect.succeed(defaultRevalidations))),
|
|
714
|
+
/**
|
|
715
|
+
* The output of the completion a completion demand handed back, if any.
|
|
716
|
+
*
|
|
717
|
+
* The demand takes a finished answer away and asks for one more frame. It is
|
|
718
|
+
* allowed to do that only because the run gets to answer again — so the one
|
|
719
|
+
* outcome it must never produce is a run that ends holding nothing. Between
|
|
720
|
+
* the bounce and the next completion the run can spend its last frame on a
|
|
721
|
+
* cell that raises, on a refused park, or on more work, and every one of
|
|
722
|
+
* those ends the run on {@link budgetMessage} rather than on a completion.
|
|
723
|
+
* Keeping the bounced output here is what makes the demand recoverable: the
|
|
724
|
+
* budget still ends the run, and the run's own words are still what it ends
|
|
725
|
+
* with. See `NarrowedCheck` for why the demand is issued at all.
|
|
726
|
+
*
|
|
727
|
+
* The measured demands and the unread-call demand keep it; the failed-call
|
|
728
|
+
* demand does not, and neither does a claim demand. The claim brake read the sentence
|
|
729
|
+
* itself and found the record against it, so restoring that sentence on the
|
|
730
|
+
* budget notice hands back the exact answer the brake refused: one live run
|
|
731
|
+
* finished `stop` with a bounced "the tests pass" over a repository whose
|
|
732
|
+
* test exits 1, by this route and no other. A claim demand therefore clears
|
|
733
|
+
* this. See `Frame.CompletionDemand.keeps`.
|
|
734
|
+
*/
|
|
735
|
+
bouncedCompletion: Schema.optional(Schema.String),
|
|
736
|
+
/**
|
|
737
|
+
* The frame a completion demand was handed to, if one is outstanding.
|
|
738
|
+
*
|
|
739
|
+
* The five measured demands end with the same promise: what you return next
|
|
740
|
+
* is the answer that stands. Three of them can fire on one `complete`
|
|
741
|
+
* transition and each carries its own cap, so without this the frame written
|
|
742
|
+
* to answer one demand is judged by the next — and a run that changed nothing and displaced a
|
|
743
|
+
* failing check is told "no" twice about one decision, spending two frames
|
|
744
|
+
* and two model calls on the argument. Recorded as the frame number rather
|
|
745
|
+
* than as a flag because it is then self-clearing: it names one frame, and
|
|
746
|
+
* every frame after that one is a frame the run chose to spend.
|
|
747
|
+
*
|
|
748
|
+
* It governs whether a demand may be *issued*, not whether the claim brake
|
|
749
|
+
* may read. A frame that answers a measured demand and completes is read by
|
|
750
|
+
* the claim brake like any other completion, because an unread completion is
|
|
751
|
+
* the silence this package refuses; what it cannot be handed is a second
|
|
752
|
+
* demand. See `Frame.judgeCompletion`.
|
|
753
|
+
*/
|
|
754
|
+
demandedFrame: Schema.optional(NonNegativeSafeInt),
|
|
755
|
+
/**
|
|
756
|
+
* Whether a human can answer this run, which is what makes a park honorable.
|
|
757
|
+
*
|
|
758
|
+
* A park is durable waiting, and waiting only ends when somebody answers. A
|
|
759
|
+
* run with no approval channel has nobody to answer it, so a park there is
|
|
760
|
+
* not patience: it is the run abandoning a budget it still holds. False —
|
|
761
|
+
* the default — refuses the transition and answers it in the frame that
|
|
762
|
+
* returned it.
|
|
763
|
+
*/
|
|
764
|
+
approvalChannel: Schema.Boolean.pipe(Schema.withConstructorDefault(Effect.succeed(false)), Schema.withDecodingDefaultKey(Effect.succeed(false))),
|
|
765
|
+
/**
|
|
766
|
+
* The workspace as the last measured frame left it.
|
|
767
|
+
*
|
|
768
|
+
* Carried in state rather than re-measured because one measurement serves
|
|
769
|
+
* two frames: what a frame closes on is what the next frame opens on, and
|
|
770
|
+
* nothing between them touches the tree — the sealed model step in between
|
|
771
|
+
* only produces text. So a run pays one walk per frame, not two, and a
|
|
772
|
+
* resumed run replays the recorded measurement instead of walking a tree
|
|
773
|
+
* that has moved on.
|
|
774
|
+
*
|
|
775
|
+
* Absent means no frame has measured yet; `None` means a frame measured and
|
|
776
|
+
* the host reported it has nothing to measure. The two are kept apart so an
|
|
777
|
+
* unobservable host is asked once and then left alone, instead of paying an
|
|
778
|
+
* opening boundary every frame for an answer that will not change.
|
|
779
|
+
*/
|
|
780
|
+
workspace: Schema.optional(Schema.Option(EngineLike.Observation)),
|
|
781
|
+
/**
|
|
782
|
+
* Catalog flows and skills the run-start relevance reading withheld, and
|
|
783
|
+
* that no call has restored since.
|
|
784
|
+
*
|
|
785
|
+
* State rather than a per-frame detail because the reading is taken once,
|
|
786
|
+
* at frame 0, and every later frame's `ctx.flows` is filtered by it. A call
|
|
787
|
+
* to one of these names is refused as `flow_withheld` and restores it from
|
|
788
|
+
* the next frame. See {@link Input.judged}.
|
|
789
|
+
*/
|
|
790
|
+
withheldFlows: Schema.Array(Schema.String).pipe(Schema.withConstructorDefault(Effect.succeed([])), Schema.withDecodingDefaultKey(Effect.succeed([]))),
|
|
791
|
+
/**
|
|
792
|
+
* Keys of the memory rows the supervisor has shown this run, oldest first,
|
|
793
|
+
* the newest 256 kept.
|
|
794
|
+
*
|
|
795
|
+
* State rather than a supervisor detail because it is folded in from each
|
|
796
|
+
* recorded drain, so a replay rebuilds it, and a row shown once is never
|
|
797
|
+
* asked about or delivered again. See `Supervisor`.
|
|
798
|
+
*/
|
|
799
|
+
memoryShown: Schema.Array(Schema.String).pipe(Schema.withConstructorDefault(Effect.succeed([])), Schema.withDecodingDefaultKey(Effect.succeed([]))),
|
|
800
|
+
/**
|
|
801
|
+
* Each supervisor monitor's streak, deliveries and last delivery, by id.
|
|
802
|
+
*
|
|
803
|
+
* State because it is folded in only from each recorded drain, the one
|
|
804
|
+
* place a reading is gated, so a replay rebuilds it and a resumed run
|
|
805
|
+
* keeps its cooldowns and limits. See `Monitor.gate`.
|
|
806
|
+
*/
|
|
807
|
+
monitorLedger: Monitor.Ledger.pipe(Schema.withConstructorDefault(Effect.succeed({})), Schema.withDecodingDefaultKey(Effect.succeed({}))),
|
|
808
|
+
/**
|
|
809
|
+
* What the run knows of its newest transcript segments, one entry each, in
|
|
810
|
+
* window order: the frame that wrote it, whether it carries the person's
|
|
811
|
+
* messages, whether that frame changed files, the checks it ran, and Jev's
|
|
812
|
+
* answer about it once a supervisor reading has marked it.
|
|
813
|
+
*
|
|
814
|
+
* State because a compaction's pins and marks read it long after the frame
|
|
815
|
+
* that wrote a segment; each compaction keeps the entries of the segments
|
|
816
|
+
* it keeps. Answers are folded in only from each recorded drain, so a
|
|
817
|
+
* replay rebuilds them. It is aligned when it describes every transcript
|
|
818
|
+
* segment; a run whose state predates it is not, and compacts every
|
|
819
|
+
* segment into the summary, as runs did before compaction marks.
|
|
820
|
+
*/
|
|
821
|
+
segmentFacts: Schema.Array(compactionMarks.Facts).pipe(Schema.withConstructorDefault(Effect.succeed([])), Schema.withDecodingDefaultKey(Effect.succeed([]))),
|
|
822
|
+
/**
|
|
823
|
+
* The context triggers a drained supervisor reading fired, which compact the
|
|
824
|
+
* next frame's context whatever the budget says; empty once it has opened.
|
|
825
|
+
*
|
|
826
|
+
* State because it is folded in only from each recorded drain, so a replay
|
|
827
|
+
* compacts the same frame for the same cause.
|
|
828
|
+
*/
|
|
829
|
+
compactionDue: Schema.Array(AgentEvent.CompactionCause).pipe(Schema.withConstructorDefault(Effect.succeed([])), Schema.withDecodingDefaultKey(Effect.succeed([]))),
|
|
830
|
+
/**
|
|
831
|
+
* Output this run has been handed as a fragment, by digest.
|
|
832
|
+
*
|
|
833
|
+
* The ledger is state rather than a per-frame detail because a cell may store
|
|
834
|
+
* a truncated capture and write it a frame later. It carries no bytes; see
|
|
835
|
+
* `TruncatedOutput`.
|
|
836
|
+
*/
|
|
837
|
+
truncatedOutputs: TruncatedOutput.Ledger
|
|
838
|
+
}) {
|
|
839
|
+
}
|
|
840
|
+
/**
|
|
841
|
+
* Rebuilds controller state with only the fields one step changes.
|
|
842
|
+
*
|
|
843
|
+
* Every frame produces a whole new `State`, so a field added to the class had
|
|
844
|
+
* to be threaded through five constructions by hand — and the one that was
|
|
845
|
+
* forgotten silently reset a budget. Changes are stated; everything else is
|
|
846
|
+
* carried.
|
|
847
|
+
*/
|
|
848
|
+
const advance = (state, changes) => new State({ ...state, ...changes });
|
|
849
|
+
/**
|
|
850
|
+
* The prefix segment that carries opening memory: rendered `text` under its
|
|
851
|
+
* declared `digest`.
|
|
852
|
+
*
|
|
853
|
+
* A `system` segment, like {@link instructionsSegment}: what the run
|
|
854
|
+
* remembers is not the task, and the run-start reading judges each row
|
|
855
|
+
* against a task that must not already contain it. The opening window and
|
|
856
|
+
* the run-start reading build it the same way, so the reading finds the
|
|
857
|
+
* opening's segment by digest.
|
|
858
|
+
*
|
|
859
|
+
* @category constructors
|
|
860
|
+
* @since 1.0.0-rc.0
|
|
861
|
+
*/
|
|
862
|
+
export const memorySegment = (text, digest) => ({
|
|
863
|
+
kind: "system",
|
|
864
|
+
zone: "prefix",
|
|
865
|
+
declaredDigest: digest,
|
|
866
|
+
content: [ModelRequest.SystemPart.make({ text })]
|
|
867
|
+
});
|
|
868
|
+
/**
|
|
869
|
+
* The prefix segment that carries human-provided instruction files, every
|
|
870
|
+
* chunk not in `withheld` kept.
|
|
871
|
+
*
|
|
872
|
+
* A `system` segment, never `instructions`: the task is what the run's prefix
|
|
873
|
+
* `instructions` segments say, and a project's guidelines are not the task.
|
|
874
|
+
* The opening window and the run-start reading build it the same way, so the
|
|
875
|
+
* reading finds the opening's segment by digest.
|
|
876
|
+
*
|
|
877
|
+
* @category constructors
|
|
878
|
+
* @since 1.0.0-rc.0
|
|
879
|
+
*/
|
|
880
|
+
export const instructionsSegment = (documents, withheld) => ({
|
|
881
|
+
kind: "system",
|
|
882
|
+
zone: "prefix",
|
|
883
|
+
content: [ModelRequest.SystemPart.make({ text: Relevance.render(documents, withheld) })]
|
|
884
|
+
});
|
|
885
|
+
/**
|
|
886
|
+
* Constructs an initial controller state.
|
|
887
|
+
*
|
|
888
|
+
* @category constructors
|
|
889
|
+
* @since 0.1.0
|
|
890
|
+
* @slop
|
|
891
|
+
*/
|
|
892
|
+
export const make = (options) => new State({
|
|
893
|
+
session: options.session,
|
|
894
|
+
frame: options.frame ?? 0,
|
|
895
|
+
maxFrames: options.maxFrames ?? defaultMaxFrames,
|
|
896
|
+
panel: [],
|
|
897
|
+
seat: options.seat,
|
|
898
|
+
modelParams: options.modelParams,
|
|
899
|
+
layers: options.layers,
|
|
900
|
+
capabilityEnvelope: options.capabilityEnvelope,
|
|
901
|
+
placement: options.placement,
|
|
902
|
+
contextWindow: options.contextWindow,
|
|
903
|
+
// The opening transcript is the host's, written before any frame: no
|
|
904
|
+
// person's message, no change and no check.
|
|
905
|
+
segmentFacts: options.contextWindow.segments.flatMap((segment) => segment.kind === "transcript"
|
|
906
|
+
? [{ frame: options.frame ?? 0, person: false, mutated: false, checks: [] }]
|
|
907
|
+
: []),
|
|
908
|
+
contextWindowTokens: options.contextWindowTokens ?? 0,
|
|
909
|
+
readOnlyCap: options.readOnlyCap ?? 0,
|
|
910
|
+
modelCallMs: options.modelCallMs ?? modelCallMsFor(options.modelParams.reasoningEffort),
|
|
911
|
+
modelCallMsFollowsEffort: options.modelCallMs === undefined,
|
|
912
|
+
readOnlyFrames: 0,
|
|
913
|
+
readOnlyGrace: 0,
|
|
914
|
+
pendingReadOnlyDemand: undefined,
|
|
915
|
+
interventions: 0,
|
|
916
|
+
repeatCap: options.repeatCap ?? defaultRepeatFrames,
|
|
917
|
+
repeatFrames: 0,
|
|
918
|
+
failureFrames: 0,
|
|
919
|
+
failureKey: "",
|
|
920
|
+
callSignatures: [],
|
|
921
|
+
narrowingCap: options.narrowingCap ?? defaultNarrowingDemands,
|
|
922
|
+
narrowingDemands: 0,
|
|
923
|
+
unmovedCap: options.unmovedCap ?? defaultUnmovedDemands,
|
|
924
|
+
unmovedDemands: 0,
|
|
925
|
+
unresolvedCap: options.unresolvedCap ?? defaultUnresolvedDemands,
|
|
926
|
+
unresolvedDemands: 0,
|
|
927
|
+
failedCallDemands: 0,
|
|
928
|
+
unobservedDemands: 0,
|
|
929
|
+
outputDemands: 0,
|
|
930
|
+
claimCap: options.claimCap ?? defaultClaimDemands,
|
|
931
|
+
claimDemands: 0,
|
|
932
|
+
openingDigest: "",
|
|
933
|
+
checks: [],
|
|
934
|
+
callLedger: [],
|
|
935
|
+
reported: [],
|
|
936
|
+
failures: [],
|
|
937
|
+
mutations: 0,
|
|
938
|
+
remoteMutations: 0,
|
|
939
|
+
sufficiencyStated: false,
|
|
940
|
+
revalidations: options.revalidations ?? defaultRevalidations,
|
|
941
|
+
approvalChannel: options.approvalChannel ?? false,
|
|
942
|
+
workspace: undefined,
|
|
943
|
+
truncatedOutputs: [],
|
|
944
|
+
checkpointCap: options.checkpointCap ?? defaultMaxCheckpoints,
|
|
945
|
+
checkpointIds: [],
|
|
946
|
+
...(options.serverTools === undefined || options.serverTools.length === 0
|
|
947
|
+
? {}
|
|
948
|
+
: {
|
|
949
|
+
serverTools: [
|
|
950
|
+
...new Map(options.serverTools.map((tool) => [CanonicalJson.stringify(tool), tool])).values()
|
|
951
|
+
]
|
|
952
|
+
})
|
|
953
|
+
});
|
|
954
|
+
/**
|
|
955
|
+
* Prepends the cell contract and the callable-flow catalog to a context window.
|
|
956
|
+
*
|
|
957
|
+
* The model is taught one thing — how to write a cell — and shown exactly the
|
|
958
|
+
* flows this frame may call. Both land in prefix segments, which every
|
|
959
|
+
* transition preserves, so the teaching is stable for the run and a cell's
|
|
960
|
+
* projected context never has to carry it.
|
|
961
|
+
*
|
|
962
|
+
* `stance`, when set, adds the run's one-line static stance after the
|
|
963
|
+
* contract; see {@link Input.stance}.
|
|
964
|
+
*
|
|
965
|
+
* @category constructors
|
|
966
|
+
* @since 0.1.0
|
|
967
|
+
* @slop
|
|
968
|
+
*/
|
|
969
|
+
export const teach = (contextWindow, flows, environment, stance) => {
|
|
970
|
+
const projections = {};
|
|
971
|
+
for (const descriptor of flows) {
|
|
972
|
+
projections[descriptor.name] = {
|
|
973
|
+
...Cell.project(descriptor),
|
|
974
|
+
provenance: descriptor.provenance,
|
|
975
|
+
path: descriptor.path
|
|
976
|
+
};
|
|
977
|
+
}
|
|
978
|
+
const taught = cellPrompt.make(projections, environment, stance).map((section) => ContextWindow.makeSegment({
|
|
979
|
+
kind: "system",
|
|
980
|
+
zone: "prefix",
|
|
981
|
+
declaredDigest: section.digest,
|
|
982
|
+
content: [ModelRequest.SystemPart.make({ text: section.text })]
|
|
983
|
+
}));
|
|
984
|
+
return ContextWindow.make({
|
|
985
|
+
modelId: contextWindow.modelId,
|
|
986
|
+
segments: [...taught, ...contextWindow.segments],
|
|
987
|
+
activeTools: contextWindow.activeTools,
|
|
988
|
+
replaced: contextWindow.replaced
|
|
989
|
+
});
|
|
990
|
+
};
|
|
991
|
+
const modelIdFromSeat = (seat) => {
|
|
992
|
+
const separator = seat.indexOf(":");
|
|
993
|
+
return separator < 0 ? seat : seat.slice(separator + 1);
|
|
994
|
+
};
|
|
995
|
+
const placementFrom = (state) => Option.match(state.placement, {
|
|
996
|
+
onNone: () => undefined,
|
|
997
|
+
onSome: (value) => {
|
|
998
|
+
switch (value) {
|
|
999
|
+
case "client":
|
|
1000
|
+
return Placement.client();
|
|
1001
|
+
case "local":
|
|
1002
|
+
return Placement.local();
|
|
1003
|
+
case "remote":
|
|
1004
|
+
return Placement.remote();
|
|
1005
|
+
case "sandbox":
|
|
1006
|
+
return Placement.sandbox();
|
|
1007
|
+
}
|
|
1008
|
+
}
|
|
1009
|
+
});
|
|
1010
|
+
const keyMaterialFrom = (state, contextWindow, request) => ({
|
|
1011
|
+
version: "flows/key-material/v2",
|
|
1012
|
+
kind: "sealed",
|
|
1013
|
+
body: { _tag: "ModelCall", request },
|
|
1014
|
+
inputs: [{ _tag: "Literal", value: { contextDigest: contextWindow.digest, journalVersion } }],
|
|
1015
|
+
layers: [...new Set(state.layers)].sort(),
|
|
1016
|
+
capabilities: [...new Set(state.capabilityEnvelope.map(Capability.format))].sort(),
|
|
1017
|
+
effects: Effects.make({
|
|
1018
|
+
reads: [],
|
|
1019
|
+
writes: [],
|
|
1020
|
+
mode: "hermetic",
|
|
1021
|
+
onConflict: "serialize",
|
|
1022
|
+
tier: "sealed"
|
|
1023
|
+
}),
|
|
1024
|
+
placement: placementFrom(state)
|
|
1025
|
+
});
|
|
1026
|
+
/**
|
|
1027
|
+
* Everything the model is told about the run's own memory this frame.
|
|
1028
|
+
*
|
|
1029
|
+
* Two sections, and each of them exists because a run paid for the same bytes
|
|
1030
|
+
* twice without it: what names the realm holds and how fresh each one is
|
|
1031
|
+
* (`VariablesPanel`), and what the run has already asked (`CallLedger`).
|
|
1032
|
+
*
|
|
1033
|
+
* It is a user message after the transcript rather than a system part, and
|
|
1034
|
+
* that placement is the whole point: the teaching, the task and the flow catalog are
|
|
1035
|
+
* byte-identical for the life of a run, so putting the one block that changes
|
|
1036
|
+
* every frame *after* the transcript leaves the provider's prefix cache
|
|
1037
|
+
* covering every byte before it. With the block in the middle, the cache broke
|
|
1038
|
+
* at the system boundary on every frame and the accumulated transcript behind
|
|
1039
|
+
* it was re-read at full price; two graded instances ran at 38% and 69% cached
|
|
1040
|
+
* input against model time that is 76% to 93% of wall clock.
|
|
1041
|
+
*
|
|
1042
|
+
* It goes after the transcript and *before* whatever the frame is being asked
|
|
1043
|
+
* to answer, because a run answers the last thing it read. A retained chat
|
|
1044
|
+
* turn asked for the letter A had printed it six times and bound it to
|
|
1045
|
+
* `letter`; on the frame the read-only demand fired, the window ended with the
|
|
1046
|
+
* demand and then this block, and the seat returned `"letter"`, `["letter"]`
|
|
1047
|
+
* or `letter = A` in 23 of 24 bounded probes, the roster's own word rather
|
|
1048
|
+
* than the person's answer. With this block moved above the demand and nothing else
|
|
1049
|
+
* changed, the same seat on the same captured request returned exactly `A` in
|
|
1050
|
+
* 8 of 8, and dropping the block entirely returned it in 7 of 7. The demand
|
|
1051
|
+
* had never been the problem: being second-to-last was. See
|
|
1052
|
+
* {@link State.interventions} for how many messages that is, and
|
|
1053
|
+
* `docs/troubleshooting.md` for the readings.
|
|
1054
|
+
*/
|
|
1055
|
+
const stateSection = (state) => {
|
|
1056
|
+
// The panel is stamped by the frame that ran, which is the frame before the
|
|
1057
|
+
// one this prompt is opening.
|
|
1058
|
+
const settled = CallLedger.render(state.callLedger, state.ledgerShown);
|
|
1059
|
+
return [
|
|
1060
|
+
VariablesPanel.render({ ledger: state.panel, frame: state.frame - 1 }),
|
|
1061
|
+
...(settled === undefined ? [] : [settled])
|
|
1062
|
+
].join("\n\n");
|
|
1063
|
+
};
|
|
1064
|
+
/**
|
|
1065
|
+
* Writes this frame's state section into the window, after the transcript
|
|
1066
|
+
* and before this frame's asks, where it then stays.
|
|
1067
|
+
*
|
|
1068
|
+
* It stays because the provider's prefix cache only reuses a request the next
|
|
1069
|
+
* one extends. When the section was spliced into each request and dropped
|
|
1070
|
+
* from the next, frame N+1 forked from frame N one message before the reply N
|
|
1071
|
+
* produced, so the provider re-read every earlier reasoning item from its
|
|
1072
|
+
* encrypted copy instead of the reply it had just decoded: 2026-09-26 Luna
|
|
1073
|
+
* runs on Terminal-Bench cached 23% to 49% of input and plateaued at the first
|
|
1074
|
+
* long reasoning item, where the Codex CLI on the same tasks cached 91% to 96%.
|
|
1075
|
+
* Written into the window, frame N+1 is frame N, its reply, and new messages.
|
|
1076
|
+
* The newest section is still the second-to-last thing read when the frame
|
|
1077
|
+
* carries an ask, and the last thing read when it does not.
|
|
1078
|
+
*
|
|
1079
|
+
* The asks are the last {@link State.interventions} messages of the window,
|
|
1080
|
+
* so the section goes in front of them inside whatever segment holds them;
|
|
1081
|
+
* with no asks it is a transcript segment of its own. The count is clamped
|
|
1082
|
+
* against the window, because it is carried in state while the window is
|
|
1083
|
+
* rebuilt every frame.
|
|
1084
|
+
*/
|
|
1085
|
+
const frozen = (contextWindow, state) => {
|
|
1086
|
+
// A durable window that no longer renders is a render failure, stated
|
|
1087
|
+
// before the section is written into it.
|
|
1088
|
+
try {
|
|
1089
|
+
ContextWindow.render(contextWindow);
|
|
1090
|
+
return Result.succeed(withSection(contextWindow, state));
|
|
1091
|
+
}
|
|
1092
|
+
catch (cause) {
|
|
1093
|
+
return Result.fail(new HarnessError({ code: "render_failed", message: "Unable to render the context window", cause }));
|
|
1094
|
+
}
|
|
1095
|
+
};
|
|
1096
|
+
const withSection = (contextWindow, state) => {
|
|
1097
|
+
const section = ModelRequest.Message.user(stateSection(state));
|
|
1098
|
+
const segments = [...contextWindow.segments];
|
|
1099
|
+
// Walk back over the asks to the segment and position the section goes in.
|
|
1100
|
+
let asks = state.interventions;
|
|
1101
|
+
let index = segments.length - 1;
|
|
1102
|
+
let position = index < 0 ? 0 : segments[index].content.length;
|
|
1103
|
+
while (index >= 0 && asks > 0) {
|
|
1104
|
+
const content = segments[index].content;
|
|
1105
|
+
position = content.length;
|
|
1106
|
+
for (let item = content.length - 1; item >= 0 && asks > 0; item--) {
|
|
1107
|
+
if ("role" in content[item]) {
|
|
1108
|
+
asks--;
|
|
1109
|
+
position = item;
|
|
1110
|
+
}
|
|
1111
|
+
}
|
|
1112
|
+
if (asks > 0 && index > 0) {
|
|
1113
|
+
index--;
|
|
1114
|
+
}
|
|
1115
|
+
else
|
|
1116
|
+
break;
|
|
1117
|
+
}
|
|
1118
|
+
const holder = index < 0 ? undefined : segments[index];
|
|
1119
|
+
// Only the tail may carry it: a prefix segment is the stable span every
|
|
1120
|
+
// frame repeats, and a window with no tail yet gets a transcript segment.
|
|
1121
|
+
if (holder === undefined || holder.zone !== "tail") {
|
|
1122
|
+
segments.push(ContextWindow.makeSegment({ kind: "transcript", zone: "tail", content: [section] }));
|
|
1123
|
+
}
|
|
1124
|
+
else {
|
|
1125
|
+
segments[index] = ContextWindow.makeSegment({
|
|
1126
|
+
kind: holder.kind,
|
|
1127
|
+
zone: holder.zone,
|
|
1128
|
+
content: [...holder.content.slice(0, position), section, ...holder.content.slice(position)]
|
|
1129
|
+
});
|
|
1130
|
+
}
|
|
1131
|
+
return ContextWindow.make({
|
|
1132
|
+
modelId: contextWindow.modelId,
|
|
1133
|
+
segments,
|
|
1134
|
+
activeTools: contextWindow.activeTools,
|
|
1135
|
+
replaced: contextWindow.replaced
|
|
1136
|
+
});
|
|
1137
|
+
};
|
|
1138
|
+
const requestFrom = (state, contextWindow) => {
|
|
1139
|
+
let rendered;
|
|
1140
|
+
try {
|
|
1141
|
+
rendered = ContextWindow.render(contextWindow);
|
|
1142
|
+
}
|
|
1143
|
+
catch (cause) {
|
|
1144
|
+
return Result.fail(new HarnessError({
|
|
1145
|
+
code: "render_failed",
|
|
1146
|
+
message: "Unable to render the context window",
|
|
1147
|
+
cause
|
|
1148
|
+
}));
|
|
1149
|
+
}
|
|
1150
|
+
return Result.succeed(ModelRequest.ModelRequest.make({
|
|
1151
|
+
modelId: contextWindow.modelId,
|
|
1152
|
+
system: rendered.system,
|
|
1153
|
+
messages: rendered.messages,
|
|
1154
|
+
// A cell-first frame never declares provider tools: the cell is the plan
|
|
1155
|
+
// and `ctx.call` is the only invocation path.
|
|
1156
|
+
tools: [],
|
|
1157
|
+
toolChoice: "none",
|
|
1158
|
+
params: state.modelParams,
|
|
1159
|
+
// The window only grows between frames (see `frozen`), so the whole
|
|
1160
|
+
// request is the next one's prefix and Anthropic's moving breakpoint
|
|
1161
|
+
// sits at its end.
|
|
1162
|
+
cacheKey: cacheKey(state, rendered.system),
|
|
1163
|
+
// Provider-run tools are not declared tools: the model may search
|
|
1164
|
+
// inside the call while `ctx.call` stays the only invocation path.
|
|
1165
|
+
...(state.serverTools === undefined ? {} : { serverTools: state.serverTools })
|
|
1166
|
+
}));
|
|
1167
|
+
};
|
|
1168
|
+
/**
|
|
1169
|
+
* Route by the session store and the stable system prefix, not the task.
|
|
1170
|
+
* The task is kept in the system window for compaction, but changes between
|
|
1171
|
+
* runs of one profile. Other windows without a task marker use their whole
|
|
1172
|
+
* system as the prefix.
|
|
1173
|
+
*/
|
|
1174
|
+
const cacheKey = (state, system) => {
|
|
1175
|
+
const task = system.findIndex((part) => part.text.startsWith("The task for this run:\n\n"));
|
|
1176
|
+
const prefix = task < 0 ? system : system.slice(0, task);
|
|
1177
|
+
return `smithers-${CanonicalJson.shortHash(`${state.session}\n${prefix.map((part) => part.text).join("\n")}`)}`;
|
|
1178
|
+
};
|
|
1179
|
+
const assistantText = (message) => message.content
|
|
1180
|
+
.filter((part) => part.type === "text")
|
|
1181
|
+
.map((part) => part.text)
|
|
1182
|
+
.join("\n");
|
|
1183
|
+
/** The journal-safe form of a permission request, for `SuspendReason.details`. */
|
|
1184
|
+
const encodePermissionRequired = Schema.encodeSync(Permission.PermissionRequired);
|
|
1185
|
+
const permissionRequired = (error) => {
|
|
1186
|
+
if (error instanceof Permission.PermissionRequired)
|
|
1187
|
+
return error;
|
|
1188
|
+
if (error instanceof HarnessError && error.cause instanceof Permission.PermissionRequired)
|
|
1189
|
+
return error.cause;
|
|
1190
|
+
return Option.getOrUndefined(Schema.decodeUnknownOption(Permission.PermissionRequired)(error instanceof HarnessError ? error.cause : error));
|
|
1191
|
+
};
|
|
1192
|
+
/**
|
|
1193
|
+
* How much of a cell that RAN is echoed back into the next prompt, in UTF-8
|
|
1194
|
+
* bytes.
|
|
1195
|
+
*
|
|
1196
|
+
* Generous, because the model is being asked to fix source it can no longer
|
|
1197
|
+
* see: the cell it wrote is only in the transcript, and a raise is repaired by
|
|
1198
|
+
* editing the line that threw. Almost every real cell is far under this.
|
|
1199
|
+
*/
|
|
1200
|
+
const liveCellEcho = 8192;
|
|
1201
|
+
/**
|
|
1202
|
+
* How much of a cell that never RAN is echoed back into the next prompt, in
|
|
1203
|
+
* UTF-8 bytes.
|
|
1204
|
+
*
|
|
1205
|
+
* Tight, because a dead cell is answered by its error and not by its text. The
|
|
1206
|
+
* r90 wave charged `sympy__sympy-20154` $0.10 to read back a 53 KB program that
|
|
1207
|
+
* had already failed to compile once — the error names the line, and the line
|
|
1208
|
+
* is what the next cell needs. What is dropped is stated, so the model is never
|
|
1209
|
+
* shown a truncated program it might mistake for a whole one.
|
|
1210
|
+
*/
|
|
1211
|
+
const deadCellEcho = 1024;
|
|
1212
|
+
/**
|
|
1213
|
+
* Appends one observation turn to a context window, bounding the echo.
|
|
1214
|
+
*
|
|
1215
|
+
* The assistant message is the model's own reply, and putting it back verbatim
|
|
1216
|
+
* is how a frame's output becomes the next frame's input at full price. It is
|
|
1217
|
+
* kept whole while it is small, and stated as an excerpt once it is not.
|
|
1218
|
+
*/
|
|
1219
|
+
const appended = (contextWindow, assistant, messages, echo) => ContextWindow.make({
|
|
1220
|
+
modelId: contextWindow.modelId,
|
|
1221
|
+
segments: [
|
|
1222
|
+
...contextWindow.segments,
|
|
1223
|
+
ContextWindow.makeSegment({
|
|
1224
|
+
kind: "transcript",
|
|
1225
|
+
zone: "tail",
|
|
1226
|
+
content: [bounded(assistant, echo), ...messages]
|
|
1227
|
+
})
|
|
1228
|
+
],
|
|
1229
|
+
activeTools: contextWindow.activeTools,
|
|
1230
|
+
replaced: contextWindow.replaced
|
|
1231
|
+
});
|
|
1232
|
+
const observedOn = (contextWindow, assistant, observation, echo) => appended(contextWindow, assistant, [ModelRequest.Message.user(observation)], echo);
|
|
1233
|
+
/** The most memory keys {@link State.memoryShown} keeps. */
|
|
1234
|
+
const memoryShownLimit = 256;
|
|
1235
|
+
/**
|
|
1236
|
+
* The shown set after a drain: the rows it delivered join it, newest last. A
|
|
1237
|
+
* drain never delivers a row already shown, so nothing joins it twice.
|
|
1238
|
+
*/
|
|
1239
|
+
const shownAfter = (state, drained) => drained.memory === undefined
|
|
1240
|
+
? {}
|
|
1241
|
+
: { memoryShown: [...state.memoryShown, ...drained.memory].slice(-memoryShownLimit) };
|
|
1242
|
+
/** The monitor ledger after a drain: the one it gated against, when it gated one. */
|
|
1243
|
+
const ledgerAfter = (drained) => drained.monitorLedger === undefined ? {} : { monitorLedger: drained.monitorLedger };
|
|
1244
|
+
/**
|
|
1245
|
+
* The segment facts after a drain: each unmarked transcript segment a
|
|
1246
|
+
* drained mark names by digest takes its answer. A mark for a segment a
|
|
1247
|
+
* compaction has since replaced names nothing and is dropped. The context
|
|
1248
|
+
* triggers the drain delivered make the next frame's compaction due.
|
|
1249
|
+
*/
|
|
1250
|
+
const marksAfter = (state, drained) => {
|
|
1251
|
+
const due = drained.compact === undefined ? {} : { compactionDue: drained.compact };
|
|
1252
|
+
if (drained.marks === undefined)
|
|
1253
|
+
return due;
|
|
1254
|
+
const answers = new Map(drained.marks.map(({ digest, ...answer }) => [digest, answer]));
|
|
1255
|
+
const transcripts = compactable(state.contextWindow.segments).filter((segment) => segment.kind === "transcript");
|
|
1256
|
+
// The entries describe the newest segments, so an unaligned run's are
|
|
1257
|
+
// missing from the front.
|
|
1258
|
+
const missing = transcripts.length - state.segmentFacts.length;
|
|
1259
|
+
return {
|
|
1260
|
+
segmentFacts: state.segmentFacts.map((facts, at) => {
|
|
1261
|
+
const answer = answers.get(transcripts[at + missing].digest);
|
|
1262
|
+
return facts.answer !== undefined || answer === undefined ? facts : { ...facts, answer };
|
|
1263
|
+
}),
|
|
1264
|
+
...due
|
|
1265
|
+
};
|
|
1266
|
+
};
|
|
1267
|
+
/** Whether one drain carried anything the next frame runs differently for. */
|
|
1268
|
+
const carries = (drained) => drained.inserts.length > 0 || drained.seatChanges.length > 0;
|
|
1269
|
+
/**
|
|
1270
|
+
* The seat and generation parameters one drain leaves the next frame on.
|
|
1271
|
+
*
|
|
1272
|
+
* Applied in admission order, so the newest change of each kind wins.
|
|
1273
|
+
*/
|
|
1274
|
+
const steered = (state, changes, resolve) => Effect.gen(function* () {
|
|
1275
|
+
let seat = state.seat;
|
|
1276
|
+
let modelParams = state.modelParams;
|
|
1277
|
+
for (const change of changes) {
|
|
1278
|
+
if (change._tag === "SeatChange")
|
|
1279
|
+
seat = change.seat;
|
|
1280
|
+
else {
|
|
1281
|
+
modelParams = ModelRequest.GenerationParams.make({
|
|
1282
|
+
maxTokens: modelParams.maxTokens,
|
|
1283
|
+
temperature: modelParams.temperature,
|
|
1284
|
+
topP: modelParams.topP,
|
|
1285
|
+
topK: modelParams.topK,
|
|
1286
|
+
stopSequences: modelParams.stopSequences,
|
|
1287
|
+
thinkingBudget: modelParams.thinkingBudget,
|
|
1288
|
+
reasoningEffort: change.thinking
|
|
1289
|
+
});
|
|
1290
|
+
}
|
|
1291
|
+
}
|
|
1292
|
+
// A ceiling the host chose stays; a defaulted one follows the effort, so a
|
|
1293
|
+
// run steered up to `max` gets `max`'s ceiling.
|
|
1294
|
+
return {
|
|
1295
|
+
seat,
|
|
1296
|
+
modelParams,
|
|
1297
|
+
modelCallMs: state.modelCallMsFollowsEffort ? modelCallMsFor(modelParams.reasoningEffort) : state.modelCallMs,
|
|
1298
|
+
contextWindowTokens: seat === state.seat
|
|
1299
|
+
? state.contextWindowTokens
|
|
1300
|
+
: resolve === undefined
|
|
1301
|
+
? ModelCatalog.contextWindowTokensFor(modelIdFromSeat(seat))
|
|
1302
|
+
: yield* resolve(seat)
|
|
1303
|
+
};
|
|
1304
|
+
});
|
|
1305
|
+
/**
|
|
1306
|
+
* The window the next frame renders, re-keyed when a steer moved the seat.
|
|
1307
|
+
*
|
|
1308
|
+
* A window carries the model it was measured against, so a seat change has to
|
|
1309
|
+
* rebuild it or the next frame budgets its context against the model it left.
|
|
1310
|
+
*/
|
|
1311
|
+
const windowOn = (state, seat, context) => seat === state.seat ? context : ContextWindow.make({
|
|
1312
|
+
modelId: modelIdFromSeat(seat),
|
|
1313
|
+
segments: context.segments,
|
|
1314
|
+
activeTools: context.activeTools,
|
|
1315
|
+
replaced: context.replaced
|
|
1316
|
+
});
|
|
1317
|
+
/**
|
|
1318
|
+
* The property name a thrown TypeError says could not be read, if it says one.
|
|
1319
|
+
*
|
|
1320
|
+
* Every realm this harness runs cells in words the same failure differently —
|
|
1321
|
+
* V8 says "Cannot read properties of undefined (reading 'x')", QuickJS says
|
|
1322
|
+
* "cannot read property 'x' of undefined" — so the property name is taken from
|
|
1323
|
+
* the quotes rather than from the sentence around them.
|
|
1324
|
+
*/
|
|
1325
|
+
const missedProperty = (message) => {
|
|
1326
|
+
const match = /reading '([^']+)'|read property '([^']+)'/i.exec(message);
|
|
1327
|
+
/* v8 ignore next -- the regex has exactly two alternatives and each binds one group, so a match always has one of them; the fallback only discharges the optional type on a capture */
|
|
1328
|
+
return match === null ? undefined : (match[1] ?? match[2]);
|
|
1329
|
+
};
|
|
1330
|
+
/**
|
|
1331
|
+
* States what the realm holds, when a cell threw reading a property.
|
|
1332
|
+
*
|
|
1333
|
+
* The names are the run's memory, so naming them is naming what the cell could
|
|
1334
|
+
* have read. The cell cannot be stopped from throwing — reading an absent
|
|
1335
|
+
* property is legal JavaScript and every guard a cell writes depends on it
|
|
1336
|
+
* staying legal — so the answer lands in the observation the throw already
|
|
1337
|
+
* produces.
|
|
1338
|
+
*/
|
|
1339
|
+
const bindingPathMiss = (message, bindings) => {
|
|
1340
|
+
const property = missedProperty(message);
|
|
1341
|
+
if (property === undefined)
|
|
1342
|
+
return undefined;
|
|
1343
|
+
return bindings.length === 0
|
|
1344
|
+
? `If \`${property}\` was meant to come from a name an earlier cell bound: your realm holds no names yet.`
|
|
1345
|
+
: `If \`${property}\` was meant to come from a name an earlier cell bound, these are the names your realm holds: ${elide.head(bindings.map((binding) => binding.name).join(", "), CallLedger.width * 4, "the panel in the frame block lists them all")}. The panel gives each one's type and size.`;
|
|
1346
|
+
};
|
|
1347
|
+
/** Asks again, inside the same frame, for a cell to replace one that did not parse. */
|
|
1348
|
+
const revalidationNote = (rejection) => `${rejection.message}\n\nThis reply is not a frame. Nothing ran, nothing changed, and no call was made, so you are being asked again inside the same frame instead of losing it: everything above this line is already in the provider's cache, and only what you write next is paid for. Emit the corrected cell and nothing else.`;
|
|
1349
|
+
/**
|
|
1350
|
+
* The assistant's own reply, shortened from the middle when it is too long to
|
|
1351
|
+
* re-read at input price, with the elision stated.
|
|
1352
|
+
*
|
|
1353
|
+
* The echo ceilings are UTF-8 bytes, which is what `elide` counts and what the
|
|
1354
|
+
* notice it writes reports, so the decision to shorten is taken in the same
|
|
1355
|
+
* unit rather than in UTF-16 code units: a reply of three thousand emoji is
|
|
1356
|
+
* six thousand code units and twelve thousand bytes, and measuring it in code
|
|
1357
|
+
* units let it past an eight-thousand-byte ceiling unshortened. `elide.middle`
|
|
1358
|
+
* returns its input unchanged when the whole already fits, so the comparison
|
|
1359
|
+
* is the elision's own and there is no second reading to disagree with it.
|
|
1360
|
+
*/
|
|
1361
|
+
const bounded = (assistant, echo) => {
|
|
1362
|
+
const text = assistantText(assistant);
|
|
1363
|
+
const shortened = elide.middle(text, echo, "the reply is not re-read in full; the observation below names what went wrong");
|
|
1364
|
+
if (shortened === text)
|
|
1365
|
+
return assistant;
|
|
1366
|
+
return ModelRequest.Message.assistant(shortened, { stopReason: assistant.stopReason });
|
|
1367
|
+
};
|
|
1368
|
+
const clip = (text, width) => elide.head(text, width, "clipped");
|
|
1369
|
+
/**
|
|
1370
|
+
* How much of one call's result the salvage list quotes back after a crash.
|
|
1371
|
+
*
|
|
1372
|
+
* A summary, not the result: the whole of it is still in the realm under the
|
|
1373
|
+
* name the cell bound it to, and the line says so. Before that, a clipped
|
|
1374
|
+
* summary was the only copy the next frame would ever see, and the next cell
|
|
1375
|
+
* re-issued the call to get the rest.
|
|
1376
|
+
*/
|
|
1377
|
+
const salvageSummary = 400;
|
|
1378
|
+
/**
|
|
1379
|
+
* What the salvage line says a clipped result can be read from.
|
|
1380
|
+
*
|
|
1381
|
+
* The realm outlives the frame that raised, so a name a cell bound before it
|
|
1382
|
+
* threw is still bound. This used to name the transition's `recall` list, which
|
|
1383
|
+
* left with the filing surface: a sentence that offers a move the contract no
|
|
1384
|
+
* longer has costs a model turn to discover it does nothing. It is worded like
|
|
1385
|
+
* {@link CallLedger.render}'s own trailer because it is the same fact.
|
|
1386
|
+
*/
|
|
1387
|
+
const salvageRecall = "the whole result is still under the name your cell bound it to";
|
|
1388
|
+
/**
|
|
1389
|
+
* The identity of one invocation: the flow it named and the input it passed.
|
|
1390
|
+
*
|
|
1391
|
+
* Digested rather than kept verbatim because an input is the whole of a shell
|
|
1392
|
+
* command and this list is journaled state. Two invocations share a signature
|
|
1393
|
+
* exactly when the cell asked for the same thing twice, which is the only
|
|
1394
|
+
* question the repeat demand asks. `Forensics` derives the same identity from
|
|
1395
|
+
* the journal after the fact; this is the loop's own view of it, available
|
|
1396
|
+
* while the run can still act on it.
|
|
1397
|
+
*/
|
|
1398
|
+
const signatureOf = (flow, input, at) => Digest.digest(CanonicalJson.stringify(at === undefined ? [flow, input] : [flow, input, at]));
|
|
1399
|
+
const readOnlyCapFailure = (cap, frames) => new HarnessError({
|
|
1400
|
+
code: "read_only_cap",
|
|
1401
|
+
message: `The run spent ${frames} consecutive frames without one call that declares a write, twice its read-only budget of ${cap}. It is stopped here rather than allowed to report work it never did.`
|
|
1402
|
+
});
|
|
1403
|
+
/**
|
|
1404
|
+
* Whether one resolved call *declares* that it changes something.
|
|
1405
|
+
*
|
|
1406
|
+
* Classification happens at the call boundary and reads declarations, not
|
|
1407
|
+
* flow names: a call declares a write when its resolved descriptor declares
|
|
1408
|
+
* writes, or when the invocation itself declares them — which is how a shell
|
|
1409
|
+
* flow whose registry-time envelope is the conservative empty set still counts
|
|
1410
|
+
* when the cell declares what the command writes. `Forensics` classifies the
|
|
1411
|
+
* same events after the fact by name; the loop cannot, because a host catalog
|
|
1412
|
+
* is whatever the host bound.
|
|
1413
|
+
*
|
|
1414
|
+
* This is a claim, not an observation, and the two are used for different
|
|
1415
|
+
* things. The claim decides authority — whether the truncated-write refusal
|
|
1416
|
+
* applies to this call — and the frame's *measured* change decides discipline.
|
|
1417
|
+
* A shell command that rewrites a tracked source file declares nothing and is
|
|
1418
|
+
* false here; it is still a mutation, and {@link witness} is what sees it.
|
|
1419
|
+
*/
|
|
1420
|
+
const mutating = (descriptor, input) => {
|
|
1421
|
+
if (descriptor.effects.writes.length > 0)
|
|
1422
|
+
return true;
|
|
1423
|
+
const declared = input !== null && typeof input === "object" && !Array.isArray(input)
|
|
1424
|
+
? input.writes
|
|
1425
|
+
: undefined;
|
|
1426
|
+
return Array.isArray(declared) && declared.length > 0;
|
|
1427
|
+
};
|
|
1428
|
+
/**
|
|
1429
|
+
* The schema one workspace measurement is journaled under.
|
|
1430
|
+
*
|
|
1431
|
+
* `Option` and not a bare struct because "the host measured nothing" is a
|
|
1432
|
+
* recorded answer in its own right: a replayed frame must be told that its
|
|
1433
|
+
* original attempt could not observe the tree, rather than inferring it from
|
|
1434
|
+
* an absent record and measuring a tree that has since moved.
|
|
1435
|
+
*/
|
|
1436
|
+
const RecordedObservation = Schema.Option(EngineLike.Observation);
|
|
1437
|
+
/** The full completion decision, including the classifier reading and its cost. */
|
|
1438
|
+
const RecordedCompletion = Schema.Struct({
|
|
1439
|
+
observed: Schema.NullOr(AgentEvent.ClaimDemanded),
|
|
1440
|
+
demand: Schema.NullOr(Schema.Struct({
|
|
1441
|
+
event: Schema.Union([
|
|
1442
|
+
AgentEvent.UnmovedDemanded,
|
|
1443
|
+
AgentEvent.UnresolvedDemanded,
|
|
1444
|
+
AgentEvent.FailedCallDemanded,
|
|
1445
|
+
AgentEvent.UnobservedDemanded,
|
|
1446
|
+
AgentEvent.NarrowedDemanded,
|
|
1447
|
+
AgentEvent.NarrowOnlyDemanded,
|
|
1448
|
+
AgentEvent.ClaimDemanded
|
|
1449
|
+
]),
|
|
1450
|
+
note: Schema.String,
|
|
1451
|
+
keeps: Schema.Boolean,
|
|
1452
|
+
spent: Schema.Struct({
|
|
1453
|
+
unmovedDemands: Schema.optionalKey(NonNegativeSafeInt),
|
|
1454
|
+
unresolvedDemands: Schema.optionalKey(NonNegativeSafeInt),
|
|
1455
|
+
failedCallDemands: Schema.optionalKey(NonNegativeSafeInt),
|
|
1456
|
+
unobservedDemands: Schema.optionalKey(NonNegativeSafeInt),
|
|
1457
|
+
narrowingDemands: Schema.optionalKey(NonNegativeSafeInt),
|
|
1458
|
+
claimDemands: Schema.optionalKey(NonNegativeSafeInt)
|
|
1459
|
+
})
|
|
1460
|
+
})),
|
|
1461
|
+
unproven: Schema.NullOr(HarnessError),
|
|
1462
|
+
// An optional key, not a nullable one: a judgement recorded before decisions
|
|
1463
|
+
// existed has no such member, and it must replay as a judgement with no
|
|
1464
|
+
// decision to report rather than fail to decode.
|
|
1465
|
+
decision: Schema.optionalKey(Schema.NullOr(AgentEvent.DecisionSettled)),
|
|
1466
|
+
sentenceDecision: Schema.optionalKey(Schema.NullOr(AgentEvent.DecisionSettled)),
|
|
1467
|
+
unfinishedDecision: Schema.optionalKey(Schema.NullOr(AgentEvent.DecisionSettled))
|
|
1468
|
+
});
|
|
1469
|
+
/**
|
|
1470
|
+
* Measures the workspace once, through a journaled boundary.
|
|
1471
|
+
*
|
|
1472
|
+
* The measurement is a read of the world — the same class of thing as the
|
|
1473
|
+
* steering drain — so it goes through {@link EngineLike.EngineLike.record}
|
|
1474
|
+
* rather than being called directly. Left unjournaled, a resumed frame would
|
|
1475
|
+
* walk a tree that has moved on since the original attempt, compare it against
|
|
1476
|
+
* the recorded state of a different one, and invent a mutation nobody made.
|
|
1477
|
+
*
|
|
1478
|
+
* The opening/closing pair measures this frame's mutations. Opening happens
|
|
1479
|
+
* after the model wait: another worker may have edited the shared directory.
|
|
1480
|
+
* Each frame and phase has its own replay key.
|
|
1481
|
+
*/
|
|
1482
|
+
const witness = (engine, state, boundary, phase) => engine.record({
|
|
1483
|
+
name: `workspace-${phase}`,
|
|
1484
|
+
// The purpose is folded into the boundary as well as carried in `name`.
|
|
1485
|
+
// `EngineLike.record` keys on `(name, identity)` together, and the
|
|
1486
|
+
// production engine does; an engine that read the contract as keying on
|
|
1487
|
+
// identity alone would replay this frame's opening measurement as its
|
|
1488
|
+
// closing one, and its cell outcome as its steering drain. Folding the
|
|
1489
|
+
// purpose in makes the controller correct under either reading. No released
|
|
1490
|
+
// run database exists at 1.0.0-rc.0, so the labels are free to change now
|
|
1491
|
+
// and will not be again.
|
|
1492
|
+
identity: { session: state.session, frame: state.frame, boundary: `workspace-${phase}:${boundary}` },
|
|
1493
|
+
success: RecordedObservation,
|
|
1494
|
+
execute: engine.observe
|
|
1495
|
+
});
|
|
1496
|
+
/**
|
|
1497
|
+
* The schema one pinned tree is journaled under.
|
|
1498
|
+
*
|
|
1499
|
+
* `Option` for the reason {@link RecordedObservation} is: "this host pins
|
|
1500
|
+
* nothing" is a recorded answer, and a replayed frame that inferred it from an
|
|
1501
|
+
* absent record would pin a tree that has since moved.
|
|
1502
|
+
*/
|
|
1503
|
+
const RecordedSnapshot = Schema.Option(EngineLike.Snapshot);
|
|
1504
|
+
/**
|
|
1505
|
+
* Pins the workspace once, through a journaled boundary.
|
|
1506
|
+
*
|
|
1507
|
+
* The same treatment {@link witness} gives a measurement, for the same reason:
|
|
1508
|
+
* a pin is a read of the world, so a resumed frame must be handed the tree its
|
|
1509
|
+
* original attempt pinned rather than pin whatever is there now. The boundary
|
|
1510
|
+
* is keyed on the cell digest and the ordinal, which is exactly the pair that
|
|
1511
|
+
* re-derives when the cell is re-executed.
|
|
1512
|
+
*/
|
|
1513
|
+
const pin = (engine, state, cell, ordinal, id, callMs) => engine.record({
|
|
1514
|
+
name: "checkpoint",
|
|
1515
|
+
identity: { session: state.session, frame: state.frame, boundary: `checkpoint:${cell}:${ordinal}` },
|
|
1516
|
+
success: RecordedSnapshot,
|
|
1517
|
+
// The per-call ceiling, applied INSIDE the record for the reason
|
|
1518
|
+
// {@link settled} applies it inside its own: a store that hung past the
|
|
1519
|
+
// budget left the cell told nothing was pinned while the pin itself was
|
|
1520
|
+
// still in flight, so the resumed frame could be handed a snapshot the
|
|
1521
|
+
// original attempt was told it never got. Cut off here, "nothing was
|
|
1522
|
+
// pinned" is what the journal holds and what every later attempt reads.
|
|
1523
|
+
execute: engine.capture({
|
|
1524
|
+
id,
|
|
1525
|
+
identity: { session: state.session, frame: state.frame, boundary: `checkpoint-capture:${cell}` }
|
|
1526
|
+
}).pipe(Effect.timeoutOrElse({ duration: callMs, orElse: () => Effect.succeed(Option.none()) }))
|
|
1527
|
+
});
|
|
1528
|
+
/**
|
|
1529
|
+
* Issues one call under the run's per-call ceiling, through a journaled
|
|
1530
|
+
* boundary.
|
|
1531
|
+
*
|
|
1532
|
+
* The ceiling is what makes this a boundary rather than a pass-through.
|
|
1533
|
+
* `EngineLike.call` is already a keyed activity, so a call that SETTLES is
|
|
1534
|
+
* durable on its own; a call the ceiling cuts off is durable nowhere, because
|
|
1535
|
+
* the activity it interrupted never settled. Left there, a re-executed cell
|
|
1536
|
+
* issues that call again against a world that has moved on, gets an answer this
|
|
1537
|
+
* time, and takes a branch the original attempt never took — and every
|
|
1538
|
+
* irreversible effect below the fork is bought twice.
|
|
1539
|
+
*
|
|
1540
|
+
* The record is written AFTER the call and read INSTEAD of it, which is the
|
|
1541
|
+
* only order available and is the one that matters. The call cannot run inside
|
|
1542
|
+
* the boundary: `EngineLike.call` is where a cell reaches a durable wait, and a
|
|
1543
|
+
* `Flow.suspend` raised inside an enclosing activity suspends that activity's
|
|
1544
|
+
* attempt rather than the run, so a cell that slept on the durable clock never
|
|
1545
|
+
* woke. So this issues the call, records what the cell is about to be told, and
|
|
1546
|
+
* on any later attempt hands back the recorded settlement whatever the re-issued
|
|
1547
|
+
* call answered this time. The cell's branch is what has to be stable, and it
|
|
1548
|
+
* is; the re-issued call is work the run pays for twice, which is what a call
|
|
1549
|
+
* the ceiling cut off already cost before this existed.
|
|
1550
|
+
*
|
|
1551
|
+
* The drive loop is told the settlement is bounded here
|
|
1552
|
+
* (`Sandbox.RealmEvaluation.bounded`): two clocks over one call would settle it
|
|
1553
|
+
* from the reading nothing keeps.
|
|
1554
|
+
*
|
|
1555
|
+
* The boundary is keyed on the cell digest and the call's ordinal, which is the
|
|
1556
|
+
* pair a re-executed cell re-derives.
|
|
1557
|
+
*/
|
|
1558
|
+
const issued = (engine, state, cell, ordinal, callMs, flow, issue, replaying, call) => Effect.gen(function* () {
|
|
1559
|
+
if (replaying) {
|
|
1560
|
+
// A terminal frame proves this prefix settled. Read its settlements
|
|
1561
|
+
// directly, including per-call timeouts whose host activity never settled.
|
|
1562
|
+
return yield* engine.record({
|
|
1563
|
+
name: "cell-call",
|
|
1564
|
+
call,
|
|
1565
|
+
identity: { session: state.session, frame: state.frame, boundary: `cell-call:${cell}:${ordinal}` },
|
|
1566
|
+
success: Cell.CallResultVariant,
|
|
1567
|
+
execute: Effect.fail(new HarnessError({
|
|
1568
|
+
code: "incompatible_journal",
|
|
1569
|
+
message: `The timed-out frame is missing settlement ${ordinal}`
|
|
1570
|
+
}))
|
|
1571
|
+
}).pipe(Effect.flatMap(Cell.decodeCallResult));
|
|
1572
|
+
}
|
|
1573
|
+
// An escape — a permission park, an abort, an engine failure — never
|
|
1574
|
+
// reaches the record at all, so nothing journals it and the attempt the
|
|
1575
|
+
// grant answers asks again. That is the whole reason the call sits outside
|
|
1576
|
+
// the boundary rather than inside its `execute`.
|
|
1577
|
+
//
|
|
1578
|
+
// Authority is decided before the clock starts: time a person spends
|
|
1579
|
+
// answering an approval is not the flow's to spend.
|
|
1580
|
+
const refused = engine.admit === undefined ? undefined : yield* engine.admit(call);
|
|
1581
|
+
const identity = { session: state.session, frame: state.frame, boundary: `cell-call:${cell}:${ordinal}` };
|
|
1582
|
+
// A guarded host parks a timed-out call instead of letting the cell read
|
|
1583
|
+
// it. The call's own timeout (a command's, a child await still running)
|
|
1584
|
+
// is reported under the same code as this boundary's. Nothing is recorded
|
|
1585
|
+
// for the parked call, so Continue issues it again and Stop refuses it
|
|
1586
|
+
// here before it runs.
|
|
1587
|
+
const subject = EngineLike.callSubject(call.identity);
|
|
1588
|
+
const guarded = refused === undefined ? engine.guard : undefined;
|
|
1589
|
+
if (guarded !== undefined)
|
|
1590
|
+
yield* guarded.admit(subject);
|
|
1591
|
+
const settlement = refused ?? (yield* issue.pipe(Effect.timeoutOrElse({ duration: callMs, orElse: () => Effect.succeed(Sandbox.callTimedOut(flow, callMs)) }), Effect.flatMap(Cell.decodeCallResult)));
|
|
1592
|
+
if (guarded !== undefined && settlement.outcome === "failure" && settlement.code === "timeout") {
|
|
1593
|
+
// The limit it ran past: this boundary's, or the one the flow's own
|
|
1594
|
+
// timeout recorded (a command's `timeoutMs` or default, an await's).
|
|
1595
|
+
yield* guarded.trip({
|
|
1596
|
+
source: "tool-call",
|
|
1597
|
+
subject,
|
|
1598
|
+
limitMillis: limitOf(settlement),
|
|
1599
|
+
message: settlement.message ?? `Flow ${flow} timed out.`
|
|
1600
|
+
});
|
|
1601
|
+
}
|
|
1602
|
+
return yield* engine.record({
|
|
1603
|
+
name: "cell-call",
|
|
1604
|
+
call,
|
|
1605
|
+
identity,
|
|
1606
|
+
success: Cell.CallResultVariant,
|
|
1607
|
+
execute: Effect.succeed(settlement)
|
|
1608
|
+
}).pipe(Effect.flatMap(Cell.decodeCallResult));
|
|
1609
|
+
});
|
|
1610
|
+
/**
|
|
1611
|
+
* The schema one settled frame is journaled under.
|
|
1612
|
+
*
|
|
1613
|
+
* The realm's own answer and execution boundary, because these are things the loop
|
|
1614
|
+
* branches on: the outcome decides the transition, the prints become the next
|
|
1615
|
+
* frame's context, and the bindings become the variables panel. The boundary
|
|
1616
|
+
* prevents a timed-out frame from reconstructing beyond its delivered prefix.
|
|
1617
|
+
*/
|
|
1618
|
+
const RecordedFrame = Schema.Struct({
|
|
1619
|
+
boundary: Schema.optional(Sandbox.FrameBoundary),
|
|
1620
|
+
outcome: Cell.Outcome,
|
|
1621
|
+
prints: Schema.String,
|
|
1622
|
+
bindings: Schema.Array(VariablesPanel.Binding)
|
|
1623
|
+
});
|
|
1624
|
+
/**
|
|
1625
|
+
* Settles one `ctx.checkpoint()` from inside a running cell.
|
|
1626
|
+
*
|
|
1627
|
+
* The id is `cp-<frame>-<ordinal>`, which is derived from the two things that
|
|
1628
|
+
* re-derive identically when a cell is re-executed, so a resumed run addresses
|
|
1629
|
+
* the same trees its original attempt did — and short enough for a model to
|
|
1630
|
+
* read back in a print. `minted` is this frame's own tally; the run's is
|
|
1631
|
+
* `State.checkpointIds`, folded in when the frame closes.
|
|
1632
|
+
*/
|
|
1633
|
+
const minter = (state, cell, engine, minted, callMs, emit) => (mint) => Effect.gen(function* () {
|
|
1634
|
+
const held = state.checkpointIds.length + minted.length;
|
|
1635
|
+
if (held >= state.checkpointCap) {
|
|
1636
|
+
return refusal("checkpoint_exhausted", `This run has pinned its ${state.checkpointCap} checkpoints and nothing was pinned here. ${held === 0 ? "" : `The ones you hold are ${[...state.checkpointIds, ...minted].join(", ")}, and `}ctx.base is always the tree this run opened on.`);
|
|
1637
|
+
}
|
|
1638
|
+
const id = `cp-${state.frame}-${mint.ordinal}`;
|
|
1639
|
+
const snapshot = yield* pin(engine, state, cell.digest, mint.ordinal, id, callMs);
|
|
1640
|
+
if (Option.isNone(snapshot)) {
|
|
1641
|
+
return refusal("checkpoint_unavailable", "This host pins no trees, so nothing was checkpointed. For a baseline, run the check before your edit, then again after it.");
|
|
1642
|
+
}
|
|
1643
|
+
minted.push(id);
|
|
1644
|
+
yield* emit(new AgentEvent.CheckpointMinted({
|
|
1645
|
+
eventType: eventType.checkpointMinted,
|
|
1646
|
+
id: snapshot.value.id,
|
|
1647
|
+
ref: snapshot.value.ref,
|
|
1648
|
+
cell: cell.digest,
|
|
1649
|
+
ordinal: mint.ordinal
|
|
1650
|
+
}));
|
|
1651
|
+
return new Cell.CallResult({ outcome: "success", value: Cell.checkpoint(snapshot.value.id) });
|
|
1652
|
+
});
|
|
1653
|
+
const emitModelProgress = (event, emit) => event.type === "retry"
|
|
1654
|
+
? emit(new AgentEvent.ModelRetried({
|
|
1655
|
+
eventType: eventType.modelRetried,
|
|
1656
|
+
attempt: event.attempt,
|
|
1657
|
+
code: event.code,
|
|
1658
|
+
delayMillis: event.delayMillis
|
|
1659
|
+
}))
|
|
1660
|
+
: event.type === "settle"
|
|
1661
|
+
? Effect.void
|
|
1662
|
+
: emit(new AgentEvent.ModelDelta({ eventType: eventType.modelDelta, delta: event }));
|
|
1663
|
+
/**
|
|
1664
|
+
* The reserved key a flow reports an invalid probe under.
|
|
1665
|
+
*
|
|
1666
|
+
* This is the one convention the loop reads off an otherwise opaque call
|
|
1667
|
+
* result. It is not a shared type — the controller must not depend on the tool
|
|
1668
|
+
* library — so it is a documented wire key. `@smthrs/std/Probe` is the
|
|
1669
|
+
* producing half and owns the taxonomy; the controller only has to know that a
|
|
1670
|
+
* result carrying this key is a result whose failure was about the command.
|
|
1671
|
+
*/
|
|
1672
|
+
const invalidProbeKey = "invalidProbe";
|
|
1673
|
+
/** What a settled call declared about whether it ran a check at all. */
|
|
1674
|
+
const invalidProbeOf = (value) => {
|
|
1675
|
+
if (value === null || typeof value !== "object" || Array.isArray(value))
|
|
1676
|
+
return undefined;
|
|
1677
|
+
const declared = value[invalidProbeKey];
|
|
1678
|
+
if (declared === null || typeof declared !== "object" || Array.isArray(declared))
|
|
1679
|
+
return undefined;
|
|
1680
|
+
const { message, reason } = declared;
|
|
1681
|
+
return typeof reason === "string" && typeof message === "string" ? { reason, message } : undefined;
|
|
1682
|
+
};
|
|
1683
|
+
/**
|
|
1684
|
+
* The reserved output key a flow reports a measured write under.
|
|
1685
|
+
*
|
|
1686
|
+
* `@smthrs/std/TreeFingerprint` is the producing half: `bash` fingerprints a
|
|
1687
|
+
* container's working directory before and after a containerised command and
|
|
1688
|
+
* sets this to whether the two differ. The controller reads a `true` here as
|
|
1689
|
+
* a standing write the host's own walk could not see, and nothing else off
|
|
1690
|
+
* it: `false` and absent both leave the call's declaration as it stands.
|
|
1691
|
+
*/
|
|
1692
|
+
const mutatedKey = "mutated";
|
|
1693
|
+
/** Whether a settled call measured, itself, that its tree moved. */
|
|
1694
|
+
const mutatedOf = (value) => value !== null && typeof value === "object" && !Array.isArray(value) &&
|
|
1695
|
+
value[mutatedKey] === true;
|
|
1696
|
+
/**
|
|
1697
|
+
* The message of a frame failure that is not already a harness error: the
|
|
1698
|
+
* fixed headline and the failure's own code and sentence, on one bounded
|
|
1699
|
+
* line, so a reader of the message alone (a receipt, a reply) learns why,
|
|
1700
|
+
* such as a provider refusing for want of credits. The typed failure stays
|
|
1701
|
+
* the error's `cause`.
|
|
1702
|
+
*/
|
|
1703
|
+
const frameFailureMessage = (error) => {
|
|
1704
|
+
const row = typeof error === "object" && error !== null ? error : {};
|
|
1705
|
+
const said = typeof row["message"] === "string" ? row["message"].replace(/\s+/g, " ").trim() : "";
|
|
1706
|
+
if (said === "")
|
|
1707
|
+
return "The cell frame failed";
|
|
1708
|
+
const code = typeof row["code"] === "string" && row["code"] !== "" ? `${row["code"].slice(0, 64)}: ` : "";
|
|
1709
|
+
return `The cell frame failed: ${code}${said.slice(0, 1_000)}`;
|
|
1710
|
+
};
|
|
1711
|
+
/**
|
|
1712
|
+
* What a run says when its frame budget, rather than the run, ended it.
|
|
1713
|
+
*
|
|
1714
|
+
* A run that never completed has only the budget to report. A run whose
|
|
1715
|
+
* completion was handed back for one of the completion demands has something
|
|
1716
|
+
* else: an answer it already gave, which the controller took away on the
|
|
1717
|
+
* promise of another frame. If that frame goes elsewhere — a cell that raises,
|
|
1718
|
+
* a refused park, more work than the budget has room for — the promise is the
|
|
1719
|
+
* only thing left, and dropping the answer would make the demand cost the run
|
|
1720
|
+
* exactly the thing it was meant to protect. Both facts are reported: the
|
|
1721
|
+
* budget ended the run, and this is what the run last said its answer was.
|
|
1722
|
+
*
|
|
1723
|
+
* Which demand took it is not named here. Three of them can, each writes its
|
|
1724
|
+
* own control event when it fires, and a sentence naming one of the three
|
|
1725
|
+
* would be wrong for the other two — as this one was, when the narrowing
|
|
1726
|
+
* demand was the only demand there was.
|
|
1727
|
+
*/
|
|
1728
|
+
const budgetMessage = (state) => state.bouncedCompletion === undefined
|
|
1729
|
+
? `The frame budget of ${state.maxFrames} is exhausted. The run stops here; the last transition was a request to continue.`
|
|
1730
|
+
: `The frame budget of ${state.maxFrames} is exhausted. This run completed once and had that completion handed back for another frame, so what it reported then stands here rather than being lost:\n\n${state.bouncedCompletion}`;
|
|
1731
|
+
/**
|
|
1732
|
+
* Resolves one cell call into a durable engine boundary.
|
|
1733
|
+
*
|
|
1734
|
+
* Resolution happens here, at the boundary, and not inside the sandbox: the
|
|
1735
|
+
* flow must exist in the catalog this frame was given, be model-invocable, and
|
|
1736
|
+
* every capability it declares must still be inside the run's narrowed envelope. Both denials are
|
|
1737
|
+
* ordinary call failures the cell can catch, which is what lets an agent
|
|
1738
|
+
* discover the shape of its authority without crashing the run.
|
|
1739
|
+
*
|
|
1740
|
+
* The truncation ledger is kept here for the same reason. The boundary is the
|
|
1741
|
+
* only party that sees both a result that declared its capture cut short and a
|
|
1742
|
+
* later call handing those exact bytes to something that writes, and a write of
|
|
1743
|
+
* a known fragment is refused rather than performed. See `TruncatedOutput`.
|
|
1744
|
+
*
|
|
1745
|
+
* `performed` collects the ordinal of every invocation that actually reached
|
|
1746
|
+
* the engine. The three refusals above return before it, and a refused call
|
|
1747
|
+
* changed nothing — so it must not read as a write to anything downstream. It
|
|
1748
|
+
* did once: the truncated-write refusal is the newest of the three and lands on
|
|
1749
|
+
* exactly the calls the read-only cap watches, so a run whose only edit was
|
|
1750
|
+
* refused had its read-only streak cleared by a write that never happened, and
|
|
1751
|
+
* the cap stayed silent through the stall it exists to break.
|
|
1752
|
+
*
|
|
1753
|
+
* `restored` collects every withheld flow a call named. The call is refused,
|
|
1754
|
+
* because the frame's catalog is frozen, and the flow is back in `ctx.flows`
|
|
1755
|
+
* from the next frame. The refusal reads only checkpointed state, so a
|
|
1756
|
+
* replayed frame restores the same names.
|
|
1757
|
+
*
|
|
1758
|
+
* `tree` counts the calls of this frame that may have written, in issue order,
|
|
1759
|
+
* and chains their digests. A sealed reading of the live tree carries both and
|
|
1760
|
+
* the run's frame clock as its `Cell.Call.epoch`, so a read after a write is a
|
|
1761
|
+
* new question and not a replay of the read before it. The count alone let two
|
|
1762
|
+
* runs on one tree — one that ran `git status`, one that edited the file —
|
|
1763
|
+
* key their next read alike, and the second replayed the first's pre-edit
|
|
1764
|
+
* text; the chain names which writes came first. Any call that is not sealed counts, not
|
|
1765
|
+
* only one that declared a write: a shell command declares nothing and writes
|
|
1766
|
+
* wherever it likes. The count advances when the call is issued, so a replayed
|
|
1767
|
+
* frame — which issues the same calls in the same order — derives the same
|
|
1768
|
+
* epochs and keys the same boundaries.
|
|
1769
|
+
*
|
|
1770
|
+
* `live` names the tree the frame opened on, which the counters alone cannot:
|
|
1771
|
+
* they start at zero in every run, so a read in a later run keyed the same as
|
|
1772
|
+
* one in an earlier run and replayed its answer over an edit made between the
|
|
1773
|
+
* two (#1948). It is the digest of a journaled measurement taken after the
|
|
1774
|
+
* model wait, so a replayed frame keys the same boundaries; without a whole
|
|
1775
|
+
* measurement it is the session and frame, and an unmeasured tree is reused
|
|
1776
|
+
* only inside the frame that read it.
|
|
1777
|
+
*/
|
|
1778
|
+
const callHandler = (state, cell, descriptors, engine, ledger, performed, restored, tree, live, callMs, replaying, emit) => (invocation) => Effect.gen(function* () {
|
|
1779
|
+
if (state.withheldFlows.includes(invocation.flow)) {
|
|
1780
|
+
restored.add(invocation.flow);
|
|
1781
|
+
return refusal("flow_withheld", `Flow ${invocation.flow} was withheld for this task. It is in ctx.flows from the next frame; reissue the call.`);
|
|
1782
|
+
}
|
|
1783
|
+
const descriptor = descriptors.get(invocation.flow);
|
|
1784
|
+
if (descriptor === undefined) {
|
|
1785
|
+
return refusal("unknown_flow", `Unknown flow ${invocation.flow}. Only the flows in ctx.flows are callable.`);
|
|
1786
|
+
}
|
|
1787
|
+
// Checked here as well as in catalog filtering and `CellCalls`, so a host
|
|
1788
|
+
// that hands the frame more than the visible set still cannot let the
|
|
1789
|
+
// model run a host-only flow.
|
|
1790
|
+
if (!descriptor.modelInvocable) {
|
|
1791
|
+
return refusal("capability_refused", `Flow ${invocation.flow} is not model-invocable.`);
|
|
1792
|
+
}
|
|
1793
|
+
const envelope = CapabilitySet.fromPatterns(state.capabilityEnvelope);
|
|
1794
|
+
const refused = descriptor.capabilities.filter((declared) => Option.match(Capability.parse(declared), {
|
|
1795
|
+
onNone: () => true,
|
|
1796
|
+
onSome: (capability) => !CapabilitySet.allows(envelope, capability)
|
|
1797
|
+
}));
|
|
1798
|
+
if (refused.length > 0) {
|
|
1799
|
+
return refusal("capability_refused", `Flow ${invocation.flow} needs ${refused.join(", ")}, which is outside this run's capability envelope.`);
|
|
1800
|
+
}
|
|
1801
|
+
// Only a call that changes something is checked. Passing a fragment to a
|
|
1802
|
+
// search, a diff, or a summary is ordinary use of what the flow returned;
|
|
1803
|
+
// handing it to something that writes is the one case that destroys a file.
|
|
1804
|
+
const changes = mutating(descriptor, invocation.input);
|
|
1805
|
+
if (changes) {
|
|
1806
|
+
const found = TruncatedOutput.reuse(invocation.input, ledger);
|
|
1807
|
+
if (found !== undefined) {
|
|
1808
|
+
return refusal("truncated_write", TruncatedOutput.refusal(invocation.flow, found));
|
|
1809
|
+
}
|
|
1810
|
+
}
|
|
1811
|
+
// Where this call runs. A checkpoint is a tree the run has already been
|
|
1812
|
+
// handed back, so it is a *reading* position and nothing else: the whole
|
|
1813
|
+
// promise the surface makes is that taking a reading against one leaves the
|
|
1814
|
+
// work alone, and a flow that declared a write would break that promise
|
|
1815
|
+
// whichever tree the host materialized. Both refusals below are fail-soft,
|
|
1816
|
+
// because both are things a cell fixes on its next line.
|
|
1817
|
+
let at;
|
|
1818
|
+
if (invocation.at !== undefined) {
|
|
1819
|
+
at = Cell.checkpointOf(invocation.at);
|
|
1820
|
+
if (at === undefined) {
|
|
1821
|
+
return refusal("invalid_input", `The at option takes a checkpoint, which is what ctx.checkpoint() resolves with and what ctx.base is. It was given ${elide.head(CanonicalJson.stringify(invocation.at), 120, "the rest is the same shape")}.`);
|
|
1822
|
+
}
|
|
1823
|
+
if (changes) {
|
|
1824
|
+
return refusal("checkpoint_readonly", `Flow ${invocation.flow} declares a write, and a checkpoint is a read-only view of a tree that has already been. Nothing was run. Make the change on the live tree, and keep at for the readings you take against ${at}.`);
|
|
1825
|
+
}
|
|
1826
|
+
}
|
|
1827
|
+
const sealed = descriptor.effects.tier === "sealed";
|
|
1828
|
+
const epoch = sealed && at === undefined
|
|
1829
|
+
? {
|
|
1830
|
+
frames: state.mutations,
|
|
1831
|
+
calls: tree.writes,
|
|
1832
|
+
...(tree.chain === undefined ? {} : { writes: tree.chain }),
|
|
1833
|
+
...live
|
|
1834
|
+
}
|
|
1835
|
+
: undefined;
|
|
1836
|
+
const call = Cell.callOf(descriptor, {
|
|
1837
|
+
input: invocation.input,
|
|
1838
|
+
identity: new Cell.CallIdentity({
|
|
1839
|
+
session: state.session,
|
|
1840
|
+
frame: state.frame,
|
|
1841
|
+
cell: cell.digest,
|
|
1842
|
+
ordinal: invocation.ordinal,
|
|
1843
|
+
declaration: Cell.declarationDigest(descriptor),
|
|
1844
|
+
layers: [...new Set(state.layers)].sort()
|
|
1845
|
+
}),
|
|
1846
|
+
...(at === undefined ? {} : { at }),
|
|
1847
|
+
...(epoch === undefined ? {} : { epoch })
|
|
1848
|
+
});
|
|
1849
|
+
performed.add(invocation.ordinal);
|
|
1850
|
+
if (!sealed) {
|
|
1851
|
+
tree.writes++;
|
|
1852
|
+
tree.chain = Digest.digest(CanonicalJson.stringify([tree.chain ?? null, signatureOf(invocation.flow, invocation.input, at)]));
|
|
1853
|
+
}
|
|
1854
|
+
yield* emit(new AgentEvent.CellCallStarted({ eventType: eventType.cellCallStarted, call }));
|
|
1855
|
+
const result = yield* issued(engine, state, cell.digest, invocation.ordinal, callMs, invocation.flow,
|
|
1856
|
+
// A flow that asks Jev journals its receipts into this run.
|
|
1857
|
+
Effect.suspend(() => engine.call(call)).pipe(Effect.provideService(AgentEvent.Journal, emit)), replaying, call);
|
|
1858
|
+
if (result.outcome === "success")
|
|
1859
|
+
ledger.push(...TruncatedOutput.captures(call.flowName, result.value));
|
|
1860
|
+
yield* emit(new AgentEvent.CellCallSettled({
|
|
1861
|
+
eventType: eventType.cellCallSettled,
|
|
1862
|
+
flowName: call.flowName,
|
|
1863
|
+
identity: call.identity,
|
|
1864
|
+
result
|
|
1865
|
+
}));
|
|
1866
|
+
return result;
|
|
1867
|
+
});
|
|
1868
|
+
/**
|
|
1869
|
+
* Journals what one model call is about to be asked.
|
|
1870
|
+
*
|
|
1871
|
+
* Emitted before the sealed step rather than beside its settlement, so a call
|
|
1872
|
+
* that never settles still leaves its request behind. What is journaled is the
|
|
1873
|
+
* request as the host will send it, which is not always the one built here:
|
|
1874
|
+
* see {@link EngineLike.EngineLike.resolve}, which is also why the answer is
|
|
1875
|
+
* asked for on every attempt and not recorded.
|
|
1876
|
+
*
|
|
1877
|
+
* A host that cannot say what it would send leaves no record. The sealed step
|
|
1878
|
+
* that follows fails on the same request, so the call the missing record would
|
|
1879
|
+
* have described is one that was never made.
|
|
1880
|
+
*/
|
|
1881
|
+
const requested = (state, engine, request, purpose, attempt, emit) => Effect.gen(function* () {
|
|
1882
|
+
const resolved = yield* EngineLike.resolve(engine, request);
|
|
1883
|
+
if (Option.isNone(resolved))
|
|
1884
|
+
return;
|
|
1885
|
+
yield* emit(new AgentEvent.ModelRequested({
|
|
1886
|
+
eventType: eventType.modelRequested,
|
|
1887
|
+
scope: state.session,
|
|
1888
|
+
frame: state.frame,
|
|
1889
|
+
attempt,
|
|
1890
|
+
purpose,
|
|
1891
|
+
seat: state.seat,
|
|
1892
|
+
binding: Option.getOrUndefined(resolved.value.binding),
|
|
1893
|
+
request: resolved.value.request
|
|
1894
|
+
}));
|
|
1895
|
+
});
|
|
1896
|
+
/** The sealed summary of what a compaction step squashes. */
|
|
1897
|
+
const summarized = (state, engine, emit, step) => Effect.gen(function* () {
|
|
1898
|
+
const summaryRequest = yield* Compaction.summaryRequest(state.contextWindow, step).pipe(Effect.orDie);
|
|
1899
|
+
const request = ModelRequest.ModelRequest.make({
|
|
1900
|
+
modelId: summaryRequest.modelId,
|
|
1901
|
+
system: summaryRequest.system,
|
|
1902
|
+
messages: summaryRequest.messages,
|
|
1903
|
+
tools: [],
|
|
1904
|
+
toolChoice: "none",
|
|
1905
|
+
params: summaryRequest.params
|
|
1906
|
+
});
|
|
1907
|
+
yield* requested(state, engine, request, "compaction", 1, emit);
|
|
1908
|
+
const events = yield* Stream.runCollect((engine.sealStepWithEvents?.({
|
|
1909
|
+
request,
|
|
1910
|
+
keyMaterial: keyMaterialFrom(state, state.contextWindow, request),
|
|
1911
|
+
modelCallMs: state.modelCallMs
|
|
1912
|
+
}, emit) ?? engine.sealStep({
|
|
1913
|
+
request,
|
|
1914
|
+
keyMaterial: keyMaterialFrom(state, state.contextWindow, request),
|
|
1915
|
+
modelCallMs: state.modelCallMs
|
|
1916
|
+
})).pipe(Stream.tap((event) => emitModelProgress(event, emit)))).pipe(Effect.map((collected) => Array.from(collected)));
|
|
1917
|
+
if (!events.some((event) => event.type === "settle")) {
|
|
1918
|
+
return yield* new HarnessError({
|
|
1919
|
+
code: "model_failed",
|
|
1920
|
+
message: "The sealed compaction step ended without a recorded settlement"
|
|
1921
|
+
});
|
|
1922
|
+
}
|
|
1923
|
+
const settled = ModelEvent.ModelEvent.settledMessage(events);
|
|
1924
|
+
const text = settled.message.content.filter((part) => part.type === "text");
|
|
1925
|
+
if (text.length === 0) {
|
|
1926
|
+
return yield* new HarnessError({
|
|
1927
|
+
code: "model_failed",
|
|
1928
|
+
message: "The sealed compaction step returned no text summary"
|
|
1929
|
+
});
|
|
1930
|
+
}
|
|
1931
|
+
return ModelRequest.Message.user(untrustedData(text.map((part) => part.text).join("\n"), "compaction summary of conversation and tool output"));
|
|
1932
|
+
});
|
|
1933
|
+
const messageText = (message) => message.content.filter((part) => part.type === "text").map((part) => part.text)
|
|
1934
|
+
.join("\n");
|
|
1935
|
+
/** One old segment as the compaction reading shows it to Jev. */
|
|
1936
|
+
const markItem = (segment) => {
|
|
1937
|
+
const messages = segment.content.filter((item) => "role" in item);
|
|
1938
|
+
const said = messages.filter((message) => message.role === "assistant").map(messageText).join("\n");
|
|
1939
|
+
const extracted = Cell.extract(said);
|
|
1940
|
+
return {
|
|
1941
|
+
tokens: segment.tokens.value,
|
|
1942
|
+
cell: extracted._tag === "Success" ? extracted.success.source.text : "",
|
|
1943
|
+
prose: Supervisor.prose(said),
|
|
1944
|
+
observed: messages.filter((message) => message.role !== "assistant").map(messageText).join("\n\n")
|
|
1945
|
+
};
|
|
1946
|
+
};
|
|
1947
|
+
/**
|
|
1948
|
+
* {@link State.segmentFacts} beside each compactable segment, absent for a
|
|
1949
|
+
* summary or steering segment; undefined when they do not describe every
|
|
1950
|
+
* transcript segment.
|
|
1951
|
+
*/
|
|
1952
|
+
const alignedFacts = (state, segments) => {
|
|
1953
|
+
const transcripts = segments.filter((segment) => segment.kind === "transcript").length;
|
|
1954
|
+
if (transcripts !== state.segmentFacts.length)
|
|
1955
|
+
return undefined;
|
|
1956
|
+
let next = 0;
|
|
1957
|
+
return segments.map((segment) => segment.kind === "transcript" ? state.segmentFacts[next++] : undefined);
|
|
1958
|
+
};
|
|
1959
|
+
/**
|
|
1960
|
+
* The transcript segments a supervisor reading is asked to mark: those with
|
|
1961
|
+
* no answer yet, each once by digest. The person's are always kept and never
|
|
1962
|
+
* asked about; a run whose facts are not aligned marks nothing.
|
|
1963
|
+
*/
|
|
1964
|
+
const unmarked = (state) => {
|
|
1965
|
+
const segments = compactable(state.contextWindow.segments);
|
|
1966
|
+
const facts = alignedFacts(state, segments);
|
|
1967
|
+
if (facts === undefined)
|
|
1968
|
+
return [];
|
|
1969
|
+
const open = segments.flatMap((segment, index) => {
|
|
1970
|
+
const known = facts[index];
|
|
1971
|
+
return known === undefined || known.person || known.answer !== undefined ? [] : [segment];
|
|
1972
|
+
});
|
|
1973
|
+
return [...new Map(open.map((segment) => [segment.digest, segment])).values()].map((segment) => ({
|
|
1974
|
+
digest: segment.digest,
|
|
1975
|
+
item: markItem(segment)
|
|
1976
|
+
}));
|
|
1977
|
+
};
|
|
1978
|
+
/**
|
|
1979
|
+
* {@link State.segmentFacts} after a compaction: the entries of the kept
|
|
1980
|
+
* prefix segments and of the suffix, in order. `marks` absent squashes every
|
|
1981
|
+
* prefix segment.
|
|
1982
|
+
*/
|
|
1983
|
+
const factsAfter = (state, prefixLength, marks) => {
|
|
1984
|
+
const transcripts = compactable(state.contextWindow.segments).flatMap((segment, index) => segment.kind === "transcript" ? [index] : []);
|
|
1985
|
+
// The entries describe the newest segments, so an unaligned run's are
|
|
1986
|
+
// missing from the front.
|
|
1987
|
+
const missing = transcripts.length - state.segmentFacts.length;
|
|
1988
|
+
return transcripts.flatMap((index, at) => {
|
|
1989
|
+
const facts = state.segmentFacts[at - missing];
|
|
1990
|
+
return facts !== undefined && (index >= prefixLength || marks?.[index] === "keep") ? [facts] : [];
|
|
1991
|
+
});
|
|
1992
|
+
};
|
|
1993
|
+
/** What the compaction-marks reading records: the mark of every prefix segment. */
|
|
1994
|
+
const MarksRecord = Schema.Array(compactionMarks.Marked);
|
|
1995
|
+
/**
|
|
1996
|
+
* The marks of a judged run's compaction, over its aligned `facts`: pins
|
|
1997
|
+
* first, then the answers the run stored for what they leave open, resolved
|
|
1998
|
+
* against the window. What is still unmarked is asked in one reading,
|
|
1999
|
+
* recorded under the prefix it marks, so a replay re-keys the same summary;
|
|
2000
|
+
* with nothing unmarked nothing is asked, and the stored answers, which
|
|
2001
|
+
* each recorded drain rebuilds, resolve the same marks on replay. Undefined,
|
|
2002
|
+
* and every segment squashed, when that reading could not be judged, which
|
|
2003
|
+
* journals `decision-unjudged`.
|
|
2004
|
+
*/
|
|
2005
|
+
const marked = (state, facts, prefixLength, engine, emit, taskOf) => Effect.gen(function* () {
|
|
2006
|
+
const segments = compactable(state.contextWindow.segments);
|
|
2007
|
+
const failing = state.checks.filter((check) => check.failing).map((check) => check.label);
|
|
2008
|
+
const pinned = compactionMarks.pins(segments, prefixLength, facts, failing);
|
|
2009
|
+
const open = compactionMarks.unpinned(pinned);
|
|
2010
|
+
const missing = open.filter((index) => facts[index]?.answer === undefined);
|
|
2011
|
+
const budget = {
|
|
2012
|
+
contextWindow: state.contextWindowTokens,
|
|
2013
|
+
reserve: defaultReserve,
|
|
2014
|
+
keepRecent: defaultKeepRecent,
|
|
2015
|
+
suffix: segments.slice(prefixLength).reduce((sum, segment) => sum + segment.tokens.value, 0),
|
|
2016
|
+
tokens: segments.slice(0, prefixLength).map((segment) => segment.tokens.value)
|
|
2017
|
+
};
|
|
2018
|
+
// `InvalidStep` is a defect: one answer per open index, by construction.
|
|
2019
|
+
const resolved = (answered) => Effect.fromResult(compactionMarks.resolve(open.map((index) => answered.get(index)), pinned, budget)).pipe(Effect.orDie);
|
|
2020
|
+
const stored = new Map(open.flatMap((index) => {
|
|
2021
|
+
const answer = facts[index]?.answer;
|
|
2022
|
+
return answer === undefined ? [] : [[index, answer]];
|
|
2023
|
+
}));
|
|
2024
|
+
if (missing.length === 0)
|
|
2025
|
+
return yield* resolved(stored);
|
|
2026
|
+
const prefix = yield* Effect.fromResult(ContextWindow.prefixDigest(state.contextWindow, prefixLength)).pipe(Effect.orDie);
|
|
2027
|
+
const record = yield* Judgement.recorded(engine, {
|
|
2028
|
+
name: "compaction-marks",
|
|
2029
|
+
identity: { session: state.session, frame: state.frame, boundary: `compaction-marks:${prefix}` },
|
|
2030
|
+
classifier: "compaction/marks",
|
|
2031
|
+
value: MarksRecord,
|
|
2032
|
+
items: missing.length
|
|
2033
|
+
}, compactionMarks.read({ task: Judgement.task(taskOf(state.contextWindow)), failing }, missing.map((index) => markItem(segments[index]))).pipe(Effect.flatMap((reading) => resolved(new Map([...stored, ...missing.map((index, at) => [index, reading.answers[at]])])).pipe(Effect.map((value) => ({
|
|
2034
|
+
value,
|
|
2035
|
+
asked: reading.asked,
|
|
2036
|
+
acted: value.some((mark) => mark.mark !== "squash")
|
|
2037
|
+
}))))));
|
|
2038
|
+
yield* Judgement.emitRecorded(emit, record);
|
|
2039
|
+
return record.value ?? undefined;
|
|
2040
|
+
});
|
|
2041
|
+
/**
|
|
2042
|
+
* Compacts the frame's context before the model is asked anything.
|
|
2043
|
+
*
|
|
2044
|
+
* Compaction is a transition of the run, not a repair applied to a request on
|
|
2045
|
+
* its way out: the summary is produced by its own sealed step, so it is keyed
|
|
2046
|
+
* and journaled like every other model call, and the settlement is emitted as
|
|
2047
|
+
* `CompactionSettled`. Without that event a replay rebuilds the uncompacted
|
|
2048
|
+
* transcript, re-crosses the same threshold, and re-keys every later frame — so
|
|
2049
|
+
* emitting it is what makes the compacted window part of the run's durable
|
|
2050
|
+
* state rather than an artifact of when the process happened to notice.
|
|
2051
|
+
*
|
|
2052
|
+
* A judged run marks each replaced segment keep, squash or remove first, and
|
|
2053
|
+
* only what it squashes is summarized; an unjudged one squashes all of them.
|
|
2054
|
+
* Marks are stored as the supervisor reads them and applied only here, when
|
|
2055
|
+
* the budget forces a compaction or a drained supervisor reading found the
|
|
2056
|
+
* context outdated or irrelevant, so the prefix the model is sent never
|
|
2057
|
+
* changes between compactions. The settlement names its `causes`. A judged
|
|
2058
|
+
* run whose facts are not aligned cannot be marked; its settlement says so as
|
|
2059
|
+
* `unaligned`.
|
|
2060
|
+
*
|
|
2061
|
+
* Nothing here is best-effort. A window that cannot be compacted stays as it
|
|
2062
|
+
* is; a compaction the model started and could not finish is a typed failure.
|
|
2063
|
+
*/
|
|
2064
|
+
const compacted = (state, engine, emit, judged, taskOf) => Effect.gen(function* () {
|
|
2065
|
+
const over = Compaction.shouldCompact({
|
|
2066
|
+
total: state.contextWindow.tokens.total,
|
|
2067
|
+
contextWindow: state.contextWindowTokens
|
|
2068
|
+
});
|
|
2069
|
+
const due = state.compactionDue;
|
|
2070
|
+
// A due compaction is this frame's alone, taken or not.
|
|
2071
|
+
const opened = due.length === 0 ? state : advance(state, { compactionDue: [] });
|
|
2072
|
+
if (!over && due.length === 0)
|
|
2073
|
+
return opened;
|
|
2074
|
+
const prefixLength = Compaction.selectPrefix(state.contextWindow);
|
|
2075
|
+
// Nothing compactable is not a failure: a window that is all prefix has
|
|
2076
|
+
// already given up everything it can, and the frame proceeds as declared.
|
|
2077
|
+
if (prefixLength === 0)
|
|
2078
|
+
return opened;
|
|
2079
|
+
const causes = [...(over ? ["budget"] : []), ...due];
|
|
2080
|
+
const facts = judged ? alignedFacts(state, compactable(state.contextWindow.segments)) : undefined;
|
|
2081
|
+
const unaligned = judged && facts === undefined;
|
|
2082
|
+
const marks = facts === undefined ? undefined : yield* marked(state, facts, prefixLength, engine, emit, taskOf);
|
|
2083
|
+
// An early compaction that would keep every segment gives nothing up.
|
|
2084
|
+
if (!over && marks !== undefined && marks.every((mark) => mark.mark === "keep"))
|
|
2085
|
+
return opened;
|
|
2086
|
+
// `InvalidStep` is discharged as a defect, not surfaced as a typed failure.
|
|
2087
|
+
// Every call below receives the same immutable window, `selectPrefix`'s
|
|
2088
|
+
// own output, marks resolved one per prefix segment, and the state's
|
|
2089
|
+
// validated generation parameters. An invalid prefix, digest, mark or
|
|
2090
|
+
// parameter value here would contradict those invariants.
|
|
2091
|
+
const step = yield* Compaction.declare(state.contextWindow, prefixLength, {
|
|
2092
|
+
identity: "flows/harness/CellTurn.compaction",
|
|
2093
|
+
modelId: state.contextWindow.modelId,
|
|
2094
|
+
params: state.modelParams
|
|
2095
|
+
}, marks?.map((mark) => mark.mark)).pipe(Effect.orDie);
|
|
2096
|
+
const summary = marks === undefined || marks.some((mark) => mark.mark === "squash")
|
|
2097
|
+
? yield* summarized(state, engine, emit, step)
|
|
2098
|
+
: undefined;
|
|
2099
|
+
const contextWindow = yield* Compaction.apply(state.contextWindow, step, summary).pipe(Effect.orDie);
|
|
2100
|
+
const segments = compactable(state.contextWindow.segments);
|
|
2101
|
+
const replaced = segments.slice(0, prefixLength);
|
|
2102
|
+
yield* emit(new AgentEvent.CompactionSettled({
|
|
2103
|
+
eventType: eventType.compactionSettled,
|
|
2104
|
+
replacedPrefixDigest: step.replacedPrefixDigest,
|
|
2105
|
+
retainedMessageCount: segments
|
|
2106
|
+
.slice(prefixLength)
|
|
2107
|
+
.flatMap((segment) => segment.content)
|
|
2108
|
+
.filter((item) => "role" in item).length,
|
|
2109
|
+
...(summary === undefined ? {} : { summary }),
|
|
2110
|
+
...(unaligned ? { unaligned } : {}),
|
|
2111
|
+
causes,
|
|
2112
|
+
...(marks === undefined ? {} : {
|
|
2113
|
+
kept: replaced
|
|
2114
|
+
.filter((_, index) => marks[index].mark === "keep")
|
|
2115
|
+
.flatMap((segment) => segment.content.filter((item) => "role" in item)),
|
|
2116
|
+
marks: replaced.map((segment, index) => ({ digest: segment.digest, ...marks[index] })),
|
|
2117
|
+
removedTokens: replaced
|
|
2118
|
+
.filter((_, index) => marks[index].mark === "remove")
|
|
2119
|
+
.reduce((sum, segment) => sum + segment.tokens.value, 0)
|
|
2120
|
+
})
|
|
2121
|
+
}));
|
|
2122
|
+
return advance(opened, {
|
|
2123
|
+
contextWindow,
|
|
2124
|
+
segmentFacts: factsAfter(state, prefixLength, marks?.map((mark) => mark.mark)),
|
|
2125
|
+
// The sections that listed earlier calls may be among what was replaced.
|
|
2126
|
+
ledgerShown: 0
|
|
2127
|
+
});
|
|
2128
|
+
});
|
|
2129
|
+
/**
|
|
2130
|
+
* The cell one answer carries, or why it carries none that can run.
|
|
2131
|
+
*
|
|
2132
|
+
* The parse the realm used to do, done here instead: a cell that cannot run is
|
|
2133
|
+
* refused before the frame commits to it.
|
|
2134
|
+
*/
|
|
2135
|
+
const parsed = (message) => {
|
|
2136
|
+
if (message.stopReason === "length") {
|
|
2137
|
+
return Result.fail(new Cell.Rejected({
|
|
2138
|
+
code: "output_truncated",
|
|
2139
|
+
message: "The model reached its output limit (length) before finishing the response. No blocks ran. Emit a shorter, complete cell."
|
|
2140
|
+
}));
|
|
2141
|
+
}
|
|
2142
|
+
const extracted = Cell.extract(assistantText(message));
|
|
2143
|
+
if (extracted._tag === "Failure")
|
|
2144
|
+
return Result.fail(extracted.failure);
|
|
2145
|
+
const validation = CellValidation.validate(extracted.success.source);
|
|
2146
|
+
return validation.rejected === undefined
|
|
2147
|
+
? Result.succeed({
|
|
2148
|
+
source: extracted.success.source,
|
|
2149
|
+
blocks: extracted.success.blocks,
|
|
2150
|
+
program: validation.compiled
|
|
2151
|
+
})
|
|
2152
|
+
: Result.fail(validation.rejected);
|
|
2153
|
+
};
|
|
2154
|
+
/**
|
|
2155
|
+
* The sealed model step, and the boundary parse of whatever it answered with.
|
|
2156
|
+
*
|
|
2157
|
+
* A cell that does not parse never ran: the world is exactly where the frame
|
|
2158
|
+
* found it, so there is nothing to record and everything to gain by asking
|
|
2159
|
+
* again inside this frame. The prefix of the re-prompt is byte-identical to the
|
|
2160
|
+
* one just sent, so the provider serves it from its cache and the answer costs
|
|
2161
|
+
* the output it writes.
|
|
2162
|
+
*/
|
|
2163
|
+
const seal = (state, engine, emit) => Effect.gen(function* () {
|
|
2164
|
+
let contextWindow = state.contextWindow;
|
|
2165
|
+
for (let attempt = 0;; attempt++) {
|
|
2166
|
+
const request = yield* Effect.fromResult(requestFrom(state, contextWindow));
|
|
2167
|
+
yield* requested(state, engine, request, "frame", attempt + 1, emit);
|
|
2168
|
+
// Timed on the injected clock, never on ambient wall time, so a test that
|
|
2169
|
+
// supplies a clock sees the duration it declared.
|
|
2170
|
+
const startedAt = yield* Clock.currentTimeMillis;
|
|
2171
|
+
const events = yield* Stream.runCollect((engine.sealStepWithEvents?.({
|
|
2172
|
+
request,
|
|
2173
|
+
keyMaterial: keyMaterialFrom(state, contextWindow, request),
|
|
2174
|
+
modelCallMs: state.modelCallMs
|
|
2175
|
+
}, emit) ?? engine.sealStep({
|
|
2176
|
+
request,
|
|
2177
|
+
keyMaterial: keyMaterialFrom(state, contextWindow, request),
|
|
2178
|
+
modelCallMs: state.modelCallMs
|
|
2179
|
+
})).pipe(Stream.tap((event) => emitModelProgress(event, emit)))).pipe(Effect.map((collected) => Array.from(collected)));
|
|
2180
|
+
const settledAt = yield* Clock.currentTimeMillis;
|
|
2181
|
+
if (!events.some((event) => event.type === "settle")) {
|
|
2182
|
+
return yield* new HarnessError({
|
|
2183
|
+
code: "model_failed",
|
|
2184
|
+
message: "The sealed model step ended without a recorded settlement"
|
|
2185
|
+
});
|
|
2186
|
+
}
|
|
2187
|
+
const settled = ModelEvent.ModelEvent.settledMessage(events);
|
|
2188
|
+
const sessionId = events.find((event) => event.type === "settle")?.sessionId;
|
|
2189
|
+
const cost = Pricing.cost(settled.usage, request.modelId, { at: settledAt });
|
|
2190
|
+
yield* emit(new AgentEvent.ModelSettled({
|
|
2191
|
+
eventType: eventType.modelSettled,
|
|
2192
|
+
message: settled.message,
|
|
2193
|
+
usage: settled.usage,
|
|
2194
|
+
...cost,
|
|
2195
|
+
...(sessionId === undefined ? {} : { sessionId }),
|
|
2196
|
+
durationMillis: settledAt - startedAt
|
|
2197
|
+
}));
|
|
2198
|
+
const cell = parsed(settled.message);
|
|
2199
|
+
if (cell._tag === "Success" || attempt >= state.revalidations) {
|
|
2200
|
+
return { contextWindow, answer: settled.message, cell };
|
|
2201
|
+
}
|
|
2202
|
+
yield* emit(new AgentEvent.CellRejectedInFrame({
|
|
2203
|
+
eventType: eventType.cellRejectedInFrame,
|
|
2204
|
+
attempt: attempt + 1,
|
|
2205
|
+
code: cell.failure.code,
|
|
2206
|
+
message: cell.failure.message
|
|
2207
|
+
}));
|
|
2208
|
+
contextWindow = observedOn(contextWindow, settled.message, revalidationNote(cell.failure), deadCellEcho);
|
|
2209
|
+
}
|
|
2210
|
+
});
|
|
2211
|
+
/**
|
|
2212
|
+
* Runs one cell in the run's realm and records the frame it settles.
|
|
2213
|
+
*
|
|
2214
|
+
* Each of the cell's calls resolves as its own keyed boundary through
|
|
2215
|
+
* {@link callHandler}, and each one it settles is observed on the way out, so
|
|
2216
|
+
* what the cell did is returned as values rather than left in variables the
|
|
2217
|
+
* frame's exits read.
|
|
2218
|
+
*/
|
|
2219
|
+
const evaluate = (input, state, produced, engine, sandbox, realm, opened, steering, emit) => Effect.gen(function* () {
|
|
2220
|
+
const cell = produced.source;
|
|
2221
|
+
const descriptors = new Map(input.flows.map((descriptor) => [descriptor.name, descriptor]));
|
|
2222
|
+
const projections = {};
|
|
2223
|
+
for (const descriptor of input.flows)
|
|
2224
|
+
projections[descriptor.name] = Cell.project(descriptor);
|
|
2225
|
+
// Seeded from what earlier frames were handed, appended to as this frame's
|
|
2226
|
+
// calls settle, and carried out through every exit that continues the run.
|
|
2227
|
+
const captures = [...state.truncatedOutputs];
|
|
2228
|
+
const calls = [];
|
|
2229
|
+
/** Ordinals of the invocations that reached the engine this frame. */
|
|
2230
|
+
const performed = new Set();
|
|
2231
|
+
/** Withheld flows this frame's calls named; see {@link callHandler}. */
|
|
2232
|
+
const restored = new Set();
|
|
2233
|
+
/** Calls of this frame that may have written; see {@link callHandler}. */
|
|
2234
|
+
const tree = { writes: 0, chain: undefined };
|
|
2235
|
+
const live = Option.isSome(opened) && opened.value.complete
|
|
2236
|
+
? { tree: opened.value.digest }
|
|
2237
|
+
: { session: state.session, frame: state.frame };
|
|
2238
|
+
// The per-call ceiling this frame enforces, resolved once. It is applied
|
|
2239
|
+
// where the settlement is recorded rather than in the drive loop, so the
|
|
2240
|
+
// number a run armed and the number its journal holds are the same one.
|
|
2241
|
+
const callMs = Sandbox.withDefaults(sandbox.capabilities, input.limits).callMs ?? Sandbox.defaultLimits.callMs;
|
|
2242
|
+
const totalMs = Sandbox.withDefaults(sandbox.capabilities, input.limits).totalMs;
|
|
2243
|
+
let replaying = false;
|
|
2244
|
+
const observing = (invocation) => {
|
|
2245
|
+
const handle = callHandler(state, cell, descriptors, engine, captures, performed, restored, tree, live, callMs, replaying, emit);
|
|
2246
|
+
const pending = engine.record({
|
|
2247
|
+
name: "steering-before-call",
|
|
2248
|
+
identity: {
|
|
2249
|
+
session: state.session,
|
|
2250
|
+
frame: state.frame,
|
|
2251
|
+
boundary: `steering-before-call:${cell.digest}:${invocation.ordinal}`
|
|
2252
|
+
},
|
|
2253
|
+
success: Schema.Boolean,
|
|
2254
|
+
execute: steering.read().pipe(Effect.map((queue) => queue.items.some((item) => item.delivery === "steer")))
|
|
2255
|
+
});
|
|
2256
|
+
// A change during a running call holds all subsequent dispatches. The
|
|
2257
|
+
// cell may finish, then its existing close drain inserts the note. No
|
|
2258
|
+
// notification is inserted into an executing cell or promoted twice.
|
|
2259
|
+
return pending.pipe(Effect.flatMap((blocked) => blocked
|
|
2260
|
+
? Effect.succeed(refusal(undefined, "Outside changes are pending; finish this cell and re-read the changed files in the next turn."))
|
|
2261
|
+
: handle(invocation)), Effect.tap((result) => Effect.sync(() => {
|
|
2262
|
+
const rendered = result.outcome === "success"
|
|
2263
|
+
? JSON.stringify(result.value)
|
|
2264
|
+
: result.message ?? "failed";
|
|
2265
|
+
// The ordinal this call will carry in the run's ledger, computed
|
|
2266
|
+
// here so the salvage line can name it: a summary the next frame
|
|
2267
|
+
// cannot expand is a summary it will pay to re-fetch.
|
|
2268
|
+
const ordinal = CallLedger.settled(state.callLedger) + calls.length + 1;
|
|
2269
|
+
const descriptor = descriptors.get(invocation.flow);
|
|
2270
|
+
const probe = result.outcome === "success" ? invalidProbeOf(result.value) : undefined;
|
|
2271
|
+
// A write the call measured itself, on a tree the host walk does
|
|
2272
|
+
// not cover. It stands beside the declaration, never instead of it.
|
|
2273
|
+
const measured = performed.has(invocation.ordinal) && result.outcome === "success" &&
|
|
2274
|
+
mutatedOf(result.value);
|
|
2275
|
+
// The tree this call actually ran against. A call the boundary
|
|
2276
|
+
// refused ran against nothing, and one whose `at` did not decode
|
|
2277
|
+
// named nothing, so neither keys anything: `performed` is the set
|
|
2278
|
+
// the boundary let through.
|
|
2279
|
+
const ranAt = performed.has(invocation.ordinal) && invocation.at !== undefined
|
|
2280
|
+
? Cell.checkpointOf(invocation.at)
|
|
2281
|
+
: undefined;
|
|
2282
|
+
calls.push({
|
|
2283
|
+
flow: invocation.flow,
|
|
2284
|
+
ok: result.outcome === "success",
|
|
2285
|
+
summary: elide.head(rendered, salvageSummary, salvageRecall),
|
|
2286
|
+
ordinal,
|
|
2287
|
+
mutates: (performed.has(invocation.ordinal) && descriptor !== undefined &&
|
|
2288
|
+
mutating(descriptor, invocation.input)) || measured,
|
|
2289
|
+
remote: measured,
|
|
2290
|
+
// Every invocation the cell issued, refused ones included: a
|
|
2291
|
+
// refusal is still a question the run has already asked, and
|
|
2292
|
+
// asking it again learns nothing either.
|
|
2293
|
+
signature: signatureOf(invocation.flow, invocation.input, ranAt),
|
|
2294
|
+
subject: signatureOf(invocation.flow, invocation.input),
|
|
2295
|
+
at: ranAt,
|
|
2296
|
+
input: invocation.input,
|
|
2297
|
+
value: result.value,
|
|
2298
|
+
message: result.message,
|
|
2299
|
+
invalidProbe: probe,
|
|
2300
|
+
failing: result.outcome === "success" && probe === undefined &&
|
|
2301
|
+
UnresolvedFailure.failed(result.value),
|
|
2302
|
+
passing: result.outcome === "success" && probe === undefined &&
|
|
2303
|
+
UnresolvedFailure.passed(result.value)
|
|
2304
|
+
});
|
|
2305
|
+
})));
|
|
2306
|
+
};
|
|
2307
|
+
// One realm per run. It is the caller's scoped resource, so the loop hands
|
|
2308
|
+
// it the cell and the call handler and nothing else about how a frame runs
|
|
2309
|
+
// depends on where the realm came from.
|
|
2310
|
+
const minted = [];
|
|
2311
|
+
const mint = minter(state, cell, engine, minted, callMs, emit);
|
|
2312
|
+
// The script the model may promote into a saved flow. It is recorded before
|
|
2313
|
+
// the realm evaluates the cell, so a cell that raises is still part of what
|
|
2314
|
+
// the run ran. A host that offers no way to save a flow binds no history and
|
|
2315
|
+
// nothing is kept.
|
|
2316
|
+
const history = yield* Effect.serviceOption(CellHistory.CellHistory);
|
|
2317
|
+
yield* Option.match(history, {
|
|
2318
|
+
onNone: () => Effect.void,
|
|
2319
|
+
onSome: (recorder) => recorder.record(cell.text)
|
|
2320
|
+
});
|
|
2321
|
+
const settled = yield* Effect.gen(function* () {
|
|
2322
|
+
// A null record marks an attempt that has not produced a frame yet.
|
|
2323
|
+
// Skip saved markers until the terminal frame is found, or append a new
|
|
2324
|
+
// marker and evaluate outside any activity. This keeps durable waits and
|
|
2325
|
+
// permission parks out of the frame record's execution context.
|
|
2326
|
+
for (let attempt = 0;; attempt++) {
|
|
2327
|
+
const identity = (slot) => ({
|
|
2328
|
+
session: state.session,
|
|
2329
|
+
frame: state.frame,
|
|
2330
|
+
boundary: `cell-frame:${cell.digest}${slot === 0 ? "" : `:attempt:${slot}`}`
|
|
2331
|
+
});
|
|
2332
|
+
let fresh = false;
|
|
2333
|
+
const recorded = yield* engine.record({
|
|
2334
|
+
name: "cell-frame",
|
|
2335
|
+
identity: identity(attempt),
|
|
2336
|
+
success: Schema.NullOr(RecordedFrame),
|
|
2337
|
+
execute: Effect.sync(() => {
|
|
2338
|
+
fresh = true;
|
|
2339
|
+
return null;
|
|
2340
|
+
})
|
|
2341
|
+
});
|
|
2342
|
+
if (recorded === null && !fresh)
|
|
2343
|
+
continue;
|
|
2344
|
+
if (recorded !== null && recorded.boundary === undefined &&
|
|
2345
|
+
recorded.outcome._tag === "rejected" && recorded.outcome.code === "limit_exceeded") {
|
|
2346
|
+
return yield* Effect.fail(new HarnessError({
|
|
2347
|
+
code: "incompatible_journal",
|
|
2348
|
+
message: "Cannot reconstruct a limited frame without its recorded execution frontier"
|
|
2349
|
+
}));
|
|
2350
|
+
}
|
|
2351
|
+
const replay = recorded !== null && recorded.boundary?.terminal === "timeout"
|
|
2352
|
+
? { boundary: recorded.boundary, outcome: yield* Cell.decodeOutcome(recorded.outcome) }
|
|
2353
|
+
: undefined;
|
|
2354
|
+
replaying = replay !== undefined;
|
|
2355
|
+
// A guarded host parks a frame past its wall-clock limit instead of
|
|
2356
|
+
// recording it. The subject is the frame, not its attempt, so Stop
|
|
2357
|
+
// refuses the re-evaluation here and Continue re-evaluates it with
|
|
2358
|
+
// its settled calls replayed.
|
|
2359
|
+
const guarded = recorded === null && engine.guard !== undefined ? engine.guard : undefined;
|
|
2360
|
+
const subject = JSON.stringify({
|
|
2361
|
+
session: state.session,
|
|
2362
|
+
frame: state.frame,
|
|
2363
|
+
boundary: `cell-frame:${cell.digest}`
|
|
2364
|
+
});
|
|
2365
|
+
if (guarded !== undefined)
|
|
2366
|
+
yield* guarded.admit(subject);
|
|
2367
|
+
const frame = yield* realm.evaluate({
|
|
2368
|
+
cell,
|
|
2369
|
+
// The boundary parse is the only one this cell gets: the realm runs
|
|
2370
|
+
// what it compiled rather than parsing the same text again.
|
|
2371
|
+
program: produced.program,
|
|
2372
|
+
frame: state.frame,
|
|
2373
|
+
// A judged run's catalog can change without a refresh: the run-start
|
|
2374
|
+
// relevance reading withholds flows and a call restores them.
|
|
2375
|
+
flows: input.refreshFlows === undefined && input.judged !== true ? undefined : projections,
|
|
2376
|
+
call: observing,
|
|
2377
|
+
mint,
|
|
2378
|
+
replay,
|
|
2379
|
+
// Settlements are bounded inside their recorded boundaries.
|
|
2380
|
+
bounded: true
|
|
2381
|
+
});
|
|
2382
|
+
if (recorded !== null)
|
|
2383
|
+
return recorded;
|
|
2384
|
+
const outcome = yield* Cell.decodeOutcome(frame.outcome);
|
|
2385
|
+
if (guarded !== undefined && frame.boundary?.terminal === "timeout" &&
|
|
2386
|
+
outcome._tag === "rejected" && outcome.code === "limit_exceeded") {
|
|
2387
|
+
yield* guarded.trip({ source: "cell", subject, limitMillis: totalMs, message: outcome.message });
|
|
2388
|
+
}
|
|
2389
|
+
return yield* engine.record({
|
|
2390
|
+
name: "cell-frame",
|
|
2391
|
+
identity: identity(attempt + 1),
|
|
2392
|
+
success: RecordedFrame,
|
|
2393
|
+
execute: Effect.succeed({ ...frame, outcome })
|
|
2394
|
+
});
|
|
2395
|
+
}
|
|
2396
|
+
});
|
|
2397
|
+
return { frame: settled, calls, minted, captures, restored };
|
|
2398
|
+
});
|
|
2399
|
+
/**
|
|
2400
|
+
* Accepted steering changes the task the completion is judged against. Keep
|
|
2401
|
+
* its provenance separate from the transcript: cell prints are also user
|
|
2402
|
+
* messages, and a compacted summary is not an instruction from the person.
|
|
2403
|
+
* The original task keeps both ends so a long history cannot displace its
|
|
2404
|
+
* newest request. Steering keeps the newest bytes, in admission order.
|
|
2405
|
+
*/
|
|
2406
|
+
const completionTask = (task, instructions) => instructions === ""
|
|
2407
|
+
? task
|
|
2408
|
+
: `${elide.middle(task, 3584, "the run record has the whole task")}
|
|
2409
|
+
|
|
2410
|
+
The person now says:
|
|
2411
|
+
|
|
2412
|
+
Later instructions accepted during this run, oldest first. Apply these changes to the task above; later changes take precedence:
|
|
2413
|
+
|
|
2414
|
+
${instructions}`;
|
|
2415
|
+
/**
|
|
2416
|
+
* Records the frame's steering delivery decision.
|
|
2417
|
+
*
|
|
2418
|
+
* Every exit records one. With no frame left to read a message, keep it pending
|
|
2419
|
+
* at the source instead of acknowledging and discarding it. Replays see the
|
|
2420
|
+
* same decision under the same boundary.
|
|
2421
|
+
*/
|
|
2422
|
+
const drain = (settling, wouldIdle) => Effect.gen(function* () {
|
|
2423
|
+
const { boundary, state } = settling;
|
|
2424
|
+
const drained = yield* settling.engine.record({
|
|
2425
|
+
name: "steering-drain",
|
|
2426
|
+
identity: { session: state.session, frame: state.frame, boundary: `steering-drain:${boundary}` },
|
|
2427
|
+
success: Steering.DrainRecord,
|
|
2428
|
+
execute: Frame.hasNextFrame(state)
|
|
2429
|
+
? settling.steering.drain({ boundary: `${state.frame}:${boundary}`, wouldIdle }).pipe(Effect.map(Steering.drainRecord),
|
|
2430
|
+
// The supervisor's nudge rides the same recorded drain, in a field
|
|
2431
|
+
// of its own and only at a boundary the run continues through: a
|
|
2432
|
+
// boundary that would idle is a completion or the last frame, and a
|
|
2433
|
+
// nudge there would turn a finished answer into a follow-up. Inside
|
|
2434
|
+
// the record, so a replayed boundary delivers what it delivered the
|
|
2435
|
+
// first time. Apart from `inserts`, because those are the person's
|
|
2436
|
+
// words and the completion brake reads them as the task; a nudge is
|
|
2437
|
+
// not. See `internal/supervision`.
|
|
2438
|
+
Effect.flatMap((record) => wouldIdle
|
|
2439
|
+
? Effect.succeed(record)
|
|
2440
|
+
: settling.supervision.take(state.frame, {
|
|
2441
|
+
ledger: state.monitorLedger,
|
|
2442
|
+
shown: state.memoryShown,
|
|
2443
|
+
deliver: settling.input.judged ?? false
|
|
2444
|
+
}).pipe(Effect.map((taken) => ({
|
|
2445
|
+
...record,
|
|
2446
|
+
...(taken.messages.length === 0 ? {} : { supervisor: taken.messages }),
|
|
2447
|
+
...(taken.memory.length === 0 ? {} : { memory: taken.memory }),
|
|
2448
|
+
...(taken.monitor === undefined ? {} : { monitor: taken.monitor }),
|
|
2449
|
+
...(taken.suppressed.length === 0 ? {} : { suppressed: taken.suppressed }),
|
|
2450
|
+
...(taken.ledger === undefined ? {} : { monitorLedger: taken.ledger }),
|
|
2451
|
+
...(taken.marks.length === 0 ? {} : { marks: taken.marks }),
|
|
2452
|
+
...(taken.compact.length === 0 ? {} : { compact: taken.compact })
|
|
2453
|
+
})))),
|
|
2454
|
+
// Executed, not replayed: the one fact that says this frame is the
|
|
2455
|
+
// run's present rather than its past.
|
|
2456
|
+
Effect.tap(() => Effect.sync(() => void (settling.live.value = true))))
|
|
2457
|
+
: Effect.succeed({ inserts: [], seatChanges: [], queued: false })
|
|
2458
|
+
});
|
|
2459
|
+
yield* settling.emit(new AgentEvent.SteeringDrained({
|
|
2460
|
+
eventType: eventType.steeringDrained,
|
|
2461
|
+
messages: drained.inserts,
|
|
2462
|
+
...(drained.supervisor === undefined || drained.supervisor.length === 0
|
|
2463
|
+
? {}
|
|
2464
|
+
: { supervisor: drained.supervisor }),
|
|
2465
|
+
...(drained.monitor === undefined ? {} : { monitor: drained.monitor }),
|
|
2466
|
+
...(drained.suppressed === undefined ? {} : { suppressed: drained.suppressed }),
|
|
2467
|
+
...(drained.memory === undefined ? {} : { memory: drained.memory })
|
|
2468
|
+
}));
|
|
2469
|
+
// The frame is offered to the supervisor from here, after its own take
|
|
2470
|
+
// and only from a live boundary the run continues through, so the
|
|
2471
|
+
// reading it produces is delivered by the boundary after this one and a
|
|
2472
|
+
// replayed frame is never read twice. A boundary that would idle offers
|
|
2473
|
+
// nothing: the run is completing or out of frames.
|
|
2474
|
+
if (settling.live.value && !wouldIdle)
|
|
2475
|
+
yield* settling.offer;
|
|
2476
|
+
return drained;
|
|
2477
|
+
});
|
|
2478
|
+
/** Closes the frame's turn, and states the run's answer when it resolves. */
|
|
2479
|
+
const close = (settling, outcome, output = budgetMessage(settling.state)) => Effect.gen(function* () {
|
|
2480
|
+
yield* settling.emit(new AgentEvent.TurnClosed({
|
|
2481
|
+
eventType: eventType.turnClosed,
|
|
2482
|
+
stopReason: settling.answer.stopReason,
|
|
2483
|
+
outcome
|
|
2484
|
+
}));
|
|
2485
|
+
if (outcome === "resolved") {
|
|
2486
|
+
yield* settling.emit(new AgentEvent.Resolved({
|
|
2487
|
+
eventType: eventType.resolved,
|
|
2488
|
+
message: ModelRequest.Message.assistant(output, { stopReason: "stop" })
|
|
2489
|
+
}));
|
|
2490
|
+
}
|
|
2491
|
+
});
|
|
2492
|
+
/** Closes the turn the way a step ends it, and hands the step on. */
|
|
2493
|
+
const finish = (settling, step) => close(settling, step._tag === "Done" ? "resolved" : step._tag === "Suspend" ? "suspended" : "continue").pipe(Effect.as(step));
|
|
2494
|
+
/**
|
|
2495
|
+
* The one way a frame continues the run: the next frame on `contextWindow`,
|
|
2496
|
+
* carrying the facts the frame measured and whatever its exit changes on top
|
|
2497
|
+
* of them.
|
|
2498
|
+
*
|
|
2499
|
+
* Every transcript segment the frame appended gets its entry in
|
|
2500
|
+
* {@link State.segmentFacts}. The last carries the frame's own pair, and the
|
|
2501
|
+
* person's messages when `person`; any before it are the in-frame re-asks of
|
|
2502
|
+
* cells that never ran.
|
|
2503
|
+
*/
|
|
2504
|
+
const continuing = (settling, contextWindow, person, changes) => {
|
|
2505
|
+
const transcripts = (window) => window.segments.filter((segment) => segment.kind === "transcript").length;
|
|
2506
|
+
const frame = settling.state.frame;
|
|
2507
|
+
const reasks = transcripts(contextWindow) - transcripts(settling.state.contextWindow) - 1;
|
|
2508
|
+
// A drain's marks answer segments the frame opened on; the frame's own
|
|
2509
|
+
// segments go after them, unmarked.
|
|
2510
|
+
const { segmentFacts = settling.state.segmentFacts, ...rest } = changes;
|
|
2511
|
+
// A frame carries no ask unless its own exit says how many it appended, so
|
|
2512
|
+
// the default is zero and the two exits that intervene state their count.
|
|
2513
|
+
const next = advance(settling.state, {
|
|
2514
|
+
frame: frame + 1,
|
|
2515
|
+
interventions: 0,
|
|
2516
|
+
// What this frame's section listed; see `frozen`.
|
|
2517
|
+
ledgerShown: CallLedger.settled(settling.state.callLedger),
|
|
2518
|
+
...settling.facts,
|
|
2519
|
+
contextWindow,
|
|
2520
|
+
segmentFacts: [
|
|
2521
|
+
...segmentFacts,
|
|
2522
|
+
...new Array(reasks).fill({ frame, person: false, mutated: false, checks: [] }),
|
|
2523
|
+
{ frame, person, ...settling.written }
|
|
2524
|
+
],
|
|
2525
|
+
...rest
|
|
2526
|
+
});
|
|
2527
|
+
// The next frame's section, written into the segment this frame appended
|
|
2528
|
+
// before anything reads that segment. See `frozen`.
|
|
2529
|
+
return {
|
|
2530
|
+
_tag: "Continue",
|
|
2531
|
+
state: advance(next, { contextWindow: withSection(next.contextWindow, next), sectioned: true })
|
|
2532
|
+
};
|
|
2533
|
+
};
|
|
2534
|
+
/**
|
|
2535
|
+
* Everything a drain delivers to the model, in the order it reads them: the
|
|
2536
|
+
* supervisor's messages, then the person's.
|
|
2537
|
+
*/
|
|
2538
|
+
const delivered = (drained) => [
|
|
2539
|
+
...(drained.supervisor ?? []),
|
|
2540
|
+
...drained.inserts
|
|
2541
|
+
];
|
|
2542
|
+
/**
|
|
2543
|
+
* Continues the run on the frame's own pair and whatever steering it was sent.
|
|
2544
|
+
*
|
|
2545
|
+
* `ask` is what the frame demands of the next one, when it demands anything.
|
|
2546
|
+
* It goes last, below whatever the drain delivered, because a run answers the
|
|
2547
|
+
* last message it read: an insert after the ask is what gets answered. With
|
|
2548
|
+
* nothing delivered the observation and the ask stay one message, as they
|
|
2549
|
+
* always were.
|
|
2550
|
+
*/
|
|
2551
|
+
const resumed = (settling, drained, changes = {}, text = settling.printed, echo = liveCellEcho, ask) => Effect.gen(function* () {
|
|
2552
|
+
const { state } = settling;
|
|
2553
|
+
const settings = yield* steered(state, drained.seatChanges, settling.input.contextWindowTokensFor);
|
|
2554
|
+
const inserts = delivered(drained);
|
|
2555
|
+
const messages = ask === undefined
|
|
2556
|
+
? [ModelRequest.Message.user(text), ...inserts]
|
|
2557
|
+
: inserts.length === 0
|
|
2558
|
+
? [ModelRequest.Message.user(text === "" ? ask : `${text}\n\n${ask}`)]
|
|
2559
|
+
: [...(text === "" ? [] : [ModelRequest.Message.user(text)]), ...inserts, ModelRequest.Message.user(ask)];
|
|
2560
|
+
const context = appended(settling.contextWindow, settling.answer, messages, echo);
|
|
2561
|
+
return continuing(settling, windowOn(state, settings.seat, context), drained.inserts.length > 0, {
|
|
2562
|
+
...settings,
|
|
2563
|
+
...shownAfter(state, drained),
|
|
2564
|
+
...ledgerAfter(drained),
|
|
2565
|
+
...marksAfter(state, drained),
|
|
2566
|
+
...changes
|
|
2567
|
+
});
|
|
2568
|
+
});
|
|
2569
|
+
/** Records an unusable frame and asks for another, budget permitting. */
|
|
2570
|
+
const observe = (settling, note, changes = {}, echo = liveCellEcho) => Effect.gen(function* () {
|
|
2571
|
+
const next = Frame.hasNextFrame(settling.state);
|
|
2572
|
+
const drained = yield* drain(settling, !next);
|
|
2573
|
+
return next
|
|
2574
|
+
? yield* resumed(settling, drained, changes, settling.printed, echo, note)
|
|
2575
|
+
: { _tag: "Done" };
|
|
2576
|
+
});
|
|
2577
|
+
const frame = (input, engine, sandbox, realm, steering, emit, readCompletion, supervision, taskOf) => Effect.gen(function* () {
|
|
2578
|
+
// Compaction happens before the turn opens, so the digest the turn records
|
|
2579
|
+
// is the one the sealed step is actually keyed on.
|
|
2580
|
+
const compactedState = yield* compacted(input.state, engine, emit, input.judged ?? false, taskOf);
|
|
2581
|
+
// A run's first frame writes its own state section, before anything reads
|
|
2582
|
+
// the window; every later one opens on the section the frame before it
|
|
2583
|
+
// wrote. See `frozen`.
|
|
2584
|
+
const sectioned = compactedState.sectioned
|
|
2585
|
+
? Result.succeed(compactedState.contextWindow)
|
|
2586
|
+
: frozen(compactedState.contextWindow, compactedState);
|
|
2587
|
+
const state = sectioned._tag === "Success"
|
|
2588
|
+
? advance(compactedState, { contextWindow: sectioned.success, sectioned: true })
|
|
2589
|
+
: compactedState;
|
|
2590
|
+
yield* emit(new AgentEvent.TurnOpened({
|
|
2591
|
+
eventType: eventType.turnOpened,
|
|
2592
|
+
seat: state.seat,
|
|
2593
|
+
modelParams: state.modelParams,
|
|
2594
|
+
activeToolNames: [],
|
|
2595
|
+
contextDigest: state.contextWindow.digest
|
|
2596
|
+
}));
|
|
2597
|
+
if (sectioned._tag === "Failure")
|
|
2598
|
+
return yield* Effect.fail(sectioned.failure);
|
|
2599
|
+
const { answer, cell: sealed, contextWindow } = yield* seal(state, engine, emit);
|
|
2600
|
+
const settling = (boundary, printed, facts, written, offer = Effect.void) => ({
|
|
2601
|
+
input,
|
|
2602
|
+
state,
|
|
2603
|
+
engine,
|
|
2604
|
+
steering,
|
|
2605
|
+
supervision,
|
|
2606
|
+
offer,
|
|
2607
|
+
live: { value: false },
|
|
2608
|
+
emit,
|
|
2609
|
+
contextWindow,
|
|
2610
|
+
answer,
|
|
2611
|
+
boundary,
|
|
2612
|
+
printed,
|
|
2613
|
+
facts,
|
|
2614
|
+
written
|
|
2615
|
+
});
|
|
2616
|
+
if (sealed._tag === "Failure") {
|
|
2617
|
+
const rejection = sealed.failure;
|
|
2618
|
+
yield* emit(new AgentEvent.CellSettled({
|
|
2619
|
+
eventType: eventType.cellSettled,
|
|
2620
|
+
cell: "",
|
|
2621
|
+
outcome: rejection
|
|
2622
|
+
}));
|
|
2623
|
+
// A rejected cell is the fourth exit that continues the run, and it is
|
|
2624
|
+
// judged by the same rule as the other three: no cell ran, so this frame
|
|
2625
|
+
// made no call and wrote nothing, and a frame that wrote nothing counts.
|
|
2626
|
+
// Leaving it frozen kept one shape of stall outside the only control
|
|
2627
|
+
// that ends one — a model that answers with prose instead of a cell
|
|
2628
|
+
// advanced no counter at all and spent the whole frame budget doing it.
|
|
2629
|
+
// Nothing to measure and nothing to demand: the cell never reached the
|
|
2630
|
+
// sandbox, so any pending demand is carried forward unanswered.
|
|
2631
|
+
const rejectedFrames = state.readOnlyFrames + 1;
|
|
2632
|
+
if (state.readOnlyCap > 0 && rejectedFrames >= state.readOnlyCap * 2) {
|
|
2633
|
+
return yield* readOnlyCapFailure(state.readOnlyCap, rejectedFrames);
|
|
2634
|
+
}
|
|
2635
|
+
const rejected = settling(contextWindow.digest, "", {
|
|
2636
|
+
truncatedOutputs: TruncatedOutput.retain(state.truncatedOutputs),
|
|
2637
|
+
readOnlyFrames: rejectedFrames
|
|
2638
|
+
}, { mutated: false, checks: [] });
|
|
2639
|
+
const step = yield* observe(rejected, rejection.message, {}, deadCellEcho);
|
|
2640
|
+
return yield* finish(rejected, step);
|
|
2641
|
+
}
|
|
2642
|
+
const cell = sealed.success.source;
|
|
2643
|
+
yield* emit(new AgentEvent.CellProduced({
|
|
2644
|
+
eventType: eventType.cellProduced,
|
|
2645
|
+
cell,
|
|
2646
|
+
blocks: sealed.success.blocks
|
|
2647
|
+
}));
|
|
2648
|
+
// A committed outside change can arrive while the model is producing this
|
|
2649
|
+
// cell. Probe only after it settles; never interrupt a model turn. Journal
|
|
2650
|
+
// the probe as well as the drain, so replay cannot execute a discarded cell
|
|
2651
|
+
// merely because the live notification queue has since been consumed.
|
|
2652
|
+
const pendingBeforeTool = yield* engine.record({
|
|
2653
|
+
name: "steering-before-tool",
|
|
2654
|
+
identity: { session: state.session, frame: state.frame, boundary: `steering-before-tool:${cell.digest}` },
|
|
2655
|
+
success: Schema.Boolean,
|
|
2656
|
+
execute: steering.read().pipe(Effect.map((queue) => queue.items.some((item) => item.delivery === "steer")))
|
|
2657
|
+
});
|
|
2658
|
+
if (pendingBeforeTool) {
|
|
2659
|
+
if (!Frame.hasNextFrame(state)) {
|
|
2660
|
+
return yield* new HarnessError({
|
|
2661
|
+
code: "engine_failed",
|
|
2662
|
+
message: "Outside changes require another model turn before tools can run"
|
|
2663
|
+
});
|
|
2664
|
+
}
|
|
2665
|
+
const beforeTool = settling(`before-tool:${cell.digest}`, "", {}, { mutated: false, checks: [] });
|
|
2666
|
+
const drained = yield* drain(beforeTool, false);
|
|
2667
|
+
if (carries(drained)) {
|
|
2668
|
+
const step = yield* resumed(beforeTool, drained, {}, "", deadCellEcho);
|
|
2669
|
+
return yield* finish(beforeTool, step);
|
|
2670
|
+
}
|
|
2671
|
+
}
|
|
2672
|
+
// The prior close predates the model wait. Measure again before this
|
|
2673
|
+
// frame so another worker's edits neither reuse stale reads nor count as
|
|
2674
|
+
// this frame's own mutations. A missing or bounded observer stays so.
|
|
2675
|
+
const opened = state.workspace !== undefined &&
|
|
2676
|
+
(Option.isNone(state.workspace) || !state.workspace.value.complete)
|
|
2677
|
+
? state.workspace
|
|
2678
|
+
: yield* witness(engine, state, cell.digest, "open");
|
|
2679
|
+
const ran = yield* evaluate(input, state, sealed.success, engine, sandbox, realm, opened, steering, emit);
|
|
2680
|
+
const outcome = yield* Cell.decodeOutcome(ran.frame.outcome);
|
|
2681
|
+
const printed = printsObservation(ran.frame.prints);
|
|
2682
|
+
yield* emit(new AgentEvent.CellPrinted({
|
|
2683
|
+
eventType: eventType.cellPrinted,
|
|
2684
|
+
cell: cell.digest,
|
|
2685
|
+
text: ran.frame.prints
|
|
2686
|
+
}));
|
|
2687
|
+
// A partial prefix cannot provide a workspace digest. Carry it forward
|
|
2688
|
+
// without paying for another walk on this or later frames.
|
|
2689
|
+
const closed = Option.isSome(opened) && !opened.value.complete
|
|
2690
|
+
? opened
|
|
2691
|
+
: yield* witness(engine, state, cell.digest, "close");
|
|
2692
|
+
yield* emit(new AgentEvent.CellSettled({
|
|
2693
|
+
eventType: eventType.cellSettled,
|
|
2694
|
+
cell: cell.digest,
|
|
2695
|
+
outcome,
|
|
2696
|
+
...(ran.frame.boundary === undefined ? {} : { boundary: ran.frame.boundary })
|
|
2697
|
+
}));
|
|
2698
|
+
// The frame's own record of what it did to the world, computed once and
|
|
2699
|
+
// carried out through every exit: a raise, a refused park and a settled
|
|
2700
|
+
// transition all continue the run on the same facts.
|
|
2701
|
+
const accounting = Frame.account({
|
|
2702
|
+
state,
|
|
2703
|
+
calls: ran.calls,
|
|
2704
|
+
opened,
|
|
2705
|
+
closed,
|
|
2706
|
+
minted: ran.minted,
|
|
2707
|
+
bindings: ran.frame.bindings,
|
|
2708
|
+
captures: ran.captures,
|
|
2709
|
+
source: cell.text
|
|
2710
|
+
});
|
|
2711
|
+
yield* emit(accounting.observed);
|
|
2712
|
+
// The supervisor's snapshot of this frame: what the frame wrote, what it
|
|
2713
|
+
// printed, how it ended, and the counts the deterministic controls keep.
|
|
2714
|
+
// Built here, on the loop's fiber, and offered by the frame's live
|
|
2715
|
+
// boundary; see `drain` for when, and `Supervisor` for what it is for.
|
|
2716
|
+
const written = Supervisor.prose(assistantText(answer));
|
|
2717
|
+
const { mutated } = accounting;
|
|
2718
|
+
const { readOnlyFrames } = accounting.facts;
|
|
2719
|
+
const failed = ran.calls.length > 0 && ran.calls.every((call) => !call.ok || call.failing);
|
|
2720
|
+
const failure = ran.calls.some((call) => call.ok && !call.failing)
|
|
2721
|
+
? ""
|
|
2722
|
+
: outcome._tag === "raised"
|
|
2723
|
+
? `cell threw ${outcome.name}: ${outcome.message}`
|
|
2724
|
+
: failed
|
|
2725
|
+
? ran.calls.map((call) => `${call.flow}: ${call.summary}`).join("; ")
|
|
2726
|
+
: "";
|
|
2727
|
+
const failureKey = failure.slice(0, 1_000);
|
|
2728
|
+
const failureFrames = failureKey === "" ? 0 : failureKey === state.failureKey ? state.failureFrames + 1 : 1;
|
|
2729
|
+
const facts = ran.restored.size === 0 ? accounting.facts : {
|
|
2730
|
+
...accounting.facts,
|
|
2731
|
+
withheldFlows: state.withheldFlows.filter((name) => !ran.restored.has(name))
|
|
2732
|
+
};
|
|
2733
|
+
// Read from the catalog this frame showed, so a skill the relevance gate
|
|
2734
|
+
// withheld is never reminded; a direct call restores it.
|
|
2735
|
+
const { capped: skillsCapped, ...offered } = Supervision.catalog(input.flows, accounting.facts.callLedger);
|
|
2736
|
+
const exit = settling(cell.digest, printed, { ...facts, failureFrames, failureKey }, { mutated, checks: accounting.frameChecks.map((check) => check.label) }, supervision.offer({
|
|
2737
|
+
frame: state.frame,
|
|
2738
|
+
digest: cell.digest,
|
|
2739
|
+
current: {
|
|
2740
|
+
frame: state.frame,
|
|
2741
|
+
cell: Supervisor.head(cell.text),
|
|
2742
|
+
prose: Supervisor.head(written),
|
|
2743
|
+
printed: Supervisor.tail(ran.frame.prints),
|
|
2744
|
+
transition: outcome._tag === "settled" ? outcome.transition._tag : outcome._tag,
|
|
2745
|
+
mutated
|
|
2746
|
+
},
|
|
2747
|
+
snapshot: {
|
|
2748
|
+
task: Judgement.task(taskOf(contextWindow)),
|
|
2749
|
+
signals: Supervision.signals(state, accounting.facts, accounting.workspaceDigest, accounting.observed.paths),
|
|
2750
|
+
...offered
|
|
2751
|
+
},
|
|
2752
|
+
shown: state.memoryShown,
|
|
2753
|
+
recent: Supervisor.head(written),
|
|
2754
|
+
skillsCapped,
|
|
2755
|
+
unmarked: input.judged === true ? unmarked(state) : [],
|
|
2756
|
+
failing: accounting.facts.checks.filter((check) => check.failing).map((check) => check.label)
|
|
2757
|
+
}));
|
|
2758
|
+
// Read-only discipline is armed for the whole frame, not for one exit: a
|
|
2759
|
+
// raise, a refused park and a settled transition all continue the run, and
|
|
2760
|
+
// each of them judges the frame by this cap.
|
|
2761
|
+
const cap = state.readOnlyCap;
|
|
2762
|
+
if (outcome._tag !== "settled") {
|
|
2763
|
+
// The frame settled no transition, but it either changed the workspace
|
|
2764
|
+
// or it did not, and that is the whole of what the streak counts.
|
|
2765
|
+
// Freezing the counter here is how a stall hid from the cap: a run that
|
|
2766
|
+
// alternates a raise with a read advances the streak once every two
|
|
2767
|
+
// frames, so a cap of twelve took twenty-four frames to reach and a run
|
|
2768
|
+
// that raised more often than it read never reached it at all. Wave 7's
|
|
2769
|
+
// own report reads the drift the other way round — the journal's
|
|
2770
|
+
// mutation streak and this counter disagreed by one per raise — and the
|
|
2771
|
+
// instance that spent its whole budget raised twice.
|
|
2772
|
+
// A demand is answered by a write or by a justification, and a frame
|
|
2773
|
+
// that settled no transition produced neither. So a raise leaves the
|
|
2774
|
+
// demand pending — its text is still in the transcript this exit appends
|
|
2775
|
+
// to — and the journal names the frame that finally answers it.
|
|
2776
|
+
if (state.pendingReadOnlyDemand !== undefined && mutated) {
|
|
2777
|
+
yield* emit(new AgentEvent.ReadOnlyDemanded({
|
|
2778
|
+
eventType: eventType.readOnlyDemanded,
|
|
2779
|
+
streak: state.pendingReadOnlyDemand.streak,
|
|
2780
|
+
cap: state.pendingReadOnlyDemand.cap,
|
|
2781
|
+
nextFrame: state.frame,
|
|
2782
|
+
nextAction: "write"
|
|
2783
|
+
}));
|
|
2784
|
+
}
|
|
2785
|
+
if (cap > 0 && readOnlyFrames >= cap * 2) {
|
|
2786
|
+
return yield* readOnlyCapFailure(cap, readOnlyFrames);
|
|
2787
|
+
}
|
|
2788
|
+
// The flow name is clipped for the same reason `CallLedger` clips it: a
|
|
2789
|
+
// call names whatever string the cell passed to `ctx.call`, a name that
|
|
2790
|
+
// matches no descriptor still settles as a failure saying so, and an
|
|
2791
|
+
// unbounded name here is an unbounded line in the notice.
|
|
2792
|
+
const salvage = ran.calls.length === 0
|
|
2793
|
+
? ""
|
|
2794
|
+
: `\nCalls this cell already completed (their results are durable, and each one is still under the name your cell bound it to — read the name instead of redoing the work):\n${ran.calls.map((call) => `- ${call.ordinal}. ${clip(call.flow, CallLedger.width)} -> ${call.ok ? "ok" : "FAILED"}: ${call.summary}`)
|
|
2795
|
+
.join("\n")}`;
|
|
2796
|
+
const alert = accounting.probeNotice === undefined ? "" : `\n\n${accounting.probeNotice}`;
|
|
2797
|
+
// A property read through an absent path is the one throw the harness can
|
|
2798
|
+
// diagnose without guessing, and it is the throw a cell reaching into
|
|
2799
|
+
// a realm binding makes. The names are already computed for the variables
|
|
2800
|
+
// panel, so naming them costs nothing and closes the loop the model would
|
|
2801
|
+
// otherwise spend a frame on. See `VariablesPanel`.
|
|
2802
|
+
const missed = outcome._tag === "raised" ? bindingPathMiss(outcome.message, ran.frame.bindings) : undefined;
|
|
2803
|
+
const note = outcome._tag === "raised"
|
|
2804
|
+
? `The cell threw ${outcome.name}: ${outcome.message}. Emit a corrected cell.${missed === undefined ? "" : `\n${missed}`}${salvage}${alert}`
|
|
2805
|
+
: `${outcome.message}${salvage}${alert}`;
|
|
2806
|
+
const step = yield* observe(exit, note);
|
|
2807
|
+
return yield* finish(exit, step);
|
|
2808
|
+
}
|
|
2809
|
+
const transition = outcome.transition;
|
|
2810
|
+
yield* emit(new AgentEvent.TransitionApplied({
|
|
2811
|
+
eventType: eventType.transitionApplied,
|
|
2812
|
+
transition
|
|
2813
|
+
}));
|
|
2814
|
+
if (transition._tag === "park") {
|
|
2815
|
+
if (state.pendingReadOnlyDemand !== undefined) {
|
|
2816
|
+
yield* emit(new AgentEvent.ReadOnlyDemanded({
|
|
2817
|
+
eventType: eventType.readOnlyDemanded,
|
|
2818
|
+
streak: state.pendingReadOnlyDemand.streak,
|
|
2819
|
+
cap: state.pendingReadOnlyDemand.cap,
|
|
2820
|
+
nextFrame: state.frame,
|
|
2821
|
+
nextAction: mutated ? "write" : "park"
|
|
2822
|
+
}));
|
|
2823
|
+
}
|
|
2824
|
+
if (!state.approvalChannel) {
|
|
2825
|
+
return yield* new HarnessError({
|
|
2826
|
+
code: "approval_unavailable",
|
|
2827
|
+
message: `The run requires an answer but has no approval channel: ${transition.message}`
|
|
2828
|
+
});
|
|
2829
|
+
}
|
|
2830
|
+
// No frame remains to consume an answer. Keep pending steering at its
|
|
2831
|
+
// source and suspend with the question instead of reporting completion.
|
|
2832
|
+
if (!Frame.hasNextFrame(state)) {
|
|
2833
|
+
return yield* finish(exit, {
|
|
2834
|
+
_tag: "Suspend",
|
|
2835
|
+
reason: new EngineLike.SuspendReason({
|
|
2836
|
+
code: transition.reason,
|
|
2837
|
+
message: transition.message
|
|
2838
|
+
})
|
|
2839
|
+
});
|
|
2840
|
+
}
|
|
2841
|
+
// The drain on the park path, and the only thing that can ever answer an
|
|
2842
|
+
// honored park.
|
|
2843
|
+
//
|
|
2844
|
+
// A parked run resumes by re-executing its own frames, so it reaches this
|
|
2845
|
+
// park again with the same cell, the same transition, and the same
|
|
2846
|
+
// question. One boundary per park would hand every later attempt the
|
|
2847
|
+
// first attempt's empty answer, the run would park again, and the
|
|
2848
|
+
// operator's message would sit in the durable queue for the life of the
|
|
2849
|
+
// run — woken, replayed, re-parked, forever. That is what makes
|
|
2850
|
+
// `waiting-input` unanswerable today.
|
|
2851
|
+
//
|
|
2852
|
+
// So the park has a LADDER of boundaries and walks it. Every rung the
|
|
2853
|
+
// queue has answered before hands back exactly what it handed back then,
|
|
2854
|
+
// which is what keeps a replay identical; the walk stops at the first
|
|
2855
|
+
// rung this run has never consulted, which is the one read a resumed
|
|
2856
|
+
// attempt is entitled to perform for real. One rung is consumed per
|
|
2857
|
+
// attempt, so a steer admitted while the run was parked is delivered by
|
|
2858
|
+
// the rung the resume reaches, and every attempt after that replays the
|
|
2859
|
+
// delivery rather than draining a queue that no longer holds it.
|
|
2860
|
+
//
|
|
2861
|
+
// The queue is the record here, and deliberately: `EngineLike.record`
|
|
2862
|
+
// would freeze the rung's answer, and a frozen `duplicate` tells every
|
|
2863
|
+
// later attempt it was the first, which is the loop this is closing.
|
|
2864
|
+
const answered = yield* Effect.gen(function* () {
|
|
2865
|
+
for (let rung = 0;; rung++) {
|
|
2866
|
+
const drained = yield* steering.drain({
|
|
2867
|
+
boundary: `${state.frame}:${cell.digest}:park:${rung}`,
|
|
2868
|
+
// A park IS the run going idle, which is the condition a queued
|
|
2869
|
+
// follow-up was admitted to wait for. The continue path passes
|
|
2870
|
+
// false because a continuing run is not idle; this is the boundary
|
|
2871
|
+
// that owes those messages their delivery.
|
|
2872
|
+
wouldIdle: true,
|
|
2873
|
+
// What the park asks, so a source with a person on the other end
|
|
2874
|
+
// can put the question to them and answer this drain.
|
|
2875
|
+
park: { reason: transition.reason, message: transition.message }
|
|
2876
|
+
});
|
|
2877
|
+
const record = Steering.drainRecord(drained);
|
|
2878
|
+
if (carries(record) || !drained.duplicate)
|
|
2879
|
+
return record;
|
|
2880
|
+
}
|
|
2881
|
+
});
|
|
2882
|
+
if (carries(answered)) {
|
|
2883
|
+
yield* emit(new AgentEvent.SteeringDrained({
|
|
2884
|
+
eventType: eventType.steeringDrained,
|
|
2885
|
+
messages: answered.inserts
|
|
2886
|
+
}));
|
|
2887
|
+
// The park was answered, so the run carries on rather than waiting for
|
|
2888
|
+
// an answer it has already been given. The frame is judged as an
|
|
2889
|
+
// honored park still is — waiting is not evasion — so the read-only
|
|
2890
|
+
// streak is carried rather than advanced.
|
|
2891
|
+
const step = yield* resumed(exit, answered, {
|
|
2892
|
+
pendingReadOnlyDemand: undefined,
|
|
2893
|
+
readOnlyFrames: state.readOnlyFrames
|
|
2894
|
+
});
|
|
2895
|
+
return yield* finish(exit, step);
|
|
2896
|
+
}
|
|
2897
|
+
return yield* finish(exit, {
|
|
2898
|
+
_tag: "Suspend",
|
|
2899
|
+
reason: new EngineLike.SuspendReason({
|
|
2900
|
+
code: transition.reason,
|
|
2901
|
+
message: transition.message
|
|
2902
|
+
})
|
|
2903
|
+
});
|
|
2904
|
+
}
|
|
2905
|
+
// Read-only discipline, applied to every frame that settled a decision.
|
|
2906
|
+
// An honored park is exempt: waiting is not evasion, and a parked run is
|
|
2907
|
+
// not reporting anything as done. A refused park is not exempt — it
|
|
2908
|
+
// continues the run — and the branch above applies the same rule to it.
|
|
2909
|
+
if (state.pendingReadOnlyDemand !== undefined) {
|
|
2910
|
+
yield* emit(new AgentEvent.ReadOnlyDemanded({
|
|
2911
|
+
eventType: eventType.readOnlyDemanded,
|
|
2912
|
+
streak: state.pendingReadOnlyDemand.streak,
|
|
2913
|
+
cap: state.pendingReadOnlyDemand.cap,
|
|
2914
|
+
nextFrame: state.frame,
|
|
2915
|
+
nextAction: mutated
|
|
2916
|
+
? "write"
|
|
2917
|
+
: (transition._tag === "continue" && (transition.justification ?? "").trim().length > 0)
|
|
2918
|
+
? "justification"
|
|
2919
|
+
: "read-only"
|
|
2920
|
+
}));
|
|
2921
|
+
}
|
|
2922
|
+
if (cap > 0 && readOnlyFrames >= cap * 2) {
|
|
2923
|
+
return yield* readOnlyCapFailure(cap, readOnlyFrames);
|
|
2924
|
+
}
|
|
2925
|
+
// `VacuousVerification` was read here, between the read-only cap and the
|
|
2926
|
+
// completion branch, and it is not read anywhere now. The module, its
|
|
2927
|
+
// tests and `AgentEvent.VacuousVerificationObserved` are kept; the arm is
|
|
2928
|
+
// off. the r93 wave report is the reason and the module's
|
|
2929
|
+
// own docblock carries it. Re-wiring is one `stored`/`find` pair here plus
|
|
2930
|
+
// the two `State` fields it needs, and it is a controlled arm of its own
|
|
2931
|
+
// wave when it happens — not a change that rides along with another.
|
|
2932
|
+
const drained = yield* drain(exit, transition._tag === "complete" || !Frame.hasNextFrame(state));
|
|
2933
|
+
if (transition._tag === "complete" && carries(drained)) {
|
|
2934
|
+
const step = yield* resumed(exit, drained, { bouncedCompletion: transition.output }, `${printed}\n\nThe completed answer before this follow-up:\n${transition.output}`);
|
|
2935
|
+
return yield* finish(exit, step);
|
|
2936
|
+
}
|
|
2937
|
+
if (transition._tag === "complete" && input.output !== undefined &&
|
|
2938
|
+
state.outputDemands < input.output.cap && Frame.handBackRoom(state, readOnlyFrames)) {
|
|
2939
|
+
// Ahead of the evidence demands and the claim brake: an answer the host
|
|
2940
|
+
// cannot read is corrected before anything spends a demand judging it.
|
|
2941
|
+
const check = input.output.check;
|
|
2942
|
+
const note = yield* engine.record({
|
|
2943
|
+
name: "output-judgement",
|
|
2944
|
+
identity: {
|
|
2945
|
+
session: state.session,
|
|
2946
|
+
frame: state.frame,
|
|
2947
|
+
boundary: `output-judgement:${cell.digest}`
|
|
2948
|
+
},
|
|
2949
|
+
success: Schema.NullOr(Schema.String),
|
|
2950
|
+
execute: Effect.map(Effect.suspend(() => check(transition.output, state.outputDemands, transition.value)), (refused) => refused ?? null)
|
|
2951
|
+
});
|
|
2952
|
+
if (note !== null) {
|
|
2953
|
+
if (exit.live.value)
|
|
2954
|
+
yield* exit.offer;
|
|
2955
|
+
yield* emit(new AgentEvent.OutputDemanded({
|
|
2956
|
+
eventType: eventType.outputDemanded,
|
|
2957
|
+
note,
|
|
2958
|
+
nextFrame: state.frame + 1
|
|
2959
|
+
}));
|
|
2960
|
+
yield* close(exit, "continue");
|
|
2961
|
+
// Shown what its own cell printed, as a continuing frame is, so the
|
|
2962
|
+
// correction never costs the work that produced the answer.
|
|
2963
|
+
return continuing(exit, appended(contextWindow, answer, [ModelRequest.Message.user(printed), ModelRequest.Message.user(note)], liveCellEcho), false, {
|
|
2964
|
+
interventions: 1,
|
|
2965
|
+
pendingReadOnlyDemand: undefined,
|
|
2966
|
+
outputDemands: state.outputDemands + 1,
|
|
2967
|
+
// An answer the host could not read is not worth restoring.
|
|
2968
|
+
bouncedCompletion: undefined
|
|
2969
|
+
});
|
|
2970
|
+
}
|
|
2971
|
+
}
|
|
2972
|
+
if (transition._tag === "complete") {
|
|
2973
|
+
// The completion's own evidence, judged once per demand; see
|
|
2974
|
+
// `Frame.judgeCompletion` for the five demands, their order, and the
|
|
2975
|
+
// three things that leave no frame to ask in.
|
|
2976
|
+
const services = yield* Effect.context();
|
|
2977
|
+
const judged = yield* engine.record({
|
|
2978
|
+
name: "completion-judgement",
|
|
2979
|
+
identity: {
|
|
2980
|
+
session: state.session,
|
|
2981
|
+
frame: state.frame,
|
|
2982
|
+
boundary: `completion-judgement:${cell.digest}`
|
|
2983
|
+
},
|
|
2984
|
+
success: RecordedCompletion,
|
|
2985
|
+
// The claim brake's reading, whole and by sentence, is paid model
|
|
2986
|
+
// work: the run's budget accounts it like any sealed step (#2681). A
|
|
2987
|
+
// bounce carries the reading as its demand and leaves `observed` empty.
|
|
2988
|
+
usage: (judgement) => paidUsage([
|
|
2989
|
+
(judgement.observed ?? (judgement.demand?.event._tag === "claim-demanded" ? judgement.demand.event : null))
|
|
2990
|
+
?.usage
|
|
2991
|
+
]),
|
|
2992
|
+
// A replay uses the entire original decision. Re-evaluating even an
|
|
2993
|
+
// accepted claim can invent a demand and execute additional work.
|
|
2994
|
+
execute: Frame.judgeCompletion(state, accounting, contextWindow, transition.output, readCompletion).pipe(Effect.provideContext(services), Effect.map((judgement) => ({
|
|
2995
|
+
observed: judgement.observed ?? null,
|
|
2996
|
+
demand: judgement.demand ?? null,
|
|
2997
|
+
unproven: judgement.unproven ?? null,
|
|
2998
|
+
decision: judgement.decision ?? null,
|
|
2999
|
+
sentenceDecision: judgement.sentenceDecision ?? null,
|
|
3000
|
+
unfinishedDecision: judgement.unfinishedDecision ?? null
|
|
3001
|
+
})))
|
|
3002
|
+
}).pipe(Effect.map((judgement) => ({
|
|
3003
|
+
observed: judgement.observed ?? undefined,
|
|
3004
|
+
demand: judgement.demand ?? undefined,
|
|
3005
|
+
unproven: judgement.unproven ?? undefined,
|
|
3006
|
+
decision: judgement.decision ?? undefined,
|
|
3007
|
+
sentenceDecision: judgement.sentenceDecision ?? undefined,
|
|
3008
|
+
unfinishedDecision: judgement.unfinishedDecision ?? undefined
|
|
3009
|
+
})));
|
|
3010
|
+
// The claim brake's reading when it issued no demand. It is the one
|
|
3011
|
+
// demand whose non-demanding readings are journaled, because it is the
|
|
3012
|
+
// one a grader cannot recompute; see `AgentEvent.ClaimDemanded`. It is
|
|
3013
|
+
// emitted before the failure below, so the reading that ended the run
|
|
3014
|
+
// is on the record the run leaves behind.
|
|
3015
|
+
if (judged.observed !== undefined)
|
|
3016
|
+
yield* emit(judged.observed);
|
|
3017
|
+
// The same reading with its evidence, from the same record, so a
|
|
3018
|
+
// replayed frame reports the decision the original attempt made. It
|
|
3019
|
+
// follows `claim-demanded` on every path but the bounce, where that
|
|
3020
|
+
// event is the demand's own and is emitted below.
|
|
3021
|
+
if (judged.decision !== undefined)
|
|
3022
|
+
yield* emit(judged.decision);
|
|
3023
|
+
if (judged.sentenceDecision !== undefined)
|
|
3024
|
+
yield* emit(judged.sentenceDecision);
|
|
3025
|
+
if (judged.unfinishedDecision !== undefined)
|
|
3026
|
+
yield* emit(judged.unfinishedDecision);
|
|
3027
|
+
// An unproven claim with no bounce left to spend, or a completion that
|
|
3028
|
+
// reports its own work unfinished. The run ends here the way
|
|
3029
|
+
// `read_only_cap` ends one, rather than returning a sentence its own
|
|
3030
|
+
// record contradicts or settling unfinished work as completed; see
|
|
3031
|
+
// `CompletionClaim.unproven` and `UnfinishedWork`.
|
|
3032
|
+
if (judged.unproven !== undefined)
|
|
3033
|
+
return yield* Effect.fail(judged.unproven);
|
|
3034
|
+
const demanded = judged.demand;
|
|
3035
|
+
if (demanded !== undefined) {
|
|
3036
|
+
// Handed back, so the run continues and the frame is supervised
|
|
3037
|
+
// after all, from the same live boundary rule `drain` applies: a
|
|
3038
|
+
// replayed judgement replays a replayed drain, and offers nothing.
|
|
3039
|
+
if (exit.live.value)
|
|
3040
|
+
yield* exit.offer;
|
|
3041
|
+
yield* emit(demanded.event);
|
|
3042
|
+
yield* close(exit, "continue");
|
|
3043
|
+
// The demand is an in-frame observation appended to what the run was
|
|
3044
|
+
// already holding, not a projected context: a completion names no
|
|
3045
|
+
// context for a next frame, and a run answering this one needs the
|
|
3046
|
+
// frame it just wrote. A demand follows a drain that carried nothing.
|
|
3047
|
+
//
|
|
3048
|
+
// A completion written before its own calls returned is shown what
|
|
3049
|
+
// they printed first, the way a continuing frame is: reading them is
|
|
3050
|
+
// the whole of what the demand asks. See `UnobservedCall`.
|
|
3051
|
+
const demandedWindow = demanded.event._tag === "unobserved-demanded"
|
|
3052
|
+
? appended(contextWindow, answer, [ModelRequest.Message.user(printed), ModelRequest.Message.user(demanded.note)], liveCellEcho)
|
|
3053
|
+
: observedOn(contextWindow, answer, demanded.note, liveCellEcho);
|
|
3054
|
+
return continuing(exit, demandedWindow, false, {
|
|
3055
|
+
// The note is the ask this frame appended; the run's memory goes
|
|
3056
|
+
// above it. See `frozen`.
|
|
3057
|
+
interventions: 1,
|
|
3058
|
+
pendingReadOnlyDemand: undefined,
|
|
3059
|
+
...demanded.spent,
|
|
3060
|
+
// The answer the demand is taking away, kept so it cannot be lost.
|
|
3061
|
+
// A frame was reserved for the run to answer in, but nothing makes
|
|
3062
|
+
// that frame end in a completion, and a run that spends it and
|
|
3063
|
+
// then runs out of budget would end on the budget notice with its
|
|
3064
|
+
// own answer discarded. See {@link budgetMessage}.
|
|
3065
|
+
//
|
|
3066
|
+
// A claim the brake read and found unsupported is not kept: the
|
|
3067
|
+
// budget notice would hand back the very sentence the brake
|
|
3068
|
+
// refused. See `Frame.CompletionDemand.keeps`.
|
|
3069
|
+
bouncedCompletion: demanded.keeps ? transition.output : undefined,
|
|
3070
|
+
// The frame this demand was handed to, which is the frame that
|
|
3071
|
+
// gets to answer it without being judged again. See
|
|
3072
|
+
// {@link State.demandedFrame}.
|
|
3073
|
+
demandedFrame: state.frame + 1
|
|
3074
|
+
});
|
|
3075
|
+
}
|
|
3076
|
+
// A completion that stands over a call its own cell failed carries the
|
|
3077
|
+
// failure in the flow's words; see `FailedCall.state`.
|
|
3078
|
+
yield* close(exit, "resolved", FailedCall.state(transition.output, FailedCall.find(accounting.calls, transition.output, accounting.source)));
|
|
3079
|
+
return { _tag: "Done" };
|
|
3080
|
+
}
|
|
3081
|
+
if (!Frame.hasNextFrame(state)) {
|
|
3082
|
+
yield* close(exit, "resolved");
|
|
3083
|
+
return { _tag: "Done" };
|
|
3084
|
+
}
|
|
3085
|
+
yield* close(exit, "continue");
|
|
3086
|
+
const { contextWindowTokens, modelCallMs, modelParams, seat } = yield* steered(state, drained.seatChanges, input.contextWindowTokensFor);
|
|
3087
|
+
// The read-only, repeat and sufficiency interventions; see
|
|
3088
|
+
// `Frame.discipline` for when each is issued and what it costs.
|
|
3089
|
+
const disciplined = Frame.discipline(state, accounting, transition.justification);
|
|
3090
|
+
for (const event of disciplined.events)
|
|
3091
|
+
yield* emit(event);
|
|
3092
|
+
// The frame's own pair, appended. The append is what keeps the provider's
|
|
3093
|
+
// prefix cache covering everything below the last frame — a transcript the
|
|
3094
|
+
// cell replaced broke the prefix on every turn, and one graded wave recorded
|
|
3095
|
+
// zero cached input tokens on an instance where an earlier wave had 5,142.
|
|
3096
|
+
const context = appended(contextWindow, answer, [ModelRequest.Message.user(printed), ...delivered(drained), ...disciplined.messages], liveCellEcho);
|
|
3097
|
+
return continuing(exit, windowOn(state, seat, context), drained.inserts.length > 0, {
|
|
3098
|
+
seat,
|
|
3099
|
+
modelParams,
|
|
3100
|
+
modelCallMs,
|
|
3101
|
+
contextWindowTokens,
|
|
3102
|
+
...shownAfter(state, drained),
|
|
3103
|
+
...ledgerAfter(drained),
|
|
3104
|
+
...marksAfter(state, drained),
|
|
3105
|
+
// Whatever this frame earned is what the next one answers, so the run's
|
|
3106
|
+
// memory goes above it. See `frozen`.
|
|
3107
|
+
interventions: disciplined.messages.length,
|
|
3108
|
+
...disciplined.changes
|
|
3109
|
+
});
|
|
3110
|
+
});
|
|
3111
|
+
/** What the run-start relevance reading records. */
|
|
3112
|
+
const RelevanceRecord = Schema.Struct({ settled: AgentEvent.RelevanceSettled });
|
|
3113
|
+
/**
|
|
3114
|
+
* The run-start relevance reading, taken once at frame 0 of a judged run.
|
|
3115
|
+
*
|
|
3116
|
+
* Every model-invocable flow and skill of the frame's catalog that is not
|
|
3117
|
+
* pinned, every chunk of the instruction files, and every opening memory row
|
|
3118
|
+
* is one item. What Jev is confident the task does not need is withheld: its
|
|
3119
|
+
* flows leave `ctx.flows`, its chunks leave the instructions segment, and its
|
|
3120
|
+
* rows leave the memory segment. A reading that could not be judged withholds
|
|
3121
|
+
* nothing, and its `decision-unjudged` row is the record.
|
|
3122
|
+
* The reading is a recorded boundary, so a replay is served what it withheld.
|
|
3123
|
+
*/
|
|
3124
|
+
const withheld = (state, catalog, input, engine, emit, taskOf) => Effect.gen(function* () {
|
|
3125
|
+
const pinned = input.pinned ?? [];
|
|
3126
|
+
const documents = input.instructions ?? [];
|
|
3127
|
+
const rows = input.memory?.rows ?? [];
|
|
3128
|
+
const items = [
|
|
3129
|
+
...catalog
|
|
3130
|
+
.filter((descriptor) => descriptor.modelInvocable && !pinned.includes(descriptor.name))
|
|
3131
|
+
.map(Relevance.flowItem)
|
|
3132
|
+
// Catalog names are unique, so no two items tie.
|
|
3133
|
+
.sort((left, right) => left.id < right.id ? -1 : 1),
|
|
3134
|
+
...Relevance.chunks(documents).map((chunk) => ({
|
|
3135
|
+
kind: "instruction",
|
|
3136
|
+
id: chunk.id,
|
|
3137
|
+
text: chunk.text
|
|
3138
|
+
})),
|
|
3139
|
+
...rows.map((row) => ({ kind: "memory", id: row.key, text: row.text }))
|
|
3140
|
+
];
|
|
3141
|
+
if (items.length === 0)
|
|
3142
|
+
return state;
|
|
3143
|
+
const record = yield* Judgement.recorded(engine, {
|
|
3144
|
+
name: "relevance",
|
|
3145
|
+
identity: { session: state.session, frame: 0, boundary: "relevance" },
|
|
3146
|
+
classifier: "relevance/unnecessary",
|
|
3147
|
+
value: RelevanceRecord,
|
|
3148
|
+
items: items.length
|
|
3149
|
+
}, Effect.map(Relevance.judge({ task: Judgement.task(taskOf(state.contextWindow)) }, items), (reading) => ({
|
|
3150
|
+
value: { settled: Relevance.settled(reading, { scope: state.session, frame: 0, source: "run" }) },
|
|
3151
|
+
asked: reading.asked,
|
|
3152
|
+
acted: reading.verdicts.some((verdict) => verdict.withheld)
|
|
3153
|
+
})));
|
|
3154
|
+
yield* Judgement.emitRecorded(emit, record);
|
|
3155
|
+
if (record.value === null)
|
|
3156
|
+
return state;
|
|
3157
|
+
const { settled } = record.value;
|
|
3158
|
+
yield* emit(settled);
|
|
3159
|
+
// A chunk is its id and its text: a replay served against an edited file
|
|
3160
|
+
// keeps the chunk now at that id, which nobody judged.
|
|
3161
|
+
const judgedChunks = new Set(settled.withheld.filter((item) => item.kind === "instruction").map((item) => `${item.id}\u0000${item.digest}`));
|
|
3162
|
+
const chunks = new Set(Relevance.chunks(documents).flatMap((chunk) => judgedChunks.has(`${chunk.id}\u0000${Digest.digest(chunk.text)}`) ? [chunk.id] : []));
|
|
3163
|
+
// A row is its key and its text, so two rows alike in both are one row.
|
|
3164
|
+
const dropped = new Set(settled.withheld.filter((item) => item.kind === "memory").map((item) => `${item.id}\u0000${item.digest}`));
|
|
3165
|
+
const replaced = new Map();
|
|
3166
|
+
if (chunks.size > 0) {
|
|
3167
|
+
const text = Relevance.render(documents, chunks);
|
|
3168
|
+
replaced.set(ContextWindow.makeSegment(instructionsSegment(documents, new Set())).digest, text === "" ? [] : [ContextWindow.makeSegment(instructionsSegment(documents, chunks))]);
|
|
3169
|
+
}
|
|
3170
|
+
if (input.memory !== undefined && dropped.size > 0) {
|
|
3171
|
+
const { digest, render } = input.memory;
|
|
3172
|
+
const text = render(rows.filter((row) => !dropped.has(`${row.key}\u0000${Digest.digest(row.text)}`)));
|
|
3173
|
+
replaced.set(ContextWindow.makeSegment(memorySegment(render(rows), digest)).digest, text === "" ? [] : [ContextWindow.makeSegment(memorySegment(text, Digest.digest(text)))]);
|
|
3174
|
+
}
|
|
3175
|
+
return advance(state, {
|
|
3176
|
+
withheldFlows: settled.withheld.filter((item) => item.kind === "flow" || item.kind === "skill").map((item) => item.id),
|
|
3177
|
+
...(replaced.size === 0 ? {} : {
|
|
3178
|
+
contextWindow: ContextWindow.make({
|
|
3179
|
+
...state.contextWindow,
|
|
3180
|
+
segments: state.contextWindow.segments.flatMap((segment) => replaced.get(segment.digest) ?? [segment])
|
|
3181
|
+
})
|
|
3182
|
+
})
|
|
3183
|
+
});
|
|
3184
|
+
});
|
|
3185
|
+
/**
|
|
3186
|
+
* Runs the cell loop until it completes, parks, or exhausts its budget.
|
|
3187
|
+
*
|
|
3188
|
+
* Cancellation is fiber interruption: interrupting this stream tears down the
|
|
3189
|
+
* sandbox through scope closure and reports one abort, without threading an
|
|
3190
|
+
* abort signal anywhere.
|
|
3191
|
+
*
|
|
3192
|
+
* @category streams
|
|
3193
|
+
* @since 0.1.0
|
|
3194
|
+
* @slop
|
|
3195
|
+
*/
|
|
3196
|
+
export const run = (input) => Stream.callback((queue) => {
|
|
3197
|
+
// A replay re-emits its recorded drains before reaching an unjudged
|
|
3198
|
+
// completion. This private evidence follows those admissions, including
|
|
3199
|
+
// through compaction, without changing public State or model-step keys.
|
|
3200
|
+
// The observer finishes its checkpoint before any evidence or frame advances.
|
|
3201
|
+
let instructions = "";
|
|
3202
|
+
const emit = (event) => Effect.flatMap(AgentEvent.Observer, (observe) => observe(event)).pipe(Effect.andThen(Effect.sync(() => {
|
|
3203
|
+
if (event._tag !== "steering-drained")
|
|
3204
|
+
return;
|
|
3205
|
+
const text = event.messages.flatMap((message) => message.content)
|
|
3206
|
+
.filter((part) => part.type === "text")
|
|
3207
|
+
.map((part) => part.text)
|
|
3208
|
+
.join("\n\n");
|
|
3209
|
+
if (text !== "")
|
|
3210
|
+
instructions = CompletionClaim.newest(`${instructions}\n\n${text}`.trim());
|
|
3211
|
+
})), Effect.andThen(Queue.offer(queue, event)), Effect.asVoid);
|
|
3212
|
+
const readCompletion = (evidence) => CompletionClaim.read({ ...evidence, task: completionTask(evidence.task, instructions) });
|
|
3213
|
+
// The supervisor reads the task the completion brake reads: the stated
|
|
3214
|
+
// task and every later instruction the run accepted from the person.
|
|
3215
|
+
const taskOf = (window) => completionTask(Frame.taskText(window), instructions);
|
|
3216
|
+
const monitors = input.monitors ?? Monitor.defaults();
|
|
3217
|
+
const judged = input.judged ?? false;
|
|
3218
|
+
const pinned = input.pinned ?? [];
|
|
3219
|
+
const loop = Effect.gen(function* () {
|
|
3220
|
+
const engine = yield* EngineLike.EngineLike;
|
|
3221
|
+
const sandbox = yield* Sandbox.Sandbox;
|
|
3222
|
+
const steering = yield* Steering.Source;
|
|
3223
|
+
let current = input.state;
|
|
3224
|
+
// The catalog the window teaches: a resumed run's window already
|
|
3225
|
+
// leaves out what the run-start reading withheld.
|
|
3226
|
+
let flows = input.flows.filter((descriptor) => !current.withheldFlows.includes(descriptor.name));
|
|
3227
|
+
if (current.journalVersion !== journalVersion) {
|
|
3228
|
+
return yield* new HarnessError({
|
|
3229
|
+
code: "incompatible_journal",
|
|
3230
|
+
message: `Controller state predates harness journal format ${journalVersion}; start a new run.`
|
|
3231
|
+
});
|
|
3232
|
+
}
|
|
3233
|
+
// Once, and only on a run's own first frame: a resumed run replays its
|
|
3234
|
+
// arming from the journal it already wrote, and a second record would
|
|
3235
|
+
// make the gate count runs instead of arming decisions.
|
|
3236
|
+
if (current.frame === 0) {
|
|
3237
|
+
const limits = Sandbox.withDefaults(sandbox.capabilities, input.limits);
|
|
3238
|
+
yield* emit(new AgentEvent.DisciplineArmed({
|
|
3239
|
+
eventType: eventType.disciplineArmed,
|
|
3240
|
+
readOnlyCap: current.readOnlyCap,
|
|
3241
|
+
maxFrames: current.maxFrames,
|
|
3242
|
+
approvalChannel: current.approvalChannel,
|
|
3243
|
+
modelCallMs: current.modelCallMs,
|
|
3244
|
+
repeatCap: current.repeatCap,
|
|
3245
|
+
narrowingCap: current.narrowingCap,
|
|
3246
|
+
unmovedCap: current.unmovedCap,
|
|
3247
|
+
unresolvedCap: current.unresolvedCap,
|
|
3248
|
+
claimCap: current.claimCap,
|
|
3249
|
+
revalidations: current.revalidations,
|
|
3250
|
+
// Written only when armed, so an unjudged journal keeps the bytes
|
|
3251
|
+
// it had before the switch existed.
|
|
3252
|
+
...(judged
|
|
3253
|
+
? {
|
|
3254
|
+
judged,
|
|
3255
|
+
relevance: { withholdAt: Relevance.withholdAt, pinned: [...pinned] },
|
|
3256
|
+
monitors: monitors.map(({ at, consecutive, cooldownFrames, id, kind, limit }) => ({
|
|
3257
|
+
id,
|
|
3258
|
+
kind,
|
|
3259
|
+
at,
|
|
3260
|
+
consecutive,
|
|
3261
|
+
cooldownFrames,
|
|
3262
|
+
limit
|
|
3263
|
+
}))
|
|
3264
|
+
}
|
|
3265
|
+
: {}),
|
|
3266
|
+
...(input.stance === undefined ? {} : { stance: input.stance }),
|
|
3267
|
+
...limits
|
|
3268
|
+
}));
|
|
3269
|
+
}
|
|
3270
|
+
// The supervisor's fiber, on this loop's scope: it reads snapshots the
|
|
3271
|
+
// frames offer and the loop never waits for it. See `Supervisor`.
|
|
3272
|
+
const supervision = yield* Supervision.open({
|
|
3273
|
+
session: current.session,
|
|
3274
|
+
engine,
|
|
3275
|
+
emit,
|
|
3276
|
+
monitors,
|
|
3277
|
+
deliver: judged
|
|
3278
|
+
});
|
|
3279
|
+
// The run's realm: an `acquireRelease` on this loop's scope rather than
|
|
3280
|
+
// on one evaluation, so teardown is still scope closure and cancellation
|
|
3281
|
+
// is still fiber interruption. A resumed run rebuilds it by re-executing
|
|
3282
|
+
// its own cells from the top — every `ctx.call` replays from its recorded
|
|
3283
|
+
// boundary, so the realm that comes out is the realm that was lost,
|
|
3284
|
+
// without anything having been serialized.
|
|
3285
|
+
//
|
|
3286
|
+
// Opened after the arming is journaled, so a run whose realm cannot be
|
|
3287
|
+
// built still has its armed budgets on the record. A failure that leaves
|
|
3288
|
+
// no record of what was asked for is a failure a grader cannot read.
|
|
3289
|
+
const realm = yield* Effect.suspend(() => {
|
|
3290
|
+
const projections = {};
|
|
3291
|
+
for (const descriptor of input.flows) {
|
|
3292
|
+
projections[descriptor.name] = Cell.project(descriptor);
|
|
3293
|
+
}
|
|
3294
|
+
return sandbox.openRealm === undefined
|
|
3295
|
+
? Effect.fail(Sandbox.realmUnsupported)
|
|
3296
|
+
: sandbox.openRealm({ flows: projections, limits: input.limits });
|
|
3297
|
+
}).pipe(Effect.mapError((cause) => new HarnessError({
|
|
3298
|
+
code: "engine_failed",
|
|
3299
|
+
message: "The run's persistent realm could not be opened",
|
|
3300
|
+
cause
|
|
3301
|
+
})));
|
|
3302
|
+
for (;;) {
|
|
3303
|
+
if (current.failureFrames >= 5) {
|
|
3304
|
+
return yield* new HarnessError({
|
|
3305
|
+
code: "model_failed",
|
|
3306
|
+
message: `Runaway guard: ${current.failureFrames} consecutive frames repeated the same failure: ${current.failureKey}`,
|
|
3307
|
+
cause: { _tag: repeatedFailureTag, frames: current.failureFrames }
|
|
3308
|
+
});
|
|
3309
|
+
}
|
|
3310
|
+
if (current.maxFrames > 0 && current.frame >= current.maxFrames) {
|
|
3311
|
+
yield* emit(new AgentEvent.Resolved({
|
|
3312
|
+
eventType: eventType.resolved,
|
|
3313
|
+
message: ModelRequest.Message.assistant(budgetMessage(current), { stopReason: "stop" })
|
|
3314
|
+
}));
|
|
3315
|
+
return;
|
|
3316
|
+
}
|
|
3317
|
+
const catalog = input.refreshFlows === undefined ? input.flows : yield* engine.record({
|
|
3318
|
+
name: "flow-catalog",
|
|
3319
|
+
identity: { session: current.session, frame: current.frame, boundary: "flow-catalog" },
|
|
3320
|
+
success: Schema.Array(Descriptor.FlowDescriptor),
|
|
3321
|
+
execute: input.refreshFlows
|
|
3322
|
+
});
|
|
3323
|
+
current = judged && current.frame === 0
|
|
3324
|
+
? yield* withheld(current, catalog, input, engine, emit, taskOf)
|
|
3325
|
+
: current;
|
|
3326
|
+
// The journaled catalog stays whole; what the frame shows leaves out
|
|
3327
|
+
// what the run-start reading withheld.
|
|
3328
|
+
const shown = catalog.filter((descriptor) => !current.withheldFlows.includes(descriptor.name));
|
|
3329
|
+
if (input.refreshFlows !== undefined || shown.length !== flows.length) {
|
|
3330
|
+
// Replace only the teaching we supplied, preserving the host's
|
|
3331
|
+
// prefix and the accumulated transcript. Rebuild before compaction
|
|
3332
|
+
// so token accounting and the sealed request use this snapshot too.
|
|
3333
|
+
const previous = teach(ContextWindow.empty(current.contextWindow.modelId), flows, undefined, input.stance);
|
|
3334
|
+
const digests = new Set(previous.segments.map((segment) => segment.digest));
|
|
3335
|
+
current = advance(current, {
|
|
3336
|
+
contextWindow: teach(ContextWindow.make({
|
|
3337
|
+
...current.contextWindow,
|
|
3338
|
+
segments: current.contextWindow.segments.filter((segment) => !digests.has(segment.digest))
|
|
3339
|
+
}), shown, undefined, input.stance)
|
|
3340
|
+
});
|
|
3341
|
+
flows = shown;
|
|
3342
|
+
}
|
|
3343
|
+
const step = yield* frame({ ...input, state: current, flows }, engine, sandbox, realm, steering, emit, readCompletion, supervision, taskOf).pipe(Effect.catch((error) => {
|
|
3344
|
+
const request = permissionRequired(error);
|
|
3345
|
+
if (request === undefined) {
|
|
3346
|
+
return Effect.fail(error instanceof HarnessError ? error : new HarnessError({
|
|
3347
|
+
code: error instanceof Sandbox.SandboxError ? "engine_failed" : "model_failed",
|
|
3348
|
+
message: frameFailureMessage(error),
|
|
3349
|
+
cause: error
|
|
3350
|
+
}));
|
|
3351
|
+
}
|
|
3352
|
+
return Effect.gen(function* () {
|
|
3353
|
+
yield* emit(new AgentEvent.PermissionRequired({
|
|
3354
|
+
eventType: eventType.permissionRequired,
|
|
3355
|
+
request
|
|
3356
|
+
}));
|
|
3357
|
+
yield* emit(new AgentEvent.TurnClosed({
|
|
3358
|
+
eventType: eventType.turnClosed,
|
|
3359
|
+
stopReason: "error",
|
|
3360
|
+
outcome: "suspended"
|
|
3361
|
+
}));
|
|
3362
|
+
return {
|
|
3363
|
+
_tag: "Suspend",
|
|
3364
|
+
reason: new EngineLike.SuspendReason({
|
|
3365
|
+
code: "permission-required",
|
|
3366
|
+
message: `Permission ${request.requestId} is required`,
|
|
3367
|
+
// Encoded, not attached: `request` is a class that extends
|
|
3368
|
+
// `Error`, and this reason is journaled inside
|
|
3369
|
+
// `AgentEvent.Suspended`. A live Error there has no JSON form,
|
|
3370
|
+
// so encoding it dies with a schema failure that replaces the
|
|
3371
|
+
// park it was carrying.
|
|
3372
|
+
details: encodePermissionRequired(request)
|
|
3373
|
+
})
|
|
3374
|
+
};
|
|
3375
|
+
});
|
|
3376
|
+
}), Effect.onInterrupt(() => Effect.gen(function* () {
|
|
3377
|
+
yield* emit(new AgentEvent.Aborted({ eventType: eventType.aborted, reason: "Cell frame interrupted" }));
|
|
3378
|
+
yield* emit(new AgentEvent.TurnClosed({
|
|
3379
|
+
eventType: eventType.turnClosed,
|
|
3380
|
+
stopReason: "aborted",
|
|
3381
|
+
outcome: "aborted"
|
|
3382
|
+
}));
|
|
3383
|
+
})));
|
|
3384
|
+
if (step._tag === "Done")
|
|
3385
|
+
return;
|
|
3386
|
+
if (step._tag === "Suspend") {
|
|
3387
|
+
yield* emit(new AgentEvent.Suspended({ eventType: eventType.suspended, reason: step.reason }));
|
|
3388
|
+
return yield* engine.suspend(step.reason);
|
|
3389
|
+
}
|
|
3390
|
+
for (const flow of current.withheldFlows) {
|
|
3391
|
+
if (step.state.withheldFlows.includes(flow))
|
|
3392
|
+
continue;
|
|
3393
|
+
yield* emit(new AgentEvent.RelevanceRestored({
|
|
3394
|
+
eventType: eventType.relevanceRestored,
|
|
3395
|
+
scope: current.session,
|
|
3396
|
+
frame: current.frame,
|
|
3397
|
+
flow
|
|
3398
|
+
}));
|
|
3399
|
+
}
|
|
3400
|
+
current = step.state;
|
|
3401
|
+
}
|
|
3402
|
+
});
|
|
3403
|
+
// The queue, not the callback's error channel, terminates the stream, so
|
|
3404
|
+
// every event already offered stays observable ahead of whatever ended the
|
|
3405
|
+
// run. An interruption is forwarded rather than swallowed: a durable park
|
|
3406
|
+
// arrives as one, and turning it into a clean end would report a suspended
|
|
3407
|
+
// run as a finished one.
|
|
3408
|
+
return Effect.onExit(Effect.scoped(loop), (exit) => Effect.asVoid(exit._tag === "Success" ? Queue.end(queue) : Queue.failCause(queue, exit.cause)));
|
|
3409
|
+
});
|
|
3410
|
+
//# sourceMappingURL=CellTurn.js.map
|