@smthrs/harness 0.0.0-stage → 1.0.0-rc.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +387 -0
- package/LICENSE +21 -0
- package/README.md +138 -2
- package/dist/cjs/AgentEvent.d.ts +3092 -0
- package/dist/cjs/AgentEvent.d.ts.map +1 -0
- package/dist/cjs/AgentEvent.js +1116 -0
- package/dist/cjs/AgentEvent.js.map +7 -0
- package/dist/cjs/CallLedger.d.ts +378 -0
- package/dist/cjs/CallLedger.d.ts.map +1 -0
- package/dist/cjs/CallLedger.js +240 -0
- package/dist/cjs/CallLedger.js.map +7 -0
- package/dist/cjs/Cell.d.ts +774 -0
- package/dist/cjs/Cell.d.ts.map +1 -0
- package/dist/cjs/Cell.js +431 -0
- package/dist/cjs/Cell.js.map +7 -0
- package/dist/cjs/CellCalls.d.ts +115 -0
- package/dist/cjs/CellCalls.d.ts.map +1 -0
- package/dist/cjs/CellCalls.js +98 -0
- package/dist/cjs/CellCalls.js.map +7 -0
- package/dist/cjs/CellHistory.d.ts +102 -0
- package/dist/cjs/CellHistory.d.ts.map +1 -0
- package/dist/cjs/CellHistory.js +54 -0
- package/dist/cjs/CellHistory.js.map +7 -0
- package/dist/cjs/CellTurn.d.ts +1071 -0
- package/dist/cjs/CellTurn.d.ts.map +1 -0
- package/dist/cjs/CellTurn.js +2618 -0
- package/dist/cjs/CellTurn.js.map +7 -0
- package/dist/cjs/CellValidation.d.ts +94 -0
- package/dist/cjs/CellValidation.d.ts.map +1 -0
- package/dist/cjs/CellValidation.js +218 -0
- package/dist/cjs/CellValidation.js.map +7 -0
- package/dist/cjs/Compaction.d.ts +158 -0
- package/dist/cjs/Compaction.d.ts.map +1 -0
- package/dist/cjs/Compaction.js +189 -0
- package/dist/cjs/Compaction.js.map +7 -0
- package/dist/cjs/CompletionClaim.d.ts +801 -0
- package/dist/cjs/CompletionClaim.d.ts.map +1 -0
- package/dist/cjs/CompletionClaim.js +303 -0
- package/dist/cjs/CompletionClaim.js.map +7 -0
- package/dist/cjs/ContextWindow.d.ts +435 -0
- package/dist/cjs/ContextWindow.d.ts.map +1 -0
- package/dist/cjs/ContextWindow.js +319 -0
- package/dist/cjs/ContextWindow.js.map +7 -0
- package/dist/cjs/EngineLike.d.ts +546 -0
- package/dist/cjs/EngineLike.d.ts.map +1 -0
- package/dist/cjs/EngineLike.js +114 -0
- package/dist/cjs/EngineLike.js.map +7 -0
- package/dist/cjs/ExternalTranscript.d.ts +349 -0
- package/dist/cjs/ExternalTranscript.d.ts.map +1 -0
- package/dist/cjs/ExternalTranscript.js +826 -0
- package/dist/cjs/ExternalTranscript.js.map +7 -0
- package/dist/cjs/FailedCall.d.ts +132 -0
- package/dist/cjs/FailedCall.d.ts.map +1 -0
- package/dist/cjs/FailedCall.js +53 -0
- package/dist/cjs/FailedCall.js.map +7 -0
- package/dist/cjs/FlowBinding.d.ts +289 -0
- package/dist/cjs/FlowBinding.d.ts.map +1 -0
- package/dist/cjs/FlowBinding.js +251 -0
- package/dist/cjs/FlowBinding.js.map +7 -0
- package/dist/cjs/HarnessError.d.ts +57 -0
- package/dist/cjs/HarnessError.d.ts.map +1 -0
- package/dist/cjs/HarnessError.js +75 -0
- package/dist/cjs/HarnessError.js.map +7 -0
- package/dist/cjs/Judgement.d.ts +289 -0
- package/dist/cjs/Judgement.d.ts.map +1 -0
- package/dist/cjs/Judgement.js +240 -0
- package/dist/cjs/Judgement.js.map +7 -0
- package/dist/cjs/Monitor.d.ts +374 -0
- package/dist/cjs/Monitor.d.ts.map +1 -0
- package/dist/cjs/Monitor.js +233 -0
- package/dist/cjs/Monitor.js.map +7 -0
- package/dist/cjs/NarrowedCheck.d.ts +523 -0
- package/dist/cjs/NarrowedCheck.d.ts.map +1 -0
- package/dist/cjs/NarrowedCheck.js +263 -0
- package/dist/cjs/NarrowedCheck.js.map +7 -0
- package/dist/cjs/Notifications.d.ts +42 -0
- package/dist/cjs/Notifications.d.ts.map +1 -0
- package/dist/cjs/Notifications.js +177 -0
- package/dist/cjs/Notifications.js.map +7 -0
- package/dist/cjs/Plan.d.ts +127 -0
- package/dist/cjs/Plan.d.ts.map +1 -0
- package/dist/cjs/Plan.js +77 -0
- package/dist/cjs/Plan.js.map +7 -0
- package/dist/cjs/QuickJSSandbox.d.ts +151 -0
- package/dist/cjs/QuickJSSandbox.d.ts.map +1 -0
- package/dist/cjs/QuickJSSandbox.js +987 -0
- package/dist/cjs/QuickJSSandbox.js.map +7 -0
- package/dist/cjs/Relevance.d.ts +213 -0
- package/dist/cjs/Relevance.d.ts.map +1 -0
- package/dist/cjs/Relevance.js +183 -0
- package/dist/cjs/Relevance.js.map +7 -0
- package/dist/cjs/Sandbox.d.ts +637 -0
- package/dist/cjs/Sandbox.d.ts.map +1 -0
- package/dist/cjs/Sandbox.js +260 -0
- package/dist/cjs/Sandbox.js.map +7 -0
- package/dist/cjs/Steering.d.ts +464 -0
- package/dist/cjs/Steering.d.ts.map +1 -0
- package/dist/cjs/Steering.js +153 -0
- package/dist/cjs/Steering.js.map +7 -0
- package/dist/cjs/StructuredOutput.d.ts +252 -0
- package/dist/cjs/StructuredOutput.d.ts.map +1 -0
- package/dist/cjs/StructuredOutput.js +266 -0
- package/dist/cjs/StructuredOutput.js.map +7 -0
- package/dist/cjs/Sufficiency.d.ts +195 -0
- package/dist/cjs/Sufficiency.d.ts.map +1 -0
- package/dist/cjs/Sufficiency.js +110 -0
- package/dist/cjs/Sufficiency.js.map +7 -0
- package/dist/cjs/Supervisor.d.ts +719 -0
- package/dist/cjs/Supervisor.d.ts.map +1 -0
- package/dist/cjs/Supervisor.js +313 -0
- package/dist/cjs/Supervisor.js.map +7 -0
- package/dist/cjs/Tokens.d.ts +86 -0
- package/dist/cjs/Tokens.d.ts.map +1 -0
- package/dist/cjs/Tokens.js +72 -0
- package/dist/cjs/Tokens.js.map +7 -0
- package/dist/cjs/Transcript.d.ts +171 -0
- package/dist/cjs/Transcript.d.ts.map +1 -0
- package/dist/cjs/Transcript.js +342 -0
- package/dist/cjs/Transcript.js.map +7 -0
- package/dist/cjs/TruncatedOutput.d.ts +186 -0
- package/dist/cjs/TruncatedOutput.d.ts.map +1 -0
- package/dist/cjs/TruncatedOutput.js +143 -0
- package/dist/cjs/TruncatedOutput.js.map +7 -0
- package/dist/cjs/UnmovedTree.d.ts +113 -0
- package/dist/cjs/UnmovedTree.d.ts.map +1 -0
- package/dist/cjs/UnmovedTree.js +38 -0
- package/dist/cjs/UnmovedTree.js.map +7 -0
- package/dist/cjs/UnresolvedFailure.d.ts +196 -0
- package/dist/cjs/UnresolvedFailure.d.ts.map +1 -0
- package/dist/cjs/UnresolvedFailure.js +77 -0
- package/dist/cjs/UnresolvedFailure.js.map +7 -0
- package/dist/cjs/VacuousVerification.d.ts +237 -0
- package/dist/cjs/VacuousVerification.d.ts.map +1 -0
- package/dist/cjs/VacuousVerification.js +91 -0
- package/dist/cjs/VacuousVerification.js.map +7 -0
- package/dist/cjs/VariablesPanel.d.ts +117 -0
- package/dist/cjs/VariablesPanel.d.ts.map +1 -0
- package/dist/cjs/VariablesPanel.js +108 -0
- package/dist/cjs/VariablesPanel.js.map +7 -0
- package/dist/cjs/index.d.ts +172 -0
- package/dist/cjs/index.d.ts.map +1 -0
- package/dist/cjs/index.js +99 -0
- package/dist/cjs/index.js.map +7 -0
- package/dist/cjs/internal/bytes.d.ts +36 -0
- package/dist/cjs/internal/bytes.d.ts.map +1 -0
- package/dist/cjs/internal/bytes.js +57 -0
- package/dist/cjs/internal/bytes.js.map +7 -0
- package/dist/cjs/internal/cellPrompt.d.ts +122 -0
- package/dist/cjs/internal/cellPrompt.d.ts.map +1 -0
- package/dist/cjs/internal/cellPrompt.js +146 -0
- package/dist/cjs/internal/cellPrompt.js.map +7 -0
- package/dist/cjs/internal/compactable.d.ts +44 -0
- package/dist/cjs/internal/compactable.d.ts.map +1 -0
- package/dist/cjs/internal/compactable.js +56 -0
- package/dist/cjs/internal/compactable.js.map +7 -0
- package/dist/cjs/internal/compactionMarks.d.ts +278 -0
- package/dist/cjs/internal/compactionMarks.d.ts.map +1 -0
- package/dist/cjs/internal/compactionMarks.js +212 -0
- package/dist/cjs/internal/compactionMarks.js.map +7 -0
- package/dist/cjs/internal/demandText.d.ts +95 -0
- package/dist/cjs/internal/demandText.d.ts.map +1 -0
- package/dist/cjs/internal/demandText.js +70 -0
- package/dist/cjs/internal/demandText.js.map +7 -0
- package/dist/cjs/internal/elide.d.ts +109 -0
- package/dist/cjs/internal/elide.d.ts.map +1 -0
- package/dist/cjs/internal/elide.js +59 -0
- package/dist/cjs/internal/elide.js.map +7 -0
- package/dist/cjs/internal/frame.d.ts +463 -0
- package/dist/cjs/internal/frame.d.ts.map +1 -0
- package/dist/cjs/internal/frame.js +514 -0
- package/dist/cjs/internal/frame.js.map +7 -0
- package/dist/cjs/internal/nonNegativeSafeInt.d.ts +20 -0
- package/dist/cjs/internal/nonNegativeSafeInt.d.ts.map +1 -0
- package/dist/cjs/internal/nonNegativeSafeInt.js +29 -0
- package/dist/cjs/internal/nonNegativeSafeInt.js.map +7 -0
- package/dist/cjs/internal/paidUsage.d.ts +52 -0
- package/dist/cjs/internal/paidUsage.d.ts.map +1 -0
- package/dist/cjs/internal/paidUsage.js +70 -0
- package/dist/cjs/internal/paidUsage.js.map +7 -0
- package/dist/cjs/internal/printChannel.d.ts +233 -0
- package/dist/cjs/internal/printChannel.d.ts.map +1 -0
- package/dist/cjs/internal/printChannel.js +165 -0
- package/dist/cjs/internal/printChannel.js.map +7 -0
- package/dist/cjs/internal/printsObservation.d.ts +26 -0
- package/dist/cjs/internal/printsObservation.d.ts.map +1 -0
- package/dist/cjs/internal/printsObservation.js +27 -0
- package/dist/cjs/internal/printsObservation.js.map +7 -0
- package/dist/cjs/internal/refusal.d.ts +40 -0
- package/dist/cjs/internal/refusal.d.ts.map +1 -0
- package/dist/cjs/internal/refusal.js +45 -0
- package/dist/cjs/internal/refusal.js.map +7 -0
- package/dist/cjs/internal/supervision.d.ts +174 -0
- package/dist/cjs/internal/supervision.d.ts.map +1 -0
- package/dist/cjs/internal/supervision.js +402 -0
- package/dist/cjs/internal/supervision.js.map +7 -0
- package/dist/cjs/internal/unfinishedWork.d.ts +99 -0
- package/dist/cjs/internal/unfinishedWork.d.ts.map +1 -0
- package/dist/cjs/internal/unfinishedWork.js +80 -0
- package/dist/cjs/internal/unfinishedWork.js.map +7 -0
- package/dist/cjs/internal/unobservedCall.d.ts +112 -0
- package/dist/cjs/internal/unobservedCall.d.ts.map +1 -0
- package/dist/cjs/internal/unobservedCall.js +345 -0
- package/dist/cjs/internal/unobservedCall.js.map +7 -0
- package/dist/cjs/internal/untrustedData.d.ts +14 -0
- package/dist/cjs/internal/untrustedData.d.ts.map +1 -0
- package/dist/cjs/internal/untrustedData.js +30 -0
- package/dist/cjs/internal/untrustedData.js.map +7 -0
- package/dist/cjs/package.json +1 -0
- package/dist/esm/AgentEvent.d.ts +3092 -0
- package/dist/esm/AgentEvent.d.ts.map +1 -0
- package/dist/esm/AgentEvent.js +1610 -0
- package/dist/esm/AgentEvent.js.map +1 -0
- package/dist/esm/CallLedger.d.ts +378 -0
- package/dist/esm/CallLedger.d.ts.map +1 -0
- package/dist/esm/CallLedger.js +506 -0
- package/dist/esm/CallLedger.js.map +1 -0
- package/dist/esm/Cell.d.ts +774 -0
- package/dist/esm/Cell.d.ts.map +1 -0
- package/dist/esm/Cell.js +772 -0
- package/dist/esm/Cell.js.map +1 -0
- package/dist/esm/CellCalls.d.ts +115 -0
- package/dist/esm/CellCalls.d.ts.map +1 -0
- package/dist/esm/CellCalls.js +97 -0
- package/dist/esm/CellCalls.js.map +1 -0
- package/dist/esm/CellHistory.d.ts +102 -0
- package/dist/esm/CellHistory.d.ts.map +1 -0
- package/dist/esm/CellHistory.js +91 -0
- package/dist/esm/CellHistory.js.map +1 -0
- package/dist/esm/CellTurn.d.ts +1071 -0
- package/dist/esm/CellTurn.d.ts.map +1 -0
- package/dist/esm/CellTurn.js +3410 -0
- package/dist/esm/CellTurn.js.map +1 -0
- package/dist/esm/CellValidation.d.ts +94 -0
- package/dist/esm/CellValidation.d.ts.map +1 -0
- package/dist/esm/CellValidation.js +335 -0
- package/dist/esm/CellValidation.js.map +1 -0
- package/dist/esm/Compaction.d.ts +158 -0
- package/dist/esm/Compaction.d.ts.map +1 -0
- package/dist/esm/Compaction.js +216 -0
- package/dist/esm/Compaction.js.map +1 -0
- package/dist/esm/CompletionClaim.d.ts +801 -0
- package/dist/esm/CompletionClaim.d.ts.map +1 -0
- package/dist/esm/CompletionClaim.js +833 -0
- package/dist/esm/CompletionClaim.js.map +1 -0
- package/dist/esm/ContextWindow.d.ts +435 -0
- package/dist/esm/ContextWindow.d.ts.map +1 -0
- package/dist/esm/ContextWindow.js +427 -0
- package/dist/esm/ContextWindow.js.map +1 -0
- package/dist/esm/EngineLike.d.ts +546 -0
- package/dist/esm/EngineLike.d.ts.map +1 -0
- package/dist/esm/EngineLike.js +218 -0
- package/dist/esm/EngineLike.js.map +1 -0
- package/dist/esm/ExternalTranscript.d.ts +349 -0
- package/dist/esm/ExternalTranscript.d.ts.map +1 -0
- package/dist/esm/ExternalTranscript.js +987 -0
- package/dist/esm/ExternalTranscript.js.map +1 -0
- package/dist/esm/FailedCall.d.ts +132 -0
- package/dist/esm/FailedCall.d.ts.map +1 -0
- package/dist/esm/FailedCall.js +131 -0
- package/dist/esm/FailedCall.js.map +1 -0
- package/dist/esm/FlowBinding.d.ts +289 -0
- package/dist/esm/FlowBinding.d.ts.map +1 -0
- package/dist/esm/FlowBinding.js +376 -0
- package/dist/esm/FlowBinding.js.map +1 -0
- package/dist/esm/HarnessError.d.ts +57 -0
- package/dist/esm/HarnessError.d.ts.map +1 -0
- package/dist/esm/HarnessError.js +85 -0
- package/dist/esm/HarnessError.js.map +1 -0
- package/dist/esm/Judgement.d.ts +289 -0
- package/dist/esm/Judgement.d.ts.map +1 -0
- package/dist/esm/Judgement.js +305 -0
- package/dist/esm/Judgement.js.map +1 -0
- package/dist/esm/Monitor.d.ts +374 -0
- package/dist/esm/Monitor.d.ts.map +1 -0
- package/dist/esm/Monitor.js +370 -0
- package/dist/esm/Monitor.js.map +1 -0
- package/dist/esm/NarrowedCheck.d.ts +523 -0
- package/dist/esm/NarrowedCheck.d.ts.map +1 -0
- package/dist/esm/NarrowedCheck.js +612 -0
- package/dist/esm/NarrowedCheck.js.map +1 -0
- package/dist/esm/Notifications.d.ts +42 -0
- package/dist/esm/Notifications.d.ts.map +1 -0
- package/dist/esm/Notifications.js +215 -0
- package/dist/esm/Notifications.js.map +1 -0
- package/dist/esm/Plan.d.ts +127 -0
- package/dist/esm/Plan.d.ts.map +1 -0
- package/dist/esm/Plan.js +97 -0
- package/dist/esm/Plan.js.map +1 -0
- package/dist/esm/QuickJSSandbox.d.ts +151 -0
- package/dist/esm/QuickJSSandbox.d.ts.map +1 -0
- package/dist/esm/QuickJSSandbox.js +1364 -0
- package/dist/esm/QuickJSSandbox.js.map +1 -0
- package/dist/esm/Relevance.d.ts +213 -0
- package/dist/esm/Relevance.d.ts.map +1 -0
- package/dist/esm/Relevance.js +254 -0
- package/dist/esm/Relevance.js.map +1 -0
- package/dist/esm/Sandbox.d.ts +637 -0
- package/dist/esm/Sandbox.d.ts.map +1 -0
- package/dist/esm/Sandbox.js +464 -0
- package/dist/esm/Sandbox.js.map +1 -0
- package/dist/esm/Steering.d.ts +464 -0
- package/dist/esm/Steering.d.ts.map +1 -0
- package/dist/esm/Steering.js +193 -0
- package/dist/esm/Steering.js.map +1 -0
- package/dist/esm/StructuredOutput.d.ts +252 -0
- package/dist/esm/StructuredOutput.d.ts.map +1 -0
- package/dist/esm/StructuredOutput.js +430 -0
- package/dist/esm/StructuredOutput.js.map +1 -0
- package/dist/esm/Sufficiency.d.ts +195 -0
- package/dist/esm/Sufficiency.d.ts.map +1 -0
- package/dist/esm/Sufficiency.js +207 -0
- package/dist/esm/Sufficiency.js.map +1 -0
- package/dist/esm/Supervisor.d.ts +719 -0
- package/dist/esm/Supervisor.d.ts.map +1 -0
- package/dist/esm/Supervisor.js +575 -0
- package/dist/esm/Supervisor.js.map +1 -0
- package/dist/esm/Tokens.d.ts +86 -0
- package/dist/esm/Tokens.d.ts.map +1 -0
- package/dist/esm/Tokens.js +92 -0
- package/dist/esm/Tokens.js.map +1 -0
- package/dist/esm/Transcript.d.ts +171 -0
- package/dist/esm/Transcript.d.ts.map +1 -0
- package/dist/esm/Transcript.js +424 -0
- package/dist/esm/Transcript.js.map +1 -0
- package/dist/esm/TruncatedOutput.d.ts +186 -0
- package/dist/esm/TruncatedOutput.d.ts.map +1 -0
- package/dist/esm/TruncatedOutput.js +257 -0
- package/dist/esm/TruncatedOutput.js.map +1 -0
- package/dist/esm/UnmovedTree.d.ts +113 -0
- package/dist/esm/UnmovedTree.d.ts.map +1 -0
- package/dist/esm/UnmovedTree.js +90 -0
- package/dist/esm/UnmovedTree.js.map +1 -0
- package/dist/esm/UnresolvedFailure.d.ts +196 -0
- package/dist/esm/UnresolvedFailure.d.ts.map +1 -0
- package/dist/esm/UnresolvedFailure.js +218 -0
- package/dist/esm/UnresolvedFailure.js.map +1 -0
- package/dist/esm/VacuousVerification.d.ts +237 -0
- package/dist/esm/VacuousVerification.d.ts.map +1 -0
- package/dist/esm/VacuousVerification.js +245 -0
- package/dist/esm/VacuousVerification.js.map +1 -0
- package/dist/esm/VariablesPanel.d.ts +117 -0
- package/dist/esm/VariablesPanel.d.ts.map +1 -0
- package/dist/esm/VariablesPanel.js +142 -0
- package/dist/esm/VariablesPanel.js.map +1 -0
- package/dist/esm/index.d.ts +172 -0
- package/dist/esm/index.d.ts.map +1 -0
- package/dist/esm/index.js +172 -0
- package/dist/esm/index.js.map +1 -0
- package/dist/esm/internal/bytes.d.ts +36 -0
- package/dist/esm/internal/bytes.d.ts.map +1 -0
- package/dist/esm/internal/bytes.js +70 -0
- package/dist/esm/internal/bytes.js.map +1 -0
- package/dist/esm/internal/cellPrompt.d.ts +122 -0
- package/dist/esm/internal/cellPrompt.d.ts.map +1 -0
- package/dist/esm/internal/cellPrompt.js +276 -0
- package/dist/esm/internal/cellPrompt.js.map +1 -0
- package/dist/esm/internal/compactable.d.ts +44 -0
- package/dist/esm/internal/compactable.d.ts.map +1 -0
- package/dist/esm/internal/compactable.js +71 -0
- package/dist/esm/internal/compactable.js.map +1 -0
- package/dist/esm/internal/compactionMarks.d.ts +278 -0
- package/dist/esm/internal/compactionMarks.d.ts.map +1 -0
- package/dist/esm/internal/compactionMarks.js +317 -0
- package/dist/esm/internal/compactionMarks.js.map +1 -0
- package/dist/esm/internal/demandText.d.ts +95 -0
- package/dist/esm/internal/demandText.d.ts.map +1 -0
- package/dist/esm/internal/demandText.js +128 -0
- package/dist/esm/internal/demandText.js.map +1 -0
- package/dist/esm/internal/elide.d.ts +109 -0
- package/dist/esm/internal/elide.d.ts.map +1 -0
- package/dist/esm/internal/elide.js +123 -0
- package/dist/esm/internal/elide.js.map +1 -0
- package/dist/esm/internal/frame.d.ts +463 -0
- package/dist/esm/internal/frame.d.ts.map +1 -0
- package/dist/esm/internal/frame.js +861 -0
- package/dist/esm/internal/frame.js.map +1 -0
- package/dist/esm/internal/nonNegativeSafeInt.d.ts +20 -0
- package/dist/esm/internal/nonNegativeSafeInt.d.ts.map +1 -0
- package/dist/esm/internal/nonNegativeSafeInt.js +20 -0
- package/dist/esm/internal/nonNegativeSafeInt.js.map +1 -0
- package/dist/esm/internal/paidUsage.d.ts +52 -0
- package/dist/esm/internal/paidUsage.d.ts.map +1 -0
- package/dist/esm/internal/paidUsage.js +72 -0
- package/dist/esm/internal/paidUsage.js.map +1 -0
- package/dist/esm/internal/printChannel.d.ts +233 -0
- package/dist/esm/internal/printChannel.d.ts.map +1 -0
- package/dist/esm/internal/printChannel.js +376 -0
- package/dist/esm/internal/printChannel.js.map +1 -0
- package/dist/esm/internal/printsObservation.d.ts +26 -0
- package/dist/esm/internal/printsObservation.d.ts.map +1 -0
- package/dist/esm/internal/printsObservation.js +29 -0
- package/dist/esm/internal/printsObservation.js.map +1 -0
- package/dist/esm/internal/refusal.d.ts +40 -0
- package/dist/esm/internal/refusal.d.ts.map +1 -0
- package/dist/esm/internal/refusal.js +47 -0
- package/dist/esm/internal/refusal.js.map +1 -0
- package/dist/esm/internal/supervision.d.ts +174 -0
- package/dist/esm/internal/supervision.d.ts.map +1 -0
- package/dist/esm/internal/supervision.js +485 -0
- package/dist/esm/internal/supervision.js.map +1 -0
- package/dist/esm/internal/unfinishedWork.d.ts +99 -0
- package/dist/esm/internal/unfinishedWork.d.ts.map +1 -0
- package/dist/esm/internal/unfinishedWork.js +85 -0
- package/dist/esm/internal/unfinishedWork.js.map +1 -0
- package/dist/esm/internal/unobservedCall.d.ts +112 -0
- package/dist/esm/internal/unobservedCall.d.ts.map +1 -0
- package/dist/esm/internal/unobservedCall.js +501 -0
- package/dist/esm/internal/unobservedCall.js.map +1 -0
- package/dist/esm/internal/untrustedData.d.ts +14 -0
- package/dist/esm/internal/untrustedData.d.ts.map +1 -0
- package/dist/esm/internal/untrustedData.js +15 -0
- package/dist/esm/internal/untrustedData.js.map +1 -0
- package/docs/README.md +129 -0
- package/docs/api.md +1869 -0
- package/docs/concepts.md +238 -0
- package/docs/external-transcripts.md +245 -0
- package/docs/guides/bind-flows.md +167 -0
- package/docs/guides/drive-the-loop.md +265 -0
- package/docs/guides/run-cells.md +262 -0
- package/docs/guides/workerd.md +144 -0
- package/docs/installation.md +69 -0
- package/docs/quickstart.md +121 -0
- package/docs/reference.md +954 -0
- package/docs/troubleshooting.md +177 -0
- package/package.json +463 -3
- package/src/AgentEvent.ts +1772 -0
- package/src/CallLedger.ts +560 -0
- package/src/Cell.ts +926 -0
- package/src/CellCalls.ts +198 -0
- package/src/CellHistory.ts +128 -0
- package/src/CellTurn.ts +4450 -0
- package/src/CellValidation.ts +382 -0
- package/src/Compaction.ts +330 -0
- package/src/CompletionClaim.ts +1013 -0
- package/src/ContextWindow.ts +669 -0
- package/src/EngineLike.ts +614 -0
- package/src/ExternalTranscript.ts +1143 -0
- package/src/FailedCall.ts +163 -0
- package/src/FlowBinding.ts +603 -0
- package/src/HarnessError.ts +98 -0
- package/src/Judgement.ts +526 -0
- package/src/Monitor.ts +579 -0
- package/src/NarrowedCheck.ts +694 -0
- package/src/Notifications.ts +262 -0
- package/src/Plan.ts +113 -0
- package/src/QuickJSSandbox.ts +1557 -0
- package/src/Relevance.ts +352 -0
- package/src/Sandbox.ts +908 -0
- package/src/Steering.ts +405 -0
- package/src/StructuredOutput.ts +484 -0
- package/src/Sufficiency.ts +247 -0
- package/src/Supervisor.ts +798 -0
- package/src/Tokens.ts +108 -0
- package/src/Transcript.ts +513 -0
- package/src/TruncatedOutput.ts +297 -0
- package/src/UnmovedTree.ts +121 -0
- package/src/UnresolvedFailure.ts +240 -0
- package/src/VacuousVerification.ts +280 -0
- package/src/VariablesPanel.ts +165 -0
- package/src/index.ts +204 -0
- package/src/internal/bytes.ts +70 -0
- package/src/internal/cellPrompt.ts +337 -0
- package/src/internal/compactable.ts +76 -0
- package/src/internal/compactionMarks.ts +454 -0
- package/src/internal/demandText.ts +145 -0
- package/src/internal/elide.ts +130 -0
- package/src/internal/frame.ts +1178 -0
- package/src/internal/nonNegativeSafeInt.ts +24 -0
- package/src/internal/paidUsage.ts +90 -0
- package/src/internal/printChannel.ts +434 -0
- package/src/internal/printsObservation.ts +31 -0
- package/src/internal/refusal.ts +53 -0
- package/src/internal/supervision.ts +652 -0
- package/src/internal/unfinishedWork.ts +123 -0
- package/src/internal/unobservedCall.ts +536 -0
- package/src/internal/untrustedData.ts +19 -0
|
@@ -0,0 +1,1178 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The decisions one controller frame takes from what it measured.
|
|
3
|
+
*
|
|
4
|
+
* `CellTurn` seals a model step and runs a cell, and both need a model, a
|
|
5
|
+
* sandbox and an engine. What the frame decides from the result needs none of
|
|
6
|
+
* them: what it did to the run's ledgers, whether a completion is handed back,
|
|
7
|
+
* and which interventions the next frame is handed are functions of the state
|
|
8
|
+
* the frame opened on and the facts it measured. Each phase here returns a
|
|
9
|
+
* value, events included, instead of writing a variable a later exit reads, so
|
|
10
|
+
* each one is read and tested without running a frame.
|
|
11
|
+
*
|
|
12
|
+
* @since 1.0.0-rc.0
|
|
13
|
+
* @private
|
|
14
|
+
*/
|
|
15
|
+
|
|
16
|
+
import { ModelRequest } from "@smthrs/model"
|
|
17
|
+
import type * as Evaluator from "@smthrs/model/Evaluator"
|
|
18
|
+
import { Effect, Option, type Schema } from "effect"
|
|
19
|
+
import * as AgentEvent from "../AgentEvent.ts"
|
|
20
|
+
import * as CallLedger from "../CallLedger.ts"
|
|
21
|
+
import type { State } from "../CellTurn.ts"
|
|
22
|
+
import * as CompletionClaim from "../CompletionClaim.ts"
|
|
23
|
+
import type * as ContextWindow from "../ContextWindow.ts"
|
|
24
|
+
import type * as EngineLike from "../EngineLike.ts"
|
|
25
|
+
import * as FailedCall from "../FailedCall.ts"
|
|
26
|
+
import type * as HarnessError from "../HarnessError.ts"
|
|
27
|
+
import * as NarrowedCheck from "../NarrowedCheck.ts"
|
|
28
|
+
import * as Sufficiency from "../Sufficiency.ts"
|
|
29
|
+
import * as TruncatedOutput from "../TruncatedOutput.ts"
|
|
30
|
+
import * as UnmovedTree from "../UnmovedTree.ts"
|
|
31
|
+
import * as UnresolvedFailure from "../UnresolvedFailure.ts"
|
|
32
|
+
import * as VariablesPanel from "../VariablesPanel.ts"
|
|
33
|
+
import * as bytes from "./bytes.ts"
|
|
34
|
+
import * as DemandText from "./demandText.ts"
|
|
35
|
+
import * as elide from "./elide.ts"
|
|
36
|
+
import { paidTogether } from "./paidUsage.ts"
|
|
37
|
+
import * as UnfinishedWork from "./unfinishedWork.ts"
|
|
38
|
+
import * as UnobservedCall from "./unobservedCall.ts"
|
|
39
|
+
|
|
40
|
+
/** The one journal-event-type table; see `AgentEvent.eventType`. */
|
|
41
|
+
const eventType = AgentEvent.eventType
|
|
42
|
+
|
|
43
|
+
/**
|
|
44
|
+
* The fields of controller state one step changes; everything else is carried.
|
|
45
|
+
*
|
|
46
|
+
* @since 1.0.0-rc.0
|
|
47
|
+
* @private
|
|
48
|
+
*/
|
|
49
|
+
export type StateChanges = Partial<ConstructorParameters<typeof State>[0]>
|
|
50
|
+
|
|
51
|
+
/**
|
|
52
|
+
* Whether the frame budget leaves a frame after this one.
|
|
53
|
+
*
|
|
54
|
+
* @since 1.0.0-rc.0
|
|
55
|
+
* @private
|
|
56
|
+
*/
|
|
57
|
+
export const hasNextFrame = (state: State): boolean => state.maxFrames === 0 || state.frame + 1 < state.maxFrames
|
|
58
|
+
|
|
59
|
+
/**
|
|
60
|
+
* Whether a completion can be handed back: a frame is left to answer in, and
|
|
61
|
+
* spending it cannot end the run at the read-only cap. See `judgeCompletion`.
|
|
62
|
+
*
|
|
63
|
+
* @since 1.0.0-rc.1
|
|
64
|
+
* @private
|
|
65
|
+
*/
|
|
66
|
+
export const handBackRoom = (state: State, readOnlyFrames: number): boolean =>
|
|
67
|
+
hasNextFrame(state) && (state.readOnlyCap === 0 || readOnlyFrames + 1 < state.readOnlyCap * 2)
|
|
68
|
+
|
|
69
|
+
/**
|
|
70
|
+
* One call a cell made this frame, as the frame's accounting reads it.
|
|
71
|
+
*
|
|
72
|
+
* Every call the frame settles is remembered so a raise can hand the model its
|
|
73
|
+
* partial work. Without this, one uncaught throw discarded the frame's reads
|
|
74
|
+
* and the next cell re-did them — often raising the same way again. Prime
|
|
75
|
+
* Agent's tool errors return stdout-so-far plus the traceback for exactly this
|
|
76
|
+
* reason.
|
|
77
|
+
*
|
|
78
|
+
* @since 1.0.0-rc.0
|
|
79
|
+
* @private
|
|
80
|
+
*/
|
|
81
|
+
export interface ObservedCall {
|
|
82
|
+
readonly flow: string
|
|
83
|
+
readonly ok: boolean
|
|
84
|
+
readonly summary: string
|
|
85
|
+
/** Where this call lands in the run's ledger, so a salvage line can name it. */
|
|
86
|
+
readonly ordinal: number
|
|
87
|
+
/**
|
|
88
|
+
* Whether the call reached the engine declaring a write, which is what
|
|
89
|
+
* breaks a read-only run. A call the boundary refused is not one: it
|
|
90
|
+
* declared a write and performed none.
|
|
91
|
+
*/
|
|
92
|
+
readonly mutates: boolean
|
|
93
|
+
/**
|
|
94
|
+
* Whether the write behind `mutates` was measured on a tree the host's
|
|
95
|
+
* workspace walk does not cover: the result carried the reserved `mutated`
|
|
96
|
+
* key, which `bash` sets by fingerprinting a container's working directory
|
|
97
|
+
* either side of the command. The read-only cap counts such a write like
|
|
98
|
+
* any other; the unmoved-tree demand and the claim judge need to know that
|
|
99
|
+
* the host digests holding still says nothing about it.
|
|
100
|
+
*/
|
|
101
|
+
readonly remote: boolean
|
|
102
|
+
/**
|
|
103
|
+
* What this invocation asked for, as `CellTurn`'s `signatureOf` names it,
|
|
104
|
+
* with the tree it asked about folded in.
|
|
105
|
+
*
|
|
106
|
+
* The repeat ledger reads this one, because the identical command against
|
|
107
|
+
* a pinned tree and against the live tree are two different questions and
|
|
108
|
+
* a run that asks both has learned twice.
|
|
109
|
+
*/
|
|
110
|
+
readonly signature: string
|
|
111
|
+
/**
|
|
112
|
+
* The same, with the tree left out: what the call asked, of whatever
|
|
113
|
+
* tree.
|
|
114
|
+
*
|
|
115
|
+
* The check ledgers read this one, because `Sufficiency` matches a
|
|
116
|
+
* failing reading against the passing reading that answered it, and the
|
|
117
|
+
* whole shape this surface exists for takes those two readings of one
|
|
118
|
+
* command over two different trees. Keyed on the tree they would never
|
|
119
|
+
* meet. For a call on the live tree the two are the same string, so
|
|
120
|
+
* nothing that existed before checkpoints re-keys.
|
|
121
|
+
*/
|
|
122
|
+
readonly subject: string
|
|
123
|
+
/** The checkpoint this call ran against, when it named one. */
|
|
124
|
+
readonly at: string | undefined
|
|
125
|
+
/** What this invocation asked for, verbatim, for the narrowing ledger. */
|
|
126
|
+
readonly input: Schema.Json
|
|
127
|
+
/** What the call resolved with, verbatim, for the call ledger's digest. */
|
|
128
|
+
readonly value: Schema.Json
|
|
129
|
+
/** What the flow said about a failure, for the call ledger's digest. */
|
|
130
|
+
readonly message: string | undefined
|
|
131
|
+
/** What the flow said about its own failure, when it said it ran nothing. */
|
|
132
|
+
readonly invalidProbe: { readonly reason: string; readonly message: string } | undefined
|
|
133
|
+
/**
|
|
134
|
+
* Whether the result reported a failing exit status about its subject.
|
|
135
|
+
*
|
|
136
|
+
* A call that declared an invalid probe is never failing here, whatever
|
|
137
|
+
* its exit status: the flow itself said the failure was about the command
|
|
138
|
+
* and not about the code, so the result is not a statement about the tree
|
|
139
|
+
* at all. See `UnresolvedFailure` `failed`.
|
|
140
|
+
*/
|
|
141
|
+
readonly failing: boolean
|
|
142
|
+
/**
|
|
143
|
+
* Whether the result reported a passing exit status about its subject.
|
|
144
|
+
*
|
|
145
|
+
* Not the negation of `failing`: a flow that reports no exit status is
|
|
146
|
+
* neither, and `Sufficiency` needs the difference between a check that
|
|
147
|
+
* passed and a call that never checked anything. An invalid probe is
|
|
148
|
+
* neither either, for the same reason it is never failing — the flow
|
|
149
|
+
* itself said the result is not a statement about the tree.
|
|
150
|
+
*/
|
|
151
|
+
readonly passing: boolean
|
|
152
|
+
}
|
|
153
|
+
|
|
154
|
+
/**
|
|
155
|
+
* States, unambiguously, that a call this frame failed about itself.
|
|
156
|
+
*
|
|
157
|
+
* The whole defect this closes is that `exitCode: 1` reads the same whether the
|
|
158
|
+
* bug reproduced or the command named a test that does not exist. The flow that
|
|
159
|
+
* ran the command is the only party that can tell, so it says so in its result;
|
|
160
|
+
* this turns that into a sentence the next frame cannot summarise away.
|
|
161
|
+
*/
|
|
162
|
+
const invalidProbeNotice = (calls: ReadonlyArray<ObservedCall>): string | undefined => {
|
|
163
|
+
const lines = calls.flatMap((call) =>
|
|
164
|
+
call.invalidProbe === undefined
|
|
165
|
+
? []
|
|
166
|
+
: [`- ${call.flow} (${call.invalidProbe.reason}): ${call.invalidProbe.message}`]
|
|
167
|
+
)
|
|
168
|
+
if (lines.length === 0) return undefined
|
|
169
|
+
return `Invalid probe — ${lines.length} call${
|
|
170
|
+
lines.length === 1 ? "" : "s"
|
|
171
|
+
} this frame failed about the command, not about the code:\n${
|
|
172
|
+
lines.join("\n")
|
|
173
|
+
}\nThat result is not a reproduction and is not a regression: it reads identically on a broken tree and on a fixed one, so it can neither prove the bug nor prove the repair. Repair the command before editing anything — find the real names first — and do not store it as \`state.verification\` or name it when you complete.`
|
|
174
|
+
}
|
|
175
|
+
|
|
176
|
+
/**
|
|
177
|
+
* How many distinct call signatures one run carries forward.
|
|
178
|
+
*
|
|
179
|
+
* The ledger is durable controller state, so it is bounded. Sixty-four covers a
|
|
180
|
+
* whole run at the rate a graded wave actually calls flows — its longest run
|
|
181
|
+
* issued 43 calls across 24 frames — so a call is recognised as a repeat
|
|
182
|
+
* however early the run first made it. A run that outlives the bound forgets
|
|
183
|
+
* its oldest distinct calls first, which can cost a demand and can never
|
|
184
|
+
* invent one.
|
|
185
|
+
*/
|
|
186
|
+
const retainedSignatures = 64
|
|
187
|
+
|
|
188
|
+
/**
|
|
189
|
+
* Folds one frame's signatures into the run's ledger, newest last and bounded.
|
|
190
|
+
*
|
|
191
|
+
* A repeated signature moves to the newest position rather than taking a
|
|
192
|
+
* second slot, so a run looping on one command cannot push everything it
|
|
193
|
+
* learned earlier out of the ledger.
|
|
194
|
+
*/
|
|
195
|
+
const remember = (
|
|
196
|
+
known: ReadonlyArray<string>,
|
|
197
|
+
made: ReadonlyArray<string>
|
|
198
|
+
): ReadonlyArray<string> => {
|
|
199
|
+
const newest = new Set(known)
|
|
200
|
+
for (const digest of made) {
|
|
201
|
+
newest.delete(digest)
|
|
202
|
+
newest.add(digest)
|
|
203
|
+
}
|
|
204
|
+
const distinct = [...newest]
|
|
205
|
+
return distinct.slice(Math.max(0, distinct.length - retainedSignatures))
|
|
206
|
+
}
|
|
207
|
+
|
|
208
|
+
/**
|
|
209
|
+
* The frame's own record of what it did to the world, computed once and
|
|
210
|
+
* carried out through every exit that continues the run.
|
|
211
|
+
*
|
|
212
|
+
* @since 1.0.0-rc.0
|
|
213
|
+
* @private
|
|
214
|
+
*/
|
|
215
|
+
export interface Accounting {
|
|
216
|
+
/** Whether the frame changed the workspace, by declaration or by measurement. */
|
|
217
|
+
readonly mutated: boolean
|
|
218
|
+
/** The journal's record of how that answer was reached. */
|
|
219
|
+
readonly observed: AgentEvent.MutationObserved
|
|
220
|
+
/**
|
|
221
|
+
* The digest the frame closed on, when its closing measurement covered the
|
|
222
|
+
* tree; empty otherwise, which makes every demand that reads it inert.
|
|
223
|
+
*/
|
|
224
|
+
readonly workspaceDigest: string
|
|
225
|
+
/** This frame's readings of the live tree, in the order they settled. */
|
|
226
|
+
readonly frameChecks: ReadonlyArray<NarrowedCheck.Check>
|
|
227
|
+
/**
|
|
228
|
+
* Every call the frame settled, verbatim, in the order they settled.
|
|
229
|
+
*
|
|
230
|
+
* The ledgers above are what a run carries forward, and they are bounded
|
|
231
|
+
* and lossy on purpose: a durable check entry keeps a clipped label and two
|
|
232
|
+
* exit-status flags, never a result. This is the one place the whole result
|
|
233
|
+
* of a call still exists, and it exists for one frame. `CompletionClaim`
|
|
234
|
+
* quotes the last check out of it; nothing else reads it.
|
|
235
|
+
*/
|
|
236
|
+
readonly calls: ReadonlyArray<ObservedCall>
|
|
237
|
+
/** The cell the frame ran, verbatim, for `FailedCall`. */
|
|
238
|
+
readonly source: string
|
|
239
|
+
/** The frame's own broken probes, stated once for whichever exit it takes. */
|
|
240
|
+
readonly probeNotice: string | undefined
|
|
241
|
+
/**
|
|
242
|
+
* Every state field the frame's measurements settle.
|
|
243
|
+
*
|
|
244
|
+
* A continuing exit states only what it changes on top of these, so a field
|
|
245
|
+
* measured here is carried whichever exit the frame leaves by. The read-only
|
|
246
|
+
* streak is one of them: when each exit passed it by hand, the frame's
|
|
247
|
+
* accounting was only as complete as the exit that remembered to.
|
|
248
|
+
*/
|
|
249
|
+
readonly facts:
|
|
250
|
+
& StateChanges
|
|
251
|
+
& Required<
|
|
252
|
+
Pick<StateChanges, "readOnlyFrames" | "repeatFrames" | "checks" | "failures" | "mutations" | "remoteMutations">
|
|
253
|
+
>
|
|
254
|
+
& Required<Pick<StateChanges, "openingDigest" | "callLedger" | "reported">>
|
|
255
|
+
}
|
|
256
|
+
|
|
257
|
+
/**
|
|
258
|
+
* What one frame that ran a cell did to the run's ledgers.
|
|
259
|
+
*
|
|
260
|
+
* Read off the calls the cell made and the two workspace measurements either
|
|
261
|
+
* side of them, and nothing else: no model, sandbox or engine is consulted, so
|
|
262
|
+
* a replayed frame accounts exactly as its original attempt did.
|
|
263
|
+
*
|
|
264
|
+
* @since 1.0.0-rc.0
|
|
265
|
+
* @private
|
|
266
|
+
*/
|
|
267
|
+
export const account = (options: {
|
|
268
|
+
/** The state the frame opened on. */
|
|
269
|
+
readonly state: State
|
|
270
|
+
/** Every call the cell made, in the order they settled. */
|
|
271
|
+
readonly calls: ReadonlyArray<ObservedCall>
|
|
272
|
+
/** The workspace before the frame's calls. */
|
|
273
|
+
readonly opened: Option.Option<EngineLike.Observation>
|
|
274
|
+
/** The workspace after them. */
|
|
275
|
+
readonly closed: Option.Option<EngineLike.Observation>
|
|
276
|
+
/** Ids of the trees the frame pinned, oldest first. */
|
|
277
|
+
readonly minted: ReadonlyArray<string>
|
|
278
|
+
/** What the realm holds after the frame, for the variables panel. */
|
|
279
|
+
readonly bindings: ReadonlyArray<VariablesPanel.Binding>
|
|
280
|
+
/** Output the run has been handed as a fragment, this frame's included. */
|
|
281
|
+
readonly captures: ReadonlyArray<TruncatedOutput.Capture>
|
|
282
|
+
/** The cell the frame ran, verbatim; omitted reads as a cell that inspects nothing. */
|
|
283
|
+
readonly source?: string | undefined
|
|
284
|
+
}): Accounting => {
|
|
285
|
+
const { calls, closed, opened, state } = options
|
|
286
|
+
|
|
287
|
+
// `declaredWrites` is what the frame's calls said about themselves and the
|
|
288
|
+
// measurement is what the workspace says. A measurement adds a mutation
|
|
289
|
+
// nothing declared — the shell redirect this exists for — and it does not
|
|
290
|
+
// take away a declaration made by a call that *succeeded*, because it does
|
|
291
|
+
// not cover the whole world: it stops at a bound, it prunes directories,
|
|
292
|
+
// and it is rooted at one path, so a real edit outside what it covers
|
|
293
|
+
// would read as an idle frame and twice the cap later the run would fail
|
|
294
|
+
// as `read_only_cap` having edited files the whole time.
|
|
295
|
+
//
|
|
296
|
+
// A declaration made by a call that *failed* is a different claim. It says
|
|
297
|
+
// what the call would have written, and a complete measurement that saw
|
|
298
|
+
// the workspace hold still contradicts it directly rather than merely
|
|
299
|
+
// failing to confirm it: nothing was written, so nothing was written
|
|
300
|
+
// outside the bound either. Wave 7 recorded two such frames on one
|
|
301
|
+
// instance — an anchor miss reporting `Failed to find expected lines`, and
|
|
302
|
+
// an edit reporting `oldString does not occur` — each of which cleared a
|
|
303
|
+
// read-only streak the run had not broken. Where the measurement is
|
|
304
|
+
// partial or absent the old rule stands, because then the digest holding
|
|
305
|
+
// still is not evidence of anything.
|
|
306
|
+
const declaredWrites = calls.filter((call) => call.mutates).length
|
|
307
|
+
const measured = Option.isSome(opened) && Option.isSome(closed)
|
|
308
|
+
const covered = measured && opened.value.complete && closed.value.complete
|
|
309
|
+
const standingWrites = calls.filter((call) => call.mutates && (call.ok || !covered)).length
|
|
310
|
+
const mutated = standingWrites > 0 || (covered && opened.value.digest !== closed.value.digest)
|
|
311
|
+
// Writes the host measurement could not have seen. A failed call measured
|
|
312
|
+
// nothing, so only a settled one counts; see `ObservedCall.remote`.
|
|
313
|
+
const remoteWrites = calls.filter((call) => call.ok && call.mutates && call.remote).length
|
|
314
|
+
const closingDigest = Option.match(closed, { onNone: () => "", onSome: (value) => value.digest })
|
|
315
|
+
|
|
316
|
+
// The frame's own repetition, measured the same way and carried the same
|
|
317
|
+
// way: a frame repeats when it issued calls, issued none this run had not
|
|
318
|
+
// already issued, and changed nothing. A signature the ledger has
|
|
319
|
+
// forgotten reads as new, which delays a demand and never fabricates one.
|
|
320
|
+
const signatures = calls.map((call) => call.signature)
|
|
321
|
+
const asked = new Set(state.callSignatures)
|
|
322
|
+
const novel = signatures.some((signature) => !asked.has(signature))
|
|
323
|
+
const repeatFrames = signatures.length === 0
|
|
324
|
+
? state.repeatFrames
|
|
325
|
+
: novel || mutated
|
|
326
|
+
? 0
|
|
327
|
+
: state.repeatFrames + 1
|
|
328
|
+
|
|
329
|
+
// The frame's own checks, and the tree they were taken over.
|
|
330
|
+
//
|
|
331
|
+
// Only a call that succeeded and declared no write is a check: a failed
|
|
332
|
+
// call observed nothing, and a call that changes the workspace is not an
|
|
333
|
+
// observation of it. The digest is the frame's own closing measurement, and
|
|
334
|
+
// only when that measurement covered the tree — under a partial or absent
|
|
335
|
+
// one a moved digest is as likely to be a bound moving as work, and this
|
|
336
|
+
// ledger's entire purpose is to say that the tree changed since a check
|
|
337
|
+
// ran. An empty digest makes an entry inert rather than wrong.
|
|
338
|
+
const workspaceDigest = covered ? closingDigest : ""
|
|
339
|
+
// Where the frame's writes sit among its calls, so a check can be stamped
|
|
340
|
+
// with the tree it actually read. `calls` settle in order, so a check with
|
|
341
|
+
// no standing write after it read the tree the frame closed on, whatever
|
|
342
|
+
// the frame did before it; a check with one after it read an earlier tree
|
|
343
|
+
// and the closing digest would be somebody else's answer. A measurement
|
|
344
|
+
// that moved with nothing declaring it cannot be placed among the calls at
|
|
345
|
+
// all, so that frame stamps nothing as read.
|
|
346
|
+
const standingAt = calls.map((call) => call.mutates && (call.ok || !covered))
|
|
347
|
+
const lastStandingWrite = standingAt.lastIndexOf(true)
|
|
348
|
+
const unattributedMutation = mutated && lastStandingWrite === -1
|
|
349
|
+
const readings = (checkpointed: boolean) =>
|
|
350
|
+
calls.flatMap((call, index) => {
|
|
351
|
+
if (!call.ok || call.mutates || (call.at !== undefined) !== checkpointed) return []
|
|
352
|
+
const recorded = NarrowedCheck.check({
|
|
353
|
+
flow: call.flow,
|
|
354
|
+
signature: call.subject,
|
|
355
|
+
input: call.input,
|
|
356
|
+
// A reading taken against a checkpoint is a reading of a tree that is
|
|
357
|
+
// not this workspace, so it carries no workspace digest: every ledger
|
|
358
|
+
// that reads one asks a question about the tree the run is standing
|
|
359
|
+
// on, and an entry stamped with this frame's digest would answer that
|
|
360
|
+
// question with somebody else's tree. Empty makes it inert there.
|
|
361
|
+
digest: checkpointed ? "" : workspaceDigest,
|
|
362
|
+
failing: call.failing,
|
|
363
|
+
passing: call.passing,
|
|
364
|
+
// Whether the tree stamped above is the tree this call read. Every
|
|
365
|
+
// write the frame declared is a call with a position among these ones,
|
|
366
|
+
// so the answer is positional: a check with no standing write after it
|
|
367
|
+
// read the closing tree, and one with a write after it read a tree
|
|
368
|
+
// that is gone. That is what lets the loop a real agent runs — edit,
|
|
369
|
+
// then run the check that was red, in one frame — hold its own
|
|
370
|
+
// evidence, while `check, then edit` still stamps nothing.
|
|
371
|
+
//
|
|
372
|
+
// A checkpointed reading is stable for a different reason: the tree it
|
|
373
|
+
// read was pinned and cannot move, so the ordering is established by
|
|
374
|
+
// the pin rather than by position — which is exactly what that surface
|
|
375
|
+
// exists to buy, and what the run used to buy by reverting its work.
|
|
376
|
+
stable: checkpointed || (!unattributedMutation && index > lastStandingWrite)
|
|
377
|
+
})
|
|
378
|
+
return recorded === undefined ? [] : [recorded]
|
|
379
|
+
})
|
|
380
|
+
const frameChecks = readings(false)
|
|
381
|
+
|
|
382
|
+
return {
|
|
383
|
+
mutated,
|
|
384
|
+
calls,
|
|
385
|
+
source: options.source ?? "",
|
|
386
|
+
observed: new AgentEvent.MutationObserved({
|
|
387
|
+
eventType: eventType.mutationObserved,
|
|
388
|
+
basis: covered ? "observed" : measured ? "partial" : "declared",
|
|
389
|
+
mutated,
|
|
390
|
+
digest: closingDigest,
|
|
391
|
+
paths: Option.match(closed, { onNone: () => 0, onSome: (value) => value.paths }),
|
|
392
|
+
declaredWrites
|
|
393
|
+
}),
|
|
394
|
+
workspaceDigest,
|
|
395
|
+
frameChecks,
|
|
396
|
+
probeNotice: invalidProbeNotice(calls),
|
|
397
|
+
facts: {
|
|
398
|
+
// Seeded from what earlier frames were handed and appended to as this
|
|
399
|
+
// frame's calls settled.
|
|
400
|
+
truncatedOutputs: TruncatedOutput.retain(options.captures),
|
|
401
|
+
workspace: closed,
|
|
402
|
+
// A frame is read-only when it changed nothing; see
|
|
403
|
+
// `State.readOnlyFrames` for why that is measured and not declared.
|
|
404
|
+
// Once the run has written something, a read-only frame that settled a
|
|
405
|
+
// call this run had not issued before holds the streak where it was: it
|
|
406
|
+
// asked a new question about work that exists, which is debugging, not a
|
|
407
|
+
// stall. Before the first write every read-only frame counts, because a
|
|
408
|
+
// run that only ever reads new things is the failure the cap was built
|
|
409
|
+
// for. Zero-call and repeat-only frames always advance it.
|
|
410
|
+
readOnlyFrames: mutated
|
|
411
|
+
? 0
|
|
412
|
+
: novel && state.mutations > 0
|
|
413
|
+
? state.readOnlyFrames
|
|
414
|
+
: state.readOnlyFrames + 1,
|
|
415
|
+
repeatFrames,
|
|
416
|
+
callSignatures: remember(state.callSignatures, signatures),
|
|
417
|
+
// Ids this frame pinned: a checkpoint a frame minted before it raised is
|
|
418
|
+
// still a tree the run holds, and forgetting it would leak a stored tree
|
|
419
|
+
// nothing can name.
|
|
420
|
+
checkpointIds: options.minted.length === 0 ? state.checkpointIds : [...state.checkpointIds, ...options.minted],
|
|
421
|
+
// Live readings only. Every consumer of this ledger — the narrowing
|
|
422
|
+
// demand, the unresolved-failure demand, the vacuous-verification notice
|
|
423
|
+
// — asks whether the tree the run is completing on was checked, and a
|
|
424
|
+
// checkpointed reading is not a reading of that tree.
|
|
425
|
+
checks: NarrowedCheck.remember(state.checks, frameChecks),
|
|
426
|
+
// What this frame asked, folded into what the run had already asked. A
|
|
427
|
+
// raise carries it too: a call that settled before the throw is work
|
|
428
|
+
// the run paid for, and the ledger is the only place the next model turn
|
|
429
|
+
// can read it.
|
|
430
|
+
callLedger: CallLedger.remember(state.callLedger, calls),
|
|
431
|
+
// Every call that reported an exit status, writes included, for the
|
|
432
|
+
// claim brake. See `CompletionClaim.record`.
|
|
433
|
+
reported: CompletionClaim.record(state.reported, calls),
|
|
434
|
+
// Which of this frame's checks failed before any change did, and how
|
|
435
|
+
// many frames have changed the workspace. Both, because this is the one
|
|
436
|
+
// ledger a checkpointed reading belongs in: its whole question is an
|
|
437
|
+
// ordering — this failed, then something changed, then that passed — and
|
|
438
|
+
// a pinned tree answers the first half honestly from a frame that also
|
|
439
|
+
// did the changing.
|
|
440
|
+
failures: Sufficiency.remember(state.failures, {
|
|
441
|
+
frame: [...readings(true), ...frameChecks],
|
|
442
|
+
epoch: state.mutations
|
|
443
|
+
}),
|
|
444
|
+
mutations: state.mutations + (mutated ? 1 : 0),
|
|
445
|
+
remoteMutations: state.remoteMutations + remoteWrites,
|
|
446
|
+
// The tree the run was handed, fixed the first time a frame measured one
|
|
447
|
+
// and never restamped. See `State.openingDigest`.
|
|
448
|
+
openingDigest: state.openingDigest !== ""
|
|
449
|
+
? state.openingDigest
|
|
450
|
+
: Option.match(opened, {
|
|
451
|
+
onNone: () => "",
|
|
452
|
+
onSome: (value) => value.complete ? value.digest : ""
|
|
453
|
+
}),
|
|
454
|
+
// The panel the next frame opens with, stamped by the frame that ran.
|
|
455
|
+
panel: VariablesPanel.stamp(state.panel, options.bindings, state.frame),
|
|
456
|
+
...(mutated ? { readOnlyGrace: 0, pendingReadOnlyDemand: undefined } : {})
|
|
457
|
+
}
|
|
458
|
+
}
|
|
459
|
+
}
|
|
460
|
+
|
|
461
|
+
/**
|
|
462
|
+
* The distinct terms of everything the harness itself put in front of the run.
|
|
463
|
+
*
|
|
464
|
+
* The prefix zone is exactly that text — the cell contract, the flow catalog,
|
|
465
|
+
* memory, and the task — and nothing else: transcript, observations, and
|
|
466
|
+
* compaction summaries all land in the tail, so no term a model wrote can
|
|
467
|
+
* reach this set. `NarrowedCheck.findOnly` reads it as the vocabulary the run
|
|
468
|
+
* was taught, which no completion may be bounced for repeating: a run that
|
|
469
|
+
* invokes the runner its task prescribes, with the flag its task prescribes,
|
|
470
|
+
* added no condition of its own. Prefix parts that carry no text — a tool
|
|
471
|
+
* declaration, a structured message — teach no terms.
|
|
472
|
+
*/
|
|
473
|
+
const taughtTerms = (window: ContextWindow.ContextWindow): ReadonlyArray<string> =>
|
|
474
|
+
NarrowedCheck.terms(
|
|
475
|
+
window.segments
|
|
476
|
+
.filter((segment) => segment.zone === "prefix")
|
|
477
|
+
.flatMap((segment) => segment.content)
|
|
478
|
+
.map((part) => "text" in part && typeof part.text === "string" ? part.text : "")
|
|
479
|
+
.join("\n")
|
|
480
|
+
)
|
|
481
|
+
|
|
482
|
+
/**
|
|
483
|
+
* One completion handed back, and what handing it back costs.
|
|
484
|
+
*
|
|
485
|
+
* @since 1.0.0-rc.0
|
|
486
|
+
* @private
|
|
487
|
+
*/
|
|
488
|
+
export interface CompletionDemand {
|
|
489
|
+
/** The control event that names the demand in the journal. */
|
|
490
|
+
readonly event:
|
|
491
|
+
| AgentEvent.UnmovedDemanded
|
|
492
|
+
| AgentEvent.UnresolvedDemanded
|
|
493
|
+
| AgentEvent.FailedCallDemanded
|
|
494
|
+
| AgentEvent.UnobservedDemanded
|
|
495
|
+
| AgentEvent.NarrowedDemanded
|
|
496
|
+
| AgentEvent.NarrowOnlyDemanded
|
|
497
|
+
| AgentEvent.ClaimDemanded
|
|
498
|
+
/** The in-frame observation the next frame answers. */
|
|
499
|
+
readonly note: string
|
|
500
|
+
/**
|
|
501
|
+
* Whether the answer this demand takes away may come back as the run's
|
|
502
|
+
* answer when the frame budget runs out; see `CellTurn.budgetMessage`.
|
|
503
|
+
*
|
|
504
|
+
* True for the measured demands except `FailedCall`, whose answer was
|
|
505
|
+
* written over a call that failed. `UnobservedCall` keeps its answer: the
|
|
506
|
+
* brake reads the source, not the sentence, so an answer it refused may be
|
|
507
|
+
* right, and a run that ends on the budget should end on it rather than on
|
|
508
|
+
* a bare notice. Each of the others says the record is
|
|
509
|
+
* missing a fact, not that the sentence is wrong, so a run that spends its
|
|
510
|
+
* last frame and never completes again is better served by the answer it
|
|
511
|
+
* wrote than by a bare budget notice.
|
|
512
|
+
*
|
|
513
|
+
* For the claim brake it depends on which of its two heights fired, because
|
|
514
|
+
* they mean different things. A reading between `CompletionClaim.unsupportedAt`
|
|
515
|
+
* and `CompletionClaim.inventedAt` is a bounce that would let the same
|
|
516
|
+
* sentence stand if the run re-stated it, so discarding it against an
|
|
517
|
+
* exhausted budget would throw away an answer the brake was never going to
|
|
518
|
+
* refuse. A reading at or above `inventedAt` is one the brake *would* refuse:
|
|
519
|
+
* restoring that sentence on the budget notice is how one measured run turned
|
|
520
|
+
* a bounced "the tests pass" into its final answer with a `stop` finish over a
|
|
521
|
+
* repository whose test exits 1. A bounce that cannot be re-judged must not be
|
|
522
|
+
* undone by the budget. See `CompletionClaim.unrecorded`.
|
|
523
|
+
*/
|
|
524
|
+
readonly keeps: boolean
|
|
525
|
+
/** The cap the demand spends. */
|
|
526
|
+
readonly spent: StateChanges
|
|
527
|
+
}
|
|
528
|
+
|
|
529
|
+
/**
|
|
530
|
+
* What judging one completion produced: at most one demand, and at most one
|
|
531
|
+
* reading to journal whichever way it went.
|
|
532
|
+
*
|
|
533
|
+
* `observed` exists for the sixth brake alone. The five before it are derived
|
|
534
|
+
* from measurements the journal already carries, so a grader recomputes them
|
|
535
|
+
* and there is nothing to write when they stay silent; the claim brake asks a
|
|
536
|
+
* model, and a reading nobody records is a reading nobody can grade. It is set
|
|
537
|
+
* only where that brake ran and issued no demand — when it does demand, the
|
|
538
|
+
* same event travels on `demand.event`, so exactly one `claim-demanded` is
|
|
539
|
+
* written per evaluation.
|
|
540
|
+
*
|
|
541
|
+
* @since 1.0.0-rc.0
|
|
542
|
+
* @private
|
|
543
|
+
*/
|
|
544
|
+
export interface CompletionJudgement {
|
|
545
|
+
/** The claim brake's reading, when it ran and issued no demand. */
|
|
546
|
+
readonly observed: AgentEvent.ClaimDemanded | undefined
|
|
547
|
+
/** The demand that hands the completion back, when one of the six issued. */
|
|
548
|
+
readonly demand: CompletionDemand | undefined
|
|
549
|
+
/**
|
|
550
|
+
* The failure the run ends with when the claim brake read a claim the
|
|
551
|
+
* evidence does not support and no bounce was left to spend. It travels on
|
|
552
|
+
* the judgement rather than as the effect's own failure so the caller
|
|
553
|
+
* journals `observed` first: a reading that ends a run is the one a grader
|
|
554
|
+
* most needs, and an effect that failed would take it with it.
|
|
555
|
+
*/
|
|
556
|
+
readonly unproven: HarnessError.HarnessError | undefined
|
|
557
|
+
/**
|
|
558
|
+
* The claim brake's whole decision, whenever it read one: the evidence, the
|
|
559
|
+
* questions and the answers behind the three numbers `observed` or the
|
|
560
|
+
* demand carries. Set on every way a reading comes out, because the record a
|
|
561
|
+
* reader reopens a decision from is needed most for the one that acted.
|
|
562
|
+
*/
|
|
563
|
+
readonly decision: AgentEvent.DecisionSettled | undefined
|
|
564
|
+
/** The per-sentence decision, when the brake asked for one. */
|
|
565
|
+
readonly sentenceDecision?: AgentEvent.DecisionSettled | undefined
|
|
566
|
+
/**
|
|
567
|
+
* Whether the completion reported its own work unfinished, when the verdict
|
|
568
|
+
* asked. See `UnfinishedWork`.
|
|
569
|
+
*/
|
|
570
|
+
readonly unfinishedDecision?: AgentEvent.DecisionSettled | undefined
|
|
571
|
+
}
|
|
572
|
+
|
|
573
|
+
/** Nothing to say about this completion: it stands. */
|
|
574
|
+
const stands: CompletionJudgement = {
|
|
575
|
+
observed: undefined,
|
|
576
|
+
demand: undefined,
|
|
577
|
+
unproven: undefined,
|
|
578
|
+
decision: undefined
|
|
579
|
+
}
|
|
580
|
+
|
|
581
|
+
/** One demand, with the brake's decision beside it when the brake issued it. */
|
|
582
|
+
const handBack = (demand: CompletionDemand, decision?: AgentEvent.DecisionSettled): CompletionJudgement => ({
|
|
583
|
+
observed: undefined,
|
|
584
|
+
demand,
|
|
585
|
+
unproven: undefined,
|
|
586
|
+
decision
|
|
587
|
+
})
|
|
588
|
+
|
|
589
|
+
/**
|
|
590
|
+
* The prose the harness itself put in front of the run as its task.
|
|
591
|
+
*
|
|
592
|
+
* The `instructions` segments of the prefix zone and nothing else: `Agent`
|
|
593
|
+
* writes the task there, while the cell contract, the flow catalog, the
|
|
594
|
+
* project's instructions and the host's memory go in as `system` and the
|
|
595
|
+
* transcript goes in the tail. So this is the closest thing the controller holds to the task as
|
|
596
|
+
* the person stated it, and it cannot pick up a sentence the model wrote.
|
|
597
|
+
*
|
|
598
|
+
* @since 1.0.0-rc.0
|
|
599
|
+
* @private
|
|
600
|
+
*/
|
|
601
|
+
export const taskText = (window: ContextWindow.ContextWindow): string =>
|
|
602
|
+
window.segments
|
|
603
|
+
.filter((segment) => segment.zone === "prefix" && segment.kind === "instructions")
|
|
604
|
+
.flatMap((segment) => segment.content)
|
|
605
|
+
.map((part) => "text" in part && typeof part.text === "string" ? part.text : "")
|
|
606
|
+
.filter((text) => text !== "")
|
|
607
|
+
.join("\n\n")
|
|
608
|
+
|
|
609
|
+
/**
|
|
610
|
+
* The last reading of the completing frame that reported an exit status.
|
|
611
|
+
*
|
|
612
|
+
* The verbatim result exists for exactly one frame, the one being judged, so
|
|
613
|
+
* this is the only check the brake can quote in full, and a completing frame
|
|
614
|
+
* that ran none sends none rather than sending a description of one.
|
|
615
|
+
*
|
|
616
|
+
* It is not the run's evidence, only the newest page of it. `CompletionClaim.record`
|
|
617
|
+
* is the rest, and the two are separate because one measured live failure was
|
|
618
|
+
* exactly the difference: a run that fixed the planted bug, ran the
|
|
619
|
+
* repository's test, and then ran `git diff` to show its work sent the
|
|
620
|
+
* `git diff` and not the test, so its true sentence reported a result nothing
|
|
621
|
+
* in the payload recorded. See `CompletionClaim.Evidence`.
|
|
622
|
+
*/
|
|
623
|
+
const lastCheck = (calls: ReadonlyArray<ObservedCall>): CompletionClaim.Check | undefined => {
|
|
624
|
+
for (let index = calls.length - 1; index >= 0; index--) {
|
|
625
|
+
const call = calls[index]!
|
|
626
|
+
if (!call.ok || call.mutates) continue
|
|
627
|
+
const status = UnresolvedFailure.exitStatus(call.value)
|
|
628
|
+
if (status === undefined) continue
|
|
629
|
+
return {
|
|
630
|
+
command: CompletionClaim.quote(call.input),
|
|
631
|
+
exitCode: status,
|
|
632
|
+
output: CompletionClaim.newest(CompletionClaim.quote(call.value))
|
|
633
|
+
}
|
|
634
|
+
}
|
|
635
|
+
return undefined
|
|
636
|
+
}
|
|
637
|
+
|
|
638
|
+
/**
|
|
639
|
+
* The demands a completion's own measurements produce, in precedence order —
|
|
640
|
+
* `FailedCall`, `UnobservedCall`, then the three below — or nothing. The
|
|
641
|
+
* `UnmovedTree` demand is not one of them: an unmoved tree is the right tree
|
|
642
|
+
* for a question, so it is issued only when the claim brake reads the
|
|
643
|
+
* completion as unsupported. See `judgeCompletion`. Every fact read here was taken by the frame that is
|
|
644
|
+
* completing or by an earlier one, so this is a pure function of the state
|
|
645
|
+
* and the accounting and it is what `judgeCompletion` consults first.
|
|
646
|
+
*
|
|
647
|
+
* @since 1.0.0-rc.0
|
|
648
|
+
* @private
|
|
649
|
+
*/
|
|
650
|
+
const measuredDemand = (
|
|
651
|
+
state: State,
|
|
652
|
+
accounting: Accounting,
|
|
653
|
+
contextWindow: ContextWindow.ContextWindow,
|
|
654
|
+
nextFrame: number,
|
|
655
|
+
claim: string
|
|
656
|
+
): CompletionDemand | undefined => {
|
|
657
|
+
const { calls, facts, frameChecks, workspaceDigest } = accounting
|
|
658
|
+
// First, because the others read a record this completion was written
|
|
659
|
+
// without: its own cell failed a call before the claim existed. It is the
|
|
660
|
+
// one measured demand whose answer is not kept, because the answer it takes
|
|
661
|
+
// away is the sentence written blind. See `FailedCall`.
|
|
662
|
+
if (state.failedCallDemands < FailedCall.cap) {
|
|
663
|
+
const failures = FailedCall.find(calls, claim, accounting.source)
|
|
664
|
+
if (failures.length > 0) {
|
|
665
|
+
return {
|
|
666
|
+
event: new AgentEvent.FailedCallDemanded({
|
|
667
|
+
eventType: eventType.failedCallDemanded,
|
|
668
|
+
failures: failures.map((failure) => ({ flow: failure.flow, message: failure.message })),
|
|
669
|
+
nextFrame
|
|
670
|
+
}),
|
|
671
|
+
note: FailedCall.demand(failures),
|
|
672
|
+
keeps: false,
|
|
673
|
+
spent: { failedCallDemands: state.failedCallDemands + 1 }
|
|
674
|
+
}
|
|
675
|
+
}
|
|
676
|
+
}
|
|
677
|
+
// Second, for the same reason: the claim was written before any result of
|
|
678
|
+
// its own cell existed, so every demand below would grade a sentence the
|
|
679
|
+
// model wrote without reading. Kept, unlike the one above: the source was
|
|
680
|
+
// read, not the sentence, so it stays the run's answer at the budget. See
|
|
681
|
+
// `UnobservedCall`.
|
|
682
|
+
if (state.unobservedDemands < UnobservedCall.cap) {
|
|
683
|
+
const unread = UnobservedCall.find(calls, accounting.source)
|
|
684
|
+
if (unread.length > 0) {
|
|
685
|
+
return {
|
|
686
|
+
event: new AgentEvent.UnobservedDemanded({
|
|
687
|
+
eventType: eventType.unobservedDemanded,
|
|
688
|
+
calls: unread,
|
|
689
|
+
nextFrame
|
|
690
|
+
}),
|
|
691
|
+
note: UnobservedCall.demand(unread),
|
|
692
|
+
keeps: true,
|
|
693
|
+
spent: { unobservedDemands: state.unobservedDemands + 1 }
|
|
694
|
+
}
|
|
695
|
+
}
|
|
696
|
+
}
|
|
697
|
+
if (state.unresolvedDemands < state.unresolvedCap) {
|
|
698
|
+
const unresolved = UnresolvedFailure.find({ ledger: facts.checks, digest: workspaceDigest })
|
|
699
|
+
if (unresolved !== undefined) {
|
|
700
|
+
return {
|
|
701
|
+
event: new AgentEvent.UnresolvedDemanded({
|
|
702
|
+
eventType: eventType.unresolvedDemanded,
|
|
703
|
+
flow: unresolved.failed.flow,
|
|
704
|
+
failed: unresolved.failed.label,
|
|
705
|
+
instead: unresolved.instead.label,
|
|
706
|
+
currentDigest: workspaceDigest,
|
|
707
|
+
nextFrame
|
|
708
|
+
}),
|
|
709
|
+
note: UnresolvedFailure.demand(unresolved),
|
|
710
|
+
keeps: true,
|
|
711
|
+
spent: { unresolvedDemands: state.unresolvedDemands + 1 }
|
|
712
|
+
}
|
|
713
|
+
}
|
|
714
|
+
}
|
|
715
|
+
if (state.narrowingDemands >= state.narrowingCap) return undefined
|
|
716
|
+
const spent = { narrowingDemands: state.narrowingDemands + 1 }
|
|
717
|
+
const narrowing = NarrowedCheck.find({ ledger: state.checks, frame: frameChecks, digest: workspaceDigest })
|
|
718
|
+
if (narrowing !== undefined) {
|
|
719
|
+
return {
|
|
720
|
+
event: new AgentEvent.NarrowedDemanded({
|
|
721
|
+
eventType: eventType.narrowedDemanded,
|
|
722
|
+
flow: narrowing.earlier.flow,
|
|
723
|
+
broader: narrowing.earlier.label,
|
|
724
|
+
narrower: narrowing.later.label,
|
|
725
|
+
broaderDigest: narrowing.earlier.digest,
|
|
726
|
+
currentDigest: workspaceDigest,
|
|
727
|
+
nextFrame
|
|
728
|
+
}),
|
|
729
|
+
note: NarrowedCheck.demand(narrowing),
|
|
730
|
+
keeps: true,
|
|
731
|
+
spent
|
|
732
|
+
}
|
|
733
|
+
}
|
|
734
|
+
const only = NarrowedCheck.findOnly({
|
|
735
|
+
ledger: facts.checks,
|
|
736
|
+
before: state.checks.map((entry) => entry.signature),
|
|
737
|
+
frame: frameChecks,
|
|
738
|
+
taught: taughtTerms(contextWindow)
|
|
739
|
+
})
|
|
740
|
+
if (only === undefined) return undefined
|
|
741
|
+
return {
|
|
742
|
+
event: new AgentEvent.NarrowOnlyDemanded({
|
|
743
|
+
eventType: eventType.narrowOnlyDemanded,
|
|
744
|
+
flow: only.later.flow,
|
|
745
|
+
check: only.later.label,
|
|
746
|
+
targets: only.targets,
|
|
747
|
+
currentDigest: workspaceDigest,
|
|
748
|
+
nextFrame
|
|
749
|
+
}),
|
|
750
|
+
note: NarrowedCheck.demandOnly(only),
|
|
751
|
+
keeps: true,
|
|
752
|
+
spent
|
|
753
|
+
}
|
|
754
|
+
}
|
|
755
|
+
|
|
756
|
+
/**
|
|
757
|
+
* Whether a `complete` transition stands, or which demand hands it back.
|
|
758
|
+
*
|
|
759
|
+
* The completion's own evidence, judged once per demand. A run gets exactly
|
|
760
|
+
* one frame wrong for free — the last one — and five things can be wrong with
|
|
761
|
+
* it, four of them read off measurements the controller already took, under
|
|
762
|
+
* four caps:
|
|
763
|
+
*
|
|
764
|
+
* 1. `UnmovedTree`: the tree it is completing on is the tree it opened on, so
|
|
765
|
+
* there is no change for any evidence to be about. This one is not read
|
|
766
|
+
* alone: a question is answered on the tree it was asked on, so it is
|
|
767
|
+
* issued only where the claim brake (5) found the completion unsupported,
|
|
768
|
+
* and then in the brake's place;
|
|
769
|
+
* 2. `UnresolvedFailure`: a check over this exact tree reported a failing exit
|
|
770
|
+
* status and the run answered it with a different reading of the same
|
|
771
|
+
* subject rather than with the check itself;
|
|
772
|
+
* 3. `NarrowedCheck.find`: this frame's check repeats an earlier, broader one
|
|
773
|
+
* and adds conditions to it, run after a change the broader one never saw;
|
|
774
|
+
* 4. `NarrowedCheck.findOnly`: the check this frame ended on is the run's only
|
|
775
|
+
* reading of what it names — nothing broader was ever taken, so there was
|
|
776
|
+
* no broader check for (3) to find — and it carries a condition the run
|
|
777
|
+
* itself added, one taught neither by the prefix this harness wrote nor by
|
|
778
|
+
* the run's own other checks.
|
|
779
|
+
*
|
|
780
|
+
* Demands 3 and 4 share one cap. They are two readings of one question —
|
|
781
|
+
* whether the evidence covers what it looks like it covers — and a run that
|
|
782
|
+
* answers either has answered the question; a second bounce would be the loop
|
|
783
|
+
* asking it twice in different words.
|
|
784
|
+
*
|
|
785
|
+
* 5. `CompletionClaim`: the last brake and the only one that is not a
|
|
786
|
+
* measurement. The four above have said nothing, which means the tree
|
|
787
|
+
* moved, no check was stepped around, and whatever the run checked it
|
|
788
|
+
* checked whole — and none of that reads the sentence the run wrote. So
|
|
789
|
+
* the claim, the task, the tree fact, every check the run took over this
|
|
790
|
+
* tree and the verbatim result of the last one go to Jev, and a claim that
|
|
791
|
+
* reports a command or a result none of that records hands the frame back
|
|
792
|
+
* from a cap of its own. It is last because it is the only one
|
|
793
|
+
* that costs a request, and because a run one of the four already named
|
|
794
|
+
* has a demand to answer: asking a model to add a second one would hand
|
|
795
|
+
* the frame two questions. It never falls back: a completion Jev could
|
|
796
|
+
* not judge — no evaluator on the host, a refusal, a deadline, an answer
|
|
797
|
+
* that does not decode — fails the turn as `completion_unjudged` carrying
|
|
798
|
+
* the reason, the way `read_only_cap` ends a run, rather than standing.
|
|
799
|
+
* `Evaluator` is therefore a required service of this function and of
|
|
800
|
+
* every turn above it.
|
|
801
|
+
*
|
|
802
|
+
* The fifth is the one that is read on *every* completion, and the three
|
|
803
|
+
* things below that take a demand away do not take the reading away. The
|
|
804
|
+
* others are recomputable from the journal, so skipping them costs a grader
|
|
805
|
+
* nothing and a skipped one lets a completion stand that the truth bar still
|
|
806
|
+
* judges. This one is a sentence being checked against the record, it is what
|
|
807
|
+
* the server banner and the operator docs promise happens to every completion,
|
|
808
|
+
* and the run's answer is the product. So when there is no bounce left to
|
|
809
|
+
* spend — the cap is used up, or none of the room below exists — a claim that
|
|
810
|
+
* reports work the record does not record ends the run as `claim_unproven`
|
|
811
|
+
* instead of standing. That is the shape `read_only_cap` already uses, and it
|
|
812
|
+
* is the only shape that keeps the promise: a cap that stops at the bounce
|
|
813
|
+
* means the second identical claim is accepted unread, which is what a live
|
|
814
|
+
* run did.
|
|
815
|
+
*
|
|
816
|
+
* The verdict is narrower than the bounce, and that is the whole of what this
|
|
817
|
+
* lane changed. A completion the brake merely finds thin is handed back once
|
|
818
|
+
* and then stands; only a completion at `CompletionClaim.inventedAt` — a
|
|
819
|
+
* sentence reporting a command or a result nothing in the run produced — is
|
|
820
|
+
* refused. Arming the verdict on "is the task done" instead killed roughly one
|
|
821
|
+
* honest run in four, including five question-shaped turns in a row and one
|
|
822
|
+
* live CI dispatch whose planted bug was fixed. `CompletionClaim`'s header
|
|
823
|
+
* carries the eighteen-state corpus that measured it.
|
|
824
|
+
*
|
|
825
|
+
* A thin completion that stands is still not a success when it says so
|
|
826
|
+
* itself. Where the verdict is reached on a completion read as not done, one
|
|
827
|
+
* more question asks whether the completion reports its own work unfinished,
|
|
828
|
+
* and one that does ends the run as `completion_incomplete` quoting that
|
|
829
|
+
* report, so a host never settles "I could not finish" as a completed run.
|
|
830
|
+
* See `UnfinishedWork`.
|
|
831
|
+
*
|
|
832
|
+
* At most one is named, in that order, because they are in descending order of
|
|
833
|
+
* how fundamental the missing thing is: there is nothing to check, then the
|
|
834
|
+
* check said no, then the check said less than it looks like it said, then
|
|
835
|
+
* nothing in the record matches what the run said it did. Naming two at once
|
|
836
|
+
* would ask the run to answer a question it has not been given a frame for.
|
|
837
|
+
*
|
|
838
|
+
* The loop names what is missing and hands the frame back; it does not re-run
|
|
839
|
+
* anything, and it does not judge the answer that comes back.
|
|
840
|
+
*
|
|
841
|
+
* Asking costs the run a frame it can answer in, so a demand is issued only
|
|
842
|
+
* where that frame exists, and three separate things take it away (and leave
|
|
843
|
+
* the claim brake reading anyway, as above):
|
|
844
|
+
*
|
|
845
|
+
* - the frame budget, which has no frame left to spend, and turning a
|
|
846
|
+
* completion into an exhausted budget would lose the run's answer to make a
|
|
847
|
+
* point about it;
|
|
848
|
+
* - the read-only cap, which is the other budget that ends a run and ends it as
|
|
849
|
+
* a typed failure carrying nothing. A run that changed nothing is exactly the
|
|
850
|
+
* run `UnmovedTree` fires on, so a completion one frame short of twice the
|
|
851
|
+
* cap would be bounced, spend that frame reading, and die as `read_only_cap`
|
|
852
|
+
* with the answer it had already written discarded — a demand turning a
|
|
853
|
+
* finished run into a failure, which is the one outcome none of these may
|
|
854
|
+
* produce;
|
|
855
|
+
* - a demand this run has already answered. Each demand ends by promising that
|
|
856
|
+
* what comes back next is the answer that stands, and three of them fire on
|
|
857
|
+
* one transition, so the frame written to answer one is never judged by the
|
|
858
|
+
* next. See `State.demandedFrame`.
|
|
859
|
+
*
|
|
860
|
+
* @since 1.0.0-rc.0
|
|
861
|
+
* @private
|
|
862
|
+
*/
|
|
863
|
+
export const judgeCompletion = (
|
|
864
|
+
state: State,
|
|
865
|
+
accounting: Accounting,
|
|
866
|
+
contextWindow: ContextWindow.ContextWindow,
|
|
867
|
+
claim: string,
|
|
868
|
+
read: typeof CompletionClaim.read = CompletionClaim.read
|
|
869
|
+
): Effect.Effect<CompletionJudgement, HarnessError.HarnessError, Evaluator.Evaluator> =>
|
|
870
|
+
Effect.gen(function*() {
|
|
871
|
+
const { calls, facts, workspaceDigest } = accounting
|
|
872
|
+
const room = handBackRoom(state, facts.readOnlyFrames) && state.demandedFrame !== state.frame
|
|
873
|
+
const nextFrame = state.frame + 1
|
|
874
|
+
if (room) {
|
|
875
|
+
const measured = measuredDemand(state, accounting, contextWindow, nextFrame, claim)
|
|
876
|
+
// A measured demand names a missing fact, so the answer it takes away
|
|
877
|
+
// is worth restoring if the budget runs out, unless the demand says the
|
|
878
|
+
// answer was written blind. See `CompletionDemand`.
|
|
879
|
+
if (measured !== undefined) return handBack(measured)
|
|
880
|
+
}
|
|
881
|
+
|
|
882
|
+
// The sixth brake, and the only one that leaves this package to decide.
|
|
883
|
+
// `claimCap` of zero disarms it: no request, no event, no failure, which
|
|
884
|
+
// is what a host that does not want a model in this path asks for.
|
|
885
|
+
if (state.claimCap === 0) return stands
|
|
886
|
+
// A host may put prior conversation before the current request. Keep
|
|
887
|
+
// both ends of the task; clipping its head alone can leave only history.
|
|
888
|
+
// Reserve the elision notice inside the task's byte budget.
|
|
889
|
+
const originalTask = taskText(contextWindow).trim()
|
|
890
|
+
const recallTask = "the run record has the whole task"
|
|
891
|
+
const task = elide.middle(
|
|
892
|
+
originalTask,
|
|
893
|
+
CompletionClaim.proseBytes - elide.noticeCost(bytes.size(originalTask), recallTask),
|
|
894
|
+
recallTask
|
|
895
|
+
)
|
|
896
|
+
const check = lastCheck(calls)
|
|
897
|
+
const unmovedTree = UnmovedTree.find({
|
|
898
|
+
opened: facts.openingDigest,
|
|
899
|
+
digest: workspaceDigest,
|
|
900
|
+
elsewhere: facts.remoteMutations
|
|
901
|
+
})
|
|
902
|
+
const evidence: CompletionClaim.Evidence = {
|
|
903
|
+
task,
|
|
904
|
+
claim: CompletionClaim.prose(claim),
|
|
905
|
+
// The `UnmovedTree` fact, read the other way round. An unmeasured tree
|
|
906
|
+
// reads as moved, which is the reading that asks for nothing.
|
|
907
|
+
treeMoved: unmovedTree === undefined,
|
|
908
|
+
checksRun: facts.reported,
|
|
909
|
+
callsRun: facts.callLedger.map((entry) => ({
|
|
910
|
+
flow: entry.flow,
|
|
911
|
+
input: entry.subject,
|
|
912
|
+
ok: entry.ok,
|
|
913
|
+
resultSummary: entry.digest
|
|
914
|
+
})),
|
|
915
|
+
...(check === undefined ? {} : { lastCheck: check })
|
|
916
|
+
}
|
|
917
|
+
const reading = yield* read(evidence)
|
|
918
|
+
if (reading === undefined) return stands
|
|
919
|
+
const found = CompletionClaim.find(reading)
|
|
920
|
+
// One bounce while the cap and a frame allow it; the verdict after that,
|
|
921
|
+
// and only over the readings the verdict is about.
|
|
922
|
+
const bounced = found !== undefined && room && state.claimDemands < state.claimCap
|
|
923
|
+
// An unmoved tree alone is not a demand: a read-only question is answered
|
|
924
|
+
// on exactly the tree it was asked on, and bouncing that answer discarded
|
|
925
|
+
// correct replies (#2937). The tree fact names what is missing only when
|
|
926
|
+
// this reading found the claim unsupported, and then its wording is the
|
|
927
|
+
// more precise of the two, so it takes the bounce while its cap lasts and
|
|
928
|
+
// the reading is journaled beside it as one that did not demand.
|
|
929
|
+
const unmoved = found !== undefined && room && state.unmovedDemands < state.unmovedCap
|
|
930
|
+
? unmovedTree
|
|
931
|
+
: undefined
|
|
932
|
+
// The third way a reading comes out, stated on the event because nothing
|
|
933
|
+
// downstream can derive it: a reading that neither stands nor hands the
|
|
934
|
+
// frame back is the one that ends the run, and a projection that could not
|
|
935
|
+
// tell it from a reading that stood wrote no card for it. See
|
|
936
|
+
// `AgentEvent.ClaimDemanded.refused`.
|
|
937
|
+
const refused = found !== undefined && !bounced && unmoved === undefined && CompletionClaim.unrecorded(found)
|
|
938
|
+
// A completion read as not done, with no bounce left and nothing to
|
|
939
|
+
// refuse, used to stand here whatever it said, so a run reporting its own
|
|
940
|
+
// work unfinished settled as a completed one (#3009). Whether it said so
|
|
941
|
+
// is one more question, asked only here; see `UnfinishedWork`.
|
|
942
|
+
const unfinished = found !== undefined && !bounced && unmoved === undefined && !refused &&
|
|
943
|
+
reading.complete <= CompletionClaim.disprovenAt
|
|
944
|
+
? yield* UnfinishedWork.read(evidence, reading.usage)
|
|
945
|
+
: undefined
|
|
946
|
+
const incomplete = unfinished !== undefined && unfinished.unfinished >= UnfinishedWork.reportedAt
|
|
947
|
+
const usage = paidTogether(reading.usage, unfinished?.usage)
|
|
948
|
+
const event = new AgentEvent.ClaimDemanded({
|
|
949
|
+
eventType: eventType.claimDemanded,
|
|
950
|
+
complete: reading.complete,
|
|
951
|
+
overclaims: reading.overclaims,
|
|
952
|
+
invented: reading.invented,
|
|
953
|
+
latencyMs: reading.latencyMs + (unfinished?.latencyMs ?? 0),
|
|
954
|
+
...(usage === undefined ? {} : { usage }),
|
|
955
|
+
demanded: bounced && unmoved === undefined,
|
|
956
|
+
refused,
|
|
957
|
+
currentDigest: workspaceDigest,
|
|
958
|
+
nextFrame
|
|
959
|
+
})
|
|
960
|
+
// The same reading with what it was a reading OF. `acted` is the two ways
|
|
961
|
+
// a reading changes what the run does next, and a reader that reported no
|
|
962
|
+
// evidence journals no decision: see `CompletionClaim.Reading.asked`.
|
|
963
|
+
const decision = reading.asked === undefined ? undefined : new AgentEvent.DecisionSettled({
|
|
964
|
+
eventType: eventType.decisionSettled,
|
|
965
|
+
scope: state.session,
|
|
966
|
+
frame: state.frame,
|
|
967
|
+
classifier: CompletionClaim.classifier.id,
|
|
968
|
+
digest: CompletionClaim.classifier.digest,
|
|
969
|
+
state: reading.asked.state,
|
|
970
|
+
questions: CompletionClaim.classifier.questions,
|
|
971
|
+
answers: reading.asked.answers,
|
|
972
|
+
latencyMs: reading.latencyMs,
|
|
973
|
+
acted: bounced || refused || unmoved !== undefined || incomplete,
|
|
974
|
+
decidedBy: "jev"
|
|
975
|
+
})
|
|
976
|
+
const unfinishedDecision = unfinished === undefined ? undefined : new AgentEvent.DecisionSettled({
|
|
977
|
+
eventType: eventType.decisionSettled,
|
|
978
|
+
scope: state.session,
|
|
979
|
+
frame: state.frame,
|
|
980
|
+
classifier: unfinished.asked.classifier,
|
|
981
|
+
digest: unfinished.asked.digest,
|
|
982
|
+
state: unfinished.asked.state,
|
|
983
|
+
questions: unfinished.asked.questions,
|
|
984
|
+
answers: unfinished.asked.answers,
|
|
985
|
+
latencyMs: unfinished.latencyMs,
|
|
986
|
+
acted: incomplete,
|
|
987
|
+
decidedBy: "jev"
|
|
988
|
+
})
|
|
989
|
+
// The per-sentence reading, journaled as its own decision because it is
|
|
990
|
+
// one: its own classifier, questions and answers. See
|
|
991
|
+
// `CompletionClaim.sentenceClassifier`.
|
|
992
|
+
const sentenceDecision = reading.sentences === undefined ? undefined : new AgentEvent.DecisionSettled({
|
|
993
|
+
eventType: eventType.decisionSettled,
|
|
994
|
+
scope: state.session,
|
|
995
|
+
frame: state.frame,
|
|
996
|
+
classifier: reading.sentences.classifier,
|
|
997
|
+
digest: reading.sentences.digest,
|
|
998
|
+
state: reading.sentences.state,
|
|
999
|
+
questions: reading.sentences.questions,
|
|
1000
|
+
answers: reading.sentences.answers,
|
|
1001
|
+
latencyMs: reading.sentences.latencyMs,
|
|
1002
|
+
acted: bounced || refused || unmoved !== undefined,
|
|
1003
|
+
decidedBy: "jev"
|
|
1004
|
+
})
|
|
1005
|
+
if (found === undefined) {
|
|
1006
|
+
return { observed: event, demand: undefined, unproven: undefined, decision, sentenceDecision }
|
|
1007
|
+
}
|
|
1008
|
+
if (unmoved !== undefined) {
|
|
1009
|
+
return {
|
|
1010
|
+
...handBack({
|
|
1011
|
+
event: new AgentEvent.UnmovedDemanded({
|
|
1012
|
+
eventType: eventType.unmovedDemanded,
|
|
1013
|
+
openedDigest: unmoved.opened,
|
|
1014
|
+
currentDigest: unmoved.closed,
|
|
1015
|
+
nextFrame
|
|
1016
|
+
}),
|
|
1017
|
+
note: UnmovedTree.demand(unmoved),
|
|
1018
|
+
keeps: !CompletionClaim.unrecorded(found),
|
|
1019
|
+
spent: { unmovedDemands: state.unmovedDemands + 1 }
|
|
1020
|
+
}, decision),
|
|
1021
|
+
observed: event,
|
|
1022
|
+
sentenceDecision
|
|
1023
|
+
}
|
|
1024
|
+
}
|
|
1025
|
+
if (bounced) {
|
|
1026
|
+
return {
|
|
1027
|
+
...handBack({
|
|
1028
|
+
event,
|
|
1029
|
+
note: CompletionClaim.demand(),
|
|
1030
|
+
// A bounce the brake would not refuse is worth restoring against an
|
|
1031
|
+
// exhausted budget; one it would refuse is not. See `CompletionDemand`.
|
|
1032
|
+
keeps: !CompletionClaim.unrecorded(found),
|
|
1033
|
+
spent: { claimDemands: state.claimDemands + 1 }
|
|
1034
|
+
}, decision),
|
|
1035
|
+
sentenceDecision
|
|
1036
|
+
}
|
|
1037
|
+
}
|
|
1038
|
+
// Out of bounces. A claim the brake only found thin stands here: it was
|
|
1039
|
+
// handed back once, the run answered, and refusing the answer as well is
|
|
1040
|
+
// the price that destroyed honest runs. Only an unrecorded claim is
|
|
1041
|
+
// refused, and only a completion reporting its own work unfinished fails.
|
|
1042
|
+
return {
|
|
1043
|
+
observed: event,
|
|
1044
|
+
demand: undefined,
|
|
1045
|
+
// Either bounce handed the claim back: the unmoved demand spends its
|
|
1046
|
+
// own cap in the claim brake's place.
|
|
1047
|
+
unproven: refused
|
|
1048
|
+
? CompletionClaim.unproven(found, state.claimDemands + state.unmovedDemands > 0, claim)
|
|
1049
|
+
: incomplete
|
|
1050
|
+
? UnfinishedWork.incomplete(reading.complete, unfinished.unfinished, claim)
|
|
1051
|
+
: undefined,
|
|
1052
|
+
decision,
|
|
1053
|
+
sentenceDecision,
|
|
1054
|
+
...(unfinishedDecision === undefined ? {} : { unfinishedDecision })
|
|
1055
|
+
}
|
|
1056
|
+
})
|
|
1057
|
+
|
|
1058
|
+
/**
|
|
1059
|
+
* The interventions one ordinary continuing frame hands to the next.
|
|
1060
|
+
*
|
|
1061
|
+
* @since 1.0.0-rc.0
|
|
1062
|
+
* @private
|
|
1063
|
+
*/
|
|
1064
|
+
export interface Discipline {
|
|
1065
|
+
/** The control events that issue them, in journal order. */
|
|
1066
|
+
readonly events: ReadonlyArray<AgentEvent.AgentEvent>
|
|
1067
|
+
/** The messages the next frame reads after the steering it was sent. */
|
|
1068
|
+
readonly messages: ReadonlyArray<ModelRequest.Message>
|
|
1069
|
+
/** What issuing them changes on top of the frame's facts. */
|
|
1070
|
+
readonly changes: StateChanges
|
|
1071
|
+
}
|
|
1072
|
+
|
|
1073
|
+
/**
|
|
1074
|
+
* The read-only, repeat and sufficiency interventions for a frame that settled
|
|
1075
|
+
* a `continue` transition and has a frame after it.
|
|
1076
|
+
*
|
|
1077
|
+
* The read-only intervention. At the cap the next frame is told, structurally,
|
|
1078
|
+
* that it must write something or say why it cannot; a justification is typed
|
|
1079
|
+
* data on the transition, is recorded, and buys a bounded quiet spell without
|
|
1080
|
+
* resetting the counter that ends the run at twice the cap.
|
|
1081
|
+
*
|
|
1082
|
+
* A justification buys that spell only when it *answers* a demand this frame
|
|
1083
|
+
* was handed. A justification volunteered by a frame nobody asked is recorded —
|
|
1084
|
+
* it is a field on the transition and the journal writes the whole transition —
|
|
1085
|
+
* and buys nothing. The two cannot be the same price, because the counter runs
|
|
1086
|
+
* regardless of which one is written: a run that volunteers one every few
|
|
1087
|
+
* frames used to renew the quiet spell before the streak could ever hand the
|
|
1088
|
+
* demand out, so the demand was never issued, never journaled, and never got
|
|
1089
|
+
* its one chance to redirect the run, while the hard stop at twice the cap —
|
|
1090
|
+
* which no grace touches — killed the run anyway. Two SWE-bench waves lost
|
|
1091
|
+
* `pydata__xarray-7393` exactly so: ten volunteered justifications, zero
|
|
1092
|
+
* `read-only-demanded` events, and death at 24 frames against a cap of 12
|
|
1093
|
+
* without the control ever speaking.
|
|
1094
|
+
*
|
|
1095
|
+
* The convergence intervention. It is journaled when it is *issued* rather than
|
|
1096
|
+
* when the next frame answers it, because what answers it is the shape of that
|
|
1097
|
+
* frame's calls — which the journal already writes one by one — and not a
|
|
1098
|
+
* field on a transition the controller has to wait for. Issuing it restarts the
|
|
1099
|
+
* count, so a run that keeps repeating is told once every `repeatCap` frames
|
|
1100
|
+
* instead of every frame.
|
|
1101
|
+
*
|
|
1102
|
+
* The counterweight, and the only notice here that asks for nothing. It is
|
|
1103
|
+
* written on the frame that completes the pair rather than at a completion,
|
|
1104
|
+
* because its whole purpose is to reach a run that is still deciding whether to
|
|
1105
|
+
* keep working — a run at its `complete` transition has already decided. See
|
|
1106
|
+
* `Sufficiency`.
|
|
1107
|
+
*
|
|
1108
|
+
* @since 1.0.0-rc.0
|
|
1109
|
+
* @private
|
|
1110
|
+
*/
|
|
1111
|
+
export const discipline = (
|
|
1112
|
+
state: State,
|
|
1113
|
+
accounting: Accounting,
|
|
1114
|
+
justification: string | undefined
|
|
1115
|
+
): Discipline => {
|
|
1116
|
+
const { facts } = accounting
|
|
1117
|
+
const cap = state.readOnlyCap
|
|
1118
|
+
const readOnly = !accounting.mutated
|
|
1119
|
+
const graceLeft = readOnly ? state.readOnlyGrace : 0
|
|
1120
|
+
const justified = readOnly && state.pendingReadOnlyDemand !== undefined &&
|
|
1121
|
+
(justification ?? "").trim().length > 0
|
|
1122
|
+
// Only a frame that advanced the streak earns the demand: a probing frame
|
|
1123
|
+
// holds the streak where it was, and repeating the demand to it every frame
|
|
1124
|
+
// would nag a run the cap has already decided is working.
|
|
1125
|
+
const advanced = readOnly && facts.readOnlyFrames > state.readOnlyFrames
|
|
1126
|
+
const demanded = cap > 0 && advanced && facts.readOnlyFrames >= cap && graceLeft === 0 && !justified
|
|
1127
|
+
const repeated = state.repeatCap > 0 && facts.repeatFrames >= state.repeatCap
|
|
1128
|
+
const sufficient = state.sufficiencyStated
|
|
1129
|
+
? undefined
|
|
1130
|
+
: Sufficiency.find({ ledger: facts.failures, frame: accounting.frameChecks, epoch: facts.mutations })
|
|
1131
|
+
const nextFrame = state.frame + 1
|
|
1132
|
+
return {
|
|
1133
|
+
events: [
|
|
1134
|
+
...(demanded
|
|
1135
|
+
? [
|
|
1136
|
+
new AgentEvent.ReadOnlyDemandIssued({
|
|
1137
|
+
eventType: eventType.readOnlyDemandIssued,
|
|
1138
|
+
streak: facts.readOnlyFrames,
|
|
1139
|
+
cap,
|
|
1140
|
+
nextFrame
|
|
1141
|
+
})
|
|
1142
|
+
]
|
|
1143
|
+
: []),
|
|
1144
|
+
...(repeated
|
|
1145
|
+
? [
|
|
1146
|
+
new AgentEvent.RepeatDemanded({
|
|
1147
|
+
eventType: eventType.repeatDemanded,
|
|
1148
|
+
frames: facts.repeatFrames,
|
|
1149
|
+
cap: state.repeatCap,
|
|
1150
|
+
nextFrame
|
|
1151
|
+
})
|
|
1152
|
+
]
|
|
1153
|
+
: []),
|
|
1154
|
+
...(sufficient === undefined ? [] : [
|
|
1155
|
+
new AgentEvent.SufficiencyObserved({
|
|
1156
|
+
eventType: eventType.sufficiencyObserved,
|
|
1157
|
+
flow: sufficient.failed.flow,
|
|
1158
|
+
failed: sufficient.failed.label,
|
|
1159
|
+
passed: sufficient.passed.label,
|
|
1160
|
+
epoch: sufficient.failed.epoch,
|
|
1161
|
+
nextFrame
|
|
1162
|
+
})
|
|
1163
|
+
])
|
|
1164
|
+
],
|
|
1165
|
+
messages: [
|
|
1166
|
+
...(accounting.probeNotice === undefined ? [] : [ModelRequest.Message.user(accounting.probeNotice)]),
|
|
1167
|
+
...(demanded ? [ModelRequest.Message.user(DemandText.readOnly(cap, facts.readOnlyFrames))] : []),
|
|
1168
|
+
...(repeated ? [ModelRequest.Message.user(DemandText.repeat(facts.repeatFrames, state.repeatCap))] : []),
|
|
1169
|
+
...(sufficient === undefined ? [] : [ModelRequest.Message.user(Sufficiency.observation(sufficient))])
|
|
1170
|
+
],
|
|
1171
|
+
changes: {
|
|
1172
|
+
readOnlyGrace: justified ? cap : Math.max(0, graceLeft - 1),
|
|
1173
|
+
repeatFrames: repeated ? 0 : facts.repeatFrames,
|
|
1174
|
+
...(sufficient === undefined ? {} : { sufficiencyStated: true }),
|
|
1175
|
+
pendingReadOnlyDemand: demanded ? { streak: facts.readOnlyFrames, cap } : undefined
|
|
1176
|
+
}
|
|
1177
|
+
}
|
|
1178
|
+
}
|