patchwork-os 0.2.0-beta.9.canary.92 → 1.1.0-beta.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.bridge.md +15 -16
- package/README.md +46 -458
- package/dist/activationMetrics.js +2 -3
- package/dist/activationMetrics.js.map +1 -1
- package/dist/activityLog.d.ts +1 -0
- package/dist/activityLog.js +8 -3
- package/dist/activityLog.js.map +1 -1
- package/dist/adapters/gemini.js +2 -2
- package/dist/adapters/gemini.js.map +1 -1
- package/dist/adapters/grok.js +6 -1
- package/dist/adapters/grok.js.map +1 -1
- package/dist/adapters/openai.js +4 -0
- package/dist/adapters/openai.js.map +1 -1
- package/dist/approvalHttp.d.ts +9 -0
- package/dist/approvalHttp.js +210 -171
- package/dist/approvalHttp.js.map +1 -1
- package/dist/approvalKpi.d.ts +102 -0
- package/dist/approvalKpi.js +282 -0
- package/dist/approvalKpi.js.map +1 -0
- package/dist/approvalQueue.d.ts +1 -1
- package/dist/approvalQueue.js.map +1 -1
- package/dist/automation.d.ts +45 -4
- package/dist/automation.js +123 -15
- package/dist/automation.js.map +1 -1
- package/dist/bridge.d.ts +10 -0
- package/dist/bridge.js +225 -46
- package/dist/bridge.js.map +1 -1
- package/dist/bridgeLockDiscovery.d.ts +16 -1
- package/dist/bridgeLockDiscovery.js +38 -4
- package/dist/bridgeLockDiscovery.js.map +1 -1
- package/dist/bridgeToolsRules.js +3 -18
- package/dist/bridgeToolsRules.js.map +1 -1
- package/dist/claudeAuthHttp.d.ts +25 -0
- package/dist/claudeAuthHttp.js +219 -0
- package/dist/claudeAuthHttp.js.map +1 -0
- package/dist/claudeDriver.js +4 -1
- package/dist/claudeDriver.js.map +1 -1
- package/dist/claudeOrchestrator.d.ts +15 -0
- package/dist/claudeOrchestrator.js +44 -0
- package/dist/claudeOrchestrator.js.map +1 -1
- package/dist/commands/connect.d.ts +47 -0
- package/dist/commands/connect.js +419 -0
- package/dist/commands/connect.js.map +1 -0
- package/dist/commands/install.js +3 -10
- package/dist/commands/install.js.map +1 -1
- package/dist/commands/launchd.d.ts +7 -0
- package/dist/commands/launchd.js +20 -2
- package/dist/commands/launchd.js.map +1 -1
- package/dist/commands/patchworkInit.d.ts +7 -0
- package/dist/commands/patchworkInit.js +26 -0
- package/dist/commands/patchworkInit.js.map +1 -1
- package/dist/commands/recipe.d.ts +59 -4
- package/dist/commands/recipe.js +206 -5
- package/dist/commands/recipe.js.map +1 -1
- package/dist/commands/recipeInstall.d.ts +9 -0
- package/dist/commands/recipeInstall.js +56 -3
- package/dist/commands/recipeInstall.js.map +1 -1
- package/dist/commands/task.d.ts +25 -0
- package/dist/commands/task.js +61 -41
- package/dist/commands/task.js.map +1 -1
- package/dist/commands/tokenEfficiency.d.ts +4 -0
- package/dist/commands/tokenEfficiency.js +23 -26
- package/dist/commands/tokenEfficiency.js.map +1 -1
- package/dist/commands/tracesExport.d.ts +15 -1
- package/dist/commands/tracesExport.js +39 -5
- package/dist/commands/tracesExport.js.map +1 -1
- package/dist/config.js +16 -0
- package/dist/config.js.map +1 -1
- package/dist/connectorRoutes.js +613 -62
- package/dist/connectorRoutes.js.map +1 -1
- package/dist/connectors/asana.js +9 -3
- package/dist/connectors/asana.js.map +1 -1
- package/dist/connectors/baseConnector.d.ts +16 -0
- package/dist/connectors/baseConnector.js +2 -0
- package/dist/connectors/baseConnector.js.map +1 -1
- package/dist/connectors/caldiy.js +34 -0
- package/dist/connectors/caldiy.js.map +1 -1
- package/dist/connectors/confluence.js +9 -2
- package/dist/connectors/confluence.js.map +1 -1
- package/dist/connectors/connectorActivity.d.ts +24 -0
- package/dist/connectors/connectorActivity.js +32 -0
- package/dist/connectors/connectorActivity.js.map +1 -0
- package/dist/connectors/connectorRedirectUri.js +22 -2
- package/dist/connectors/connectorRedirectUri.js.map +1 -1
- package/dist/connectors/connectorRegistry.d.ts +0 -3
- package/dist/connectors/connectorRegistry.js +8 -8
- package/dist/connectors/connectorRegistry.js.map +1 -1
- package/dist/connectors/datadog.js +8 -1
- package/dist/connectors/datadog.js.map +1 -1
- package/dist/connectors/discord.js +4 -3
- package/dist/connectors/discord.js.map +1 -1
- package/dist/connectors/elasticsearch.js +22 -0
- package/dist/connectors/elasticsearch.js.map +1 -1
- package/dist/connectors/github.d.ts +38 -0
- package/dist/connectors/github.js +91 -5
- package/dist/connectors/github.js.map +1 -1
- package/dist/connectors/gitlab.js +19 -16
- package/dist/connectors/gitlab.js.map +1 -1
- package/dist/connectors/gmail.d.ts +5 -0
- package/dist/connectors/gmail.js +53 -11
- package/dist/connectors/gmail.js.map +1 -1
- package/dist/connectors/googleCalendar.js +5 -4
- package/dist/connectors/googleCalendar.js.map +1 -1
- package/dist/connectors/googleDocs.js +7 -6
- package/dist/connectors/googleDocs.js.map +1 -1
- package/dist/connectors/googleDrive.js +15 -8
- package/dist/connectors/googleDrive.js.map +1 -1
- package/dist/connectors/grafana.js +39 -3
- package/dist/connectors/grafana.js.map +1 -1
- package/dist/connectors/jira.d.ts +1 -1
- package/dist/connectors/jira.js +19 -4
- package/dist/connectors/jira.js.map +1 -1
- package/dist/connectors/linear.js +2 -2
- package/dist/connectors/linear.js.map +1 -1
- package/dist/connectors/mcpClient.d.ts +29 -1
- package/dist/connectors/mcpClient.js +114 -44
- package/dist/connectors/mcpClient.js.map +1 -1
- package/dist/connectors/mcpOAuth.d.ts +1 -0
- package/dist/connectors/mcpOAuth.js +49 -4
- package/dist/connectors/mcpOAuth.js.map +1 -1
- package/dist/connectors/monday.js +5 -4
- package/dist/connectors/monday.js.map +1 -1
- package/dist/connectors/mongodb.js +32 -5
- package/dist/connectors/mongodb.js.map +1 -1
- package/dist/connectors/notion.d.ts +1 -1
- package/dist/connectors/notion.js +22 -6
- package/dist/connectors/notion.js.map +1 -1
- package/dist/connectors/oauthError.d.ts +17 -0
- package/dist/connectors/oauthError.js +50 -0
- package/dist/connectors/oauthError.js.map +1 -0
- package/dist/connectors/postgres.d.ts +7 -0
- package/dist/connectors/postgres.js +32 -2
- package/dist/connectors/postgres.js.map +1 -1
- package/dist/connectors/posthog.js +26 -0
- package/dist/connectors/posthog.js.map +1 -1
- package/dist/connectors/redis.d.ts +1 -0
- package/dist/connectors/redis.js +47 -20
- package/dist/connectors/redis.js.map +1 -1
- package/dist/connectors/salesforce.js +66 -5
- package/dist/connectors/salesforce.js.map +1 -1
- package/dist/connectors/sentry.js +2 -2
- package/dist/connectors/sentry.js.map +1 -1
- package/dist/connectors/slack.js +37 -14
- package/dist/connectors/slack.js.map +1 -1
- package/dist/connectors/snowflake.js +24 -0
- package/dist/connectors/snowflake.js.map +1 -1
- package/dist/connectors/stripe.js +9 -2
- package/dist/connectors/stripe.js.map +1 -1
- package/dist/connectors/supabase.js +34 -0
- package/dist/connectors/supabase.js.map +1 -1
- package/dist/connectors/tokenStorage.d.ts +20 -0
- package/dist/connectors/tokenStorage.js +196 -64
- package/dist/connectors/tokenStorage.js.map +1 -1
- package/dist/connectors/woocommerce.js +34 -0
- package/dist/connectors/woocommerce.js.map +1 -1
- package/dist/copilot/parseIntent.d.ts +58 -0
- package/dist/copilot/parseIntent.js +149 -0
- package/dist/copilot/parseIntent.js.map +1 -0
- package/dist/decisionReplay.d.ts +8 -6
- package/dist/decisionReplay.js +35 -11
- package/dist/decisionReplay.js.map +1 -1
- package/dist/decisionTraceLog.d.ts +9 -0
- package/dist/decisionTraceLog.js +6 -0
- package/dist/decisionTraceLog.js.map +1 -1
- package/dist/drivers/claude/api.js +49 -7
- package/dist/drivers/claude/api.js.map +1 -1
- package/dist/drivers/claude/envSanitizer.d.ts +9 -1
- package/dist/drivers/claude/envSanitizer.js +41 -3
- package/dist/drivers/claude/envSanitizer.js.map +1 -1
- package/dist/drivers/claude/streamParser.d.ts +13 -0
- package/dist/drivers/claude/streamParser.js.map +1 -1
- package/dist/drivers/claude/subprocess.js +131 -16
- package/dist/drivers/claude/subprocess.js.map +1 -1
- package/dist/drivers/claude/subprocessSettings.d.ts +1 -1
- package/dist/drivers/claude/subprocessSettings.js +18 -4
- package/dist/drivers/claude/subprocessSettings.js.map +1 -1
- package/dist/drivers/gemini/index.d.ts +13 -0
- package/dist/drivers/gemini/index.js +211 -67
- package/dist/drivers/gemini/index.js.map +1 -1
- package/dist/drivers/local/index.d.ts +2 -17
- package/dist/drivers/local/index.js +5 -92
- package/dist/drivers/local/index.js.map +1 -1
- package/dist/drivers/openai/index.js +69 -9
- package/dist/drivers/openai/index.js.map +1 -1
- package/dist/drivers/outputCap.d.ts +27 -0
- package/dist/drivers/outputCap.js +50 -0
- package/dist/drivers/outputCap.js.map +1 -0
- package/dist/embeddings/cosine.d.ts +23 -0
- package/dist/embeddings/cosine.js +59 -0
- package/dist/embeddings/cosine.js.map +1 -0
- package/dist/embeddings/index.d.ts +19 -0
- package/dist/embeddings/index.js +35 -0
- package/dist/embeddings/index.js.map +1 -0
- package/dist/embeddings/localEmbeddings.d.ts +29 -0
- package/dist/embeddings/localEmbeddings.js +103 -0
- package/dist/embeddings/localEmbeddings.js.map +1 -0
- package/dist/embeddings/localEndpoints.d.ts +24 -0
- package/dist/embeddings/localEndpoints.js +37 -0
- package/dist/embeddings/localEndpoints.js.map +1 -0
- package/dist/embeddings/types.d.ts +32 -0
- package/dist/embeddings/types.js +12 -0
- package/dist/embeddings/types.js.map +1 -0
- package/dist/featureFlags.d.ts +18 -0
- package/dist/featureFlags.js +32 -0
- package/dist/featureFlags.js.map +1 -1
- package/dist/fp/automationInterpreter.js +118 -39
- package/dist/fp/automationInterpreter.js.map +1 -1
- package/dist/fp/automationProgram.d.ts +10 -0
- package/dist/fp/automationProgram.js.map +1 -1
- package/dist/fp/automationState.d.ts +6 -0
- package/dist/fp/automationState.js +38 -0
- package/dist/fp/automationState.js.map +1 -1
- package/dist/fp/commandDescription.d.ts +6 -0
- package/dist/fp/commandDescription.js +18 -5
- package/dist/fp/commandDescription.js.map +1 -1
- package/dist/fp/interpreterContext.d.ts +16 -5
- package/dist/fp/interpreterContext.js +59 -3
- package/dist/fp/interpreterContext.js.map +1 -1
- package/dist/fp/policyParser.js +18 -8
- package/dist/fp/policyParser.js.map +1 -1
- package/dist/haltPushDispatch.d.ts +4 -0
- package/dist/haltPushDispatch.js +7 -22
- package/dist/haltPushDispatch.js.map +1 -1
- package/dist/inboxRoutes.js +2 -2
- package/dist/inboxRoutes.js.map +1 -1
- package/dist/index.js +672 -135
- package/dist/index.js.map +1 -1
- package/dist/installGuard.js +6 -0
- package/dist/installGuard.js.map +1 -1
- package/dist/localEndpointGuard.d.ts +25 -0
- package/dist/localEndpointGuard.js +101 -0
- package/dist/localEndpointGuard.js.map +1 -0
- package/dist/mcpRoutes.js +1 -1
- package/dist/mcpRoutes.js.map +1 -1
- package/dist/oauth.d.ts +13 -0
- package/dist/oauth.js +53 -9
- package/dist/oauth.js.map +1 -1
- package/dist/orchestrator/childBridgeClient.d.ts +7 -0
- package/dist/orchestrator/childBridgeClient.js +42 -1
- package/dist/orchestrator/childBridgeClient.js.map +1 -1
- package/dist/orchestrator/orchestratorBridge.d.ts +33 -0
- package/dist/orchestrator/orchestratorBridge.js +75 -9
- package/dist/orchestrator/orchestratorBridge.js.map +1 -1
- package/dist/patchworkConfig.d.ts +4 -0
- package/dist/patchworkConfig.js +33 -4
- package/dist/patchworkConfig.js.map +1 -1
- package/dist/probe.js +39 -5
- package/dist/probe.js.map +1 -1
- package/dist/recipeOrchestration.d.ts +59 -0
- package/dist/recipeOrchestration.js +314 -60
- package/dist/recipeOrchestration.js.map +1 -1
- package/dist/recipeRoutes.d.ts +93 -0
- package/dist/recipeRoutes.js +664 -27
- package/dist/recipeRoutes.js.map +1 -1
- package/dist/recipes/RecipeOrchestrator.d.ts +14 -0
- package/dist/recipes/RecipeOrchestrator.js +42 -2
- package/dist/recipes/RecipeOrchestrator.js.map +1 -1
- package/dist/recipes/agentExecutor.d.ts +51 -2
- package/dist/recipes/agentExecutor.js +86 -13
- package/dist/recipes/agentExecutor.js.map +1 -1
- package/dist/recipes/chainedRunner.d.ts +111 -3
- package/dist/recipes/chainedRunner.js +440 -45
- package/dist/recipes/chainedRunner.js.map +1 -1
- package/dist/recipes/compiler.js +42 -29
- package/dist/recipes/compiler.js.map +1 -1
- package/dist/recipes/connectorPreflight.js +31 -0
- package/dist/recipes/connectorPreflight.js.map +1 -1
- package/dist/recipes/dependencyGraph.d.ts +16 -0
- package/dist/recipes/dependencyGraph.js +37 -2
- package/dist/recipes/dependencyGraph.js.map +1 -1
- package/dist/recipes/eventTriggerPrograms.d.ts +36 -0
- package/dist/recipes/eventTriggerPrograms.js +158 -0
- package/dist/recipes/eventTriggerPrograms.js.map +1 -0
- package/dist/recipes/githubInstallSource.d.ts +9 -2
- package/dist/recipes/githubInstallSource.js +36 -6
- package/dist/recipes/githubInstallSource.js.map +1 -1
- package/dist/recipes/haltCategory.d.ts +10 -0
- package/dist/recipes/haltCategory.js +11 -1
- package/dist/recipes/haltCategory.js.map +1 -1
- package/dist/recipes/idempotencyKey.d.ts +30 -3
- package/dist/recipes/idempotencyKey.js +89 -5
- package/dist/recipes/idempotencyKey.js.map +1 -1
- package/dist/recipes/installer.js +12 -35
- package/dist/recipes/installer.js.map +1 -1
- package/dist/recipes/judgeVerdict.d.ts +11 -1
- package/dist/recipes/judgeVerdict.js +47 -7
- package/dist/recipes/judgeVerdict.js.map +1 -1
- package/dist/recipes/names.d.ts +5 -0
- package/dist/recipes/names.js +10 -5
- package/dist/recipes/names.js.map +1 -1
- package/dist/recipes/parser.js +74 -10
- package/dist/recipes/parser.js.map +1 -1
- package/dist/recipes/pricing/costRouter.d.ts +43 -0
- package/dist/recipes/pricing/costRouter.js +44 -0
- package/dist/recipes/pricing/costRouter.js.map +1 -0
- package/dist/recipes/pricing/priceTable.d.ts +76 -0
- package/dist/recipes/pricing/priceTable.js +144 -0
- package/dist/recipes/pricing/priceTable.js.map +1 -0
- package/dist/recipes/replayRun.js +5 -2
- package/dist/recipes/replayRun.js.map +1 -1
- package/dist/recipes/resolveRecipePath.js +15 -1
- package/dist/recipes/resolveRecipePath.js.map +1 -1
- package/dist/recipes/runBudget.d.ts +93 -32
- package/dist/recipes/runBudget.js +213 -49
- package/dist/recipes/runBudget.js.map +1 -1
- package/dist/recipes/runRegistry.d.ts +32 -0
- package/dist/recipes/runRegistry.js +54 -0
- package/dist/recipes/runRegistry.js.map +1 -0
- package/dist/recipes/scheduler.js +6 -2
- package/dist/recipes/scheduler.js.map +1 -1
- package/dist/recipes/schema.d.ts +59 -6
- package/dist/recipes/schemaGenerator.d.ts +9 -0
- package/dist/recipes/schemaGenerator.js +446 -77
- package/dist/recipes/schemaGenerator.js.map +1 -1
- package/dist/recipes/simulation/aggregateRunRisk.d.ts +46 -0
- package/dist/recipes/simulation/aggregateRunRisk.js +120 -0
- package/dist/recipes/simulation/aggregateRunRisk.js.map +1 -0
- package/dist/recipes/simulation/costProjector.d.ts +32 -0
- package/dist/recipes/simulation/costProjector.js +194 -0
- package/dist/recipes/simulation/costProjector.js.map +1 -0
- package/dist/recipes/simulation/sideEffects.d.ts +32 -0
- package/dist/recipes/simulation/sideEffects.js +62 -0
- package/dist/recipes/simulation/sideEffects.js.map +1 -0
- package/dist/recipes/simulation/simulate.d.ts +36 -0
- package/dist/recipes/simulation/simulate.js +264 -0
- package/dist/recipes/simulation/simulate.js.map +1 -0
- package/dist/recipes/simulation/simulateMockedRun.d.ts +52 -0
- package/dist/recipes/simulation/simulateMockedRun.js +84 -0
- package/dist/recipes/simulation/simulateMockedRun.js.map +1 -0
- package/dist/recipes/simulation/synthesizeMockedOutputs.d.ts +31 -0
- package/dist/recipes/simulation/synthesizeMockedOutputs.js +50 -0
- package/dist/recipes/simulation/synthesizeMockedOutputs.js.map +1 -0
- package/dist/recipes/simulation/types.d.ts +198 -0
- package/dist/recipes/simulation/types.js +30 -0
- package/dist/recipes/simulation/types.js.map +1 -0
- package/dist/recipes/stepObservation.d.ts +12 -0
- package/dist/recipes/stepObservation.js +21 -1
- package/dist/recipes/stepObservation.js.map +1 -1
- package/dist/recipes/toolRegistry.js +61 -24
- package/dist/recipes/toolRegistry.js.map +1 -1
- package/dist/recipes/tools/airtable.d.ts +15 -0
- package/dist/recipes/tools/airtable.js +240 -0
- package/dist/recipes/tools/airtable.js.map +1 -0
- package/dist/recipes/tools/caldiy.d.ts +13 -0
- package/dist/recipes/tools/caldiy.js +215 -0
- package/dist/recipes/tools/caldiy.js.map +1 -0
- package/dist/recipes/tools/circleci.d.ts +10 -0
- package/dist/recipes/tools/circleci.js +204 -0
- package/dist/recipes/tools/circleci.js.map +1 -0
- package/dist/recipes/tools/cloudflare.d.ts +13 -0
- package/dist/recipes/tools/cloudflare.js +211 -0
- package/dist/recipes/tools/cloudflare.js.map +1 -0
- package/dist/recipes/tools/confluence.js +11 -10
- package/dist/recipes/tools/confluence.js.map +1 -1
- package/dist/recipes/tools/datadog.js +13 -12
- package/dist/recipes/tools/datadog.js.map +1 -1
- package/dist/recipes/tools/docs.d.ts +18 -0
- package/dist/recipes/tools/docs.js +95 -0
- package/dist/recipes/tools/docs.js.map +1 -0
- package/dist/recipes/tools/elasticsearch.d.ts +11 -0
- package/dist/recipes/tools/elasticsearch.js +157 -0
- package/dist/recipes/tools/elasticsearch.js.map +1 -0
- package/dist/recipes/tools/fanOut.js +4 -4
- package/dist/recipes/tools/fanOut.js.map +1 -1
- package/dist/recipes/tools/figma.d.ts +12 -0
- package/dist/recipes/tools/figma.js +195 -0
- package/dist/recipes/tools/figma.js.map +1 -0
- package/dist/recipes/tools/file.js +43 -22
- package/dist/recipes/tools/file.js.map +1 -1
- package/dist/recipes/tools/github.js +162 -5
- package/dist/recipes/tools/github.js.map +1 -1
- package/dist/recipes/tools/gmail.js +45 -33
- package/dist/recipes/tools/gmail.js.map +1 -1
- package/dist/recipes/tools/grafana.d.ts +11 -0
- package/dist/recipes/tools/grafana.js +216 -0
- package/dist/recipes/tools/grafana.js.map +1 -0
- package/dist/recipes/tools/http.d.ts +1 -1
- package/dist/recipes/tools/http.js +10 -28
- package/dist/recipes/tools/http.js.map +1 -1
- package/dist/recipes/tools/hubspot.js +13 -12
- package/dist/recipes/tools/hubspot.js.map +1 -1
- package/dist/recipes/tools/index.d.ts +28 -0
- package/dist/recipes/tools/index.js +28 -0
- package/dist/recipes/tools/index.js.map +1 -1
- package/dist/recipes/tools/intercom.js +11 -10
- package/dist/recipes/tools/intercom.js.map +1 -1
- package/dist/recipes/tools/jira.js +13 -0
- package/dist/recipes/tools/jira.js.map +1 -1
- package/dist/recipes/tools/meetingNotes.js +2 -2
- package/dist/recipes/tools/monday.d.ts +22 -0
- package/dist/recipes/tools/monday.js +242 -0
- package/dist/recipes/tools/monday.js.map +1 -0
- package/dist/recipes/tools/mongodb.d.ts +19 -0
- package/dist/recipes/tools/mongodb.js +194 -0
- package/dist/recipes/tools/mongodb.js.map +1 -0
- package/dist/recipes/tools/notion.js +11 -10
- package/dist/recipes/tools/notion.js.map +1 -1
- package/dist/recipes/tools/obsidian.d.ts +15 -0
- package/dist/recipes/tools/obsidian.js +173 -0
- package/dist/recipes/tools/obsidian.js.map +1 -0
- package/dist/recipes/tools/outcomes.d.ts +15 -0
- package/dist/recipes/tools/outcomes.js +111 -0
- package/dist/recipes/tools/outcomes.js.map +1 -0
- package/dist/recipes/tools/paystack.d.ts +11 -0
- package/dist/recipes/tools/paystack.js +212 -0
- package/dist/recipes/tools/paystack.js.map +1 -0
- package/dist/recipes/tools/pipedrive.d.ts +16 -0
- package/dist/recipes/tools/pipedrive.js +234 -0
- package/dist/recipes/tools/pipedrive.js.map +1 -0
- package/dist/recipes/tools/postgres.d.ts +15 -0
- package/dist/recipes/tools/postgres.js +186 -0
- package/dist/recipes/tools/postgres.js.map +1 -0
- package/dist/recipes/tools/posthog.d.ts +16 -0
- package/dist/recipes/tools/posthog.js +219 -0
- package/dist/recipes/tools/posthog.js.map +1 -0
- package/dist/recipes/tools/redis.d.ts +9 -0
- package/dist/recipes/tools/redis.js +141 -0
- package/dist/recipes/tools/redis.js.map +1 -0
- package/dist/recipes/tools/resend.d.ts +8 -0
- package/dist/recipes/tools/resend.js +153 -0
- package/dist/recipes/tools/resend.js.map +1 -0
- package/dist/recipes/tools/salesforce.d.ts +16 -0
- package/dist/recipes/tools/salesforce.js +184 -0
- package/dist/recipes/tools/salesforce.js.map +1 -0
- package/dist/recipes/tools/sendgrid.d.ts +9 -0
- package/dist/recipes/tools/sendgrid.js +175 -0
- package/dist/recipes/tools/sendgrid.js.map +1 -0
- package/dist/recipes/tools/shopify.d.ts +16 -0
- package/dist/recipes/tools/shopify.js +266 -0
- package/dist/recipes/tools/shopify.js.map +1 -0
- package/dist/recipes/tools/snowflake.d.ts +16 -0
- package/dist/recipes/tools/snowflake.js +173 -0
- package/dist/recipes/tools/snowflake.js.map +1 -0
- package/dist/recipes/tools/stripe.js +13 -12
- package/dist/recipes/tools/stripe.js.map +1 -1
- package/dist/recipes/tools/supabase.d.ts +9 -0
- package/dist/recipes/tools/supabase.js +133 -0
- package/dist/recipes/tools/supabase.js.map +1 -0
- package/dist/recipes/tools/todoist.d.ts +15 -0
- package/dist/recipes/tools/todoist.js +228 -0
- package/dist/recipes/tools/todoist.js.map +1 -0
- package/dist/recipes/tools/twilio.d.ts +11 -0
- package/dist/recipes/tools/twilio.js +180 -0
- package/dist/recipes/tools/twilio.js.map +1 -0
- package/dist/recipes/tools/vercel.d.ts +9 -0
- package/dist/recipes/tools/vercel.js +146 -0
- package/dist/recipes/tools/vercel.js.map +1 -0
- package/dist/recipes/tools/webflow.d.ts +11 -0
- package/dist/recipes/tools/webflow.js +244 -0
- package/dist/recipes/tools/webflow.js.map +1 -0
- package/dist/recipes/tools/woocommerce.d.ts +18 -0
- package/dist/recipes/tools/woocommerce.js +260 -0
- package/dist/recipes/tools/woocommerce.js.map +1 -0
- package/dist/recipes/tools/wrapConnectorExecute.d.ts +25 -0
- package/dist/recipes/tools/wrapConnectorExecute.js +37 -0
- package/dist/recipes/tools/wrapConnectorExecute.js.map +1 -0
- package/dist/recipes/tools/zendesk.js +11 -10
- package/dist/recipes/tools/zendesk.js.map +1 -1
- package/dist/recipes/triggerVars.d.ts +15 -0
- package/dist/recipes/triggerVars.js +57 -0
- package/dist/recipes/triggerVars.js.map +1 -0
- package/dist/recipes/validation.d.ts +1 -1
- package/dist/recipes/validation.js +536 -12
- package/dist/recipes/validation.js.map +1 -1
- package/dist/recipes/yamlPositions.d.ts +1 -1
- package/dist/recipes/yamlPositions.js +7 -2
- package/dist/recipes/yamlPositions.js.map +1 -1
- package/dist/recipes/yamlRunner.d.ts +216 -4
- package/dist/recipes/yamlRunner.js +1030 -96
- package/dist/recipes/yamlRunner.js.map +1 -1
- package/dist/recipesHttp.d.ts +5 -0
- package/dist/recipesHttp.js +68 -6
- package/dist/recipesHttp.js.map +1 -1
- package/dist/resources.js +14 -1
- package/dist/resources.js.map +1 -1
- package/dist/riskSignals.d.ts +42 -0
- package/dist/riskSignals.js +197 -0
- package/dist/riskSignals.js.map +1 -0
- package/dist/riskTier.d.ts +9 -0
- package/dist/riskTier.js +48 -0
- package/dist/riskTier.js.map +1 -1
- package/dist/runLog.d.ts +53 -0
- package/dist/runLog.js +75 -16
- package/dist/runLog.js.map +1 -1
- package/dist/schemas/recipe.v1.json +360 -16
- package/dist/server.d.ts +60 -0
- package/dist/server.js +192 -31
- package/dist/server.js.map +1 -1
- package/dist/sessionCheckpoint.js +4 -1
- package/dist/sessionCheckpoint.js.map +1 -1
- package/dist/sessionDetail.d.ts +43 -0
- package/dist/sessionDetail.js +59 -0
- package/dist/sessionDetail.js.map +1 -0
- package/dist/settingsEnv.d.ts +47 -0
- package/dist/settingsEnv.js +88 -0
- package/dist/settingsEnv.js.map +1 -0
- package/dist/streamableHttp.js +44 -10
- package/dist/streamableHttp.js.map +1 -1
- package/dist/ta/backtest/abnday.d.ts +53 -0
- package/dist/ta/backtest/abnday.js +123 -0
- package/dist/ta/backtest/abnday.js.map +1 -0
- package/dist/ta/backtest/directional.d.ts +48 -0
- package/dist/ta/backtest/directional.js +103 -0
- package/dist/ta/backtest/directional.js.map +1 -0
- package/dist/ta/backtest/fundx.d.ts +67 -0
- package/dist/ta/backtest/fundx.js +264 -0
- package/dist/ta/backtest/fundx.js.map +1 -0
- package/dist/ta/backtest/fvg.d.ts +62 -0
- package/dist/ta/backtest/fvg.js +236 -0
- package/dist/ta/backtest/fvg.js.map +1 -0
- package/dist/ta/backtest/ichimoku.d.ts +41 -0
- package/dist/ta/backtest/ichimoku.js +146 -0
- package/dist/ta/backtest/ichimoku.js.map +1 -0
- package/dist/ta/backtest/ingest.d.ts +20 -0
- package/dist/ta/backtest/ingest.js +104 -0
- package/dist/ta/backtest/ingest.js.map +1 -0
- package/dist/ta/backtest/msb.d.ts +49 -0
- package/dist/ta/backtest/msb.js +162 -0
- package/dist/ta/backtest/msb.js.map +1 -0
- package/dist/ta/backtest/scoring.d.ts +33 -0
- package/dist/ta/backtest/scoring.js +82 -0
- package/dist/ta/backtest/scoring.js.map +1 -0
- package/dist/ta/backtest/summary.d.ts +20 -0
- package/dist/ta/backtest/summary.js +42 -0
- package/dist/ta/backtest/summary.js.map +1 -0
- package/dist/ta/backtest/tsmom.d.ts +72 -0
- package/dist/ta/backtest/tsmom.js +202 -0
- package/dist/ta/backtest/tsmom.js.map +1 -0
- package/dist/ta/backtest/volsq.d.ts +34 -0
- package/dist/ta/backtest/volsq.js +154 -0
- package/dist/ta/backtest/volsq.js.map +1 -0
- package/dist/ta/backtest/walkForward.d.ts +45 -0
- package/dist/ta/backtest/walkForward.js +136 -0
- package/dist/ta/backtest/walkForward.js.map +1 -0
- package/dist/ta/cycles.d.ts +15 -0
- package/dist/ta/cycles.js +31 -0
- package/dist/ta/cycles.js.map +1 -0
- package/dist/ta/desk/accrualEmitter.d.ts +59 -0
- package/dist/ta/desk/accrualEmitter.js +136 -0
- package/dist/ta/desk/accrualEmitter.js.map +1 -0
- package/dist/ta/desk/cellBacktest.d.ts +179 -0
- package/dist/ta/desk/cellBacktest.js +514 -0
- package/dist/ta/desk/cellBacktest.js.map +1 -0
- package/dist/ta/desk/cells/wpMaRejection.d.ts +49 -0
- package/dist/ta/desk/cells/wpMaRejection.js +141 -0
- package/dist/ta/desk/cells/wpMaRejection.js.map +1 -0
- package/dist/ta/desk/cells/wpVolumeClimax.d.ts +57 -0
- package/dist/ta/desk/cells/wpVolumeClimax.js +193 -0
- package/dist/ta/desk/cells/wpVolumeClimax.js.map +1 -0
- package/dist/ta/desk/collectors.d.ts +18 -0
- package/dist/ta/desk/collectors.js +450 -0
- package/dist/ta/desk/collectors.js.map +1 -0
- package/dist/ta/desk/contract.d.ts +51 -0
- package/dist/ta/desk/contract.js +261 -0
- package/dist/ta/desk/contract.js.map +1 -0
- package/dist/ta/desk/deskLedger.d.ts +36 -0
- package/dist/ta/desk/deskLedger.js +239 -0
- package/dist/ta/desk/deskLedger.js.map +1 -0
- package/dist/ta/desk/liqTape.d.ts +37 -0
- package/dist/ta/desk/liqTape.js +99 -0
- package/dist/ta/desk/liqTape.js.map +1 -0
- package/dist/ta/desk/post.d.ts +20 -0
- package/dist/ta/desk/post.js +65 -0
- package/dist/ta/desk/post.js.map +1 -0
- package/dist/ta/desk/render.d.ts +25 -0
- package/dist/ta/desk/render.js +398 -0
- package/dist/ta/desk/render.js.map +1 -0
- package/dist/ta/desk/surfaces.d.ts +216 -0
- package/dist/ta/desk/surfaces.js +472 -0
- package/dist/ta/desk/surfaces.js.map +1 -0
- package/dist/ta/desk/types.d.ts +174 -0
- package/dist/ta/desk/types.js +104 -0
- package/dist/ta/desk/types.js.map +1 -0
- package/dist/ta/index.d.ts +23 -0
- package/dist/ta/index.js +54 -0
- package/dist/ta/index.js.map +1 -0
- package/dist/ta/ledger.d.ts +44 -0
- package/dist/ta/ledger.js +95 -0
- package/dist/ta/ledger.js.map +1 -0
- package/dist/ta/levels.d.ts +21 -0
- package/dist/ta/levels.js +41 -0
- package/dist/ta/levels.js.map +1 -0
- package/dist/ta/swings.d.ts +30 -0
- package/dist/ta/swings.js +83 -0
- package/dist/ta/swings.js.map +1 -0
- package/dist/ta/types.d.ts +81 -0
- package/dist/ta/types.js +20 -0
- package/dist/ta/types.js.map +1 -0
- package/dist/telemetry.js +20 -9
- package/dist/telemetry.js.map +1 -1
- package/dist/tokenUsageTracker.d.ts +2 -1
- package/dist/tokenUsageTracker.js +38 -20
- package/dist/tokenUsageTracker.js.map +1 -1
- package/dist/tools/contextBundle.js +13 -9
- package/dist/tools/contextBundle.js.map +1 -1
- package/dist/tools/ctxGetTaskContext.js +8 -3
- package/dist/tools/ctxGetTaskContext.js.map +1 -1
- package/dist/tools/ctxQueryTraces.d.ts +18 -1
- package/dist/tools/ctxQueryTraces.js +176 -27
- package/dist/tools/ctxQueryTraces.js.map +1 -1
- package/dist/tools/enrichCommit.js +5 -2
- package/dist/tools/enrichCommit.js.map +1 -1
- package/dist/tools/getDependencyTree.js +2 -2
- package/dist/tools/getDependencyTree.js.map +1 -1
- package/dist/tools/getDiagnostics.js +3 -0
- package/dist/tools/getDiagnostics.js.map +1 -1
- package/dist/tools/getDocumentSymbols.js +5 -1
- package/dist/tools/getDocumentSymbols.js.map +1 -1
- package/dist/tools/getGitHotspots.js +12 -1
- package/dist/tools/getGitHotspots.js.map +1 -1
- package/dist/tools/getGitLog.js +5 -2
- package/dist/tools/getGitLog.js.map +1 -1
- package/dist/tools/getGitStatus.js +6 -1
- package/dist/tools/getGitStatus.js.map +1 -1
- package/dist/tools/getProjectContext.js +3 -1
- package/dist/tools/getProjectContext.js.map +1 -1
- package/dist/tools/getToolCapabilities.js +39 -4
- package/dist/tools/getToolCapabilities.js.map +1 -1
- package/dist/tools/gitWrite.js +12 -3
- package/dist/tools/gitWrite.js.map +1 -1
- package/dist/tools/github/composite.d.ts +1 -1
- package/dist/tools/github/composite.js +7 -7
- package/dist/tools/github/composite.js.map +1 -1
- package/dist/tools/github/pr.d.ts +6 -6
- package/dist/tools/github/pr.js +18 -11
- package/dist/tools/github/pr.js.map +1 -1
- package/dist/tools/httpClient.js +55 -3
- package/dist/tools/httpClient.js.map +1 -1
- package/dist/tools/index.d.ts +2 -1
- package/dist/tools/index.js +19 -12
- package/dist/tools/index.js.map +1 -1
- package/dist/tools/lsp.d.ts +6 -0
- package/dist/tools/lsp.js +81 -12
- package/dist/tools/lsp.js.map +1 -1
- package/dist/tools/previewEdit.js +16 -1
- package/dist/tools/previewEdit.js.map +1 -1
- package/dist/tools/resumeClaudeTask.js +17 -0
- package/dist/tools/resumeClaudeTask.js.map +1 -1
- package/dist/tools/searchAndReplace.js +9 -5
- package/dist/tools/searchAndReplace.js.map +1 -1
- package/dist/tools/searchTools.d.ts +26 -0
- package/dist/tools/searchTools.js +42 -2
- package/dist/tools/searchTools.js.map +1 -1
- package/dist/tools/terminal.js +23 -51
- package/dist/tools/terminal.js.map +1 -1
- package/dist/tools/transaction.d.ts +9 -1
- package/dist/tools/transaction.js +59 -4
- package/dist/tools/transaction.js.map +1 -1
- package/dist/tools/utils.d.ts +16 -0
- package/dist/tools/utils.js +77 -7
- package/dist/tools/utils.js.map +1 -1
- package/dist/transport.d.ts +31 -0
- package/dist/transport.js +117 -14
- package/dist/transport.js.map +1 -1
- package/dist/wireHaltPushDispatch.js +12 -0
- package/dist/wireHaltPushDispatch.js.map +1 -1
- package/dist/workerGateDecisionLog.d.ts +141 -0
- package/dist/workerGateDecisionLog.js +361 -0
- package/dist/workerGateDecisionLog.js.map +1 -0
- package/dist/workers/actionClass.d.ts +38 -0
- package/dist/workers/actionClass.js +155 -0
- package/dist/workers/actionClass.js.map +1 -0
- package/dist/workers/backtest.d.ts +68 -0
- package/dist/workers/backtest.js +113 -0
- package/dist/workers/backtest.js.map +1 -0
- package/dist/workers/contextRisk.d.ts +35 -0
- package/dist/workers/contextRisk.js +20 -0
- package/dist/workers/contextRisk.js.map +1 -0
- package/dist/workers/contextRiskScorer.d.ts +48 -0
- package/dist/workers/contextRiskScorer.js +94 -0
- package/dist/workers/contextRiskScorer.js.map +1 -0
- package/dist/workers/graduation.d.ts +61 -0
- package/dist/workers/graduation.js +112 -0
- package/dist/workers/graduation.js.map +1 -0
- package/dist/workers/outcomeStore.d.ts +98 -0
- package/dist/workers/outcomeStore.js +146 -0
- package/dist/workers/outcomeStore.js.map +1 -0
- package/dist/workers/outcomesCli.d.ts +34 -0
- package/dist/workers/outcomesCli.js +91 -0
- package/dist/workers/outcomesCli.js.map +1 -0
- package/dist/workers/runWorkerShadow.d.ts +90 -0
- package/dist/workers/runWorkerShadow.js +323 -0
- package/dist/workers/runWorkerShadow.js.map +1 -0
- package/dist/workers/shadowGate.d.ts +30 -0
- package/dist/workers/shadowGate.js +58 -0
- package/dist/workers/shadowGate.js.map +1 -0
- package/dist/workers/shadowObserver.d.ts +165 -0
- package/dist/workers/shadowObserver.js +203 -0
- package/dist/workers/shadowObserver.js.map +1 -0
- package/dist/workers/shadowReport.d.ts +16 -0
- package/dist/workers/shadowReport.js +61 -0
- package/dist/workers/shadowReport.js.map +1 -0
- package/dist/workers/shadowRun.d.ts +38 -0
- package/dist/workers/shadowRun.js +44 -0
- package/dist/workers/shadowRun.js.map +1 -0
- package/dist/workers/trustLevel.d.ts +82 -0
- package/dist/workers/trustLevel.js +100 -0
- package/dist/workers/trustLevel.js.map +1 -0
- package/dist/workers/worker.d.ts +49 -0
- package/dist/workers/worker.js +88 -0
- package/dist/workers/worker.js.map +1 -0
- package/dist/workers/workerGate.d.ts +101 -0
- package/dist/workers/workerGate.js +210 -0
- package/dist/workers/workerGate.js.map +1 -0
- package/dist/workers/workerLevelStore.d.ts +33 -0
- package/dist/workers/workerLevelStore.js +87 -0
- package/dist/workers/workerLevelStore.js.map +1 -0
- package/dist/workers/workerLoader.d.ts +7 -0
- package/dist/workers/workerLoader.js +31 -0
- package/dist/workers/workerLoader.js.map +1 -0
- package/dist/writeFileAtomic.js +40 -13
- package/dist/writeFileAtomic.js.map +1 -1
- package/package.json +14 -6
- package/scripts/mcp-stdio-shim.cjs +247 -87
- package/templates/CLAUDE.bridge.md +1 -1
- package/templates/recipes/dependency-bump.yaml +33 -0
- package/templates/recipes/incident-to-pr.yaml +187 -0
- package/templates/recipes/outcome-ingester.yaml +53 -0
- package/templates/recipes/release-notes.yaml +30 -0
- package/templates/recipes/triage-failing-tests-autofile.yaml +152 -0
- package/templates/recipes/triage-failing-tests.yaml +38 -0
- package/templates/workers/dependency-upkeep.worker.yaml +26 -0
- package/templates/workers/release-notes.worker.yaml +17 -0
- package/templates/workers/test-guardian.worker.yaml +24 -0
- package/deploy/README.md +0 -172
- package/deploy/bootstrap-new-vps.sh +0 -364
- package/deploy/bootstrap-vps.sh +0 -192
- package/deploy/claude-ide-bridge.service.template +0 -67
- package/deploy/claude-ide-bridge@.service +0 -31
- package/deploy/deploy-dashboard.sh +0 -198
- package/deploy/deploy-landing.sh +0 -136
- package/deploy/ecosystem.config.js.example +0 -36
- package/deploy/install-vps-service.sh +0 -240
- package/deploy/macos/README.md +0 -153
- package/deploy/macos/com.patchwork.bridge.plist.template +0 -54
- package/deploy/macos/com.patchwork.tunnel.plist.template +0 -76
- package/deploy/macos/install-mac-bridge.sh +0 -244
- package/deploy/macos/uninstall-mac-bridge.sh +0 -22
- package/deploy/nginx-claude-bridge.conf.template +0 -129
|
@@ -31,8 +31,10 @@ import { fileURLToPath } from "node:url";
|
|
|
31
31
|
import { parse as parseYaml } from "yaml";
|
|
32
32
|
import { captureFixture } from "../connectors/fixtureRecorder.js";
|
|
33
33
|
import { sanitizeEnv } from "../drivers/claude/envSanitizer.js";
|
|
34
|
+
import { isLoopbackOrPrivateEndpoint } from "../localEndpointGuard.js";
|
|
34
35
|
import { loadConfig as loadPatchworkConfigSync } from "../patchworkConfig.js";
|
|
35
36
|
import { findYamlRecipePath } from "../recipesHttp.js";
|
|
37
|
+
import { classifyTool } from "../riskTier.js";
|
|
36
38
|
/**
|
|
37
39
|
* Local alias for `sanitizeParsedJson` from `src/sanitizeParsedJson.ts`.
|
|
38
40
|
* Kept under the old name so the existing callsites in this file don't
|
|
@@ -42,14 +44,18 @@ import { findYamlRecipePath } from "../recipesHttp.js";
|
|
|
42
44
|
*/
|
|
43
45
|
import { sanitizeParsedJson as sanitizeParsed } from "../sanitizeParsedJson.js";
|
|
44
46
|
import { ensureCmdShim } from "../winShim.js";
|
|
47
|
+
import { mergeAgentDisallowedTools } from "../workers/workerGate.js";
|
|
45
48
|
import { executeAgent as _executeAgent, } from "./agentExecutor.js";
|
|
46
49
|
import { categoriseHaltReason } from "./haltCategory.js";
|
|
47
50
|
import { assertValidManualRunId, deriveScopeKey, WriteEffectLedger, } from "./idempotencyKey.js";
|
|
48
51
|
import { buildJudgeArtefactBlock, JUDGE_PROMPT_SUFFIX, parseJudgeVerdict, } from "./judgeVerdict.js";
|
|
49
52
|
import { defaultDeprecationWarn, normalizeRecipeForRuntime, } from "./migrations/index.js";
|
|
53
|
+
import { costRouter } from "./pricing/costRouter.js";
|
|
54
|
+
import { loadPriceTable, costUsd as priceCostUsd, } from "./pricing/priceTable.js";
|
|
50
55
|
import { resolveRecipePath } from "./resolveRecipePath.js";
|
|
51
56
|
import { RunBudget } from "./runBudget.js";
|
|
52
|
-
import {
|
|
57
|
+
import { registerRun, unregisterRun } from "./runRegistry.js";
|
|
58
|
+
import { detectSilentFail, redactSecretsForPrompt } from "./stepObservation.js";
|
|
53
59
|
// Import tool registry and trigger tool self-registration
|
|
54
60
|
import { applyToolOutputContext, executeTool, getTool, hasTool, registerPluginTools, } from "./toolRegistry.js";
|
|
55
61
|
import { resolveWorkspaceRoot } from "./workspaceRoot.js";
|
|
@@ -134,6 +140,13 @@ export function evaluateExpect(result, expect) {
|
|
|
134
140
|
* without schema assertions don't pay the import/compile cost.
|
|
135
141
|
*/
|
|
136
142
|
let _stepExpectAjv;
|
|
143
|
+
// Process-scoped probe cache for `claude --version`. Avoids spawning the .cmd
|
|
144
|
+
// shim (300–700 ms on Windows) on every recipe step when no claudeFn is
|
|
145
|
+
// configured. Exported for tests that need to reset between cases.
|
|
146
|
+
let _claudeCliProbeCache;
|
|
147
|
+
export function resetProbeCliCache() {
|
|
148
|
+
_claudeCliProbeCache = undefined;
|
|
149
|
+
}
|
|
137
150
|
async function getStepExpectAjv() {
|
|
138
151
|
if (!_stepExpectAjv) {
|
|
139
152
|
const { createAjv2020 } = await import("../ajv2020.js");
|
|
@@ -314,6 +327,14 @@ export async function loadRecipeServers(specs) {
|
|
|
314
327
|
debug: (_msg) => { },
|
|
315
328
|
};
|
|
316
329
|
for (const spec of toLoad) {
|
|
330
|
+
// Mark the spec as loaded OPTIMISTICALLY before the async load so two
|
|
331
|
+
// concurrent recipe runs sharing a `servers:` spec don't both pass the
|
|
332
|
+
// `filter` dedup above and double-register the same plugin tools (the
|
|
333
|
+
// registry does not guard re-registration). On failure we remove it so a
|
|
334
|
+
// later run can retry.
|
|
335
|
+
if (loadedPluginSpecs.has(spec))
|
|
336
|
+
continue;
|
|
337
|
+
loadedPluginSpecs.add(spec);
|
|
317
338
|
try {
|
|
318
339
|
const loaded = await loadPluginsFull([spec], minimalConfig, minimalLogger);
|
|
319
340
|
let toolCount = 0;
|
|
@@ -325,39 +346,172 @@ export async function loadRecipeServers(specs) {
|
|
|
325
346
|
}));
|
|
326
347
|
toolCount += registerPluginTools(pluginTools);
|
|
327
348
|
}
|
|
328
|
-
loadedPluginSpecs.add(spec);
|
|
329
349
|
if (toolCount > 0) {
|
|
330
350
|
console.info(`[recipe servers] loaded "${spec}" — ${toolCount} tool(s) registered`);
|
|
331
351
|
}
|
|
332
352
|
}
|
|
333
353
|
catch (err) {
|
|
354
|
+
loadedPluginSpecs.delete(spec);
|
|
334
355
|
console.warn(`[recipe servers] failed to load "${spec}": ${err instanceof Error ? err.message : String(err)}`);
|
|
335
356
|
}
|
|
336
357
|
}
|
|
337
358
|
}
|
|
359
|
+
/**
|
|
360
|
+
* P1 cost/token corpus — drivers that incur real, metered, per-token API
|
|
361
|
+
* billing (the only ones whose spend is real money and thus priceable here).
|
|
362
|
+
* Mirrors `BILLABLE_DRIVERS` in runBudget.ts (kept local — that set is private
|
|
363
|
+
* and runBudget.ts is enforcement-critical / must not be modified for P1).
|
|
364
|
+
* `local` reports usage but costs no real money, so it is NOT billable.
|
|
365
|
+
*/
|
|
366
|
+
const COST_BILLABLE_DRIVERS = new Set([
|
|
367
|
+
"anthropic",
|
|
368
|
+
"openai",
|
|
369
|
+
"grok",
|
|
370
|
+
"gemini",
|
|
371
|
+
"gemini-api",
|
|
372
|
+
]);
|
|
373
|
+
function newStepUsageAccumulator() {
|
|
374
|
+
return { inputTokens: 0, outputTokens: 0, measured: false };
|
|
375
|
+
}
|
|
376
|
+
/**
|
|
377
|
+
* Fold one agent call's usage into a per-step accumulator. Adds tokens when
|
|
378
|
+
* `usage` is present; adds USD only when the served model is billable AND
|
|
379
|
+
* present in the price table (NEVER a `0` placeholder for the unpriced case).
|
|
380
|
+
*/
|
|
381
|
+
function accumulateAgentUsage(acc, usage, servedBy, priceTable) {
|
|
382
|
+
if (!usage)
|
|
383
|
+
return;
|
|
384
|
+
acc.measured = true;
|
|
385
|
+
acc.inputTokens += usage.inputTokens;
|
|
386
|
+
acc.outputTokens += usage.outputTokens;
|
|
387
|
+
const driver = servedBy?.driver;
|
|
388
|
+
const model = servedBy?.model;
|
|
389
|
+
if (driver && model && COST_BILLABLE_DRIVERS.has(driver)) {
|
|
390
|
+
const cost = priceCostUsd(model, usage, priceTable);
|
|
391
|
+
if (typeof cost === "number") {
|
|
392
|
+
acc.costUsd = (acc.costUsd ?? 0) + cost;
|
|
393
|
+
}
|
|
394
|
+
}
|
|
395
|
+
}
|
|
396
|
+
/**
|
|
397
|
+
* Build the optional token fields for a step result from its accumulator.
|
|
398
|
+
* Returns an empty object (no fields) when the step reported no usage, so a
|
|
399
|
+
* tool step or unmeasured-driver step round-trips with the fields ABSENT.
|
|
400
|
+
*/
|
|
401
|
+
function stepUsageFields(acc) {
|
|
402
|
+
if (!acc.measured)
|
|
403
|
+
return {};
|
|
404
|
+
return {
|
|
405
|
+
inputTokens: acc.inputTokens,
|
|
406
|
+
outputTokens: acc.outputTokens,
|
|
407
|
+
...(typeof acc.costUsd === "number" ? { costUsd: acc.costUsd } : {}),
|
|
408
|
+
};
|
|
409
|
+
}
|
|
410
|
+
/**
|
|
411
|
+
* P1 — single-call usage → persisted-step usage fields. Exported for the
|
|
412
|
+
* chained runner, whose agent steps make exactly one agent call (no
|
|
413
|
+
* judge→refine loop), so a per-call computation suffices. Returns undefined
|
|
414
|
+
* when the driver reported no usage (fields stay ABSENT). `costUsd` set only
|
|
415
|
+
* for a billable driver + priced model; never a `0` placeholder.
|
|
416
|
+
*/
|
|
417
|
+
export function computeAgentCallUsage(usage, servedBy, priceTable = loadPriceTable()) {
|
|
418
|
+
if (!usage)
|
|
419
|
+
return undefined;
|
|
420
|
+
const driver = servedBy?.driver;
|
|
421
|
+
const model = servedBy?.model;
|
|
422
|
+
let cost;
|
|
423
|
+
if (driver && model && COST_BILLABLE_DRIVERS.has(driver)) {
|
|
424
|
+
const c = priceCostUsd(model, usage, priceTable);
|
|
425
|
+
if (typeof c === "number")
|
|
426
|
+
cost = c;
|
|
427
|
+
}
|
|
428
|
+
return {
|
|
429
|
+
inputTokens: usage.inputTokens,
|
|
430
|
+
outputTokens: usage.outputTokens,
|
|
431
|
+
...(typeof cost === "number" ? { costUsd: cost } : {}),
|
|
432
|
+
};
|
|
433
|
+
}
|
|
434
|
+
function newRunUsageAccumulator() {
|
|
435
|
+
return { inputTokens: 0, outputTokens: 0, measured: false };
|
|
436
|
+
}
|
|
437
|
+
function foldStepIntoRun(run, step) {
|
|
438
|
+
if (!step.measured)
|
|
439
|
+
return;
|
|
440
|
+
run.measured = true;
|
|
441
|
+
run.inputTokens += step.inputTokens;
|
|
442
|
+
run.outputTokens += step.outputTokens;
|
|
443
|
+
if (typeof step.costUsd === "number") {
|
|
444
|
+
run.costUsd = (run.costUsd ?? 0) + step.costUsd;
|
|
445
|
+
}
|
|
446
|
+
}
|
|
447
|
+
/** Build the optional `tokenTotals` for a run, or undefined when none measured. */
|
|
448
|
+
function runTokenTotals(run) {
|
|
449
|
+
if (!run.measured)
|
|
450
|
+
return undefined;
|
|
451
|
+
return {
|
|
452
|
+
inputTokens: run.inputTokens,
|
|
453
|
+
outputTokens: run.outputTokens,
|
|
454
|
+
...(typeof run.costUsd === "number" ? { costUsd: run.costUsd } : {}),
|
|
455
|
+
};
|
|
456
|
+
}
|
|
457
|
+
/**
|
|
458
|
+
* Extract ONLY the env vars a recipe explicitly declares via a
|
|
459
|
+
* `context: [{ type: "env", keys: [...] }]` block. Both the flat runner AND the
|
|
460
|
+
* chained/replay paths MUST use this so undeclared process-level secrets never
|
|
461
|
+
* reach `{{env.X}}` template expressions.
|
|
462
|
+
*
|
|
463
|
+
* Audit 2026-06-08 (recipe-support-3): the chained dispatch and replay paths
|
|
464
|
+
* previously spread the entire `process.env` into the template context, silently
|
|
465
|
+
* diverging from the flat runner's allowlist and exposing every process secret
|
|
466
|
+
* (API keys, OAuth/connector tokens, TLS material) to any chained recipe author.
|
|
467
|
+
*/
|
|
468
|
+
export function declaredRecipeEnv(recipe, processEnv = process.env) {
|
|
469
|
+
const out = {};
|
|
470
|
+
const blocks = recipe?.context;
|
|
471
|
+
if (!Array.isArray(blocks))
|
|
472
|
+
return out;
|
|
473
|
+
for (const block of blocks) {
|
|
474
|
+
const b = block;
|
|
475
|
+
if (b.type === "env" && Array.isArray(b.keys)) {
|
|
476
|
+
for (const key of b.keys) {
|
|
477
|
+
if (typeof key !== "string")
|
|
478
|
+
continue;
|
|
479
|
+
const v = processEnv[key];
|
|
480
|
+
if (v !== undefined)
|
|
481
|
+
out[key] = v;
|
|
482
|
+
}
|
|
483
|
+
}
|
|
484
|
+
}
|
|
485
|
+
return out;
|
|
486
|
+
}
|
|
338
487
|
export async function runYamlRecipe(recipe, deps = {}, seedContext = {}) {
|
|
339
488
|
if (recipe.servers?.length) {
|
|
340
489
|
await loadRecipeServers(recipe.servers);
|
|
341
490
|
}
|
|
342
491
|
const now = deps.now ? deps.now() : new Date();
|
|
343
|
-
// Resolve recipe-level context blocks (type: env) into seed context
|
|
344
|
-
|
|
345
|
-
|
|
346
|
-
|
|
347
|
-
|
|
348
|
-
|
|
349
|
-
|
|
350
|
-
|
|
351
|
-
|
|
352
|
-
|
|
353
|
-
envCtx[key] = v;
|
|
354
|
-
}
|
|
355
|
-
}
|
|
356
|
-
}
|
|
357
|
-
}
|
|
492
|
+
// Resolve recipe-level context blocks (type: env) into seed context via the
|
|
493
|
+
// shared declared-keys allowlist (also used by the chained/replay paths).
|
|
494
|
+
const envCtx = declaredRecipeEnv(recipe);
|
|
495
|
+
// SECRETS-IN-VARS: track which ctx keys came from a `type: env` block so the
|
|
496
|
+
// agent (LLM-facing) prompt can redact them. Their raw values still flow to
|
|
497
|
+
// TOOL steps (an http header / DB password legitimately needs the secret),
|
|
498
|
+
// but they must never reach the model verbatim — the secure default is
|
|
499
|
+
// redaction. See PR body / docs/recipe-feature-investigation-2026-06-05.md.
|
|
500
|
+
const secretKeys = new Set(Object.keys(envCtx));
|
|
501
|
+
const iso = now.toISOString();
|
|
358
502
|
const ctx = {
|
|
359
|
-
date:
|
|
503
|
+
date: iso.slice(0, 10),
|
|
360
504
|
time: now.toTimeString().slice(0, 5),
|
|
505
|
+
// Built-in date/time tokens, injected (not phantom) so {{YYYY-MM-DD}} etc.
|
|
506
|
+
// render real values at run time AND pass template-ref lint. Keep in sync
|
|
507
|
+
// with builtinKeys in validation.ts. (audit 2026-06-10 recipe-validation-1)
|
|
508
|
+
YYYY: iso.slice(0, 4),
|
|
509
|
+
"YYYY-MM": iso.slice(0, 7),
|
|
510
|
+
"YYYY-MM-DD": iso.slice(0, 10),
|
|
511
|
+
ISO_NOW: iso,
|
|
512
|
+
HH: iso.slice(11, 13),
|
|
513
|
+
MM: iso.slice(14, 16),
|
|
514
|
+
SS: iso.slice(17, 19),
|
|
361
515
|
...envCtx,
|
|
362
516
|
...seedContext,
|
|
363
517
|
};
|
|
@@ -470,10 +624,38 @@ export async function runYamlRecipe(recipe, deps = {}, seedContext = {}) {
|
|
|
470
624
|
// Non-fatal — run-log failures must never break recipe execution.
|
|
471
625
|
}
|
|
472
626
|
}
|
|
627
|
+
// Register this run so POST /runs/:seq/cancel can abort it (H11).
|
|
628
|
+
// Mirrors chainedRunner.ts:1277 — only the top-level run registers.
|
|
629
|
+
const runController = runSeq !== undefined ? registerRun(runSeq) : undefined;
|
|
630
|
+
// L1 (review #1028): the LIVE cancel handle is runController.signal (aborted
|
|
631
|
+
// by POST /runs/:seq/cancel); deps.signal is the external caller signal
|
|
632
|
+
// (absent on the production flat path). Combine both so a cancelled run aborts
|
|
633
|
+
// a pending approval wait instead of hanging the full TTL — forwarding only
|
|
634
|
+
// deps.signal left the flat path's L1 goal unmet. Mirrors the dual-signal
|
|
635
|
+
// next-step check below.
|
|
636
|
+
const effectiveRunSignal = runController?.signal && deps.signal
|
|
637
|
+
? AbortSignal.any([runController.signal, deps.signal])
|
|
638
|
+
: (runController?.signal ?? deps.signal);
|
|
473
639
|
const outputs = [];
|
|
474
640
|
const stepResults = [];
|
|
641
|
+
// P1 cost/token corpus. The price table is loaded once per run (fail-open).
|
|
642
|
+
// `currentStepUsage` accumulates usage across all agent calls of the CURRENT
|
|
643
|
+
// agent step (including judge→refine re-runs via `runAgentText`); `runUsage`
|
|
644
|
+
// sums measured steps into the run-level total.
|
|
645
|
+
const priceTable = loadPriceTable();
|
|
646
|
+
const runUsage = newRunUsageAccumulator();
|
|
647
|
+
let currentStepUsage = newStepUsageAccumulator();
|
|
475
648
|
let stepsRun = 0;
|
|
476
649
|
let runError;
|
|
650
|
+
// Bug (2): the flat runner historically recorded the first non-optional
|
|
651
|
+
// failure in `runError` but kept executing later steps — diverging from
|
|
652
|
+
// chainedRunner, which aborts on a fatal failure. This flag is set ONLY
|
|
653
|
+
// when a failure is fatal (non-optional AND fail-open semantics do not
|
|
654
|
+
// apply via step.optional / on_error.fallback=log_only|deliver_original).
|
|
655
|
+
// The loop checks it at the top and breaks, matching chainedRunner's
|
|
656
|
+
// abort-on-failure contract. Fail-open failures never set it, so
|
|
657
|
+
// log_only/deliver_original/optional steps still let the run continue.
|
|
658
|
+
let haltAfterFailure = false;
|
|
477
659
|
// Live-tail SSE broadcaster. Wrapped in a try/catch on every call so a
|
|
478
660
|
// misbehaving listener can never break the run (mirrors chainedRunner).
|
|
479
661
|
// No-ops when `activityLog` isn't wired (CLI runs, tests, mocks).
|
|
@@ -543,6 +725,245 @@ export async function runYamlRecipe(recipe, deps = {}, seedContext = {}) {
|
|
|
543
725
|
ts: Date.now(),
|
|
544
726
|
});
|
|
545
727
|
};
|
|
728
|
+
// ── OPT-IN judge → refine loop (helper closure) ──────────────────────────
|
|
729
|
+
//
|
|
730
|
+
// ⚠️ INVARIANT DEPARTURE — this drives a bounded revise→re-judge loop and
|
|
731
|
+
// MAY gate the run on exhaustion. It departs the augment-only invariant in
|
|
732
|
+
// judgeVerdict.ts, but is reachable ONLY when the judge step opts in via
|
|
733
|
+
// `agent.max_revisions > 0`. The augment-only PR3a path is untouched.
|
|
734
|
+
//
|
|
735
|
+
// `runAgentText` mirrors the main agent path's text processing exactly
|
|
736
|
+
// (strip leading narration, then JSON-fence parse + sanitize, else use the
|
|
737
|
+
// raw string) so a revised draft commits to ctx the same way a first-pass
|
|
738
|
+
// agent step would. It returns `{ value, ok }`; `ok: false` signals a
|
|
739
|
+
// failed / silent-fail / empty agent response — the caller stops the loop
|
|
740
|
+
// and treats it as exhausted (we don't re-judge a non-result).
|
|
741
|
+
const runAgentText = async (prompt, driver, model, mcpAccess, downshift, providerOptions,
|
|
742
|
+
// P0-5: carry the reviewed/judge step's opt-in tool sandbox into refine-loop
|
|
743
|
+
// re-runs so a sandboxed step STAYS sandboxed across revisions/re-judges.
|
|
744
|
+
sandboxOpts) => {
|
|
745
|
+
// Phase 4: route revisions too, so a downshift on the reviewed step also
|
|
746
|
+
// applies to its refine-loop re-runs (no-op when downshift is absent).
|
|
747
|
+
const routed = resolveRouting({ driver, model }, downshift, prompt, runBudget);
|
|
748
|
+
const agentReturn = await _executeAgent({
|
|
749
|
+
prompt,
|
|
750
|
+
driver: routed.driver === "api" ? "anthropic" : routed.driver,
|
|
751
|
+
model: routed.model,
|
|
752
|
+
...(mcpAccess !== undefined && { mcpAccess }),
|
|
753
|
+
...(sandboxOpts?.sandbox !== undefined && {
|
|
754
|
+
sandbox: sandboxOpts.sandbox,
|
|
755
|
+
}),
|
|
756
|
+
...(sandboxOpts?.tools !== undefined && {
|
|
757
|
+
allowedTools: sandboxOpts.tools,
|
|
758
|
+
}),
|
|
759
|
+
// Worker.autonomy: a sandboxed step STAYS sandboxed across re-runs AND
|
|
760
|
+
// inherits the worker's agent-step deny list (same merge as the primary
|
|
761
|
+
// agent branch), so refine-loop re-runs can't bypass the gate either.
|
|
762
|
+
...(() => {
|
|
763
|
+
const merged = mergeAgentDisallowedTools(sandboxOpts?.disallowedTools, deps.agentDisallowedTools);
|
|
764
|
+
return merged !== undefined ? { disallowedTools: merged } : {};
|
|
765
|
+
})(),
|
|
766
|
+
// Fail closed if a worker sandbox can't be enforced on the chosen driver.
|
|
767
|
+
...(deps.agentDisallowedTools?.length && { enforceSandbox: true }),
|
|
768
|
+
...(providerOptions && { providerOptions }),
|
|
769
|
+
}, buildAgentExecutorDeps(stepDeps, deps));
|
|
770
|
+
runBudget.reconcile(
|
|
771
|
+
// Prefer the driver executeAgent actually resolved+ran; fall back to
|
|
772
|
+
// the routed value only when servedBy is absent (non-executeAgent
|
|
773
|
+
// callers). Stops auto-detected runs being mis-attributed to "auto".
|
|
774
|
+
agentReturn.servedBy?.driver ??
|
|
775
|
+
(routed.driver === "api" ? "anthropic" : (routed.driver ?? "auto")), agentReturn.usage,
|
|
776
|
+
// Resolved model for USD pricing (Phase 3). Absent → unpriced → the USD
|
|
777
|
+
// cap fails open for this call.
|
|
778
|
+
agentReturn.servedBy?.model,
|
|
779
|
+
// Char counts for the opt-in unmeasured-driver ≈$ estimate (warn-only).
|
|
780
|
+
{ inputChars: prompt.length, outputChars: agentReturn.text.length });
|
|
781
|
+
// P1: fold this refine-loop agent call into the current step's usage.
|
|
782
|
+
accumulateAgentUsage(currentStepUsage, agentReturn.usage, agentReturn.servedBy, priceTable);
|
|
783
|
+
const text = agentReturn.text;
|
|
784
|
+
// Same failure detection as the main agent branch: explicit failure
|
|
785
|
+
// marker or silent-fail patterns ⇒ not a usable result.
|
|
786
|
+
if (text.startsWith("[agent step failed:") || detectSilentFail(text)) {
|
|
787
|
+
return { value: text, ok: false };
|
|
788
|
+
}
|
|
789
|
+
const stripped = stripLeadingNarration(text);
|
|
790
|
+
if (!stripped.trim()) {
|
|
791
|
+
return { value: stripped, ok: false };
|
|
792
|
+
}
|
|
793
|
+
try {
|
|
794
|
+
const jsonMatch = /```(?:json)?\s*([\s\S]*?)```/.exec(stripped) ?? [
|
|
795
|
+
null,
|
|
796
|
+
stripped,
|
|
797
|
+
];
|
|
798
|
+
const parsed = sanitizeParsed(JSON.parse((jsonMatch[1] ?? "").trim()));
|
|
799
|
+
return { value: parsed, ok: true };
|
|
800
|
+
}
|
|
801
|
+
catch {
|
|
802
|
+
return { value: stripped, ok: true };
|
|
803
|
+
}
|
|
804
|
+
};
|
|
805
|
+
const runJudgeRefineLoop = async (params) => {
|
|
806
|
+
const { agentCfg, reviewsKey, maxRevisions, judgeStepId, firstVerdict, judgeStepResult, failOpenAgent, } = params;
|
|
807
|
+
// Find the agent step whose output the judge reviews. A judge that
|
|
808
|
+
// reviews a tool step or a seed var (no agent to re-run) cannot be
|
|
809
|
+
// refined — skip the loop gracefully, leaving the augment-only verdict
|
|
810
|
+
// already stashed on the judge step result untouched.
|
|
811
|
+
const reviewedStep = recipe.steps.find((s) => s.agent && (s.agent.into ?? "agent_output") === reviewsKey);
|
|
812
|
+
if (!reviewedStep?.agent) {
|
|
813
|
+
return { haltAfterFailure: false };
|
|
814
|
+
}
|
|
815
|
+
const reviewedAgent = reviewedStep.agent;
|
|
816
|
+
let currentVerdict = firstVerdict;
|
|
817
|
+
let revisions = 0;
|
|
818
|
+
while (revisions < maxRevisions &&
|
|
819
|
+
currentVerdict.verdict === "request_changes") {
|
|
820
|
+
// Budget gate: never exceed the run's token budget. If admission is
|
|
821
|
+
// refused, stop early (treat as exhausted) — the budget halt is
|
|
822
|
+
// surfaced by the next top-of-loop admission check for later steps.
|
|
823
|
+
const admission = runBudget.admit();
|
|
824
|
+
if (!admission.admitted) {
|
|
825
|
+
break;
|
|
826
|
+
}
|
|
827
|
+
// REVISE: re-run the reviewed agent with the prior draft + fixList.
|
|
828
|
+
const priorDraft = ctx[reviewsKey];
|
|
829
|
+
const fixList = currentVerdict.fixList ?? [];
|
|
830
|
+
const revisionBlock = `\n\n<revision-request>\n` +
|
|
831
|
+
`A reviewer requested changes to your previous draft. Address every` +
|
|
832
|
+
` item, then return the full revised draft only.\n\n` +
|
|
833
|
+
`<previous-draft>\n${typeof priorDraft === "string" ? priorDraft : JSON.stringify(priorDraft, null, 2)}\n</previous-draft>\n\n` +
|
|
834
|
+
`<fix-list>\n${fixList.length > 0 ? fixList.map((f) => `- ${f}`).join("\n") : "- (no explicit fix list provided)"}\n</fix-list>\n` +
|
|
835
|
+
`</revision-request>`;
|
|
836
|
+
const revisionPrompt = render(reviewedAgent.prompt, redactSecretsForPrompt(ctx, secretKeys)) +
|
|
837
|
+
revisionBlock;
|
|
838
|
+
// Quality-aware escalation: on the Nth revision, re-run the reviewed step
|
|
839
|
+
// with the Nth more-capable candidate (`escalate[revisions]`) instead of
|
|
840
|
+
// the base model — local/cheap first, escalate to cloud only when the
|
|
841
|
+
// judge keeps rejecting. `revisions` is 0-based here (incremented after
|
|
842
|
+
// the re-judge below). When escalating we drop `downshift` for this call:
|
|
843
|
+
// escalation means "go stronger", so it must not be re-downshifted.
|
|
844
|
+
const escalateTo = reviewedAgent.escalate?.[revisions];
|
|
845
|
+
const revised = await runAgentText(revisionPrompt, escalateTo?.driver ?? reviewedAgent.driver, escalateTo?.model ?? reviewedAgent.model, reviewedAgent.mcpAccess, escalateTo ? undefined : reviewedAgent.downshift, undefined,
|
|
846
|
+
// P0-5: the revision re-runs the REVIEWED step → keep its sandbox.
|
|
847
|
+
{
|
|
848
|
+
...(reviewedAgent.sandbox !== undefined && {
|
|
849
|
+
sandbox: reviewedAgent.sandbox,
|
|
850
|
+
}),
|
|
851
|
+
...(reviewedAgent.tools !== undefined && {
|
|
852
|
+
tools: reviewedAgent.tools,
|
|
853
|
+
}),
|
|
854
|
+
...(reviewedAgent.disallowedTools !== undefined && {
|
|
855
|
+
disallowedTools: reviewedAgent.disallowedTools,
|
|
856
|
+
}),
|
|
857
|
+
});
|
|
858
|
+
if (!revised.ok) {
|
|
859
|
+
// A failed / empty revision can't be re-judged — stop and treat the
|
|
860
|
+
// loop as exhausted with the last good verdict still in place.
|
|
861
|
+
break;
|
|
862
|
+
}
|
|
863
|
+
// R4 #1 (HIGH): stage the revised draft locally — do NOT commit to ctx
|
|
864
|
+
// yet. Committing before the verdict resolves leaves an UNAPPROVED draft
|
|
865
|
+
// in ctx on any loop break (unparseable verdict, budget denial, failed
|
|
866
|
+
// re-judge), so downstream steps treat it as approved. The revised value
|
|
867
|
+
// is only promoted to ctx once a verdict accepts it (approve, or
|
|
868
|
+
// exhaustion with on_exhausted: "proceed").
|
|
869
|
+
const pendingRevised = revised.value;
|
|
870
|
+
// Budget gate (post-revise, pre-re-judge): the revise call may have
|
|
871
|
+
// exhausted the token budget. Check again before firing the re-judge so
|
|
872
|
+
// we don't make one extra LLM call over budget — audit 2026-06-03 LOW #2.
|
|
873
|
+
const postReviseAdmission = runBudget.admit();
|
|
874
|
+
if (!postReviseAdmission.admitted) {
|
|
875
|
+
// Audit 2026-06-08 (recipe-flat-1): the revision was produced but the
|
|
876
|
+
// budget ran out before we could re-judge it. On on_exhausted:"proceed"
|
|
877
|
+
// the user opted to accept best-effort output, so promote the revision
|
|
878
|
+
// instead of silently discarding it and keeping the stale pre-revision
|
|
879
|
+
// draft. On "halt" the run errors below, so leave ctx untouched.
|
|
880
|
+
if ((agentCfg.on_exhausted ?? "halt") === "proceed") {
|
|
881
|
+
ctx[reviewsKey] = pendingRevised;
|
|
882
|
+
}
|
|
883
|
+
break;
|
|
884
|
+
}
|
|
885
|
+
// RE-JUDGE: rebuild the judge prompt against the revised artefact. The
|
|
886
|
+
// judge reviews the STAGED draft, not ctx (which still holds the prior
|
|
887
|
+
// accepted value).
|
|
888
|
+
const reJudgePrompt = render(agentCfg.prompt, redactSecretsForPrompt(ctx, secretKeys)) +
|
|
889
|
+
buildJudgeArtefactBlock(pendingRevised) +
|
|
890
|
+
JUDGE_PROMPT_SUFFIX;
|
|
891
|
+
const judged = await runAgentText(reJudgePrompt, agentCfg.driver, agentCfg.model, agentCfg.mcpAccess,
|
|
892
|
+
// M32: pass the judge step's downshift so cost-aware routing applies
|
|
893
|
+
// to re-judge calls in the refine loop, not just the initial judge.
|
|
894
|
+
agentCfg.downshift,
|
|
895
|
+
// Re-judge is a judge call → enforce JSON on supporting drivers.
|
|
896
|
+
{ responseFormat: { type: "json_object" } },
|
|
897
|
+
// P0-5: the re-judge re-runs the JUDGE step → keep its sandbox.
|
|
898
|
+
{
|
|
899
|
+
...(agentCfg.sandbox !== undefined && { sandbox: agentCfg.sandbox }),
|
|
900
|
+
...(agentCfg.tools !== undefined && { tools: agentCfg.tools }),
|
|
901
|
+
...(agentCfg.disallowedTools !== undefined && {
|
|
902
|
+
disallowedTools: agentCfg.disallowedTools,
|
|
903
|
+
}),
|
|
904
|
+
});
|
|
905
|
+
if (!judged.ok) {
|
|
906
|
+
// Audit 2026-06-03 (MEDIUM #17): a failed / silent-fail / empty
|
|
907
|
+
// RE-JUDGE can't yield a trustworthy verdict. Mirror the revise-
|
|
908
|
+
// failure break above: stop and KEEP the last good verdict. Parsing
|
|
909
|
+
// the failure/empty text would have produced a bogus verdict (usually
|
|
910
|
+
// "unparseable"), silently dropping the request_changes signal and
|
|
911
|
+
// skipping the on_exhausted gate — the run would proceed as if the
|
|
912
|
+
// (unvalidated) revised draft had been approved.
|
|
913
|
+
break;
|
|
914
|
+
}
|
|
915
|
+
const judgedText = typeof judged.value === "string"
|
|
916
|
+
? judged.value
|
|
917
|
+
: JSON.stringify(judged.value);
|
|
918
|
+
currentVerdict = parseJudgeVerdict(stripLeadingNarration(judgedText));
|
|
919
|
+
revisions++;
|
|
920
|
+
// R4 #2 (HIGH): an UNPARSEABLE verdict exits the while-loop (only
|
|
921
|
+
// "request_changes" continues it), but the exhaustion gate below fires
|
|
922
|
+
// ONLY on "request_changes" — so an unparseable verdict would leave the
|
|
923
|
+
// run 'ok' with the unvalidated draft never committed and no error.
|
|
924
|
+
// Treat it as a hard, non-ok stop (distinct from the failed-re-judge
|
|
925
|
+
// break above, which keeps the prior good verdict). Do NOT promote the
|
|
926
|
+
// staged draft.
|
|
927
|
+
if (currentVerdict.verdict === "unparseable") {
|
|
928
|
+
const reason = `judge "${judgeStepId}" returned an unparseable verdict after revision`;
|
|
929
|
+
judgeStepResult.judgeVerdict = currentVerdict;
|
|
930
|
+
judgeStepResult.revisions = revisions;
|
|
931
|
+
judgeStepResult.status = "error";
|
|
932
|
+
judgeStepResult.error = reason;
|
|
933
|
+
judgeStepResult.haltReason = reason;
|
|
934
|
+
judgeStepResult.haltCategory = "judge_revisions_exhausted";
|
|
935
|
+
return {
|
|
936
|
+
runError: reason,
|
|
937
|
+
haltAfterFailure: !failOpenAgent,
|
|
938
|
+
};
|
|
939
|
+
}
|
|
940
|
+
// R4 #1: verdict accepted the revision (approve, or non-exhausted
|
|
941
|
+
// continuation). Promote the staged draft to ctx so downstream steps and
|
|
942
|
+
// the next iteration see the improved, judged value.
|
|
943
|
+
ctx[reviewsKey] = pendingRevised;
|
|
944
|
+
}
|
|
945
|
+
// Record the FINAL verdict + the revision count on the judge step result.
|
|
946
|
+
judgeStepResult.judgeVerdict = currentVerdict;
|
|
947
|
+
judgeStepResult.revisions = revisions;
|
|
948
|
+
// EXHAUSTION: still requesting changes after the loop.
|
|
949
|
+
if (currentVerdict.verdict === "request_changes") {
|
|
950
|
+
const onExhausted = agentCfg.on_exhausted ?? "halt";
|
|
951
|
+
if (onExhausted === "halt") {
|
|
952
|
+
const reason = `judge "${judgeStepId}" did not approve after ${maxRevisions} revisions`;
|
|
953
|
+
judgeStepResult.status = "error";
|
|
954
|
+
judgeStepResult.error = reason;
|
|
955
|
+
judgeStepResult.haltReason = reason;
|
|
956
|
+
judgeStepResult.haltCategory = "judge_revisions_exhausted";
|
|
957
|
+
return {
|
|
958
|
+
runError: reason,
|
|
959
|
+
// Respect fail-open like other agent failures.
|
|
960
|
+
haltAfterFailure: !failOpenAgent,
|
|
961
|
+
};
|
|
962
|
+
}
|
|
963
|
+
// "proceed": leave status ok, keep the recorded (unapproved) verdict.
|
|
964
|
+
}
|
|
965
|
+
return { haltAfterFailure: false };
|
|
966
|
+
};
|
|
546
967
|
// The step loop is wrapped so an uncaught throw from any unguarded
|
|
547
968
|
// call site (a `when`/prompt render on a malformed step, a path-jail
|
|
548
969
|
// re-check, etc.) cannot escape `runYamlRecipe` and strand the
|
|
@@ -551,6 +972,26 @@ export async function runYamlRecipe(recipe, deps = {}, seedContext = {}) {
|
|
|
551
972
|
// finalization path, which marks the run "error".
|
|
552
973
|
try {
|
|
553
974
|
for (const step of recipe.steps) {
|
|
975
|
+
// Bug (2): abort on a prior fatal failure. chainedRunner throws (and
|
|
976
|
+
// stops) when a non-optional step fails; the flat runner used to keep
|
|
977
|
+
// going. Break here so later steps don't run on top of a failed
|
|
978
|
+
// dependency. Fail-open failures (step.optional / on_error.fallback=
|
|
979
|
+
// log_only|deliver_original) never set `haltAfterFailure`, so they
|
|
980
|
+
// still let the run continue exactly as before.
|
|
981
|
+
if (haltAfterFailure)
|
|
982
|
+
break;
|
|
983
|
+
// Run-level cancel: abort when the registry controller fires (H11) OR
|
|
984
|
+
// when a caller-provided signal is aborted (#850 parity — external
|
|
985
|
+
// cancellation, e.g. POST /runs/:seq/cancel). An in-flight step is
|
|
986
|
+
// allowed to finish; the next step is not dispatched.
|
|
987
|
+
if (runController?.signal.aborted || deps.signal?.aborted) {
|
|
988
|
+
runError = runError ?? "recipe run cancelled";
|
|
989
|
+
break;
|
|
990
|
+
}
|
|
991
|
+
// Pick up a `~/.patchwork/prices.json` update mid-run for long-running
|
|
992
|
+
// recipes (honours the refreshPrices() contract). No-op unless a usdMax
|
|
993
|
+
// cap is set; never disturbs injected (unit-test) price tables.
|
|
994
|
+
runBudget.refreshPrices();
|
|
554
995
|
const stepIdForEmit = step.into ?? step.agent?.into ?? `step_${stepsRun}`;
|
|
555
996
|
const stepTs = Date.now();
|
|
556
997
|
stepStartTs.set(stepIdForEmit, stepTs);
|
|
@@ -567,13 +1008,23 @@ export async function runYamlRecipe(recipe, deps = {}, seedContext = {}) {
|
|
|
567
1008
|
// A falsy guard records the step as `skipped`, increments stepsRun, and
|
|
568
1009
|
// continues — it is NOT a failure. Bridge-dev iMessage recipes rely on
|
|
569
1010
|
// this to suppress the iMessage agent step when phone is empty.
|
|
570
|
-
if (
|
|
571
|
-
|
|
572
|
-
const
|
|
573
|
-
|
|
574
|
-
|
|
575
|
-
|
|
576
|
-
|
|
1011
|
+
if (step.when === false ||
|
|
1012
|
+
(typeof step.when === "string" && step.when.length > 0)) {
|
|
1013
|
+
const rendered = step.when === false
|
|
1014
|
+
? "false"
|
|
1015
|
+
: render(step.when, ctx).trim().toLowerCase();
|
|
1016
|
+
// Falsy if the WHOLE value is a falsy token OR its LAST token is. The
|
|
1017
|
+
// last-token check is what makes a `when` fed an agent's free-text
|
|
1018
|
+
// decision gate correctly: a step like `decide_file` emits paragraphs of
|
|
1019
|
+
// reasoning that END in "true"/"false" (into: should_file), and a bare
|
|
1020
|
+
// "non-empty ⇒ truthy" check treated that prose as truthy and ran the
|
|
1021
|
+
// guarded step even on a "false" verdict. Single-token guards (the common
|
|
1022
|
+
// case — {{phone}}, {{repo}}, "0") are unchanged: last token === whole
|
|
1023
|
+
// value. Trailing punctuation/backticks/quotes are stripped so `` `false`. ``
|
|
1024
|
+
// still reads false.
|
|
1025
|
+
const FALSY = new Set(["", "0", "false", "null", "undefined"]);
|
|
1026
|
+
const lastToken = (rendered.split(/\s+/).pop() ?? "").replace(/[^a-z0-9]/g, "");
|
|
1027
|
+
const truthy = !!rendered && !FALSY.has(rendered) && !FALSY.has(lastToken);
|
|
577
1028
|
if (!truthy) {
|
|
578
1029
|
const skipId = step.into ?? step.agent?.into ?? `step_${stepsRun}`;
|
|
579
1030
|
stepResults.push({
|
|
@@ -596,6 +1047,80 @@ export async function runYamlRecipe(recipe, deps = {}, seedContext = {}) {
|
|
|
596
1047
|
continue;
|
|
597
1048
|
}
|
|
598
1049
|
}
|
|
1050
|
+
// Bug (3): per-recipe token budget gates ALL step types, not just
|
|
1051
|
+
// agent steps. The admission check used to live inside the
|
|
1052
|
+
// `if (step.agent)` branch, so once the budget was breached the run
|
|
1053
|
+
// kept executing tool steps unbounded. Gate here — after the `when:`
|
|
1054
|
+
// guard resolves truthy, before the agent/tool split — so a breach
|
|
1055
|
+
// halts the run regardless of the next step's kind. Subscription
|
|
1056
|
+
// drivers report no usage and fail open inside RunBudget, so this is
|
|
1057
|
+
// a no-op until a measured agent step actually breaches the cap.
|
|
1058
|
+
const budgetAdmission = runBudget.admit();
|
|
1059
|
+
if (!budgetAdmission.admitted) {
|
|
1060
|
+
const reason = budgetAdmission.reason ??
|
|
1061
|
+
"Run exceeded its token budget — budget_exceeded.";
|
|
1062
|
+
runError = runError ?? reason;
|
|
1063
|
+
haltAfterFailure = true;
|
|
1064
|
+
const budgetStepId = step.into ?? step.agent?.into ?? `step_${stepsRun}`;
|
|
1065
|
+
stepResults.push({
|
|
1066
|
+
id: budgetStepId,
|
|
1067
|
+
tool: step.agent ? "agent" : step.tool,
|
|
1068
|
+
status: "error",
|
|
1069
|
+
error: reason,
|
|
1070
|
+
haltReason: reason,
|
|
1071
|
+
haltCategory: "budget_exceeded",
|
|
1072
|
+
durationMs: 0,
|
|
1073
|
+
});
|
|
1074
|
+
stepsRun++;
|
|
1075
|
+
persistLiveStepResults();
|
|
1076
|
+
emitStepDone(stepIdForEmit);
|
|
1077
|
+
continue;
|
|
1078
|
+
}
|
|
1079
|
+
// M3 — flat-runner approval gate. Safe-by-default: engages for
|
|
1080
|
+
// `manual`-triggered runs (cron/webhook/recipe runs never block
|
|
1081
|
+
// mid-flight) and only when the bridge injected `requireApprovalFn`
|
|
1082
|
+
// (i.e. approvalGate != "off"). Per-recipe opt-out via
|
|
1083
|
+
// `requireApproval: false`. The injected fn applies the tier threshold
|
|
1084
|
+
// itself and returns `true` for steps that don't need sign-off; a
|
|
1085
|
+
// `false` result is an explicit human rejection → halt the run.
|
|
1086
|
+
//
|
|
1087
|
+
// worker.autonomy: when `gateAutomatedRuns` is set the gate ALSO engages
|
|
1088
|
+
// on automated triggers (that's how workers run), and `requireApprovalFn`
|
|
1089
|
+
// is the worker-aware fn — reversible actions pass, risky-unearned ones
|
|
1090
|
+
// queue. Off → manual-only, byte-identical to pre-flip behaviour.
|
|
1091
|
+
if (deps.requireApprovalFn &&
|
|
1092
|
+
(recipeTriggerKind === "manual" || deps.gateAutomatedRuns) &&
|
|
1093
|
+
recipe.requireApproval !== false) {
|
|
1094
|
+
const approvalToolId = step.agent ? "agent" : (step.tool ?? "unknown");
|
|
1095
|
+
const approved = await deps.requireApprovalFn({
|
|
1096
|
+
toolId: approvalToolId,
|
|
1097
|
+
tier: classifyTool(approvalToolId),
|
|
1098
|
+
summary: step.agent
|
|
1099
|
+
? `agent step${step.agent.into ? ` → ${step.agent.into}` : ""}`
|
|
1100
|
+
: `tool ${approvalToolId}`,
|
|
1101
|
+
params: step.agent ? undefined : step,
|
|
1102
|
+
...(effectiveRunSignal && { signal: effectiveRunSignal }), // L1
|
|
1103
|
+
});
|
|
1104
|
+
if (!approved) {
|
|
1105
|
+
const reason = `Step rejected by approval gate — approval_rejected.`;
|
|
1106
|
+
runError = runError ?? reason;
|
|
1107
|
+
haltAfterFailure = true;
|
|
1108
|
+
const rejId = step.into ?? step.agent?.into ?? `step_${stepsRun}`;
|
|
1109
|
+
stepResults.push({
|
|
1110
|
+
id: rejId,
|
|
1111
|
+
tool: step.agent ? "agent" : step.tool,
|
|
1112
|
+
status: "error",
|
|
1113
|
+
error: reason,
|
|
1114
|
+
haltReason: reason,
|
|
1115
|
+
haltCategory: "approval_rejected",
|
|
1116
|
+
durationMs: 0,
|
|
1117
|
+
});
|
|
1118
|
+
stepsRun++;
|
|
1119
|
+
persistLiveStepResults();
|
|
1120
|
+
emitStepDone(stepIdForEmit);
|
|
1121
|
+
continue;
|
|
1122
|
+
}
|
|
1123
|
+
}
|
|
599
1124
|
// Handle agent steps separately
|
|
600
1125
|
if (step.agent) {
|
|
601
1126
|
const agentCfg = step.agent;
|
|
@@ -603,7 +1128,7 @@ export async function runYamlRecipe(recipe, deps = {}, seedContext = {}) {
|
|
|
603
1128
|
// PR3a: judge prompt convention. Append the structured-verdict
|
|
604
1129
|
// suffix and, when `reviews: <stepId>` is set, inject the
|
|
605
1130
|
// upstream step's output as an <artefact> block.
|
|
606
|
-
let renderedPrompt = render(agentCfg.prompt, ctx);
|
|
1131
|
+
let renderedPrompt = render(agentCfg.prompt, redactSecretsForPrompt(ctx, secretKeys));
|
|
607
1132
|
if (isJudge) {
|
|
608
1133
|
if (agentCfg.reviews) {
|
|
609
1134
|
renderedPrompt += buildJudgeArtefactBlock(ctx[agentCfg.reviews]);
|
|
@@ -613,44 +1138,81 @@ export async function runYamlRecipe(recipe, deps = {}, seedContext = {}) {
|
|
|
613
1138
|
const intoKey = agentCfg.into ?? "agent_output";
|
|
614
1139
|
const stepId = intoKey;
|
|
615
1140
|
const stepStart = Date.now();
|
|
1141
|
+
// P1: fresh per-step usage accumulator for this agent step (and any
|
|
1142
|
+
// judge→refine re-runs it spawns via runAgentText, which share it).
|
|
1143
|
+
currentStepUsage = newStepUsageAccumulator();
|
|
616
1144
|
let agentResult;
|
|
617
|
-
//
|
|
618
|
-
//
|
|
619
|
-
//
|
|
620
|
-
//
|
|
621
|
-
//
|
|
622
|
-
|
|
623
|
-
|
|
624
|
-
|
|
625
|
-
|
|
626
|
-
|
|
627
|
-
|
|
628
|
-
|
|
629
|
-
|
|
630
|
-
|
|
631
|
-
|
|
632
|
-
haltReason: reason,
|
|
633
|
-
haltCategory: "budget_exceeded",
|
|
634
|
-
durationMs: 0,
|
|
635
|
-
});
|
|
636
|
-
stepsRun++;
|
|
637
|
-
persistLiveStepResults();
|
|
638
|
-
emitStepDone(stepIdForEmit);
|
|
639
|
-
continue;
|
|
640
|
-
}
|
|
1145
|
+
// Bug (2): fail-open semantics for THIS agent step. Mirrors the
|
|
1146
|
+
// tool-branch `failOpen` (step.optional OR recipe-level
|
|
1147
|
+
// on_error.fallback=log_only|deliver_original). Used to decide
|
|
1148
|
+
// whether an agent failure is fatal (sets `haltAfterFailure`, which
|
|
1149
|
+
// aborts the run at the next loop top) or fail-open (records the
|
|
1150
|
+
// error but lets the run continue, as before).
|
|
1151
|
+
const agentFallback = recipe.on_error?.fallback;
|
|
1152
|
+
const agentFallbackFailOpen = agentFallback === "log_only" || agentFallback === "deliver_original";
|
|
1153
|
+
const failOpenAgent = step.optional === true || agentFallbackFailOpen;
|
|
1154
|
+
// PR2b: per-recipe token budget. Admission is now checked once at the
|
|
1155
|
+
// top of the loop (Bug (3)) so it gates tool steps too; here we only
|
|
1156
|
+
// reconcile actual consumption after the call. Subscription drivers
|
|
1157
|
+
// (Claude CLI, provider subprocess) report `usage === undefined` —
|
|
1158
|
+
// `RunBudget.reconcile` records a fail-open warning per driver per
|
|
1159
|
+
// run and continues.
|
|
641
1160
|
try {
|
|
1161
|
+
// Phase 4: opt-in cost-aware routing. No-op (returns preferred) when
|
|
1162
|
+
// the step has no `downshift` list or no USD cap is set.
|
|
1163
|
+
const routed = resolveRouting({ driver: agentCfg.driver, model: agentCfg.model }, agentCfg.downshift, renderedPrompt, runBudget);
|
|
1164
|
+
// Worker.autonomy: fold the worker's agent-step sandbox into this
|
|
1165
|
+
// step's own deny list so the subprocess can't bypass the gate.
|
|
1166
|
+
const agentDisallowed = mergeAgentDisallowedTools(agentCfg.disallowedTools, deps.agentDisallowedTools);
|
|
642
1167
|
const agentReturn = await _executeAgent({
|
|
643
1168
|
prompt: renderedPrompt,
|
|
644
|
-
driver:
|
|
645
|
-
model:
|
|
1169
|
+
driver: routed.driver === "api" ? "anthropic" : routed.driver,
|
|
1170
|
+
model: routed.model,
|
|
646
1171
|
...(agentCfg.mcpAccess !== undefined && {
|
|
647
1172
|
mcpAccess: agentCfg.mcpAccess,
|
|
648
1173
|
}),
|
|
1174
|
+
// P0-5 opt-in tool sandbox: thread sandbox + allow/deny lists onto
|
|
1175
|
+
// the executor input so the subprocess driver can enforce them via
|
|
1176
|
+
// --allowed-tools / --disallowed-tools / --permission-mode dontAsk.
|
|
1177
|
+
...(agentCfg.sandbox !== undefined && {
|
|
1178
|
+
sandbox: agentCfg.sandbox,
|
|
1179
|
+
}),
|
|
1180
|
+
...(agentCfg.tools !== undefined && {
|
|
1181
|
+
allowedTools: agentCfg.tools,
|
|
1182
|
+
}),
|
|
1183
|
+
...(agentDisallowed !== undefined && {
|
|
1184
|
+
disallowedTools: agentDisallowed,
|
|
1185
|
+
}),
|
|
1186
|
+
// Worker sandbox is enforceable only on the subprocess driver;
|
|
1187
|
+
// fail closed on any other driver rather than run un-sandboxed.
|
|
1188
|
+
...(deps.agentDisallowedTools?.length && {
|
|
1189
|
+
enforceSandbox: true,
|
|
1190
|
+
}),
|
|
1191
|
+
// Constrained decoding: enforce a pure-JSON verdict on judge steps
|
|
1192
|
+
// (OpenAI-compatible drivers honor it; others ignore it). Pairs
|
|
1193
|
+
// with the pure-JSON JUDGE_PROMPT_SUFFIX + tolerant parser.
|
|
1194
|
+
...(isJudge && {
|
|
1195
|
+
providerOptions: { responseFormat: { type: "json_object" } },
|
|
1196
|
+
}),
|
|
649
1197
|
}, buildAgentExecutorDeps(stepDeps, deps));
|
|
650
1198
|
agentResult = agentReturn.text;
|
|
651
|
-
runBudget.reconcile(
|
|
652
|
-
|
|
653
|
-
|
|
1199
|
+
runBudget.reconcile(
|
|
1200
|
+
// Prefer the driver executeAgent actually resolved+ran; the routed
|
|
1201
|
+
// value is only the fallback for non-executeAgent callers (it is
|
|
1202
|
+
// often undefined → previously logged "auto").
|
|
1203
|
+
agentReturn.servedBy?.driver ??
|
|
1204
|
+
(routed.driver === "api"
|
|
1205
|
+
? "anthropic"
|
|
1206
|
+
: (routed.driver ?? "auto")), agentReturn.usage,
|
|
1207
|
+
// Resolved model for USD pricing (Phase 3); absent → fail open.
|
|
1208
|
+
agentReturn.servedBy?.model,
|
|
1209
|
+
// Char counts for the opt-in unmeasured-driver ≈$ estimate.
|
|
1210
|
+
{
|
|
1211
|
+
inputChars: renderedPrompt.length,
|
|
1212
|
+
outputChars: agentReturn.text.length,
|
|
1213
|
+
});
|
|
1214
|
+
// P1: fold this primary agent call into the current step's usage.
|
|
1215
|
+
accumulateAgentUsage(currentStepUsage, agentReturn.usage, agentReturn.servedBy, priceTable);
|
|
654
1216
|
// Catch both `[agent step failed: ...]` (existing) and the
|
|
655
1217
|
// silent-fail patterns `[agent step skipped: ...]` etc. via the
|
|
656
1218
|
// shared detector. Per-step opt-out via `silentFailDetection: false`.
|
|
@@ -663,6 +1225,8 @@ export async function runYamlRecipe(recipe, deps = {}, seedContext = {}) {
|
|
|
663
1225
|
? `silent-fail detected (${agentSilentFail.reason}): ${agentSilentFail.matched}`
|
|
664
1226
|
: agentResult;
|
|
665
1227
|
runError = runError ?? reason;
|
|
1228
|
+
if (!failOpenAgent)
|
|
1229
|
+
haltAfterFailure = true;
|
|
666
1230
|
stepResults.push({
|
|
667
1231
|
id: stepId,
|
|
668
1232
|
tool: "agent",
|
|
@@ -680,6 +1244,8 @@ export async function runYamlRecipe(recipe, deps = {}, seedContext = {}) {
|
|
|
680
1244
|
if (!stripped.trim()) {
|
|
681
1245
|
const errMsg = `[agent step failed: ${agentCfg.driver ?? "agent"} returned only narration or whitespace — no content]`;
|
|
682
1246
|
runError = runError ?? errMsg;
|
|
1247
|
+
if (!failOpenAgent)
|
|
1248
|
+
haltAfterFailure = true;
|
|
683
1249
|
stepResults.push({
|
|
684
1250
|
id: stepId,
|
|
685
1251
|
tool: "agent",
|
|
@@ -695,12 +1261,15 @@ export async function runYamlRecipe(recipe, deps = {}, seedContext = {}) {
|
|
|
695
1261
|
try {
|
|
696
1262
|
const jsonMatch = /```(?:json)?\s*([\s\S]*?)```/.exec(stripped) ?? [null, stripped];
|
|
697
1263
|
const parsed = sanitizeParsed(JSON.parse((jsonMatch[1] ?? "").trim()));
|
|
698
|
-
|
|
1264
|
+
if (!isJudge)
|
|
1265
|
+
ctx[intoKey] = parsed;
|
|
699
1266
|
}
|
|
700
1267
|
catch {
|
|
701
|
-
|
|
1268
|
+
if (!isJudge)
|
|
1269
|
+
ctx[intoKey] = stripped;
|
|
702
1270
|
}
|
|
703
|
-
|
|
1271
|
+
if (!isJudge)
|
|
1272
|
+
outputs.push(intoKey);
|
|
704
1273
|
// PR3a: parse + stash the judge verdict on the step result.
|
|
705
1274
|
// Augment-only: a `request_changes` verdict still yields
|
|
706
1275
|
// `status: "ok"`. The verdict surfaces via the runlog +
|
|
@@ -708,13 +1277,43 @@ export async function runYamlRecipe(recipe, deps = {}, seedContext = {}) {
|
|
|
708
1277
|
const judgeVerdict = isJudge
|
|
709
1278
|
? parseJudgeVerdict(stripped)
|
|
710
1279
|
: undefined;
|
|
711
|
-
|
|
1280
|
+
const judgeStepResult = {
|
|
712
1281
|
id: stepId,
|
|
713
1282
|
tool: "agent",
|
|
714
1283
|
status: "ok",
|
|
715
1284
|
...(judgeVerdict !== undefined && { judgeVerdict }),
|
|
716
1285
|
durationMs: Date.now() - stepStart,
|
|
717
|
-
}
|
|
1286
|
+
};
|
|
1287
|
+
stepResults.push(judgeStepResult);
|
|
1288
|
+
// ── OPT-IN judge → refine loop ───────────────────────────────
|
|
1289
|
+
// ⚠️ INVARIANT DEPARTURE: when the judge step opts in via
|
|
1290
|
+
// `max_revisions > 0`, a `request_changes` verdict now DRIVES a
|
|
1291
|
+
// bounded revise→re-judge loop instead of merely stashing the
|
|
1292
|
+
// verdict. This deliberately departs the augment-only invariant
|
|
1293
|
+
// (see judgeVerdict.ts) — but ONLY when the opt-in fields are
|
|
1294
|
+
// present. With them absent the block below is skipped entirely
|
|
1295
|
+
// and behavior is byte-identical to the PR3a augment-only path.
|
|
1296
|
+
if (isJudge &&
|
|
1297
|
+
agentCfg.reviews &&
|
|
1298
|
+
typeof agentCfg.max_revisions === "number" &&
|
|
1299
|
+
agentCfg.max_revisions > 0 &&
|
|
1300
|
+
judgeVerdict?.verdict === "request_changes") {
|
|
1301
|
+
const loopOutcome = await runJudgeRefineLoop({
|
|
1302
|
+
agentCfg,
|
|
1303
|
+
reviewsKey: agentCfg.reviews,
|
|
1304
|
+
maxRevisions: agentCfg.max_revisions,
|
|
1305
|
+
judgeStepId: stepId,
|
|
1306
|
+
firstVerdict: judgeVerdict,
|
|
1307
|
+
judgeStepResult,
|
|
1308
|
+
failOpenAgent,
|
|
1309
|
+
});
|
|
1310
|
+
if (loopOutcome.runError !== undefined) {
|
|
1311
|
+
runError = runError ?? loopOutcome.runError;
|
|
1312
|
+
}
|
|
1313
|
+
if (loopOutcome.haltAfterFailure) {
|
|
1314
|
+
haltAfterFailure = true;
|
|
1315
|
+
}
|
|
1316
|
+
}
|
|
718
1317
|
// Slice 2 — per-step expect eval. Runs on the value just
|
|
719
1318
|
// committed to ctx[intoKey]. Halt failure flips the just-pushed
|
|
720
1319
|
// result to error and rolls back the ctx commit so downstream
|
|
@@ -730,11 +1329,9 @@ export async function runYamlRecipe(recipe, deps = {}, seedContext = {}) {
|
|
|
730
1329
|
last.error = `expect_failed: ${failures.join("; ")}`;
|
|
731
1330
|
last.haltReason = `expect_failed in step "${stepId}": ${failures.join("; ")}`;
|
|
732
1331
|
last.haltCategory = "expect_failed";
|
|
733
|
-
const fbk = recipe.on_error?.fallback;
|
|
734
|
-
const fbkOpen = fbk === "log_only" || fbk === "deliver_original";
|
|
735
|
-
const failOpenAgent = step.optional === true || fbkOpen;
|
|
736
1332
|
if (!failOpenAgent) {
|
|
737
1333
|
runError = runError ?? last.haltReason;
|
|
1334
|
+
haltAfterFailure = true;
|
|
738
1335
|
}
|
|
739
1336
|
delete ctx[intoKey];
|
|
740
1337
|
}
|
|
@@ -750,6 +1347,8 @@ export async function runYamlRecipe(recipe, deps = {}, seedContext = {}) {
|
|
|
750
1347
|
catch (err) {
|
|
751
1348
|
const msg = err instanceof Error ? err.message : String(err);
|
|
752
1349
|
runError = runError ?? `agent step "${stepId}" failed: ${msg}`;
|
|
1350
|
+
if (!failOpenAgent)
|
|
1351
|
+
haltAfterFailure = true;
|
|
753
1352
|
stepResults.push({
|
|
754
1353
|
id: stepId,
|
|
755
1354
|
tool: "agent",
|
|
@@ -760,6 +1359,14 @@ export async function runYamlRecipe(recipe, deps = {}, seedContext = {}) {
|
|
|
760
1359
|
durationMs: Date.now() - stepStart,
|
|
761
1360
|
});
|
|
762
1361
|
}
|
|
1362
|
+
// P1: attach this agent step's summed token usage (across primary +
|
|
1363
|
+
// any judge→refine re-runs) to the result just pushed, and fold it
|
|
1364
|
+
// into the run-level total. Fields are ABSENT when no usage measured.
|
|
1365
|
+
const pushedAgentResult = stepResults[stepResults.length - 1];
|
|
1366
|
+
if (pushedAgentResult) {
|
|
1367
|
+
Object.assign(pushedAgentResult, stepUsageFields(currentStepUsage));
|
|
1368
|
+
}
|
|
1369
|
+
foldStepIntoRun(runUsage, currentStepUsage);
|
|
763
1370
|
stepsRun++;
|
|
764
1371
|
persistLiveStepResults();
|
|
765
1372
|
emitStepDone(stepIdForEmit);
|
|
@@ -768,10 +1375,22 @@ export async function runYamlRecipe(recipe, deps = {}, seedContext = {}) {
|
|
|
768
1375
|
const stepStart = Date.now();
|
|
769
1376
|
const stepId = step.into ?? `step_${stepsRun}`;
|
|
770
1377
|
// Resolve retry policy: step-level overrides recipe-level.
|
|
771
|
-
|
|
1378
|
+
// Clamp to 0 as a safety net against negative values slipping past
|
|
1379
|
+
// schema validation (M31: negative retry loops 0 times, skipping step).
|
|
1380
|
+
const retryCount = Math.max(0, step.retry ?? recipe.on_error?.retry ?? 0);
|
|
772
1381
|
const retryDelayMs = step.retryDelay ?? recipe.on_error?.retryDelay ?? 1000;
|
|
773
1382
|
let result = null;
|
|
774
1383
|
let stepError;
|
|
1384
|
+
// Bug (2): distinguish a HARD tool error (a thrown error or a
|
|
1385
|
+
// `{ok:false}` JSON envelope) from a SOFT silent-fail detection
|
|
1386
|
+
// (`{count:0,error}` connector envelopes, string placeholders). Only
|
|
1387
|
+
// hard failures abort the run; soft silent-fail detections keep the
|
|
1388
|
+
// run going so connector health-check recipes can still deliver the
|
|
1389
|
+
// degraded payload downstream (a long-standing, tested contract —
|
|
1390
|
+
// see "linear.list_issues — returns error payload" tests). Silent-fail
|
|
1391
|
+
// detection is an observability augment; it was never meant to gate
|
|
1392
|
+
// delivery for these envelopes.
|
|
1393
|
+
let stepErrorIsSilentFail = false;
|
|
775
1394
|
let thrownError;
|
|
776
1395
|
let thrownErrorCode;
|
|
777
1396
|
for (let attempt = 0; attempt <= retryCount; attempt++) {
|
|
@@ -779,6 +1398,7 @@ export async function runYamlRecipe(recipe, deps = {}, seedContext = {}) {
|
|
|
779
1398
|
await new Promise((r) => setTimeout(r, retryDelayMs));
|
|
780
1399
|
}
|
|
781
1400
|
stepError = undefined;
|
|
1401
|
+
stepErrorIsSilentFail = false;
|
|
782
1402
|
thrownError = undefined;
|
|
783
1403
|
thrownErrorCode = undefined;
|
|
784
1404
|
try {
|
|
@@ -835,6 +1455,7 @@ export async function runYamlRecipe(recipe, deps = {}, seedContext = {}) {
|
|
|
835
1455
|
const detected = detectSilentFail(result);
|
|
836
1456
|
if (detected) {
|
|
837
1457
|
stepError = `silent-fail detected (${detected.reason}): ${detected.matched}`;
|
|
1458
|
+
stepErrorIsSilentFail = true;
|
|
838
1459
|
}
|
|
839
1460
|
}
|
|
840
1461
|
}
|
|
@@ -850,6 +1471,21 @@ export async function runYamlRecipe(recipe, deps = {}, seedContext = {}) {
|
|
|
850
1471
|
}
|
|
851
1472
|
if (!stepError && !thrownError)
|
|
852
1473
|
break;
|
|
1474
|
+
// Audit 2026-06-10 recipe-runners-2: do NOT retry on a step_timeout.
|
|
1475
|
+
// The timed-out attempt's underlying tool call keeps running in the
|
|
1476
|
+
// background (Promise.race only abandons the wait, it does not cancel
|
|
1477
|
+
// the call). Re-issuing the step here is, at best, pointless — for a
|
|
1478
|
+
// write tool the in-flight idempotency ledger short-circuits the retry
|
|
1479
|
+
// to the SAME promise (no second side effect, but also no progress) —
|
|
1480
|
+
// and, at worst, a second side effect for any tool the ledger cannot
|
|
1481
|
+
// dedup (non-write tools, or a write tool whose first attempt already
|
|
1482
|
+
// committed its effect then threw). A true cancel needs an AbortSignal
|
|
1483
|
+
// threaded through every tool/connector call, which is out of scope
|
|
1484
|
+
// here; until then, refusing to retry on timeout is the safe contract.
|
|
1485
|
+
// (Genuine transient failures — non-timeout throws / {ok:false} — still
|
|
1486
|
+
// retry below.)
|
|
1487
|
+
if (thrownError?.startsWith("step_timeout:"))
|
|
1488
|
+
break;
|
|
853
1489
|
}
|
|
854
1490
|
// Recipe-level fallback: log_only / deliver_original treat step failure
|
|
855
1491
|
// as non-fatal (fail-open) — same semantics as step-level optional: true.
|
|
@@ -872,6 +1508,7 @@ export async function runYamlRecipe(recipe, deps = {}, seedContext = {}) {
|
|
|
872
1508
|
});
|
|
873
1509
|
if (!failOpen) {
|
|
874
1510
|
runError = runError ?? `${step.tool} failed: ${thrownError}`;
|
|
1511
|
+
haltAfterFailure = true;
|
|
875
1512
|
}
|
|
876
1513
|
else if (fallbackFailOpen && !step.optional) {
|
|
877
1514
|
console.warn(`step ${stepId} failed but on_error.fallback=${fallback} — treating as non-fatal: ${thrownError}`);
|
|
@@ -880,6 +1517,24 @@ export async function runYamlRecipe(recipe, deps = {}, seedContext = {}) {
|
|
|
880
1517
|
else {
|
|
881
1518
|
const finalStatus = result === null ? "skipped" : stepError ? "error" : "ok";
|
|
882
1519
|
const retryNote = retryCount > 0 ? ` after ${retryCount + 1} attempts` : "";
|
|
1520
|
+
// Outcome attribution: capture the filed-issue URL on github.create_issue
|
|
1521
|
+
// steps so trust-replay can look up the issue's eventual disposition in
|
|
1522
|
+
// the outcome store (confirmed/junk/unknown). Targeted to this one tool
|
|
1523
|
+
// only — storing all tool outputs would bloat the run log unnecessarily.
|
|
1524
|
+
let stepOutput;
|
|
1525
|
+
if (finalStatus === "ok" &&
|
|
1526
|
+
result !== null &&
|
|
1527
|
+
step.tool === "github.create_issue") {
|
|
1528
|
+
try {
|
|
1529
|
+
const parsed = JSON.parse(result);
|
|
1530
|
+
if (typeof parsed.url === "string") {
|
|
1531
|
+
stepOutput = { url: parsed.url, issueNumber: parsed.issueNumber };
|
|
1532
|
+
}
|
|
1533
|
+
}
|
|
1534
|
+
catch {
|
|
1535
|
+
/* non-JSON or missing url — skip output capture */
|
|
1536
|
+
}
|
|
1537
|
+
}
|
|
883
1538
|
stepResults.push({
|
|
884
1539
|
id: stepId,
|
|
885
1540
|
tool: step.tool,
|
|
@@ -891,11 +1546,17 @@ export async function runYamlRecipe(recipe, deps = {}, seedContext = {}) {
|
|
|
891
1546
|
haltCategory: "tool_error",
|
|
892
1547
|
}
|
|
893
1548
|
: {}),
|
|
1549
|
+
...(stepOutput !== undefined ? { output: stepOutput } : {}),
|
|
894
1550
|
durationMs: Date.now() - stepStart,
|
|
895
1551
|
});
|
|
896
1552
|
if (stepError) {
|
|
897
1553
|
if (!failOpen) {
|
|
898
1554
|
runError = runError ?? `${step.tool} failed: ${stepError}`;
|
|
1555
|
+
// Soft silent-fail detections (connector error envelopes) record
|
|
1556
|
+
// the error but must NOT abort the run — see the
|
|
1557
|
+
// `stepErrorIsSilentFail` note above. Hard `{ok:false}` errors do.
|
|
1558
|
+
if (!stepErrorIsSilentFail)
|
|
1559
|
+
haltAfterFailure = true;
|
|
899
1560
|
}
|
|
900
1561
|
else if (fallbackFailOpen && !step.optional) {
|
|
901
1562
|
console.warn(`step ${stepId} failed but on_error.fallback=${fallback} — treating as non-fatal: ${stepError}`);
|
|
@@ -932,6 +1593,7 @@ export async function runYamlRecipe(recipe, deps = {}, seedContext = {}) {
|
|
|
932
1593
|
last.haltCategory = "expect_failed";
|
|
933
1594
|
if (!failOpen) {
|
|
934
1595
|
runError = runError ?? last.haltReason;
|
|
1596
|
+
haltAfterFailure = true;
|
|
935
1597
|
}
|
|
936
1598
|
result = null;
|
|
937
1599
|
}
|
|
@@ -967,6 +1629,13 @@ export async function runYamlRecipe(recipe, deps = {}, seedContext = {}) {
|
|
|
967
1629
|
const msg = err instanceof Error ? err.message : String(err);
|
|
968
1630
|
runError = runError ?? `recipe run aborted: ${msg}`;
|
|
969
1631
|
}
|
|
1632
|
+
finally {
|
|
1633
|
+
// Drop the run from the registry (success, failure, or cancel) so
|
|
1634
|
+
// the seq can't be cancelled post-hoc and the map doesn't leak (H11).
|
|
1635
|
+
if (runController !== undefined && runSeq !== undefined) {
|
|
1636
|
+
unregisterRun(runSeq);
|
|
1637
|
+
}
|
|
1638
|
+
}
|
|
970
1639
|
// Evaluate expect block before persisting so failures are stored in the
|
|
971
1640
|
// run log. Guarded: a throw here must not skip finalization and strand
|
|
972
1641
|
// the run at "running".
|
|
@@ -998,8 +1667,21 @@ export async function runYamlRecipe(recipe, deps = {}, seedContext = {}) {
|
|
|
998
1667
|
...(s.haltReason ? { haltReason: s.haltReason } : {}),
|
|
999
1668
|
...(s.haltCategory ? { haltCategory: s.haltCategory } : {}),
|
|
1000
1669
|
...(s.judgeVerdict ? { judgeVerdict: s.judgeVerdict } : {}),
|
|
1670
|
+
// P1: carry per-step token usage through to the persisted run row.
|
|
1671
|
+
// Absent for tool / unmeasured-driver steps (round-trips unchanged).
|
|
1672
|
+
...(typeof s.inputTokens === "number"
|
|
1673
|
+
? { inputTokens: s.inputTokens }
|
|
1674
|
+
: {}),
|
|
1675
|
+
...(typeof s.outputTokens === "number"
|
|
1676
|
+
? { outputTokens: s.outputTokens }
|
|
1677
|
+
: {}),
|
|
1678
|
+
...(typeof s.costUsd === "number" ? { costUsd: s.costUsd } : {}),
|
|
1001
1679
|
durationMs: s.durationMs,
|
|
1002
1680
|
}));
|
|
1681
|
+
// P1: run-level token aggregate + budget totals (latter only when a
|
|
1682
|
+
// budget was configured — never persist all-zero no-budget totals).
|
|
1683
|
+
const tokenTotals = runTokenTotals(runUsage);
|
|
1684
|
+
const budgetTotals = recipe.budget ? runBudget.totals() : undefined;
|
|
1003
1685
|
if (deps.runLog && runSeq !== undefined) {
|
|
1004
1686
|
deps.runLog.completeRun(runSeq, {
|
|
1005
1687
|
status: runError ? "error" : "done",
|
|
@@ -1010,6 +1692,11 @@ export async function runYamlRecipe(recipe, deps = {}, seedContext = {}) {
|
|
|
1010
1692
|
...(runError !== undefined && { errorMessage: runError }),
|
|
1011
1693
|
...(assertionFailures.length > 0 ? { assertionFailures } : {}),
|
|
1012
1694
|
...(inboxOutputs.length > 0 ? { inboxOutputs } : {}),
|
|
1695
|
+
...(runBudget.finalWarnings().length > 0
|
|
1696
|
+
? { budgetWarnings: runBudget.finalWarnings() }
|
|
1697
|
+
: {}),
|
|
1698
|
+
...(tokenTotals ? { tokenTotals } : {}),
|
|
1699
|
+
...(budgetTotals ? { budgetTotals } : {}),
|
|
1013
1700
|
});
|
|
1014
1701
|
emit("recipe_done", {
|
|
1015
1702
|
runSeq,
|
|
@@ -1047,6 +1734,8 @@ export async function runYamlRecipe(recipe, deps = {}, seedContext = {}) {
|
|
|
1047
1734
|
stepResults: finalStepResults,
|
|
1048
1735
|
...(assertionFailures.length > 0 ? { assertionFailures } : {}),
|
|
1049
1736
|
...(inboxOutputs.length > 0 ? { inboxOutputs } : {}),
|
|
1737
|
+
...(tokenTotals ? { tokenTotals } : {}),
|
|
1738
|
+
...(budgetTotals ? { budgetTotals } : {}),
|
|
1050
1739
|
});
|
|
1051
1740
|
}
|
|
1052
1741
|
}
|
|
@@ -1093,6 +1782,14 @@ export async function runYamlRecipe(recipe, deps = {}, seedContext = {}) {
|
|
|
1093
1782
|
stepResults,
|
|
1094
1783
|
errorMessage: runError,
|
|
1095
1784
|
...(assertionFailures.length > 0 ? { assertionFailures } : {}),
|
|
1785
|
+
...(runBudget.finalWarnings().length > 0
|
|
1786
|
+
? { budgetWarnings: runBudget.finalWarnings() }
|
|
1787
|
+
: {}),
|
|
1788
|
+
// P1: forward run-level token aggregate to callers / persisters.
|
|
1789
|
+
...(() => {
|
|
1790
|
+
const tt = runTokenTotals(runUsage);
|
|
1791
|
+
return tt ? { tokenTotals: tt } : {};
|
|
1792
|
+
})(),
|
|
1096
1793
|
};
|
|
1097
1794
|
}
|
|
1098
1795
|
export async function executeStep(step, ctx, deps) {
|
|
@@ -1296,6 +1993,18 @@ export function defaultGitStaleBranches(days, workdir) {
|
|
|
1296
1993
|
return "(git branches unavailable)";
|
|
1297
1994
|
}
|
|
1298
1995
|
}
|
|
1996
|
+
/**
|
|
1997
|
+
* True when running under the vitest harness (same VITEST / NODE_ENV signal
|
|
1998
|
+
* `src/recipes/migrations/index.ts` guards on). Used only to DEFAULT `testMode` on so a
|
|
1999
|
+
* bare `runYamlRecipe(...)` in a unit test never appends a synthetic row to the
|
|
2000
|
+
* operator's real `~/.patchwork/runs.jsonl` — which is also the de-facto
|
|
2001
|
+
* worker-trust store and rotates at 1 MB / 10k lines, so test rows would evict
|
|
2002
|
+
* real trust evidence and pollute every operator halt surface. An explicit
|
|
2003
|
+
* `deps.testMode` (true or false) always wins over this default.
|
|
2004
|
+
*/
|
|
2005
|
+
function isVitestEnv() {
|
|
2006
|
+
return process.env.VITEST != null || process.env.NODE_ENV === "test";
|
|
2007
|
+
}
|
|
1299
2008
|
/** Resolve all RunnerDeps to concrete StepDeps with production defaults filled in. */
|
|
1300
2009
|
function resolveStepDeps(deps, scope) {
|
|
1301
2010
|
const workdir = deps.workdir ?? process.cwd();
|
|
@@ -1354,7 +2063,8 @@ function resolveStepDeps(deps, scope) {
|
|
|
1354
2063
|
return getValidAccessToken();
|
|
1355
2064
|
}),
|
|
1356
2065
|
logDir: deps.logDir,
|
|
1357
|
-
|
|
2066
|
+
activityLog: deps.activityLog,
|
|
2067
|
+
testMode: deps.testMode ?? isVitestEnv(),
|
|
1358
2068
|
// PR5a/b: per-attempt idempotency ledger. Disk-backed when
|
|
1359
2069
|
// `ledgerDir` + `manualRunId` + recipe name are all available so a
|
|
1360
2070
|
// retry of the same logical attempt re-uses prior records (resume
|
|
@@ -1383,12 +2093,19 @@ function buildAgentExecutorDeps(stepDeps, runnerDeps, claudeCodeFnOverride) {
|
|
|
1383
2093
|
const claudeCliFn = claudeCodeFnOverride ?? stepDeps.claudeCodeFn;
|
|
1384
2094
|
return {
|
|
1385
2095
|
anthropicFn: async (prompt, model) => toAgentResult(await stepDeps.claudeFn(prompt, model)),
|
|
1386
|
-
providerDriverFn: async (driver, prompt, model) => toAgentResult(
|
|
2096
|
+
providerDriverFn: async (driver, prompt, model, providerOptions) => toAgentResult(
|
|
2097
|
+
// Keep the 3-arg call shape when unconstrained (backward-compatible
|
|
2098
|
+
// with deps.providerDriverFn mocks that assert exact arity).
|
|
2099
|
+
providerOptions
|
|
2100
|
+
? await stepDeps.providerDriverFn(driver, prompt, model, providerOptions)
|
|
2101
|
+
: await stepDeps.providerDriverFn(driver, prompt, model)),
|
|
1387
2102
|
claudeCliFn: async (prompt, opts) => toAgentResult(await claudeCliFn(prompt, opts)),
|
|
1388
2103
|
localFn: async (prompt, model) => toAgentResult(await stepDeps.localFn(prompt, model)),
|
|
1389
2104
|
probeClaudeCli: () => {
|
|
1390
2105
|
if (runnerDeps.claudeFn !== undefined)
|
|
1391
2106
|
return false;
|
|
2107
|
+
if (_claudeCliProbeCache !== undefined)
|
|
2108
|
+
return _claudeCliProbeCache.result;
|
|
1392
2109
|
// Use the same resolution as defaultClaudeCodeFn so the auto-detect
|
|
1393
2110
|
// branch in agentExecutor.ts doesn't probe "claude" via PATH and
|
|
1394
2111
|
// then later fail to spawn the configured override (or vice versa).
|
|
@@ -1396,7 +2113,8 @@ function buildAgentExecutorDeps(stepDeps, runnerDeps, claudeCodeFnOverride) {
|
|
|
1396
2113
|
encoding: "utf-8",
|
|
1397
2114
|
timeout: 5000,
|
|
1398
2115
|
});
|
|
1399
|
-
|
|
2116
|
+
_claudeCliProbeCache = { result: !probe.error };
|
|
2117
|
+
return _claudeCliProbeCache.result;
|
|
1400
2118
|
},
|
|
1401
2119
|
loadPatchworkConfig: () => {
|
|
1402
2120
|
// Synchronous static import — earlier `require()` form silently failed
|
|
@@ -1454,19 +2172,35 @@ export function defaultClaudeCodeFn(prompt, opts) {
|
|
|
1454
2172
|
if (opts?.mcpAccess === true) {
|
|
1455
2173
|
return Promise.resolve("[agent step failed: recipe_mcp_unsupported — defaultClaudeCodeFn does not support mcpAccess:true; route via SubprocessDriver or unset the mcpAccess flag on this step]");
|
|
1456
2174
|
}
|
|
2175
|
+
// P0-5 opt-in tool sandbox on the `recipe run --local` / non-bridge path.
|
|
2176
|
+
// Without this the sandbox would be silently ignored here (a one-path gap);
|
|
2177
|
+
// mirror the SubprocessDriver argv rule (§3): filter argv-injection values,
|
|
2178
|
+
// run in --permission-mode dontAsk + --allowed-tools when sandbox is active,
|
|
2179
|
+
// and always apply --disallowed-tools regardless of mode.
|
|
2180
|
+
const sandboxAllowed = (Array.isArray(opts?.allowedTools) ? opts.allowedTools : []).filter((t) => typeof t === "string" && t.length > 0 && !t.startsWith("-"));
|
|
2181
|
+
const sandboxDenied = (Array.isArray(opts?.disallowedTools) ? opts.disallowedTools : []).filter((t) => typeof t === "string" && t.length > 0 && !t.startsWith("-"));
|
|
2182
|
+
const localArgs = [
|
|
2183
|
+
"-p",
|
|
2184
|
+
prompt,
|
|
2185
|
+
// --strict-mcp-config: never load ~/.claude.json or .mcp.json. Recipes
|
|
2186
|
+
// are sandboxed by default (mcpAccess defaults to false above). This
|
|
2187
|
+
// also prevents accidental session attachment when the parent process
|
|
2188
|
+
// had a bridge MCP entry in ~/.claude.json.
|
|
2189
|
+
"--strict-mcp-config",
|
|
2190
|
+
"--system-prompt",
|
|
2191
|
+
"You are a helpful assistant processing a recipe task. Use ONLY the data explicitly provided in the user message — treat it as ground truth. Do not call tools to look up git history, emails, or any other information; all necessary data is already included.",
|
|
2192
|
+
"--no-session-persistence",
|
|
2193
|
+
];
|
|
2194
|
+
if (opts?.sandbox === true && sandboxAllowed.length > 0) {
|
|
2195
|
+
localArgs.push("--permission-mode", "dontAsk");
|
|
2196
|
+
localArgs.push("--allowed-tools", ...sandboxAllowed);
|
|
2197
|
+
}
|
|
2198
|
+
// Deny rules apply in ANY mode.
|
|
2199
|
+
if (sandboxDenied.length > 0) {
|
|
2200
|
+
localArgs.push("--disallowed-tools", ...sandboxDenied);
|
|
2201
|
+
}
|
|
1457
2202
|
try {
|
|
1458
|
-
const result = spawnSync(binary,
|
|
1459
|
-
"-p",
|
|
1460
|
-
prompt,
|
|
1461
|
-
// --strict-mcp-config: never load ~/.claude.json or .mcp.json. Recipes
|
|
1462
|
-
// are sandboxed by default (mcpAccess defaults to false above). This
|
|
1463
|
-
// also prevents accidental session attachment when the parent process
|
|
1464
|
-
// had a bridge MCP entry in ~/.claude.json.
|
|
1465
|
-
"--strict-mcp-config",
|
|
1466
|
-
"--system-prompt",
|
|
1467
|
-
"You are a helpful assistant processing a recipe task. Use ONLY the data explicitly provided in the user message — treat it as ground truth. Do not call tools to look up git history, emails, or any other information; all necessary data is already included.",
|
|
1468
|
-
"--no-session-persistence",
|
|
1469
|
-
], {
|
|
2203
|
+
const result = spawnSync(binary, localArgs, {
|
|
1470
2204
|
cwd: workspace.path,
|
|
1471
2205
|
// sanitizeEnv strips CLAUDECODE / CLAUDE_CODE_* / MCP_* from the
|
|
1472
2206
|
// child so the spawn doesn't re-authenticate as, or nest under,
|
|
@@ -1493,10 +2227,74 @@ export function defaultClaudeCodeFn(prompt, opts) {
|
|
|
1493
2227
|
return Promise.resolve(`[agent step failed: ${err instanceof Error ? err.message : String(err)}]`);
|
|
1494
2228
|
}
|
|
1495
2229
|
}
|
|
2230
|
+
/**
|
|
2231
|
+
* Map a driver's `providerMeta` to AgentUsage. Returns undefined unless BOTH
|
|
2232
|
+
* token counts are present as numbers — a half-populated count would mislead
|
|
2233
|
+
* RunBudget. Pure + exported for tests.
|
|
2234
|
+
*/
|
|
2235
|
+
export function providerMetaToUsage(meta) {
|
|
2236
|
+
if (!meta)
|
|
2237
|
+
return undefined;
|
|
2238
|
+
const inputTokens = meta.inputTokens;
|
|
2239
|
+
const outputTokens = meta.outputTokens;
|
|
2240
|
+
if (typeof inputTokens === "number" && typeof outputTokens === "number") {
|
|
2241
|
+
// Reject NaN/Infinity/negative counts: a negative count would price to a
|
|
2242
|
+
// negative cost and silently *reduce* usdSpent, defeating the usdMax cap.
|
|
2243
|
+
if (!Number.isFinite(inputTokens) ||
|
|
2244
|
+
inputTokens < 0 ||
|
|
2245
|
+
!Number.isFinite(outputTokens) ||
|
|
2246
|
+
outputTokens < 0) {
|
|
2247
|
+
return undefined;
|
|
2248
|
+
}
|
|
2249
|
+
return { inputTokens, outputTokens };
|
|
2250
|
+
}
|
|
2251
|
+
return undefined;
|
|
2252
|
+
}
|
|
2253
|
+
const ROUTER_CHARS_PER_TOKEN = 4;
|
|
2254
|
+
/**
|
|
2255
|
+
* Empirical output:input token ratio used for pre-dispatch cost estimates.
|
|
2256
|
+
* LLMs typically produce far fewer output tokens than they consume on input for
|
|
2257
|
+
* most agentic tasks (completion, classification, summarisation). The old 1:1
|
|
2258
|
+
* assumption made models appear 2–5× more expensive than reality, causing
|
|
2259
|
+
* unnecessary downshifts to cheaper models.
|
|
2260
|
+
*
|
|
2261
|
+
* 0.3 is a deliberately-conservative upper bound (real ratios are often 0.1–0.2
|
|
2262
|
+
* for short-form steps). Using a higher-than-typical value avoids under-estimating
|
|
2263
|
+
* cost and over-spending, while still being far more accurate than 1:1.
|
|
2264
|
+
*
|
|
2265
|
+
* The real cost is always reconciled after the call (see the cost-routing ADR),
|
|
2266
|
+
* so this estimate only affects routing decisions, never final billing.
|
|
2267
|
+
*/
|
|
2268
|
+
const ROUTER_OUTPUT_RATIO = 0.3;
|
|
2269
|
+
/**
|
|
2270
|
+
* Apply opt-in cost-aware routing (Phase 4) to choose the driver/model for an
|
|
2271
|
+
* agent dispatch. Returns `preferred` UNCHANGED when there is no downshift list
|
|
2272
|
+
* or no USD cap is set (byte-identical to no routing). The output-token figure
|
|
2273
|
+
* uses a 0.3:1 output:input estimate (conservative upper bound; the 1:1 default
|
|
2274
|
+
* doubled apparent cost and caused unnecessary model downshifts — audit
|
|
2275
|
+
* 2026-06-03 LOW #7). The real cost is reconciled after the call.
|
|
2276
|
+
* Exported for unit testing.
|
|
2277
|
+
*/
|
|
2278
|
+
export function resolveRouting(preferred, downshift, promptText, budget) {
|
|
2279
|
+
if (!downshift || downshift.length === 0)
|
|
2280
|
+
return preferred;
|
|
2281
|
+
const remainingUsd = budget.remainingUsd();
|
|
2282
|
+
if (remainingUsd === undefined)
|
|
2283
|
+
return preferred; // no USD cap → no routing
|
|
2284
|
+
const estInputTokens = Math.ceil(promptText.length / ROUTER_CHARS_PER_TOKEN);
|
|
2285
|
+
// Fix (audit 2026-06-03 LOW #7): use a realistic 0.3:1 output:input ratio
|
|
2286
|
+
// instead of 1:1. LLMs produce far fewer output tokens than input for most
|
|
2287
|
+
// tasks; 0.3 is a conservative upper bound that avoids under-estimating cost.
|
|
2288
|
+
const estOutputTokens = Math.ceil(estInputTokens * ROUTER_OUTPUT_RATIO);
|
|
2289
|
+
return costRouter(preferred, downshift, {
|
|
2290
|
+
remainingUsd,
|
|
2291
|
+
quote: (driver, model) => budget.quoteUsd(driver, model, estInputTokens, estOutputTokens),
|
|
2292
|
+
});
|
|
2293
|
+
}
|
|
1496
2294
|
/** Returns a providerDriverFn with a per-run driver cache (not shared across runs). */
|
|
1497
|
-
function makeProviderDriverFn() {
|
|
2295
|
+
export function makeProviderDriverFn() {
|
|
1498
2296
|
const cache = new Map();
|
|
1499
|
-
return async function defaultProviderDriverFn(driverName, prompt, model) {
|
|
2297
|
+
return async function defaultProviderDriverFn(driverName, prompt, model, providerOptions) {
|
|
1500
2298
|
try {
|
|
1501
2299
|
let driver = cache.get(driverName);
|
|
1502
2300
|
if (!driver) {
|
|
@@ -1520,14 +2318,45 @@ function makeProviderDriverFn() {
|
|
|
1520
2318
|
startupTimeoutMs,
|
|
1521
2319
|
signal: controller.signal,
|
|
1522
2320
|
model,
|
|
2321
|
+
...(providerOptions && { providerOptions }),
|
|
1523
2322
|
});
|
|
1524
2323
|
if (result.exitCode !== undefined && result.exitCode !== 0) {
|
|
1525
2324
|
const detail = result.stderrTail ?? result.text ?? "";
|
|
1526
2325
|
return `[agent step failed: ${driverName} exited ${result.exitCode}${detail ? ` — ${detail.slice(0, 200)}` : ""}]`;
|
|
1527
2326
|
}
|
|
2327
|
+
// API drivers (OpenAI / Grok) never set exitCode. On failure they
|
|
2328
|
+
// resolve with `{ text: "", wasAborted?/errorMessage }` — surface the
|
|
2329
|
+
// real cause (timeout / 401 / 429) instead of the generic
|
|
2330
|
+
// "empty output" branch below, which swallows the actual reason.
|
|
2331
|
+
if (result.wasAborted) {
|
|
2332
|
+
return `[agent step failed: ${driverName} timed out or was cancelled]`;
|
|
2333
|
+
}
|
|
2334
|
+
if (result.errorMessage) {
|
|
2335
|
+
return `[agent step failed: ${driverName} — ${result.errorMessage.slice(0, 200)}]`;
|
|
2336
|
+
}
|
|
1528
2337
|
if (!result.text) {
|
|
1529
2338
|
return `[agent step failed: ${driverName} returned empty output (possible timeout or auth error)]`;
|
|
1530
2339
|
}
|
|
2340
|
+
// Forward token usage (when the driver reported it) so RunBudget can
|
|
2341
|
+
// enforce a real budget for openai/grok/gemini instead of failing
|
|
2342
|
+
// open. No usage → bare string, normalised to {text} downstream.
|
|
2343
|
+
const usage = providerMetaToUsage(result.providerMeta);
|
|
2344
|
+
// Carry the model the driver ACTUALLY resolved+billed (providerMeta.
|
|
2345
|
+
// model, e.g. openai's "gpt-4o" default when the step omitted model) so
|
|
2346
|
+
// RunBudget prices the real model. executeAgent's stamp() is idempotent
|
|
2347
|
+
// — it preserves this servedBy rather than re-deriving from raw input.
|
|
2348
|
+
const resolvedModel = typeof result.providerMeta?.model === "string"
|
|
2349
|
+
? result.providerMeta.model
|
|
2350
|
+
: undefined;
|
|
2351
|
+
if (usage || resolvedModel) {
|
|
2352
|
+
return {
|
|
2353
|
+
text: result.text,
|
|
2354
|
+
...(usage ? { usage } : {}),
|
|
2355
|
+
...(resolvedModel
|
|
2356
|
+
? { servedBy: { driver: driverName, model: resolvedModel } }
|
|
2357
|
+
: {}),
|
|
2358
|
+
};
|
|
2359
|
+
}
|
|
1531
2360
|
return result.text;
|
|
1532
2361
|
}
|
|
1533
2362
|
finally {
|
|
@@ -1539,13 +2368,30 @@ function makeProviderDriverFn() {
|
|
|
1539
2368
|
}
|
|
1540
2369
|
};
|
|
1541
2370
|
}
|
|
1542
|
-
|
|
2371
|
+
/** Default Anthropic API request timeout. Mirrors the provider path (300s). */
|
|
2372
|
+
const DEFAULT_CLAUDE_API_TIMEOUT_MS = 300_000;
|
|
2373
|
+
/**
|
|
2374
|
+
* R4 #4 (HIGH): default max output tokens. The old hard-coded 1024 silently
|
|
2375
|
+
* truncated structured JSON (judge verdicts, multi-field agent outputs).
|
|
2376
|
+
*/
|
|
2377
|
+
const DEFAULT_CLAUDE_MAX_TOKENS = 4096;
|
|
2378
|
+
export async function defaultClaudeFn(prompt, model, opts) {
|
|
1543
2379
|
const apiKey = process.env.ANTHROPIC_API_KEY;
|
|
1544
2380
|
if (!apiKey)
|
|
1545
2381
|
return { text: "[agent step skipped: ANTHROPIC_API_KEY not set]" };
|
|
2382
|
+
const maxTokens = typeof opts?.maxTokens === "number" && opts.maxTokens > 0
|
|
2383
|
+
? opts.maxTokens
|
|
2384
|
+
: DEFAULT_CLAUDE_MAX_TOKENS;
|
|
2385
|
+
// R4 #3 (HIGH): abort a stalled gateway instead of hanging the run forever.
|
|
2386
|
+
const timeoutMs = typeof opts?.timeoutMs === "number" && opts.timeoutMs > 0
|
|
2387
|
+
? opts.timeoutMs
|
|
2388
|
+
: DEFAULT_CLAUDE_API_TIMEOUT_MS;
|
|
2389
|
+
const controller = new AbortController();
|
|
2390
|
+
const timeout = setTimeout(() => controller.abort(), timeoutMs);
|
|
1546
2391
|
try {
|
|
1547
2392
|
const res = await fetch("https://api.anthropic.com/v1/messages", {
|
|
1548
2393
|
method: "POST",
|
|
2394
|
+
signal: controller.signal,
|
|
1549
2395
|
headers: {
|
|
1550
2396
|
"x-api-key": apiKey,
|
|
1551
2397
|
"anthropic-version": "2023-06-01",
|
|
@@ -1553,7 +2399,7 @@ async function defaultClaudeFn(prompt, model) {
|
|
|
1553
2399
|
},
|
|
1554
2400
|
body: JSON.stringify({
|
|
1555
2401
|
model,
|
|
1556
|
-
max_tokens:
|
|
2402
|
+
max_tokens: maxTokens,
|
|
1557
2403
|
messages: [
|
|
1558
2404
|
{
|
|
1559
2405
|
role: "user",
|
|
@@ -1571,25 +2417,57 @@ async function defaultClaudeFn(prompt, model) {
|
|
|
1571
2417
|
// versions) and downstream (subscription/CLI driver returns
|
|
1572
2418
|
// undefined here).
|
|
1573
2419
|
const data = (await res.json());
|
|
1574
|
-
|
|
2420
|
+
let text = data.content?.[0]?.text ?? "[agent step failed: empty response]";
|
|
2421
|
+
// R4 #4: detect+warn when the response was cut off at the token cap so a
|
|
2422
|
+
// truncated (likely unparseable) JSON payload isn't silently trusted.
|
|
2423
|
+
if (data.stop_reason === "max_tokens") {
|
|
2424
|
+
text = `[warning: response truncated at max_tokens=${maxTokens}; raise max_tokens]\n${text}`;
|
|
2425
|
+
}
|
|
1575
2426
|
const inputTokens = data.usage?.input_tokens;
|
|
1576
2427
|
const outputTokens = data.usage?.output_tokens;
|
|
1577
|
-
if (typeof inputTokens === "number" &&
|
|
2428
|
+
if (typeof inputTokens === "number" &&
|
|
2429
|
+
typeof outputTokens === "number" &&
|
|
2430
|
+
Number.isFinite(inputTokens) &&
|
|
2431
|
+
inputTokens >= 0 &&
|
|
2432
|
+
Number.isFinite(outputTokens) &&
|
|
2433
|
+
outputTokens >= 0) {
|
|
1578
2434
|
return { text, usage: { inputTokens, outputTokens } };
|
|
1579
2435
|
}
|
|
1580
2436
|
return { text };
|
|
1581
2437
|
}
|
|
1582
2438
|
catch (err) {
|
|
2439
|
+
const aborted = controller.signal.aborted ||
|
|
2440
|
+
(err instanceof Error && err.name === "AbortError");
|
|
2441
|
+
if (aborted) {
|
|
2442
|
+
return {
|
|
2443
|
+
text: `[agent step failed: Anthropic API request timed out after ${timeoutMs}ms]`,
|
|
2444
|
+
};
|
|
2445
|
+
}
|
|
1583
2446
|
return {
|
|
1584
2447
|
text: `[agent step failed: ${err instanceof Error ? err.message : String(err)}]`,
|
|
1585
2448
|
};
|
|
1586
2449
|
}
|
|
2450
|
+
finally {
|
|
2451
|
+
clearTimeout(timeout);
|
|
2452
|
+
}
|
|
1587
2453
|
}
|
|
1588
|
-
async function defaultLocalFn(prompt, model) {
|
|
2454
|
+
export async function defaultLocalFn(prompt, model) {
|
|
1589
2455
|
try {
|
|
1590
2456
|
const { createLocalAdapter } = await import("../adapters/local.js");
|
|
1591
2457
|
const { loadConfig: loadPatchworkConfig } = await import("../patchworkConfig.js");
|
|
1592
2458
|
const cfg = loadPatchworkConfig();
|
|
2459
|
+
// Anti-SSRF: the local adapter streams the prompt to `cfg.localEndpoint`
|
|
2460
|
+
// (dashboard/config-controlled). A `driver: local` recipe must not be
|
|
2461
|
+
// able to POST the prompt to an arbitrary public host. Mirror the
|
|
2462
|
+
// LocalApiDriver gate (src/drivers/local/index.ts): reject any non
|
|
2463
|
+
// loopback/private endpoint unless LOCAL_ENDPOINT_ALLOW_REMOTE=1.
|
|
2464
|
+
if (cfg.localEndpoint &&
|
|
2465
|
+
process.env.LOCAL_ENDPOINT_ALLOW_REMOTE !== "1" &&
|
|
2466
|
+
!isLoopbackOrPrivateEndpoint(cfg.localEndpoint)) {
|
|
2467
|
+
return {
|
|
2468
|
+
text: "[agent step failed: localEndpoint is a public host; set LOCAL_ENDPOINT_ALLOW_REMOTE=1 to override]",
|
|
2469
|
+
};
|
|
2470
|
+
}
|
|
1593
2471
|
const adapter = createLocalAdapter({
|
|
1594
2472
|
endpoint: cfg.localEndpoint,
|
|
1595
2473
|
defaultModel: cfg.localModel ?? model,
|
|
@@ -1665,17 +2543,38 @@ export function buildChainedDeps(runnerDeps, claudeCodeFnOverride) {
|
|
|
1665
2543
|
const result = await executeStep(step, {}, stepDeps);
|
|
1666
2544
|
return result ?? "";
|
|
1667
2545
|
};
|
|
1668
|
-
const executeAgent = async (prompt, model, driver,
|
|
1669
|
-
//
|
|
1670
|
-
//
|
|
1671
|
-
//
|
|
1672
|
-
|
|
2546
|
+
const executeAgent = async (prompt, model, driver, opts) => {
|
|
2547
|
+
// Surface the FULL AgentResult (text + usage + servedBy) so the chained
|
|
2548
|
+
// runner can reconcile real spend against the run budget — alignment with
|
|
2549
|
+
// the flat path, which already reads `.usage`. (Previously this closure
|
|
2550
|
+
// discarded everything but `.text`, leaving the chained path's budget
|
|
2551
|
+
// unenforced — the S1 SECURITY finding.)
|
|
2552
|
+
//
|
|
2553
|
+
// P0-5 + parity fix: the prior 4th param was `mcpAccess?: boolean`, but the
|
|
2554
|
+
// AgentExecutor type was 3-arg and the chained call site passed only 3 args
|
|
2555
|
+
// → chained recipes silently dropped mcpAccess (and would have dropped the
|
|
2556
|
+
// new sandbox fields too). Threading an opts object closes both gaps.
|
|
2557
|
+
return _executeAgent({
|
|
1673
2558
|
prompt,
|
|
1674
2559
|
model,
|
|
1675
|
-
driver,
|
|
1676
|
-
...(mcpAccess !== undefined && { mcpAccess }),
|
|
2560
|
+
driver: driver === "api" ? "anthropic" : driver,
|
|
2561
|
+
...(opts?.mcpAccess !== undefined && { mcpAccess: opts.mcpAccess }),
|
|
2562
|
+
...(opts?.sandbox !== undefined && { sandbox: opts.sandbox }),
|
|
2563
|
+
...(opts?.allowedTools !== undefined && {
|
|
2564
|
+
allowedTools: opts.allowedTools,
|
|
2565
|
+
}),
|
|
2566
|
+
// Worker.autonomy: single chokepoint for the CHAINED path — fold the
|
|
2567
|
+
// worker's agent-step deny list into every chained agent call so the
|
|
2568
|
+
// subprocess can't bypass the per-step gate (mirrors the flat branch).
|
|
2569
|
+
...(() => {
|
|
2570
|
+
const merged = mergeAgentDisallowedTools(opts?.disallowedTools, runnerDeps.agentDisallowedTools);
|
|
2571
|
+
return merged !== undefined ? { disallowedTools: merged } : {};
|
|
2572
|
+
})(),
|
|
2573
|
+
// Fail closed if a worker sandbox can't be enforced on the chosen driver.
|
|
2574
|
+
...(runnerDeps.agentDisallowedTools?.length && {
|
|
2575
|
+
enforceSandbox: true,
|
|
2576
|
+
}),
|
|
1677
2577
|
}, buildAgentExecutorDeps(stepDeps, runnerDeps, claudeCodeFnOverride));
|
|
1678
|
-
return result.text;
|
|
1679
2578
|
};
|
|
1680
2579
|
// ---------------------------------------------------------------------
|
|
1681
2580
|
// BEGIN A-PR2 EDIT BLOCK — `loadNestedRecipe` jail (dogfood F-04).
|
|
@@ -1768,7 +2667,17 @@ export function buildChainedDeps(runnerDeps, claudeCodeFnOverride) {
|
|
|
1768
2667
|
}
|
|
1769
2668
|
return null;
|
|
1770
2669
|
};
|
|
1771
|
-
return {
|
|
2670
|
+
return {
|
|
2671
|
+
executeTool,
|
|
2672
|
+
executeAgent,
|
|
2673
|
+
loadNestedRecipe,
|
|
2674
|
+
// Tier-1 #4 (audit 2026-06-22): forward the approval gate into the chained
|
|
2675
|
+
// path so it is no longer flat-only. Undefined when the bridge didn't
|
|
2676
|
+
// inject one (approvalGate == "off") — the chained gate then no-ops.
|
|
2677
|
+
...(runnerDeps.requireApprovalFn && {
|
|
2678
|
+
requireApprovalFn: runnerDeps.requireApprovalFn,
|
|
2679
|
+
}),
|
|
2680
|
+
};
|
|
1772
2681
|
}
|
|
1773
2682
|
/**
|
|
1774
2683
|
* Dispatch a loaded recipe to the appropriate runner.
|
|
@@ -1787,10 +2696,21 @@ export async function dispatchRecipe(recipe, deps, seedContext = {}) {
|
|
|
1787
2696
|
const chainedRecipe = recipe;
|
|
1788
2697
|
const now = deps.now ? deps.now() : new Date();
|
|
1789
2698
|
const options = {
|
|
2699
|
+
// Audit 2026-06-08 (recipe-support-3): only the recipe's declared env
|
|
2700
|
+
// keys reach the template context — NOT the full process.env. Parity with
|
|
2701
|
+
// the flat runner; prevents undeclared-secret exposure via {{env.X}}.
|
|
1790
2702
|
env: {
|
|
1791
|
-
...
|
|
2703
|
+
...declaredRecipeEnv(chainedRecipe),
|
|
1792
2704
|
DATE: now.toISOString().slice(0, 10),
|
|
1793
2705
|
TIME: now.toTimeString().slice(0, 5),
|
|
2706
|
+
// Built-in date/time tokens (parity with the flat runner ctx + lint).
|
|
2707
|
+
YYYY: now.toISOString().slice(0, 4),
|
|
2708
|
+
"YYYY-MM": now.toISOString().slice(0, 7),
|
|
2709
|
+
"YYYY-MM-DD": now.toISOString().slice(0, 10),
|
|
2710
|
+
ISO_NOW: now.toISOString(),
|
|
2711
|
+
HH: now.toISOString().slice(11, 13),
|
|
2712
|
+
MM: now.toISOString().slice(14, 16),
|
|
2713
|
+
SS: now.toISOString().slice(17, 19),
|
|
1794
2714
|
...seedContext,
|
|
1795
2715
|
},
|
|
1796
2716
|
maxConcurrency: Math.max(1, chainedRecipe.maxConcurrency ?? 4),
|
|
@@ -1804,6 +2724,20 @@ export async function dispatchRecipe(recipe, deps, seedContext = {}) {
|
|
|
1804
2724
|
activityLog: deps.chainedOptions?.activityLog,
|
|
1805
2725
|
mockedOutputs: deps.chainedOptions?.mockedOutputs,
|
|
1806
2726
|
taskIdPrefix: deps.chainedOptions?.taskIdPrefix,
|
|
2727
|
+
// Parity (#850): forward the run-level budget, price table, and
|
|
2728
|
+
// cancellation signal that the chained runner honours. Without these the
|
|
2729
|
+
// chained path silently diverged from the flat path —
|
|
2730
|
+
// - `budget` lets a caller inject a shared RunBudget (and is the
|
|
2731
|
+
// hook the chained runner uses to enforce usdMax).
|
|
2732
|
+
// - `priceTable` reuses an already-loaded table instead of forcing the
|
|
2733
|
+
// chained RunBudget to re-load it from disk.
|
|
2734
|
+
// - `signal` wires AbortSignal-based cancellation into the run; a
|
|
2735
|
+
// pre-aborted signal now prevents dispatch on the
|
|
2736
|
+
// chained path too (parity target — flat cancellation
|
|
2737
|
+
// is still a separate gap).
|
|
2738
|
+
budget: deps.chainedOptions?.budget,
|
|
2739
|
+
priceTable: deps.chainedOptions?.priceTable,
|
|
2740
|
+
signal: deps.chainedOptions?.signal,
|
|
1807
2741
|
};
|
|
1808
2742
|
if (!deps.chainedDeps) {
|
|
1809
2743
|
throw new Error("chainedDeps required for chained recipes (provide executeTool, executeAgent, loadNestedRecipe)");
|