@cursor/july 0.1.29 → 0.1.31
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/AGENTS.md +1 -1
- package/README.md +10 -5
- package/dist/bin/agent-serve.js +4 -4
- package/dist/channels/github/index.d.ts +1 -1
- package/dist/channels/github/index.js +1 -1
- package/dist/channels/slack/bot-mentions.d.ts +1 -1
- package/dist/channels/slack/bot-mentions.d.ts.map +1 -1
- package/dist/channels/slack/bot-mentions.js +1 -1
- package/dist/channels/slack/eval-directive.js +1 -1
- package/dist/channels/slack/external-policy.d.ts +1 -1
- package/dist/channels/slack/external-policy.js +1 -1
- package/dist/channels/slack/log.d.ts +1 -1
- package/dist/channels/slack/log.js +1 -1
- package/dist/channels/slack/setup.js +1 -1
- package/dist/channels/slack/socket-mode.js +1 -1
- package/dist/docs/404.html +3 -3
- package/dist/docs/ab.html +11 -11
- package/dist/docs/assets/{ab.md.BMCZ6Hd7.js → ab.md.DAQoJ-up.js} +6 -6
- package/dist/docs/assets/{ab.md.BMCZ6Hd7.lean.js → ab.md.DAQoJ-up.lean.js} +1 -1
- package/dist/docs/assets/{app.QdunVQKg.js → app.DigB_9cQ.js} +1 -1
- package/dist/docs/assets/building-with-agents.md.CnHqvYDd.js +13 -0
- package/dist/docs/assets/{building-with-agents.md.CJCtZCyi.lean.js → building-with-agents.md.CnHqvYDd.lean.js} +1 -1
- package/dist/docs/assets/chunks/@localSearchIndexroot.CnFFl07y.js +1 -0
- package/dist/docs/assets/chunks/{VPLocalSearchBox.B6EpUXYf.js → VPLocalSearchBox.BCMX25Bv.js} +1 -1
- package/dist/docs/assets/chunks/{theme.BOTJVqh7.js → theme.BDDWeELx.js} +2 -2
- package/dist/docs/assets/concepts.md.DFaQEFkA.js +4 -0
- package/dist/docs/assets/concepts.md.DFaQEFkA.lean.js +1 -0
- package/dist/docs/assets/deployment.md.9MYBuKM1.js +55 -0
- package/dist/docs/assets/deployment.md.9MYBuKM1.lean.js +1 -0
- package/dist/docs/assets/{evals.md.DYOjkRCX.js → evals.md.BIUoVZ6X.js} +13 -13
- package/dist/docs/assets/evals.md.BIUoVZ6X.lean.js +1 -0
- package/dist/docs/assets/{example-agents_approval-buddy.md.DFGBYLcc.js → example-agents_approval-buddy.md.BhEfleVx.js} +4 -4
- package/dist/docs/assets/{example-agents_approval-buddy.md.DFGBYLcc.lean.js → example-agents_approval-buddy.md.BhEfleVx.lean.js} +1 -1
- package/dist/docs/assets/example-agents_benny.md.2Et1qa8f.js +7 -0
- package/dist/docs/assets/{example-agents_bugbot.md.DelIdhxB.js → example-agents_bugbot.md.ByUexi5i.js} +5 -5
- package/dist/docs/assets/{example-agents_codebase-wiki.md.DC6sgwn0.js → example-agents_codebase-wiki.md.B4y-7ZVW.js} +6 -6
- package/dist/docs/assets/{example-agents_codebase-wiki.md.DC6sgwn0.lean.js → example-agents_codebase-wiki.md.B4y-7ZVW.lean.js} +1 -1
- package/dist/docs/assets/{example-agents_codeowners-review.md.Ku_tG2RY.js → example-agents_codeowners-review.md.D6ay4nvf.js} +6 -6
- package/dist/docs/assets/{example-agents_codeowners-review.md.Ku_tG2RY.lean.js → example-agents_codeowners-review.md.D6ay4nvf.lean.js} +1 -1
- package/dist/docs/assets/{example-agents_concierge.md.4rQTSMXt.js → example-agents_concierge.md.lL8rhYlj.js} +7 -7
- package/dist/docs/assets/{example-agents_fsd.md.CzgUrDfi.js → example-agents_fsd.md.DfNKQTHz.js} +5 -5
- package/dist/docs/assets/example-agents_index.md.DgGBwckv.js +2 -0
- package/dist/docs/assets/example-agents_index.md.DgGBwckv.lean.js +1 -0
- package/dist/docs/assets/{example-agents_knowledge-base.md.BPJiVueF.js → example-agents_knowledge-base.md.CzyZ2DCr.js} +5 -5
- package/dist/docs/assets/{example-agents_knowledge-base.md.BPJiVueF.lean.js → example-agents_knowledge-base.md.CzyZ2DCr.lean.js} +1 -1
- package/dist/docs/assets/example-agents_oncall.md.wFFXXEyW.js +10 -0
- package/dist/docs/assets/{example-agents_security-reviewer.md.Dhj_m7_B.js → example-agents_security-reviewer.md.Dkf1gyo6.js} +8 -8
- package/dist/docs/assets/example-agents_slack-agent.md.DvgvT4nn.js +5 -0
- package/dist/docs/assets/example-agents_weather-agent.md.Dmrcphhl.js +24 -0
- package/dist/docs/assets/example-agents_weather-agent.md.Dmrcphhl.lean.js +1 -0
- package/dist/docs/assets/{guides_agent-to-agent.md.Bpzgq2Pq.js → guides_agent-to-agent.md.Bmbxy-FA.js} +4 -4
- package/dist/docs/assets/guides_cloud-runtime.md.BZ2GA7Es.js +9 -0
- package/dist/docs/assets/{guides_github.md.DOOCpqsW.js → guides_github.md.R2QlpR75.js} +5 -5
- package/dist/docs/assets/{guides_human-in-the-loop.md.DlUqsp1S.js → guides_human-in-the-loop.md.BWvT7UqY.js} +1 -1
- package/dist/docs/assets/{guides_mcp-oauth.md.Dd8EgSem.js → guides_mcp-oauth.md.C7G7IykG.js} +5 -5
- package/dist/docs/assets/guides_mcp-oauth.md.C7G7IykG.lean.js +1 -0
- package/dist/docs/assets/guides_slack.md.zriQpU_9.js +47 -0
- package/dist/docs/assets/guides_slack.md.zriQpU_9.lean.js +1 -0
- package/dist/docs/assets/{guides_webhooks.md.wSOYas3X.js → guides_webhooks.md.DiAwSR42.js} +1 -1
- package/dist/docs/assets/{hillclimbing.md.DHNast08.js → hillclimbing.md.D9Y1_bYh.js} +1 -1
- package/dist/docs/assets/index.md.CZqbBJPB.js +20 -0
- package/dist/docs/assets/index.md.CZqbBJPB.lean.js +1 -0
- package/dist/docs/assets/{quickstart.md.BU6Iwi_9.js → quickstart.md.TnEXYgYW.js} +12 -12
- package/dist/docs/assets/{reference_agent-config.md.DrW2JUM8.js → reference_agent-config.md.kuN6-OxK.js} +1 -1
- package/dist/docs/assets/reference_cli.md.sD-IUWjg.js +73 -0
- package/dist/docs/assets/{reference_cli.md.ccoKOoXt.lean.js → reference_cli.md.sD-IUWjg.lean.js} +1 -1
- package/dist/docs/assets/{reference_connections.md.B9Q3TOve.js → reference_connections.md.DGqAsFXb.js} +3 -3
- package/dist/docs/assets/reference_http-api.md.CfVM_ICa.js +11 -0
- package/dist/docs/assets/{reference_project-layout.md.Bd_CKtNS.js → reference_project-layout.md.D8E6ZmHJ.js} +4 -4
- package/dist/docs/assets/{reference_project-layout.md.Bd_CKtNS.lean.js → reference_project-layout.md.D8E6ZmHJ.lean.js} +1 -1
- package/dist/docs/assets/{reference_schedules.md.w_F2mXB6.js → reference_schedules.md.gmfYzf_I.js} +1 -1
- package/dist/docs/assets/{reference_sessions.md.DLd6mvbv.js → reference_sessions.md.C_ouF_uf.js} +3 -3
- package/dist/docs/assets/{reference_tools.md.BRSDnTbN.js → reference_tools.md.BswAQM41.js} +3 -3
- package/dist/docs/assets/{scaffolding-agents.md.C3pTrmoE.js → scaffolding-agents.md.Bsr9Pwzu.js} +1 -1
- package/dist/docs/assets/{scaffolding-agents.md.C3pTrmoE.lean.js → scaffolding-agents.md.Bsr9Pwzu.lean.js} +1 -1
- package/dist/docs/assets/{storage.md.DRTdnFvd.js → storage.md.xZoiGM58.js} +3 -3
- package/dist/docs/assets/storage.md.xZoiGM58.lean.js +1 -0
- package/dist/docs/assets/troubleshooting.md.B5RVX_tL.js +1 -0
- package/dist/docs/assets/{troubleshooting.md.CmQkmnzC.lean.js → troubleshooting.md.B5RVX_tL.lean.js} +1 -1
- package/dist/docs/building-with-agents.html +12 -12
- package/dist/docs/concepts.html +6 -6
- package/dist/docs/deployment.html +32 -32
- package/dist/docs/evals.html +19 -19
- package/dist/docs/example-agents/approval-buddy.html +9 -9
- package/dist/docs/example-agents/benny.html +11 -11
- package/dist/docs/example-agents/bugbot.html +10 -10
- package/dist/docs/example-agents/codebase-wiki.html +11 -11
- package/dist/docs/example-agents/codeowners-review.html +11 -11
- package/dist/docs/example-agents/concierge.html +12 -12
- package/dist/docs/example-agents/fsd.html +10 -10
- package/dist/docs/example-agents/index.html +7 -7
- package/dist/docs/example-agents/knowledge-base.html +10 -10
- package/dist/docs/example-agents/oncall.html +10 -10
- package/dist/docs/example-agents/security-reviewer.html +13 -13
- package/dist/docs/example-agents/slack-agent.html +10 -10
- package/dist/docs/example-agents/weather-agent.html +16 -16
- package/dist/docs/guides/agent-to-agent.html +9 -9
- package/dist/docs/guides/cloud-runtime.html +7 -7
- package/dist/docs/guides/github.html +10 -10
- package/dist/docs/guides/human-in-the-loop.html +7 -7
- package/dist/docs/guides/mcp-oauth.html +11 -11
- package/dist/docs/guides/slack.html +22 -17
- package/dist/docs/guides/webhooks.html +7 -7
- package/dist/docs/hashmap.json +1 -1
- package/dist/docs/hillclimbing.html +7 -7
- package/dist/docs/index.html +9 -9
- package/dist/docs/quickstart.html +17 -17
- package/dist/docs/reference/agent-config.html +7 -7
- package/dist/docs/reference/channels.html +5 -5
- package/dist/docs/reference/cli.html +56 -48
- package/dist/docs/reference/connections.html +9 -9
- package/dist/docs/reference/hooks.html +5 -5
- package/dist/docs/reference/http-api.html +7 -7
- package/dist/docs/reference/instructions.html +5 -5
- package/dist/docs/reference/playground.html +5 -5
- package/dist/docs/reference/project-layout.html +9 -9
- package/dist/docs/reference/prompt.html +5 -5
- package/dist/docs/reference/schedules.html +7 -7
- package/dist/docs/reference/sessions.html +8 -8
- package/dist/docs/reference/skills.html +5 -5
- package/dist/docs/reference/subagents.html +5 -5
- package/dist/docs/reference/tools.html +9 -9
- package/dist/docs/scaffolding-agents.html +6 -6
- package/dist/docs/storage.html +9 -9
- package/dist/docs/troubleshooting.html +6 -6
- package/dist/evals/reporters.d.ts +1 -1
- package/dist/evals/reporters.js +1 -1
- package/dist/evals.d.ts +2 -2
- package/dist/files-backends/agent-store-presigned-url.d.ts +100 -0
- package/dist/files-backends/agent-store-presigned-url.d.ts.map +1 -0
- package/dist/files-backends/agent-store-presigned-url.js +347 -0
- package/dist/files-backends/cursor-hosted.d.ts +87 -0
- package/dist/files-backends/cursor-hosted.d.ts.map +1 -0
- package/dist/files-backends/cursor-hosted.js +540 -0
- package/dist/files-backends/local-fs.d.ts +33 -0
- package/dist/files-backends/local-fs.d.ts.map +1 -0
- package/dist/files-backends/local-fs.js +199 -0
- package/dist/files.d.ts +139 -0
- package/dist/files.d.ts.map +1 -0
- package/dist/files.js +89 -0
- package/dist/internal/ab-collector.js +2 -2
- package/dist/internal/ab-fold.js +1 -1
- package/dist/internal/cli-ax.d.ts +2 -2
- package/dist/internal/cli-ax.js +2 -2
- package/dist/internal/cli-deploy.d.ts +1 -1
- package/dist/internal/cli-deploy.d.ts.map +1 -1
- package/dist/internal/cli-deploy.js +7 -2
- package/dist/internal/cli-docs.d.ts +1 -1
- package/dist/internal/cli-docs.js +1 -1
- package/dist/internal/cli-github.js +13 -13
- package/dist/internal/cli-mcp-oauth.d.ts +1 -1
- package/dist/internal/cli-mcp-oauth.js +2 -2
- package/dist/internal/cli-mcp.d.ts +3 -3
- package/dist/internal/cli-mcp.js +4 -4
- package/dist/internal/cli-skills.js +1 -1
- package/dist/internal/cloud-turn-cost.d.ts +9 -1
- package/dist/internal/cloud-turn-cost.d.ts.map +1 -1
- package/dist/internal/cloud-turn-cost.js +14 -4
- package/dist/internal/cursor/account-mcp.js +5 -5
- package/dist/internal/cursor/backend-client.js +1 -1
- package/dist/internal/cursor/github-credentials.js +3 -3
- package/dist/internal/cursor-account-mcp-auth.d.ts +1 -1
- package/dist/internal/cursor-account-mcp-auth.js +1 -1
- package/dist/internal/cursor-event-relay.js +1 -1
- package/dist/internal/cursor-relay-core.d.ts +1 -1
- package/dist/internal/cursor-relay-core.d.ts.map +1 -1
- package/dist/internal/cursor-slack-relay.js +1 -1
- package/dist/internal/deploy-client.d.ts +6 -1
- package/dist/internal/deploy-client.d.ts.map +1 -1
- package/dist/internal/deploy-client.js +3 -2
- package/dist/internal/deploy-source.d.ts +2 -2
- package/dist/internal/deploy-source.js +2 -2
- package/dist/internal/discovery.js +2 -2
- package/dist/internal/distribution.d.ts +5 -5
- package/dist/internal/distribution.d.ts.map +1 -1
- package/dist/internal/distribution.js +6 -6
- package/dist/internal/docs-site.js +5 -5
- package/dist/internal/eval-run-store.js +8 -8
- package/dist/internal/evals-client.d.ts +1 -1
- package/dist/internal/evals-client.js +1 -1
- package/dist/internal/github-fanout.js +2 -2
- package/dist/internal/host-files.d.ts +29 -0
- package/dist/internal/host-files.d.ts.map +1 -0
- package/dist/internal/host-files.js +283 -0
- package/dist/internal/host-platforms.js +5 -5
- package/dist/internal/hosting.d.ts +1 -1
- package/dist/internal/hosting.js +1 -1
- package/dist/internal/http-channel.d.ts.map +1 -1
- package/dist/internal/http-channel.js +26 -0
- package/dist/internal/install-cursor-skills.d.ts +1 -1
- package/dist/internal/install-cursor-skills.d.ts.map +1 -1
- package/dist/internal/install-cursor-skills.js +33 -4
- package/dist/internal/local-control-plane.js +2 -2
- package/dist/internal/logs-client.d.ts +2 -2
- package/dist/internal/logs-client.js +2 -2
- package/dist/internal/mcp-endpoint.js +1 -1
- package/dist/internal/mcp-oauth.js +2 -2
- package/dist/internal/platform-schedule-sync.js +2 -2
- package/dist/internal/playground/toolchain.js +6 -6
- package/dist/internal/reminder-runner.js +13 -13
- package/dist/internal/resolved-connections.js +6 -6
- package/dist/internal/schedule-runner.js +1 -1
- package/dist/internal/sdk-runner.js +5 -5
- package/dist/internal/server.js +16 -16
- package/dist/internal/session-cost.d.ts +3 -2
- package/dist/internal/session-cost.d.ts.map +1 -1
- package/dist/internal/session-cost.js +7 -6
- package/dist/internal/session-engine.d.ts +31 -1
- package/dist/internal/session-engine.d.ts.map +1 -1
- package/dist/internal/session-engine.js +119 -31
- package/dist/internal/slack-provision-client.js +1 -1
- package/dist/internal/storage-coordinator.js +7 -7
- package/dist/internal/trajectory.js +2 -2
- package/dist/internal/turn-cost.d.ts +27 -0
- package/dist/internal/turn-cost.d.ts.map +1 -0
- package/dist/internal/turn-cost.js +83 -0
- package/dist/internal/update-check.d.ts +3 -3
- package/dist/internal/update-check.d.ts.map +1 -1
- package/dist/internal/update-check.js +5 -5
- package/dist/memory.d.ts +1 -1
- package/dist/memory.js +1 -1
- package/dist/playground/assets/index-50PKeJlG.css +1 -0
- package/dist/playground/assets/index-C0f2Wl1q.js +85 -0
- package/dist/playground/index.html +2 -2
- package/dist/storage.d.ts +1 -1
- package/dist/storage.js +1 -1
- package/dist/types.d.ts +120 -16
- package/dist/types.d.ts.map +1 -1
- package/docs/README.md +23 -23
- package/docs/ab.md +12 -12
- package/docs/building-with-agents.md +9 -9
- package/docs/concepts.md +11 -11
- package/docs/deployment.md +55 -43
- package/docs/evals.md +20 -20
- package/docs/example-agents/approval-buddy.md +9 -9
- package/docs/example-agents/benny.md +13 -13
- package/docs/example-agents/bugbot.md +6 -6
- package/docs/example-agents/codebase-wiki.md +8 -8
- package/docs/example-agents/codeowners-review.md +8 -8
- package/docs/example-agents/concierge.md +10 -10
- package/docs/example-agents/fsd.md +7 -7
- package/docs/example-agents/index.md +8 -8
- package/docs/example-agents/knowledge-base.md +9 -9
- package/docs/example-agents/oncall.md +7 -7
- package/docs/example-agents/security-reviewer.md +12 -12
- package/docs/example-agents/slack-agent.md +10 -10
- package/docs/example-agents/weather-agent.md +30 -26
- package/docs/guides/agent-to-agent.md +4 -4
- package/docs/guides/cloud-runtime.md +3 -3
- package/docs/guides/github.md +6 -6
- package/docs/guides/human-in-the-loop.md +1 -1
- package/docs/guides/mcp-oauth.md +9 -9
- package/docs/guides/slack.md +94 -21
- package/docs/guides/webhooks.md +1 -1
- package/docs/hillclimbing.md +3 -3
- package/docs/quickstart.md +18 -18
- package/docs/reference/agent-config.md +1 -1
- package/docs/reference/cli.md +110 -79
- package/docs/reference/connections.md +3 -3
- package/docs/reference/http-api.md +2 -2
- package/docs/reference/project-layout.md +7 -7
- package/docs/reference/schedules.md +1 -1
- package/docs/reference/sessions.md +7 -7
- package/docs/reference/tools.md +3 -3
- package/docs/scaffolding-agents.md +2 -2
- package/docs/storage.md +5 -5
- package/docs/troubleshooting.md +11 -10
- package/package.json +3 -1
- package/skills/ab/SKILL.md +5 -5
- package/skills/create-agent/SKILL.md +17 -17
- package/skills/debug/SKILL.md +10 -10
- package/skills/evals/SKILL.md +13 -13
- package/skills/framework-map/SKILL.md +3 -3
- package/skills/github/SKILL.md +11 -11
- package/skills/hillclimb/SKILL.md +9 -9
- package/skills/mcp-auth/SKILL.md +11 -11
- package/skills/setup-slack/SKILL.md +28 -28
- package/src/bin/agent-serve.ts +4 -4
- package/src/channels/github/index.ts +1 -1
- package/src/channels/slack/bot-mentions.ts +1 -1
- package/src/channels/slack/eval-directive.ts +1 -1
- package/src/channels/slack/external-policy.ts +1 -1
- package/src/channels/slack/log.ts +1 -1
- package/src/channels/slack/setup.ts +1 -1
- package/src/channels/slack/socket-mode.ts +1 -1
- package/src/evals/reporters.ts +1 -1
- package/src/evals.ts +2 -2
- package/src/files-backends/agent-store-presigned-url.ts +402 -0
- package/src/files-backends/cursor-hosted.ts +698 -0
- package/src/files-backends/local-fs.ts +178 -0
- package/src/files.ts +195 -0
- package/src/internal/ab-collector.ts +2 -2
- package/src/internal/ab-fold.ts +1 -1
- package/src/internal/cli-ax.ts +2 -2
- package/src/internal/cli-deploy.ts +7 -2
- package/src/internal/cli-docs.ts +1 -1
- package/src/internal/cli-github.ts +13 -13
- package/src/internal/cli-mcp-oauth.ts +2 -2
- package/src/internal/cli-mcp.ts +4 -4
- package/src/internal/cli-skills.ts +1 -1
- package/src/internal/cloud-turn-cost.ts +20 -4
- package/src/internal/cursor/account-mcp.ts +5 -5
- package/src/internal/cursor/backend-client.ts +1 -1
- package/src/internal/cursor/github-credentials.ts +3 -3
- package/src/internal/cursor-account-mcp-auth.ts +1 -1
- package/src/internal/cursor-event-relay.ts +1 -1
- package/src/internal/cursor-relay-core.ts +1 -1
- package/src/internal/cursor-slack-relay.ts +1 -1
- package/src/internal/deploy-client.ts +8 -2
- package/src/internal/deploy-source.ts +2 -2
- package/src/internal/discovery.ts +2 -2
- package/src/internal/distribution.ts +6 -6
- package/src/internal/docs-site.ts +5 -5
- package/src/internal/eval-run-store.ts +8 -8
- package/src/internal/evals-client.ts +1 -1
- package/src/internal/github-fanout.ts +2 -2
- package/src/internal/host-files.ts +372 -0
- package/src/internal/host-platforms.ts +5 -5
- package/src/internal/hosting.ts +1 -1
- package/src/internal/http-channel.ts +30 -0
- package/src/internal/install-cursor-skills.ts +31 -3
- package/src/internal/local-control-plane.ts +2 -2
- package/src/internal/logs-client.ts +2 -2
- package/src/internal/mcp-endpoint.ts +1 -1
- package/src/internal/mcp-oauth.ts +2 -2
- package/src/internal/platform-schedule-sync.ts +2 -2
- package/src/internal/playground/toolchain.ts +6 -6
- package/src/internal/reminder-runner.ts +13 -13
- package/src/internal/resolved-connections.ts +6 -6
- package/src/internal/schedule-runner.ts +1 -1
- package/src/internal/sdk-runner.ts +5 -5
- package/src/internal/server.ts +16 -16
- package/src/internal/session-cost.ts +7 -6
- package/src/internal/session-engine.ts +141 -28
- package/src/internal/slack-provision-client.ts +1 -1
- package/src/internal/storage-coordinator.ts +7 -7
- package/src/internal/trajectory.ts +2 -2
- package/src/internal/turn-cost.ts +110 -0
- package/src/internal/update-check.ts +6 -6
- package/src/memory.ts +1 -1
- package/src/storage.ts +1 -1
- package/src/types.ts +148 -16
- package/dist/docs/assets/building-with-agents.md.CJCtZCyi.js +0 -13
- package/dist/docs/assets/chunks/@localSearchIndexroot.DGe61XHA.js +0 -1
- package/dist/docs/assets/concepts.md.Cfb9b-k1.js +0 -4
- package/dist/docs/assets/concepts.md.Cfb9b-k1.lean.js +0 -1
- package/dist/docs/assets/deployment.md.TecHo0_2.js +0 -55
- package/dist/docs/assets/deployment.md.TecHo0_2.lean.js +0 -1
- package/dist/docs/assets/evals.md.DYOjkRCX.lean.js +0 -1
- package/dist/docs/assets/example-agents_benny.md.B0gjhI-p.js +0 -7
- package/dist/docs/assets/example-agents_index.md.D2PEVSXl.js +0 -2
- package/dist/docs/assets/example-agents_index.md.D2PEVSXl.lean.js +0 -1
- package/dist/docs/assets/example-agents_oncall.md.BG_sUMly.js +0 -10
- package/dist/docs/assets/example-agents_slack-agent.md.buLbgvBf.js +0 -5
- package/dist/docs/assets/example-agents_weather-agent.md.C9Qv-W0o.js +0 -24
- package/dist/docs/assets/example-agents_weather-agent.md.C9Qv-W0o.lean.js +0 -1
- package/dist/docs/assets/guides_cloud-runtime.md.gVzabdQL.js +0 -9
- package/dist/docs/assets/guides_mcp-oauth.md.Dd8EgSem.lean.js +0 -1
- package/dist/docs/assets/guides_slack.md.CjmJSvZS.js +0 -42
- package/dist/docs/assets/guides_slack.md.CjmJSvZS.lean.js +0 -1
- package/dist/docs/assets/index.md.CH_s5uZe.js +0 -20
- package/dist/docs/assets/index.md.CH_s5uZe.lean.js +0 -1
- package/dist/docs/assets/reference_cli.md.ccoKOoXt.js +0 -65
- package/dist/docs/assets/reference_http-api.md.BncLd3PZ.js +0 -11
- package/dist/docs/assets/storage.md.DRTdnFvd.lean.js +0 -1
- package/dist/docs/assets/troubleshooting.md.CmQkmnzC.js +0 -1
- package/dist/internal/model-pricing.d.ts +0 -49
- package/dist/internal/model-pricing.d.ts.map +0 -1
- package/dist/internal/model-pricing.js +0 -377
- package/dist/playground/assets/index-CquRB-l0.js +0 -85
- package/dist/playground/assets/index-Dj0bWpkn.css +0 -1
- package/src/internal/model-pricing.ts +0 -426
- /package/dist/docs/assets/{example-agents_benny.md.B0gjhI-p.lean.js → example-agents_benny.md.2Et1qa8f.lean.js} +0 -0
- /package/dist/docs/assets/{example-agents_bugbot.md.DelIdhxB.lean.js → example-agents_bugbot.md.ByUexi5i.lean.js} +0 -0
- /package/dist/docs/assets/{example-agents_concierge.md.4rQTSMXt.lean.js → example-agents_concierge.md.lL8rhYlj.lean.js} +0 -0
- /package/dist/docs/assets/{example-agents_fsd.md.CzgUrDfi.lean.js → example-agents_fsd.md.DfNKQTHz.lean.js} +0 -0
- /package/dist/docs/assets/{example-agents_oncall.md.BG_sUMly.lean.js → example-agents_oncall.md.wFFXXEyW.lean.js} +0 -0
- /package/dist/docs/assets/{example-agents_security-reviewer.md.Dhj_m7_B.lean.js → example-agents_security-reviewer.md.Dkf1gyo6.lean.js} +0 -0
- /package/dist/docs/assets/{example-agents_slack-agent.md.buLbgvBf.lean.js → example-agents_slack-agent.md.DvgvT4nn.lean.js} +0 -0
- /package/dist/docs/assets/{guides_agent-to-agent.md.Bpzgq2Pq.lean.js → guides_agent-to-agent.md.Bmbxy-FA.lean.js} +0 -0
- /package/dist/docs/assets/{guides_cloud-runtime.md.gVzabdQL.lean.js → guides_cloud-runtime.md.BZ2GA7Es.lean.js} +0 -0
- /package/dist/docs/assets/{guides_github.md.DOOCpqsW.lean.js → guides_github.md.R2QlpR75.lean.js} +0 -0
- /package/dist/docs/assets/{guides_human-in-the-loop.md.DlUqsp1S.lean.js → guides_human-in-the-loop.md.BWvT7UqY.lean.js} +0 -0
- /package/dist/docs/assets/{guides_webhooks.md.wSOYas3X.lean.js → guides_webhooks.md.DiAwSR42.lean.js} +0 -0
- /package/dist/docs/assets/{hillclimbing.md.DHNast08.lean.js → hillclimbing.md.D9Y1_bYh.lean.js} +0 -0
- /package/dist/docs/assets/{quickstart.md.BU6Iwi_9.lean.js → quickstart.md.TnEXYgYW.lean.js} +0 -0
- /package/dist/docs/assets/{reference_agent-config.md.DrW2JUM8.lean.js → reference_agent-config.md.kuN6-OxK.lean.js} +0 -0
- /package/dist/docs/assets/{reference_connections.md.B9Q3TOve.lean.js → reference_connections.md.DGqAsFXb.lean.js} +0 -0
- /package/dist/docs/assets/{reference_http-api.md.BncLd3PZ.lean.js → reference_http-api.md.CfVM_ICa.lean.js} +0 -0
- /package/dist/docs/assets/{reference_schedules.md.w_F2mXB6.lean.js → reference_schedules.md.gmfYzf_I.lean.js} +0 -0
- /package/dist/docs/assets/{reference_sessions.md.DLd6mvbv.lean.js → reference_sessions.md.C_ouF_uf.lean.js} +0 -0
- /package/dist/docs/assets/{reference_tools.md.BRSDnTbN.lean.js → reference_tools.md.BswAQM41.lean.js} +0 -0
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import{_ as i,c as a,o as e,ag as t}from"./chunks/framework.CAZyNGu9.js";const c=JSON.parse('{"title":"Evals","description":"Define repeatable checks with defineEval, run them with
|
|
1
|
+
import{_ as i,c as a,o as e,ag as t}from"./chunks/framework.CAZyNGu9.js";const c=JSON.parse('{"title":"Evals","description":"Define repeatable checks with defineEval, run them with agent-sdk eval, and use them as regression checks.","frontmatter":{"title":"Evals","description":"Define repeatable checks with defineEval, run them with agent-sdk eval, and use them as regression checks."},"headers":[],"relativePath":"evals.md","filePath":"evals.md"}'),n={name:"evals.md"};function l(h,s,p,d,o,r){return e(),a("div",null,[...s[0]||(s[0]=[t(`<h1 id="evals" tabindex="-1">Evals <a class="header-anchor" href="#evals" aria-label="Permalink to "Evals""></a></h1><p>An eval is a repeatable check that runs your agent against a fixed input and gates the recorded trajectory: the run completed, the right tool ran, the reply has the right shape. Evals are how you know a prompt tweak helped, a refactor didn't regress the agent, and last month's fix is still holding.</p><p>Evals exercise the same surface your users hit. The runner starts (or targets) a real agent server, drives sessions over the public API, and grades what comes back. A passing eval means the agent started, accepted a message, and did what you asserted.</p><div class="note custom-block github-alert"><p class="custom-block-title">NOTE</p><p>Import paths here use <code>@cursor/july/evals</code>. On projects still using <code>@cursor/july</code>, swap the import and run <code>agent-serve eval</code>. See <a href="./README.html#run-the-cli">Run the CLI</a> for the full rename table.</p></div><h2 id="define-evals-with-defineeval" tabindex="-1">Define evals with <code>defineEval</code> <a class="header-anchor" href="#define-evals-with-defineeval" aria-label="Permalink to "Define evals with \`defineEval\`""></a></h2><p>The Agent SDK discovers evals under the project-root <code>evals/</code> directory, in <code>.eval.ts</code> or <code>.eval.js</code> files. That's a sibling of <code>agent/</code>, never inside it (<code>agent/evals/</code> is silently ignored). TypeScript is the normal authoring format.</p><p>The file path is the eval's identity, so you don't author an id. Directories group related evals: <code>evals/builds/api.eval.ts</code> becomes id <code>builds/api</code>. An <code>index</code> filename collapses to its directory, so <code>evals/builds/index.eval.ts</code> becomes <code>builds</code>.</p><p>An eval is a single <code>async test(t)</code>. You drive the agent with <code>t</code> and assert on the run with the same <code>t</code>:</p><div class="language-ts vp-adaptive-theme"><button title="Copy Code" class="copy"></button><span class="lang">ts</span><pre class="shiki shiki-themes github-light github-dark vp-code" tabindex="0"><code><span class="line"><span style="--shiki-light:#6A737D;--shiki-dark:#6A737D;">// evals/readiness.eval.ts</span></span>
|
|
2
2
|
<span class="line"><span style="--shiki-light:#D73A49;--shiki-dark:#F97583;">import</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;"> { defineEval, includes } </span><span style="--shiki-light:#D73A49;--shiki-dark:#F97583;">from</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> "@cursor/july/evals"</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">;</span></span>
|
|
3
3
|
<span class="line"></span>
|
|
4
4
|
<span class="line"><span style="--shiki-light:#D73A49;--shiki-dark:#F97583;">export</span><span style="--shiki-light:#D73A49;--shiki-dark:#F97583;"> default</span><span style="--shiki-light:#6F42C1;--shiki-dark:#B392F0;"> defineEval</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">({</span></span>
|
|
@@ -58,15 +58,15 @@ import{_ as i,c as a,o as e,ag as t}from"./chunks/framework.CAZyNGu9.js";const c
|
|
|
58
58
|
<span class="line"><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">t.</span><span style="--shiki-light:#6F42C1;--shiki-dark:#B392F0;">check</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">(</span></span>
|
|
59
59
|
<span class="line"><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;"> toolResults.</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;">length</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">,</span></span>
|
|
60
60
|
<span class="line"><span style="--shiki-light:#6F42C1;--shiki-dark:#B392F0;"> satisfies</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">((</span><span style="--shiki-light:#E36209;--shiki-dark:#FFAB70;">n</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">) </span><span style="--shiki-light:#D73A49;--shiki-dark:#F97583;">=></span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;"> (n </span><span style="--shiki-light:#D73A49;--shiki-dark:#F97583;">as</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> number</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">) </span><span style="--shiki-light:#D73A49;--shiki-dark:#F97583;"><=</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> 4</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">, </span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;">"at most 4 tool calls"</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">)</span></span>
|
|
61
|
-
<span class="line"><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">);</span></span></code></pre></div><p>A case with no explicit gates falls back to whether at least one turn completed successfully. Add <code>t.succeeded()</code> and behavior-specific gates anyway. They make the contract visible during review.</p><h2 id="run-evals-from-the-cli" tabindex="-1">Run evals from the CLI <a class="header-anchor" href="#run-evals-from-the-cli" aria-label="Permalink to "Run evals from the CLI""></a></h2><p>The <code>eval</code> command discovers, filters, and runs cases.</p><p>Run the CLI under Node 22.13 or newer. Do not use Bun. Its HTTP/2 client breaks tool-result streams and causes eval turns to fail.</p><div class="language-bash vp-adaptive-theme"><button title="Copy Code" class="copy"></button><span class="lang">bash</span><pre class="shiki shiki-themes github-light github-dark vp-code" tabindex="0"><code><span class="line"><span style="--shiki-light:#6F42C1;--shiki-dark:#B392F0;">
|
|
62
|
-
<span class="line"><span style="--shiki-light:#6F42C1;--shiki-dark:#B392F0;">
|
|
63
|
-
<span class="line"><span style="--shiki-light:#6F42C1;--shiki-dark:#B392F0;">
|
|
64
|
-
<span class="line"><span style="--shiki-light:#6F42C1;--shiki-dark:#B392F0;">
|
|
65
|
-
<span class="line"><span style="--shiki-light:#6F42C1;--shiki-dark:#B392F0;">
|
|
66
|
-
<span class="line"><span style="--shiki-light:#6F42C1;--shiki-dark:#B392F0;">
|
|
67
|
-
<span class="line"><span style="--shiki-light:#6F42C1;--shiki-dark:#B392F0;">
|
|
61
|
+
<span class="line"><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">);</span></span></code></pre></div><p>A case with no explicit gates falls back to whether at least one turn completed successfully. Add <code>t.succeeded()</code> and behavior-specific gates anyway. They make the contract visible during review.</p><h2 id="run-evals-from-the-cli" tabindex="-1">Run evals from the CLI <a class="header-anchor" href="#run-evals-from-the-cli" aria-label="Permalink to "Run evals from the CLI""></a></h2><p>The <code>eval</code> command discovers, filters, and runs cases.</p><p>Run the CLI under Node 22.13 or newer. Do not use Bun. Its HTTP/2 client breaks tool-result streams and causes eval turns to fail.</p><div class="language-bash vp-adaptive-theme"><button title="Copy Code" class="copy"></button><span class="lang">bash</span><pre class="shiki shiki-themes github-light github-dark vp-code" tabindex="0"><code><span class="line"><span style="--shiki-light:#6F42C1;--shiki-dark:#B392F0;">agent-sdk</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> eval</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> --dir</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> .</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> --list</span><span style="--shiki-light:#6A737D;--shiki-dark:#6A737D;"> # discover only</span></span>
|
|
62
|
+
<span class="line"><span style="--shiki-light:#6F42C1;--shiki-dark:#B392F0;">agent-sdk</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> eval</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> --dir</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> .</span><span style="--shiki-light:#6A737D;--shiki-dark:#6A737D;"> # run all</span></span>
|
|
63
|
+
<span class="line"><span style="--shiki-light:#6F42C1;--shiki-dark:#B392F0;">agent-sdk</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> eval</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> --dir</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> .</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> builds/checkout</span><span style="--shiki-light:#6A737D;--shiki-dark:#6A737D;"> # one datapoint</span></span>
|
|
64
|
+
<span class="line"><span style="--shiki-light:#6F42C1;--shiki-dark:#B392F0;">agent-sdk</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> eval</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> --dir</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> .</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> builds</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> search</span><span style="--shiki-light:#6A737D;--shiki-dark:#6A737D;"> # several ids or prefixes</span></span>
|
|
65
|
+
<span class="line"><span style="--shiki-light:#6F42C1;--shiki-dark:#B392F0;">agent-sdk</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> eval</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> --dir</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> .</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> --tag</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> smoke</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> --tag</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> pull-request</span><span style="--shiki-light:#6A737D;--shiki-dark:#6A737D;"> # any matching tag</span></span>
|
|
66
|
+
<span class="line"><span style="--shiki-light:#6F42C1;--shiki-dark:#B392F0;">agent-sdk</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> eval</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> --dir</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> .</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> --json</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> --no-stream</span><span style="--shiki-light:#6A737D;--shiki-dark:#6A737D;"> # machine-readable results</span></span>
|
|
67
|
+
<span class="line"><span style="--shiki-light:#6F42C1;--shiki-dark:#B392F0;">agent-sdk</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> eval</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> --dir</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> .</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> --verbose</span><span style="--shiki-light:#6A737D;--shiki-dark:#6A737D;"> # logs + reply snippets</span></span></code></pre></div><p>Id filters use OR semantics. Each filter selects an exact id and its descendants. For example, <code>builds</code> selects <code>builds</code>, <code>builds/checkout</code>, and every other case below that path. Repeated tags also use OR semantics. When you provide both ids and tags, a case must match both groups.</p><p><code>eval</code> boots an ephemeral server on port 0 with a temp state root outside the project, so cases don't inherit ambient monorepo rules and don't pollute <code>.agent-sdk/</code>. Point <code>--url</code> at a running server to eval a live agent instead:</p><div class="language-bash vp-adaptive-theme"><button title="Copy Code" class="copy"></button><span class="lang">bash</span><pre class="shiki shiki-themes github-light github-dark vp-code" tabindex="0"><code><span class="line"><span style="--shiki-light:#6F42C1;--shiki-dark:#B392F0;">agent-sdk</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> eval</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> --dir</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> .</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> \\</span></span>
|
|
68
68
|
<span class="line"><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> --url</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> http://127.0.0.1:3000/weather-agent</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> \\</span></span>
|
|
69
|
-
<span class="line"><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> --bearer-token</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> "</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">$AGENT_TOKEN</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;">"</span></span></code></pre></div><p>The eval definitions still come from <code>--dir</code>; <code>--url</code> only changes the agent that receives the turns. For a locally mounted multi-agent directory, <code>--slug weather-agent</code> chooses the target. Use <code>--state-root</code> to keep ephemeral session state at a chosen path, <code>--timeout-ms</code> to override the project timeout, and <code>--no-stream</code> to keep live progress off stderr. A TTY streams turn progress by default. <code>--verbose</code> still writes <code>t.log</code> lines to stderr and adds reply snippets to text results.</p><p>Model turns need a Cursor credential from <code>
|
|
69
|
+
<span class="line"><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> --bearer-token</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> "</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">$AGENT_TOKEN</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;">"</span></span></code></pre></div><p>The eval definitions still come from <code>--dir</code>; <code>--url</code> only changes the agent that receives the turns. For a locally mounted multi-agent directory, <code>--slug weather-agent</code> chooses the target. Use <code>--state-root</code> to keep ephemeral session state at a chosen path, <code>--timeout-ms</code> to override the project timeout, and <code>--no-stream</code> to keep live progress off stderr. A TTY streams turn progress by default. <code>--verbose</code> still writes <code>t.log</code> lines to stderr and adds reply snippets to text results.</p><p>Model turns need a Cursor credential from <code>agent-sdk login</code> or <code>CURSOR_API_KEY</code>.</p><p>The exit code is <code>0</code> when every selected case passes, <code>1</code> when any case fails, and <code>2</code> when no case matches. <code>--list</code> exits <code>0</code>, including when it finds no cases.</p><p>For a compact command index, see <a href="./reference/cli.html#eval">CLI: eval</a>.</p><h3 id="json-results" tabindex="-1">JSON results <a class="header-anchor" href="#json-results" aria-label="Permalink to "JSON results""></a></h3><p>Use <code>--json --no-stream</code> in scripts and CI. The top-level result carries the totals and one result per case:</p><div class="language-json vp-adaptive-theme"><button title="Copy Code" class="copy"></button><span class="lang">json</span><pre class="shiki shiki-themes github-light github-dark vp-code" tabindex="0"><code><span class="line"><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">{</span></span>
|
|
70
70
|
<span class="line"><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> "ok"</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">: </span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;">true</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">,</span></span>
|
|
71
71
|
<span class="line"><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> "passed"</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">: </span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;">1</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">,</span></span>
|
|
72
72
|
<span class="line"><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> "failed"</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">: </span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;">0</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">,</span></span>
|
|
@@ -82,10 +82,10 @@ import{_ as i,c as a,o as e,ag as t}from"./chunks/framework.CAZyNGu9.js";const c
|
|
|
82
82
|
<span class="line"><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> "durationMs"</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">: </span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;">12340</span></span>
|
|
83
83
|
<span class="line"><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;"> }</span></span>
|
|
84
84
|
<span class="line"><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;"> ]</span></span>
|
|
85
|
-
<span class="line"><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">}</span></span></code></pre></div><p>Each case result can also include <code>description</code>, <code>finalText</code>, <code>tools</code>, <code>error</code>, and tool arguments or output. This shape lets CI report the failed assertion without parsing terminal text.</p><h2 id="run-evals-in-the-playground" tabindex="-1">Run evals in the playground <a class="header-anchor" href="#run-evals-in-the-playground" aria-label="Permalink to "Run evals in the playground""></a></h2><p>Start the server with <code>--dev</code>, open the playground, and choose <strong>Evals</strong>. You can run every case or one case, watch progress, and open the resulting session trace.</p><div class="language-bash vp-adaptive-theme"><button title="Copy Code" class="copy"></button><span class="lang">bash</span><pre class="shiki shiki-themes github-light github-dark vp-code" tabindex="0"><code><span class="line"><span style="--shiki-light:#6F42C1;--shiki-dark:#B392F0;">
|
|
85
|
+
<span class="line"><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">}</span></span></code></pre></div><p>Each case result can also include <code>description</code>, <code>finalText</code>, <code>tools</code>, <code>error</code>, and tool arguments or output. This shape lets CI report the failed assertion without parsing terminal text.</p><h2 id="run-evals-in-the-playground" tabindex="-1">Run evals in the playground <a class="header-anchor" href="#run-evals-in-the-playground" aria-label="Permalink to "Run evals in the playground""></a></h2><p>Start the server with <code>--dev</code>, open the playground, and choose <strong>Evals</strong>. You can run every case or one case, watch progress, and open the resulting session trace.</p><div class="language-bash vp-adaptive-theme"><button title="Copy Code" class="copy"></button><span class="lang">bash</span><pre class="shiki shiki-themes github-light github-dark vp-code" tabindex="0"><code><span class="line"><span style="--shiki-light:#6F42C1;--shiki-dark:#B392F0;">agent-sdk</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> serve</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> --dir</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> .</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> --dev</span></span></code></pre></div><p>Playground runs target the live server instead of an ephemeral one. Their sessions appear in the session list. One eval batch can run at a time. By default those batches are <strong>in-memory only</strong> (capped by <code>maxPlaygroundRuns</code>); set <code>persistRuns</code> in <code>evals.config.ts</code> if you need them after a serve restart — see <a href="#configure-eval-runs">Configure eval runs</a>.</p><p>The UI uses the playground eval routes (available without <code>--dev</code>): <code>GET /v1/dev/evals</code> lists datapoints and config (includes <code>maxPlaygroundRuns</code> / <code>durableRuns</code>), <code>GET /v1/dev/evals/runs</code> rehydrates recent batches after navigation, <code>POST /v1/dev/evals/runs</code> starts a batch (returns an <strong>Eval ID</strong> / <code>runId</code>), <code>GET /v1/dev/evals/runs/:runId</code> polls it, and <code>POST /v1/dev/evals/runs/:runId/cancel</code> cancels a running batch. See <a href="./reference/http-api.html#playground-eval-routes">Playground eval routes</a>. The start request returns <code>202</code> while cases run in the background. Poll until the snapshot status becomes <code>completed</code>, <code>failed</code>, or <code>cancelled</code>. Configuration errors appear on a failed snapshot.</p><p>On <code>--prod</code> / <code>--url</code>, the CLI prints the Eval ID as soon as the batch is accepted (and a Playground deep link with <code>?view=evals&evalRunId=…</code>):</p><div class="language-bash vp-adaptive-theme"><button title="Copy Code" class="copy"></button><span class="lang">bash</span><pre class="shiki shiki-themes github-light github-dark vp-code" tabindex="0"><code><span class="line"><span style="--shiki-light:#6F42C1;--shiki-dark:#B392F0;">agent-sdk</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> eval</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> --prod</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> --slug</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> vulnerability-scanner</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> --tag</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> deepsec</span></span>
|
|
86
86
|
<span class="line"><span style="--shiki-light:#6A737D;--shiki-dark:#6A737D;"># Eval ID: evalrun_…</span></span>
|
|
87
|
-
<span class="line"><span style="--shiki-light:#6A737D;--shiki-dark:#6A737D;"># Cancel:
|
|
87
|
+
<span class="line"><span style="--shiki-light:#6A737D;--shiki-dark:#6A737D;"># Cancel: agent-sdk eval cancel evalrun_… --prod --slug vulnerability-scanner</span></span>
|
|
88
88
|
<span class="line"><span style="--shiki-light:#6A737D;--shiki-dark:#6A737D;"># Playground: https://…/playground?view=evals&evalRunId=evalrun_…</span></span>
|
|
89
89
|
<span class="line"></span>
|
|
90
|
-
<span class="line"><span style="--shiki-light:#6F42C1;--shiki-dark:#B392F0;">
|
|
91
|
-
<span class="line"><span style="--shiki-light:#6F42C1;--shiki-dark:#B392F0;">
|
|
90
|
+
<span class="line"><span style="--shiki-light:#6F42C1;--shiki-dark:#B392F0;">agent-sdk</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> eval</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> cancel</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> evalrun_…</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> --prod</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> --slug</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> vulnerability-scanner</span></span>
|
|
91
|
+
<span class="line"><span style="--shiki-light:#6F42C1;--shiki-dark:#B392F0;">agent-sdk</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> eval</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> status</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> evalrun_…</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> --prod</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> --slug</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> vulnerability-scanner</span></span></code></pre></div><p>The Evals tab prefers the server’s in-flight batch (<code>activeRunId</code>) over a stale tab-local remembered id, so CLI / Slack kicks show up without an incognito window.</p><h2 id="what-good-cases-assert" tabindex="-1">What good cases assert <a class="header-anchor" href="#what-good-cases-assert" aria-label="Permalink to "What good cases assert""></a></h2><p>Gate decisions and shape, not prose. Model wording varies run to run. Tool choice, tool avoidance, and output structure are the stable contract.</p><ol><li><code>t.succeeded()</code>: always, first.</li><li>The tool decision: <code>calledTool</code> for the intended path, <code>notCalledTool</code> for the likely wrong alternative. The pair is stronger than either alone.</li><li>Output shape: a regex for the contract (<code>/ready|blocked/i</code>, a JSON marker, a findings-block fence), never exact sentences.</li><li>For structured output, parse <code>t.reply</code> and check fields with <code>satisfies</code> instead of substring-matching JSON.</li></ol><p>The common failure modes: asserting exact phrasing, packing more than about five gates into one case (split it), and cases that depend on live external state that drifts (pin the input; see fixtures).</p><h2 id="pick-fixtures-by-agent-type" tabindex="-1">Pick fixtures by agent type <a class="header-anchor" href="#pick-fixtures-by-agent-type" aria-label="Permalink to "Pick fixtures by agent type""></a></h2><p>The right fixture depends on the surface under test.</p><table tabindex="0"><thead><tr><th>Agent surface</th><th>Fixture</th></tr></thead><tbody><tr><td>Chat / domain assistant</td><td>A canonical prompt string, chosen once and frozen</td></tr><tr><td>Tool-heavy</td><td>Run <code>agent-sdk call <tool></code> first to pin what the tool returns, then freeze the prompt that triggers it</td></tr><tr><td>GitHub webhook</td><td><code>agent-sdk github replay <pr> --events '*' --dry-run --out fixtures/github</code> snapshots real payloads for offline replay (<a href="./guides/github.html">GitHub guide</a>)</td></tr><tr><td>PR reviewer with host preparation</td><td>Diff, metadata, and gold labels pinned to commit SHAs; keep any live PR matrix small</td></tr><tr><td>Workspace-dependent</td><td><code>workspaceFiles</code> in <code>t.send</code> options, never developer-machine paths</td></tr></tbody></table><p>Tag the fast, reliably passing core <code>smoke</code> and run <code>--tag smoke</code> in the inner loop. Leave slow or flaky-prone cases untagged for explicit runs.</p><h3 id="materialize-api-backed-fixtures" tabindex="-1">Materialize API-backed fixtures <a class="header-anchor" href="#materialize-api-backed-fixtures" aria-label="Permalink to "Materialize API-backed fixtures""></a></h3><p>An input that only points at external data, such as a pull request URL, snapshot id, or pair of commit SHAs, is not self-contained. Fetch it once and commit the rendered fixture before you expand the suite.</p><ol><li>Save the diff, metadata, and labels under <code>fixtures/</code> at pinned revisions.</li><li>Seed those files with <code>workspaceFiles</code>, or read them from the fixture directory.</li><li>Assert decisions and output shape against the saved evidence.</li><li>Keep a small <code>smoke</code> subset for any remaining live pipeline checks.</li></ol><p><code>maxConcurrency</code> limits parallel datapoints. It does not limit model or API fan-out inside one datapoint. Materialized fixtures prevent a large suite from exhausting provider and GitHub rate limits. The <a href="./../skills/evals/SKILL.html">evals skill</a> has the full fixture workflow.</p><h2 id="keep-improvements-with-regression-evals" tabindex="-1">Keep improvements with regression evals <a class="header-anchor" href="#keep-improvements-with-regression-evals" aria-label="Permalink to "Keep improvements with regression evals""></a></h2><p>Every <a href="./hillclimbing.html">hillclimb</a> round that keeps a change must land an eval that would have failed before the change. If you can't express the improvement as a gate (a <code>calledTool</code> shift, a bounded <code>action.result</code> count, an output-shape regex), the improvement is unverified, and it'll regress silently.</p><p>The rule cuts the other way too: never weaken an existing gate to make a round pass. That's the freeze line moving, and it turns your regression suite into a list of checks that no longer protect anything.</p><h2 id="compare-variants-on-live-traffic" tabindex="-1">Compare variants on live traffic <a class="header-anchor" href="#compare-variants-on-live-traffic" aria-label="Permalink to "Compare variants on live traffic""></a></h2><p>Use <code>defineAB</code> to compare variant metrics on live sessions. It is not a test runner and has no <code>agent-sdk ab</code> command. Keep <code>defineEval</code> as the regression ratchet. Eval sessions do not enroll or change live metrics. See <a href="./ab.html">Live A/B metrics</a> for assignment, behavior, collection, and inspection.</p><h2 id="what-s-next" tabindex="-1">What's next <a class="header-anchor" href="#what-s-next" aria-label="Permalink to "What's next""></a></h2><p>Continue with these pages:</p><ul><li><a href="./ab.html">Live A/B metrics</a>: sticky variants and cumulative metrics on live sessions</li><li><a href="./hillclimbing.html">Hillclimbing</a>: the loop evals make trustworthy</li><li><a href="./building-with-agents.html">Building agents with agents</a>: have a coding agent write the first suite</li><li><a href="./guides/github.html">GitHub guide</a>: deterministic webhook fixtures with <code>github replay</code></li><li><a href="./reference/sessions.html">Sessions and streaming</a>: the events <code>t.events</code> contains</li></ul>`,77)])])}const g=i(n,[["render",l]]);export{c as __pageData,g as default};
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
import{_ as i,c as a,o as e,ag as t}from"./chunks/framework.CAZyNGu9.js";const c=JSON.parse('{"title":"Evals","description":"Define repeatable checks with defineEval, run them with agent-sdk eval, and use them as regression checks.","frontmatter":{"title":"Evals","description":"Define repeatable checks with defineEval, run them with agent-sdk eval, and use them as regression checks."},"headers":[],"relativePath":"evals.md","filePath":"evals.md"}'),n={name:"evals.md"};function l(h,s,p,d,o,r){return e(),a("div",null,[...s[0]||(s[0]=[t("",77)])])}const g=i(n,[["render",l]]);export{c as __pageData,g as default};
|
|
@@ -1,10 +1,10 @@
|
|
|
1
|
-
import{_ as a,c as t,o as
|
|
2
|
-
<span class="line"><span style="--shiki-light:#6F42C1;--shiki-dark:#B392F0;">
|
|
1
|
+
import{_ as a,c as t,o as s,ag as i}from"./chunks/framework.CAZyNGu9.js";const u=JSON.parse('{"title":"Keep PR approval policy deterministic with Approval Buddy","description":"Separate code-owned eligibility from model-owned review, then connect GitHub, Slack, subagents, durable storage, and evals.","frontmatter":{"title":"Keep PR approval policy deterministic with Approval Buddy","description":"Separate code-owned eligibility from model-owned review, then connect GitHub, Slack, subagents, durable storage, and evals."},"headers":[],"relativePath":"example-agents/approval-buddy.md","filePath":"example-agents/approval-buddy.md"}'),o={name:"example-agents/approval-buddy.md"};function r(l,e,n,d,p,h){return s(),t("div",null,[...e[0]||(e[0]=[i(`<h1 id="keep-pr-approval-policy-deterministic-with-approval-buddy" tabindex="-1">Keep PR approval policy deterministic with Approval Buddy <a class="header-anchor" href="#keep-pr-approval-policy-deterministic-with-approval-buddy" aria-label="Permalink to "Keep PR approval policy deterministic with Approval Buddy""></a></h1><p>Approval Buddy approves eligible pull requests from a fixed roster and declines every other request. GitHub still blocks self-approval when the stamp identity authored the PR. Code decides eligibility. The model prepares evidence, runs two specialist reviews, and passes their findings to the approval tool without changing the policy decision.</p><p>Use this example when an agent can make a judgment inside a workflow, but authorization and the final side effect must stay in deterministic code.</p><p><a href="./../../examples/approval-buddy/">Browse the Approval Buddy source.</a></p><h2 id="keep-approval-policy-in-code" tabindex="-1">Keep approval policy in code <a class="header-anchor" href="#keep-approval-policy-in-code" aria-label="Permalink to "Keep approval policy in code""></a></h2><p>Approval Buddy draws three hard boundaries:</p><ul><li><code>prepare_review</code> and <code>approve_pr</code> re-read the live PR and apply the same eligibility rules.</li><li>Two subagents inspect prepared evidence, but their findings never grant or block approval.</li><li>Only <code>approve_pr</code> posts the GitHub review.</li></ul><p>A spoofed webhook, Slack message, or model claim can't add someone to the buddy roster. The mutating tool checks the source of truth immediately before it acts.</p><h2 id="follow-the-intended-stamp-flow" tabindex="-1">Follow the intended stamp flow <a class="header-anchor" href="#follow-the-intended-stamp-flow" aria-label="Permalink to "Follow the intended stamp flow""></a></h2><p>The root instructions ask the model to run this sequence for a qualifying PR:</p><ol><li>A non-draft <code>pull_request</code> event arrives with action <code>opened</code>, <code>reopened</code>, or <code>ready_for_review</code>.</li><li>The GitHub channel checks its repository allowlist and starts a session.</li><li><code>turn.started</code> posts a pending commit status.</li><li>The model calls <code>prepare_review</code>.</li><li>Host code fetches the live PR. It checks the author, open state, merged state, and draft state.</li><li>A qualifying PR gets <code>pr/MANIFEST.md</code>, <code>pr/meta.json</code>, and <code>pr/diff.patch</code> in the session workspace. Diffs above 2,000,000 characters are truncated and marked in metadata.</li><li>The model calls both review subagents through the built-in <code>task</code> tool.</li><li>It concatenates their contracted replies and calls <code>approve_pr</code>.</li><li><code>approve_pr</code> re-runs eligibility, posts an <code>APPROVE</code> review, and returns the outcome.</li><li>The channel posts a final commit status. A self-approval block also gets a short timeline comment because no approval review can appear.</li></ol><p>Ineligible PRs skip evidence and subagents. The model still calls <code>approve_pr</code> so the deterministic tool returns the formal decline reason.</p><p>Steps 4 through 9 are prompt-driven. The channel doesn't enforce tool order or prove both subagents ran, and <code>approve_pr</code> accepts missing findings. A failed turn clears the pending status with a green non-blocking result without approving the PR.</p><h2 id="map-the-framework-features" tabindex="-1">Map the framework features <a class="header-anchor" href="#map-the-framework-features" aria-label="Permalink to "Map the framework features""></a></h2><table tabindex="0"><thead><tr><th>Capability</th><th>Source</th><th>Role</th></tr></thead><tbody><tr><td>Root agent and policy prompt</td><td><a href="../../examples/approval-buddy/agent/agent.ts"><code>agent/agent.ts</code></a>, <a href="./../../examples/approval-buddy/agent/instructions.html"><code>agent/instructions.md</code></a></td><td>Configure the local agent and describe orchestration order.</td></tr><tr><td>GitHub channel</td><td><a href="../../examples/approval-buddy/agent/channels/github.ts"><code>agent/channels/github.ts</code></a></td><td>Filter wakes, lease GitHub access, and publish status events.</td></tr><tr><td>Slack channel</td><td><a href="../../examples/approval-buddy/agent/channels/slack.ts"><code>agent/channels/slack.ts</code></a></td><td>Accept approval-bot stamp and qualification requests.</td></tr><tr><td>Server tools</td><td><a href="./../../examples/approval-buddy/agent/tools/"><code>agent/tools/</code></a></td><td>Prepare evidence, approve, list buddies, and search GIFs.</td></tr><tr><td>Deterministic policy</td><td><a href="../../examples/approval-buddy/agent/lib/approve.ts"><code>agent/lib/approve.ts</code></a>, <a href="../../examples/approval-buddy/agent/lib/buddies.ts"><code>agent/lib/buddies.ts</code></a></td><td>Own the roster and live eligibility checks.</td></tr><tr><td>Review subagents</td><td><a href="./../../examples/approval-buddy/agent/subagents/"><code>agent/subagents/</code></a></td><td>Run deep audit and code-quality passes over the same evidence.</td></tr><tr><td>Storage</td><td><a href="../../examples/approval-buddy/agent/storage.ts"><code>agent/storage.ts</code></a></td><td>Persist sessions and events with <code>cursorHostedStorage</code> (Bugbot <code>agent_serve_*</code>).</td></tr><tr><td>Evals and unit tests</td><td><a href="./../../examples/approval-buddy/evals/"><code>evals/</code></a>, <a href="./../../examples/approval-buddy/agent/lib/"><code>agent/lib/</code></a></td><td>Protect routing, output contracts, policy, and GitHub behavior.</td></tr></tbody></table><p>There are no authored skills, MCP connections, schedules, reminders, hooks, A/B experiments, sandbox seeds, or tool approvals.</p><h2 id="prepare-credentials" tabindex="-1">Prepare credentials <a class="header-anchor" href="#prepare-credentials" aria-label="Permalink to "Prepare credentials""></a></h2><p>You need:</p><ul><li>Node 22.13 or newer.</li><li>An agent-runtime credential.</li><li>GitHub access to read PRs, post reviews, create commit statuses, and post the self-approval visibility comment.</li></ul><p>Optional GIF selection uses:</p><ul><li><code>GIPHY_API_KEY</code> or <code>APPROVAL_BUDDY_GIPHY_API_KEY</code>,</li><li><code>APPROVAL_BUDDY_STAMP_GIF</code>, or</li><li>severity-specific <code>APPROVAL_BUDDY_STAMP_GIF_<LEVEL></code> variables.</li></ul><p>If you enable Giphy in a hosted copy, declare its secret and <code>api.giphy.com</code> egress.</p><h2 id="validate-without-approving-a-pr" tabindex="-1">Validate without approving a PR <a class="header-anchor" href="#validate-without-approving-a-pr" aria-label="Permalink to "Validate without approving a PR""></a></h2><div class="language-bash vp-adaptive-theme"><button title="Copy Code" class="copy"></button><span class="lang">bash</span><pre class="shiki shiki-themes github-light github-dark vp-code" tabindex="0"><code><span class="line"><span style="--shiki-light:#6F42C1;--shiki-dark:#B392F0;">agent-sdk</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> validate</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> --dir</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> examples/approval-buddy</span></span>
|
|
2
|
+
<span class="line"><span style="--shiki-light:#6F42C1;--shiki-dark:#B392F0;">agent-sdk</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> info</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> --dir</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> examples/approval-buddy</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> --json</span></span></code></pre></div><p>List the deterministic roster:</p><div class="language-bash vp-adaptive-theme"><button title="Copy Code" class="copy"></button><span class="lang">bash</span><pre class="shiki shiki-themes github-light github-dark vp-code" tabindex="0"><code><span class="line"><span style="--shiki-light:#6F42C1;--shiki-dark:#B392F0;">agent-sdk</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> call</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> list_buddies</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> \\</span></span>
|
|
3
3
|
<span class="line"><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> --dir</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> examples/approval-buddy</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> \\</span></span>
|
|
4
4
|
<span class="line"><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> --input</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> '{}'</span></span></code></pre></div><p>Set a known merged PR, then run the read-only precheck:</p><div class="language-bash vp-adaptive-theme"><button title="Copy Code" class="copy"></button><span class="lang">bash</span><pre class="shiki shiki-themes github-light github-dark vp-code" tabindex="0"><code><span class="line"><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">MERGED_PR_URL</span><span style="--shiki-light:#D73A49;--shiki-dark:#F97583;">=</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;">https://github.com/your-org/your-repo/pull/123</span></span>
|
|
5
|
-
<span class="line"><span style="--shiki-light:#6F42C1;--shiki-dark:#B392F0;">
|
|
5
|
+
<span class="line"><span style="--shiki-light:#6F42C1;--shiki-dark:#B392F0;">agent-sdk</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> call</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> prepare_review</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> \\</span></span>
|
|
6
6
|
<span class="line"><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> --dir</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> examples/approval-buddy</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> \\</span></span>
|
|
7
|
-
<span class="line"><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> --input</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> "{</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;">\\"</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;">prUrl</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;">\\"</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;">:</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;">\\"</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">$MERGED_PR_URL</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;">\\"</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;">}"</span></span></code></pre></div><p>The result should decline because the PR is no longer open. <code>prepare_review</code> never posts an approval.</p><div class="caution custom-block github-alert"><p class="custom-block-title">CAUTION</p><p>Don't use <code>
|
|
7
|
+
<span class="line"><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> --input</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> "{</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;">\\"</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;">prUrl</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;">\\"</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;">:</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;">\\"</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">$MERGED_PR_URL</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;">\\"</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;">}"</span></span></code></pre></div><p>The result should decline because the PR is no longer open. <code>prepare_review</code> never posts an approval.</p><div class="caution custom-block github-alert"><p class="custom-block-title">CAUTION</p><p>Don't use <code>agent-sdk call approve_pr</code> as a smoke test. The tool has no <code>needsApproval</code> gate and posts a real GitHub review when the PR qualifies.</p></div><h2 id="see-why-preparation-is-separate" tabindex="-1">See why preparation is separate <a class="header-anchor" href="#see-why-preparation-is-separate" aria-label="Permalink to "See why preparation is separate""></a></h2><p><code>prepare_review</code> is read-only. It checks policy before fetching a large diff, so declined requests don't spend review-agent work.</p><p>Direct calls return the evidence file map because their scratch workspace is deleted after the call. In-session calls write the tree to <code>ctx.workspaceDir</code>, where both subagents can read it.</p><p><code>approve_pr</code> repeats the live check instead of trusting preparation. A PR can close, merge, become a draft, or change author-related context between the two steps. Revalidation keeps the final write bound to current state.</p><p>This is a reusable two-tool pattern:</p><ul><li>a read-only tool prepares and explains the decision,</li><li>a mutating tool repeats policy at the side-effect boundary.</li></ul><h2 id="fan-out-two-review-contracts" tabindex="-1">Fan out two review contracts <a class="header-anchor" href="#fan-out-two-review-contracts" aria-label="Permalink to "Fan out two review contracts""></a></h2><p>The two discovered subagents have different contracts:</p><ul><li>The security reviewer reports bugs, breaking changes, and security findings with <code>High</code>, <code>Medium</code>, or <code>Low</code> tags.</li><li>The code-quality reviewer reports maintainability and structure concerns with <code>Blocker</code>, <code>Major</code>, or <code>Minor</code> tags.</li></ul><p>The parent calls both through the harness <code>task</code> tool. They inherit the root agent's execution surface and read the same <code>pr/</code> workspace. The prompt asks the parent not to rewrite either reply. The review body trims the combined text and caps it at 16,000 characters.</p><p>Findings are informational. A high-severity finding doesn't veto the stamp. That policy is explicit in the root instructions and approval code.</p><h2 id="trace-github-channel-behavior" tabindex="-1">Trace GitHub channel behavior <a class="header-anchor" href="#trace-github-channel-behavior" aria-label="Permalink to "Trace GitHub channel behavior""></a></h2><p>The channel uses <code>githubChannel</code> with:</p><ul><li>a configured repository allowlist on the account-linked GitHub transport,</li><li>a second optional <code>APPROVAL_BUDDY_REPOS</code> wake filter,</li><li><code>deliverReplies: false</code>,</li><li>progress reactions disabled, and</li><li>event handlers for turn start, <code>approve_pr</code> results, and failed turns.</li></ul><p>The source requests <code>contents-write</code>, even though the documented workflow posts reviews, statuses, and comments. When adapting the example, start with <code>pr-write</code> and opt up only if a tool must push code.</p><p>Every terminal status is green by design. Declines and crashed turns are informational, not merge-blocking. This is a product decision in the example, not an Agent SDK default.</p><p>A successful turn that never calls <code>approve_pr</code> leaves the pending status in place. The channel clears it on <code>approve_pr</code> results and <code>turn.failed</code>, but has no <code>turn.completed</code> fallback.</p><p><code>github replay</code> reaches the same channel and can post a real approval, status, or comment. Use replay only against a repository and PR created for this test.</p><h2 id="use-slack-for-explicit-requests" tabindex="-1">Use Slack for explicit requests <a class="header-anchor" href="#use-slack-for-explicit-requests" aria-label="Permalink to "Use Slack for explicit requests""></a></h2><p>Start the dev server:</p><div class="language-bash vp-adaptive-theme"><button title="Copy Code" class="copy"></button><span class="lang">bash</span><pre class="shiki shiki-themes github-light github-dark vp-code" tabindex="0"><code><span class="line"><span style="--shiki-light:#6F42C1;--shiki-dark:#B392F0;">agent-sdk</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> dev</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> examples/approval-buddy</span></span></code></pre></div><p>Then ask through the signed-in account-linked Slack connection:</p><blockquote><p>Would this PR qualify for a stamp?</p></blockquote><p>The instructions route qualification questions to <code>prepare_review</code> only. A stamp request runs the complete flow and may approve the PR.</p><p>This channel uses the account-linked transport instead of a dedicated Socket Mode app.</p><h2 id="see-how-durable-storage-fits" tabindex="-1">See how durable storage fits <a class="header-anchor" href="#see-how-durable-storage-fits" aria-label="Permalink to "See how durable storage fits""></a></h2><p><code>defineStorage</code> replaces the default local session store with a shared, durable key-value adapter. Approval Buddy chooses:</p><ul><li>a 15-second write debounce,</li><li>startup restoration for up to 200 sessions, and</li><li>a 14-day restore window.</li></ul><p>That policy fits long-lived Slack threads and a small webhook fleet. The security reviewer uses the same adapter with lazy restore, which fits its shorter sessions.</p><h2 id="run-the-regression-suite" tabindex="-1">Run the regression suite <a class="header-anchor" href="#run-the-regression-suite" aria-label="Permalink to "Run the regression suite""></a></h2><p>List the four eval cases:</p><div class="language-bash vp-adaptive-theme"><button title="Copy Code" class="copy"></button><span class="lang">bash</span><pre class="shiki shiki-themes github-light github-dark vp-code" tabindex="0"><code><span class="line"><span style="--shiki-light:#6F42C1;--shiki-dark:#B392F0;">agent-sdk</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> eval</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> --dir</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> examples/approval-buddy</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> --list</span></span></code></pre></div><p>The suite covers:</p><ul><li>buddy-list routing,</li><li>declining a merged PR,</li><li>using only <code>prepare_review</code> for a qualification question, and</li><li>the combined findings headings and severity format over seeded evidence.</li></ul><p>Run the safe qualification case:</p><div class="language-bash vp-adaptive-theme"><button title="Copy Code" class="copy"></button><span class="lang">bash</span><pre class="shiki shiki-themes github-light github-dark vp-code" tabindex="0"><code><span class="line"><span style="--shiki-light:#6F42C1;--shiki-dark:#B392F0;">agent-sdk</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> eval</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> \\</span></span>
|
|
8
8
|
<span class="line"><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> --dir</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> examples/approval-buddy</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> \\</span></span>
|
|
9
9
|
<span class="line"><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> qualify/merged-pr-question</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> \\</span></span>
|
|
10
10
|
<span class="line"><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> --json</span></span></code></pre></div><p>The qualification case reads a live merged PR. The seeded format case shown by <code>--list</code> uses a planted auth-bypass diff and checks for a <code>task</code> call, both headings, and severity tags. It doesn't prove both named subagents ran or whether their output reached <code>approve_pr</code>. Unit tests under <code>agent/lib/</code> cover policy, self-approval handling, evidence limits, status mapping, GIF selection, and severity parsing.</p><h2 id="reuse-the-policy-boundary" tabindex="-1">Reuse the policy boundary <a class="header-anchor" href="#reuse-the-policy-boundary" aria-label="Permalink to "Reuse the policy boundary""></a></h2><p>Keep these properties when you replace the buddy policy:</p><ol><li>Put authorization in typed code.</li><li>Fetch the source of truth inside both prepare and mutate steps.</li><li>Give the model evidence only after the request qualifies.</li><li>Treat specialist findings as data, not authority.</li><li>Keep the core domain mutation in one named tool. Treat channel status and visibility writes as separate, audited effects.</li><li>Add a human approval gate if your policy still needs operator consent.</li><li>Test read-only routing separately from mutation.</li></ol><h2 id="where-to-go-next" tabindex="-1">Where to go next <a class="header-anchor" href="#where-to-go-next" aria-label="Permalink to "Where to go next""></a></h2><ul><li><a href="./../guides/github.html">GitHub</a></li><li><a href="./../reference/tools.html">Tools</a></li><li><a href="./../reference/subagents.html">Subagents</a></li><li><a href="./../storage.html">Storage</a></li><li><a href="./../guides/slack.html">Slack</a></li><li><a href="./../evals.html">Evals</a></li></ul>`,72)])])}const k=a(o,[["render",r]]);export{u as __pageData,k as default};
|
|
@@ -1 +1 @@
|
|
|
1
|
-
import{_ as a,c as t,o as
|
|
1
|
+
import{_ as a,c as t,o as s,ag as i}from"./chunks/framework.CAZyNGu9.js";const u=JSON.parse('{"title":"Keep PR approval policy deterministic with Approval Buddy","description":"Separate code-owned eligibility from model-owned review, then connect GitHub, Slack, subagents, durable storage, and evals.","frontmatter":{"title":"Keep PR approval policy deterministic with Approval Buddy","description":"Separate code-owned eligibility from model-owned review, then connect GitHub, Slack, subagents, durable storage, and evals."},"headers":[],"relativePath":"example-agents/approval-buddy.md","filePath":"example-agents/approval-buddy.md"}'),o={name:"example-agents/approval-buddy.md"};function r(l,e,n,d,p,h){return s(),t("div",null,[...e[0]||(e[0]=[i("",72)])])}const k=a(o,[["render",r]]);export{u as __pageData,k as default};
|
|
@@ -0,0 +1,7 @@
|
|
|
1
|
+
import{_ as a,c as t,o as s,ag as i}from"./chunks/framework.CAZyNGu9.js";const k=JSON.parse('{"title":"Route Slack work through repository playbooks","description":"Combine account-linked chat, allowlisted Socket Mode channel watching, inherited repository skills, and a custom local workspace.","frontmatter":{"title":"Route Slack work through repository playbooks","description":"Combine account-linked chat, allowlisted Socket Mode channel watching, inherited repository skills, and a custom local workspace."},"headers":[],"relativePath":"example-agents/benny.md","filePath":"example-agents/benny.md"}'),o={name:"example-agents/benny.md"};function n(l,e,r,h,d,p){return s(),t("div",null,[...e[0]||(e[0]=[i(`<h1 id="route-slack-work-through-repository-playbooks" tabindex="-1">Route Slack work through repository playbooks <a class="header-anchor" href="#route-slack-work-through-repository-playbooks" aria-label="Permalink to "Route Slack work through repository playbooks""></a></h1><p>This agent is a Slack teammate for a product team. Mentions and direct messages reach it through an account-linked transport. New top-level posts in an allowlisted issue channel reach it through a dedicated Slack app, even without a mention. The agent then selects a repository playbook for triage, reproduction, fixes, reviews, on-call work, or design critique.</p><p>Use this example when Slack is the intake surface and your durable procedures already live as repository skills.</p><p><a href="./../../examples/benny/">Browse the current playbook-router source.</a></p><h2 id="combine-two-slack-transports-with-repo-skills" tabindex="-1">Combine two Slack transports with repo skills <a class="header-anchor" href="#combine-two-slack-transports-with-repo-skills" aria-label="Permalink to "Combine two Slack transports with repo skills""></a></h2><p>The playbook router uniquely combines three decisions:</p><ul><li>Two Slack transports serve different engagement modes.</li><li><code>local.cwd</code> keeps session workspaces inside the monorepo.</li><li>Instructions route work to inherited repository playbooks instead of authored <code>agent/skills/</code>.</li></ul><p>The result is a thin agent project over a mature procedure library.</p><h2 id="follow-an-issue-report" tabindex="-1">Follow an issue report <a class="header-anchor" href="#follow-an-issue-report" aria-label="Permalink to "Follow an issue report""></a></h2><ol><li>A teammate creates a top-level post in the allowlisted issue channel.</li><li>The dedicated Socket Mode channel accepts the allowlisted channel.</li><li>A 15-second debounce lets edits settle. Deleting the post during that window cancels the dispatch.</li><li>The Agent SDK creates a thread-scoped session and sends the report to the playbook router.</li><li>The instructions select the matching triage playbook.</li><li>The harness finds the repository root, opens the inherited playbook, and follows its procedure.</li><li>The agent posts only in the source thread and reports the evidence it gathered.</li></ol><p>Mentions and direct messages follow the same agent instructions. They don't need the watched-channel path.</p><h2 id="map-the-playbook-router-files" tabindex="-1">Map the playbook router files <a class="header-anchor" href="#map-the-playbook-router-files" aria-label="Permalink to "Map the playbook router files""></a></h2><table tabindex="0"><thead><tr><th>File</th><th>Purpose</th></tr></thead><tbody><tr><td><a href="../../examples/benny/agent/agent.ts"><code>agent/agent.ts</code></a></td><td>Names the agent, selects its model, and keeps the harness under <code>.agent-serve/harness</code>.</td></tr><tr><td><a href="./../../examples/benny/agent/instructions.html"><code>agent/instructions.md</code></a></td><td>Defines engagement rules, evidence policy, and the playbook routing map.</td></tr><tr><td><a href="../../examples/benny/agent/channels/slack.ts"><code>agent/channels/slack.ts</code></a></td><td>Handles account-linked mentions and direct messages.</td></tr><tr><td><a href="../../examples/benny/agent/channels/slack-app.ts"><code>agent/channels/slack-app.ts</code></a></td><td>Runs the dedicated app and watches one allowlisted channel.</td></tr><tr><td><a href="../../examples/benny/evals/smoke.eval.ts"><code>evals/smoke.eval.ts</code></a></td><td>Checks the agent identity and expected triage route.</td></tr></tbody></table><p>The playbook router authors no tools, MCP connections, subagents, schedules, hooks, A/B experiments, or sandbox seeds.</p><h2 id="see-why-local-cwd-matters" tabindex="-1">See why <code>local.cwd</code> matters <a class="header-anchor" href="#see-why-local-cwd-matters" aria-label="Permalink to "See why \`local.cwd\` matters""></a></h2><p>The Agent SDK normally keeps an ephemeral <code>run</code> or <code>eval</code> workspace outside a large monorepo. This prevents ancestor instruction and repository-rule files from leaking into an unrelated agent.</p><p>The playbook router needs the opposite. Its procedures live at the repository root, so <code>agent.ts</code> sets:</p><div class="language-ts vp-adaptive-theme"><button title="Copy Code" class="copy"></button><span class="lang">ts</span><pre class="shiki shiki-themes github-light github-dark vp-code" tabindex="0"><code><span class="line"><span style="--shiki-light:#6F42C1;--shiki-dark:#B392F0;">local</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">: {</span></span>
|
|
2
|
+
<span class="line"><span style="--shiki-light:#6F42C1;--shiki-dark:#B392F0;"> cwd</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">: </span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;">".agent-serve/harness"</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">,</span></span>
|
|
3
|
+
<span class="line"><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">}</span></span></code></pre></div><p>Each harness workspace lands under <code>examples/benny/.agent-serve/harness/<sessionId></code>. Walking up the directory tree reaches the host repository and its inherited playbook directory.</p><p>Those playbooks are inherited context. <code>agent-sdk info</code> reports zero authored skills for the agent. Copying this project into another repository removes its main procedures unless you copy or replace the skill library too.</p><h2 id="connect-both-slack-paths" tabindex="-1">Connect both Slack paths <a class="header-anchor" href="#connect-both-slack-paths" aria-label="Permalink to "Connect both Slack paths""></a></h2><p>The account-linked path needs an agent-runtime login and a connected Slack account:</p><div class="language-bash vp-adaptive-theme"><button title="Copy Code" class="copy"></button><span class="lang">bash</span><pre class="shiki shiki-themes github-light github-dark vp-code" tabindex="0"><code><span class="line"><span style="--shiki-light:#6F42C1;--shiki-dark:#B392F0;">agent-sdk</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> login</span></span>
|
|
4
|
+
<span class="line"><span style="--shiki-light:#6F42C1;--shiki-dark:#B392F0;">agent-sdk</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> whoami</span></span></code></pre></div><p>It routes explicit mentions without a dedicated Slack token on the host.</p><p>For the watched-channel path, configure a dedicated Socket Mode app with:</p><ul><li>subscribe to <code>message.channels</code> and <code>message.groups</code>,</li><li>have an App-Level Token with <code>connections:write</code>, and</li><li>be a member of the watched channel.</li></ul><p>Run <code>agent-sdk slack setup</code> for the guided app workflow. Generate the project manifest with <code>--channel-posts</code> when you create a new copy, then validate the configured channel prefix with <code>agent-sdk slack doctor</code>.</p><p>Missing dedicated-app tokens leave that channel idle. They don't stop the account-linked channel.</p><h2 id="validate-and-start-the-server" tabindex="-1">Validate and start the server <a class="header-anchor" href="#validate-and-start-the-server" aria-label="Permalink to "Validate and start the server""></a></h2><div class="language-bash vp-adaptive-theme"><button title="Copy Code" class="copy"></button><span class="lang">bash</span><pre class="shiki shiki-themes github-light github-dark vp-code" tabindex="0"><code><span class="line"><span style="--shiki-light:#6F42C1;--shiki-dark:#B392F0;">agent-sdk</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> validate</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> --dir</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> examples/benny</span></span>
|
|
5
|
+
<span class="line"><span style="--shiki-light:#6F42C1;--shiki-dark:#B392F0;">agent-sdk</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> info</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> --dir</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> examples/benny</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> --json</span></span>
|
|
6
|
+
<span class="line"><span style="--shiki-light:#6F42C1;--shiki-dark:#B392F0;">agent-sdk</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> dev</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> examples/benny</span></span></code></pre></div><p>The info output should show two Slack channels and no authored skill. That combination confirms the example is using inherited playbooks.</p><h2 id="exercise-each-engagement-mode" tabindex="-1">Exercise each engagement mode <a class="header-anchor" href="#exercise-each-engagement-mode" aria-label="Permalink to "Exercise each engagement mode""></a></h2><p>Test the explicit account-linked path by asking:</p><blockquote><p>Which playbook would you use to triage a product UI bug?</p></blockquote><p>Test the dedicated app:</p><ol><li>Create a top-level post in the allowlisted issue channel.</li><li>Don't mention the bot.</li><li>Wait for the debounce window.</li><li>Confirm the agent replies in the post's thread.</li></ol><p>Thread replies don't trigger the proactive watch. Mentions still use Slack's normal mention path. Bot-authored posts are ignored to prevent loops.</p><p>The channel uses the default handler after filtering. It doesn't apply a second code-level classifier, so every accepted top-level post spends a model turn and reaches the prompt.</p><h2 id="inspect-thread-continuity" tabindex="-1">Inspect thread continuity <a class="header-anchor" href="#inspect-thread-continuity" aria-label="Permalink to "Inspect thread continuity""></a></h2><p>The Agent SDK keys Slack sessions by channel and thread timestamp. A follow-up in the same thread resumes the conversation and workspace. A new top-level issue gets a new session.</p><p>This lets a playbook gather evidence over several turns without mixing two reports. The playground shows both the account-linked and dedicated-app sessions while the dev server runs.</p><h2 id="run-the-smoke-eval" tabindex="-1">Run the smoke eval <a class="header-anchor" href="#run-the-smoke-eval" aria-label="Permalink to "Run the smoke eval""></a></h2><div class="language-bash vp-adaptive-theme"><button title="Copy Code" class="copy"></button><span class="lang">bash</span><pre class="shiki shiki-themes github-light github-dark vp-code" tabindex="0"><code><span class="line"><span style="--shiki-light:#6F42C1;--shiki-dark:#B392F0;">agent-sdk</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> eval</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> --dir</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> examples/benny</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> --list</span></span>
|
|
7
|
+
<span class="line"><span style="--shiki-light:#6F42C1;--shiki-dark:#B392F0;">agent-sdk</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> eval</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> --dir</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> examples/benny</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> smoke</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> --json</span></span></code></pre></div><p>The case asks for the agent identity and the playbook used for issue triage. It checks the configured identity and route label.</p><p>This is a lexical smoke test. It doesn't prove Slack delivery, skill selection, skill loading, procedure execution, or thread-only behavior. Add fixture-backed evals around the playbooks when you reuse this design.</p><h2 id="build-a-playbook-routed-teammate" tabindex="-1">Build a playbook-routed teammate <a class="header-anchor" href="#build-a-playbook-routed-teammate" aria-label="Permalink to "Build a playbook-routed teammate""></a></h2><p>Use this structure when your organization already has tested skills:</p><ol><li>Put the playbooks under a stable repository path.</li><li>Set <code>local.cwd</code> so harness workspaces can inherit that path.</li><li>Write a short routing table in <code>instructions.md</code>.</li><li>Use account-linked Slack for explicit requests.</li><li>Add a dedicated app only for allowlisted proactive intake.</li><li>Keep the channel allowlist narrow and debounce edited posts.</li><li>Add an eval for every important request-to-playbook route.</li></ol><p>If the procedures should ship with the agent, put them under <code>agent/skills/</code> instead. Authored skills appear in the manifest and travel with the project.</p><h2 id="where-to-go-next" tabindex="-1">Where to go next <a class="header-anchor" href="#where-to-go-next" aria-label="Permalink to "Where to go next""></a></h2><ul><li><a href="./../guides/slack.html">Slack</a></li><li><a href="./../reference/agent-config.html">Agent config</a></li><li><a href="./../reference/skills.html">Skills</a></li><li><a href="./../reference/sessions.html">Sessions and streaming</a></li><li><a href="./../evals.html">Evals</a></li></ul>`,51)])])}const u=a(o,[["render",n]]);export{k as __pageData,u as default};
|
|
@@ -1,11 +1,11 @@
|
|
|
1
|
-
import{_ as a,c as t,o as s,ag as i}from"./chunks/framework.CAZyNGu9.js";const u=JSON.parse('{"title":"Review prepared pull-request evidence","description":"Fetch a PR on the host, seed a trimmed diff-first workspace, and run a GitHub-read-only review through HTTP, Slack, or an eval.","frontmatter":{"title":"Review prepared pull-request evidence","description":"Fetch a PR on the host, seed a trimmed diff-first workspace, and run a GitHub-read-only review through HTTP, Slack, or an eval."},"headers":[],"relativePath":"example-agents/bugbot.md","filePath":"example-agents/bugbot.md"}'),r={name:"example-agents/bugbot.md"};function n(l,e,o,h,d,p){return s(),t("div",null,[...e[0]||(e[0]=[i(`<h1 id="review-prepared-pull-request-evidence" tabindex="-1">Review prepared pull-request evidence <a class="header-anchor" href="#review-prepared-pull-request-evidence" aria-label="Permalink to "Review prepared pull-request evidence""></a></h1><p>This GitHub-read-only reviewer uses host code to fetch the PR with <code>gh</code> and <code>git</code>, builds a trimmed <code>pr/</code> evidence tree, then hands that tree to the model. The model reads the diff, loads a review skill, and returns at most three high-confidence findings.</p><p>Use this example when the host should control evidence collection and the model shouldn't browse or mutate the source repository.</p><p><a href="./../../examples/bugbot/">Browse the current reviewer source.</a></p><h2 id="separate-evidence-preparation-from-review" tabindex="-1">Separate evidence preparation from review <a class="header-anchor" href="#separate-evidence-preparation-from-review" aria-label="Permalink to "Separate evidence preparation from review""></a></h2><p>The reviewer separates preparation from judgment:</p><ul><li>Host code owns GitHub and Git access.</li><li>A server tool turns untrusted PR input into bounded workspace files.</li><li>A custom channel seeds those files before the model starts.</li><li>An on-demand skill defines the review procedure and output contract.</li><li>The model returns chat text. No path posts a GitHub review.</li></ul><p>This architecture gives the model a purpose-built evidence package instead of a checkout.</p><h2 id="follow-a-review" tabindex="-1">Follow a review <a class="header-anchor" href="#follow-a-review" aria-label="Permalink to "Follow a review""></a></h2><p>The custom HTTP path runs this sequence:</p><ol><li><code>POST /v1/channels/review/</code> receives a PR reference.</li><li>The handler calls <code>prepare_pr</code> without a model turn.</li><li>Host code reads PR metadata and the unified diff.</li><li>It reuses a matching checkout, force-fetching the PR ref there when the commit is missing. Without a matching checkout, it uses a temporary bare cache.</li><li>It creates <code>pr/MANIFEST.md</code>, <code>pr/meta.json</code>, <code>pr/diff.patch</code>, and selected small files and rules.</li><li><code>send({ workspaceFiles })</code> creates the model session with that evidence.</li><li>The model reads the manifest and diff, then loads <code>pr-review</code>.</li><li>The channel returns session and playground URLs while the review streams.</li></ol><p>If a normal chat starts without evidence, the model can call <code>prepare_pr</code> mid-turn. That form writes the same files into the active session workspace.</p><h2 id="map-the-evidence-review-files" tabindex="-1">Map the evidence-review files <a class="header-anchor" href="#map-the-evidence-review-files" aria-label="Permalink to "Map the evidence-review files""></a></h2><table tabindex="0"><thead><tr><th>File</th><th>Purpose</th></tr></thead><tbody><tr><td><a href="../../examples/bugbot/agent/agent.ts"><code>agent/agent.ts</code></a></td><td>Selects the local runtime and model.</td></tr><tr><td><a href="./../../examples/bugbot/agent/instructions.html"><code>agent/instructions.md</code></a></td><td>Requires diff-first review and confines model work to <code>pr/</code>.</td></tr><tr><td><a href="../../examples/bugbot/agent/tools/prepare_pr.ts"><code>agent/tools/prepare_pr.ts</code></a></td><td>Exposes host preparation as a typed server tool.</td></tr><tr><td><a href="../../examples/bugbot/agent/lib/prepare-pr.ts"><code>agent/lib/prepare-pr.ts</code></a></td><td>Parses PR references, runs <code>gh</code> and <code>git</code>, and builds the evidence map.</td></tr><tr><td><a href="../../examples/bugbot/agent/channels/review.ts"><code>agent/channels/review.ts</code></a></td><td>Provides the loopback-only prepare-and-send HTTP route.</td></tr><tr><td><a href="../../examples/bugbot/agent/channels/slack.ts"><code>agent/channels/slack.ts</code></a></td><td>Extracts PR references and prepares evidence for mentions and direct messages.</td></tr><tr><td><a href="./../../examples/bugbot/agent/skills/pr-review.html"><code>agent/skills/pr-review.md</code></a></td><td>Sets finding limits, severities, and the machine-readable review format.</td></tr><tr><td><a href="../../examples/bugbot/evals/review/smoke.eval.ts"><code>evals/review/smoke.eval.ts</code></a></td><td>Seeds fake evidence and checks the review path without GitHub.</td></tr></tbody></table><p>There is no authored GitHub channel, MCP connection, subagent, schedule, hook, A/B experiment, approval, or custom storage.</p><h2 id="prepare-the-host" tabindex="-1">Prepare the host <a class="header-anchor" href="#prepare-the-host" aria-label="Permalink to "Prepare the host""></a></h2><p>You need:</p><ul><li>Node 22.13 or newer.</li><li>An agent-runtime credential for model turns and account-linked Slack.</li><li><code>gh</code> and <code>git</code> on <code>PATH</code>.</li><li><code>gh</code> access to the target PR.</li><li>Network access to GitHub and a writable temporary directory.</li></ul><p>The preparer can prefer a configured local checkout. Its <code>origin</code> must match the target repository. Otherwise the reviewer uses its bare cache. It never checks out the PR into the serve host's working tree.</p><h2 id="validate-the-surface" tabindex="-1">Validate the surface <a class="header-anchor" href="#validate-the-surface" aria-label="Permalink to "Validate the surface""></a></h2><div class="language-bash vp-adaptive-theme"><button title="Copy Code" class="copy"></button><span class="lang">bash</span><pre class="shiki shiki-themes github-light github-dark vp-code" tabindex="0"><code><span class="line"><span style="--shiki-light:#6F42C1;--shiki-dark:#B392F0;">
|
|
2
|
-
<span class="line"><span style="--shiki-light:#6F42C1;--shiki-dark:#B392F0;">
|
|
1
|
+
import{_ as a,c as t,o as s,ag as i}from"./chunks/framework.CAZyNGu9.js";const u=JSON.parse('{"title":"Review prepared pull-request evidence","description":"Fetch a PR on the host, seed a trimmed diff-first workspace, and run a GitHub-read-only review through HTTP, Slack, or an eval.","frontmatter":{"title":"Review prepared pull-request evidence","description":"Fetch a PR on the host, seed a trimmed diff-first workspace, and run a GitHub-read-only review through HTTP, Slack, or an eval."},"headers":[],"relativePath":"example-agents/bugbot.md","filePath":"example-agents/bugbot.md"}'),r={name:"example-agents/bugbot.md"};function n(l,e,o,h,d,p){return s(),t("div",null,[...e[0]||(e[0]=[i(`<h1 id="review-prepared-pull-request-evidence" tabindex="-1">Review prepared pull-request evidence <a class="header-anchor" href="#review-prepared-pull-request-evidence" aria-label="Permalink to "Review prepared pull-request evidence""></a></h1><p>This GitHub-read-only reviewer uses host code to fetch the PR with <code>gh</code> and <code>git</code>, builds a trimmed <code>pr/</code> evidence tree, then hands that tree to the model. The model reads the diff, loads a review skill, and returns at most three high-confidence findings.</p><p>Use this example when the host should control evidence collection and the model shouldn't browse or mutate the source repository.</p><p><a href="./../../examples/bugbot/">Browse the current reviewer source.</a></p><h2 id="separate-evidence-preparation-from-review" tabindex="-1">Separate evidence preparation from review <a class="header-anchor" href="#separate-evidence-preparation-from-review" aria-label="Permalink to "Separate evidence preparation from review""></a></h2><p>The reviewer separates preparation from judgment:</p><ul><li>Host code owns GitHub and Git access.</li><li>A server tool turns untrusted PR input into bounded workspace files.</li><li>A custom channel seeds those files before the model starts.</li><li>An on-demand skill defines the review procedure and output contract.</li><li>The model returns chat text. No path posts a GitHub review.</li></ul><p>This architecture gives the model a purpose-built evidence package instead of a checkout.</p><h2 id="follow-a-review" tabindex="-1">Follow a review <a class="header-anchor" href="#follow-a-review" aria-label="Permalink to "Follow a review""></a></h2><p>The custom HTTP path runs this sequence:</p><ol><li><code>POST /v1/channels/review/</code> receives a PR reference.</li><li>The handler calls <code>prepare_pr</code> without a model turn.</li><li>Host code reads PR metadata and the unified diff.</li><li>It reuses a matching checkout, force-fetching the PR ref there when the commit is missing. Without a matching checkout, it uses a temporary bare cache.</li><li>It creates <code>pr/MANIFEST.md</code>, <code>pr/meta.json</code>, <code>pr/diff.patch</code>, and selected small files and rules.</li><li><code>send({ workspaceFiles })</code> creates the model session with that evidence.</li><li>The model reads the manifest and diff, then loads <code>pr-review</code>.</li><li>The channel returns session and playground URLs while the review streams.</li></ol><p>If a normal chat starts without evidence, the model can call <code>prepare_pr</code> mid-turn. That form writes the same files into the active session workspace.</p><h2 id="map-the-evidence-review-files" tabindex="-1">Map the evidence-review files <a class="header-anchor" href="#map-the-evidence-review-files" aria-label="Permalink to "Map the evidence-review files""></a></h2><table tabindex="0"><thead><tr><th>File</th><th>Purpose</th></tr></thead><tbody><tr><td><a href="../../examples/bugbot/agent/agent.ts"><code>agent/agent.ts</code></a></td><td>Selects the local runtime and model.</td></tr><tr><td><a href="./../../examples/bugbot/agent/instructions.html"><code>agent/instructions.md</code></a></td><td>Requires diff-first review and confines model work to <code>pr/</code>.</td></tr><tr><td><a href="../../examples/bugbot/agent/tools/prepare_pr.ts"><code>agent/tools/prepare_pr.ts</code></a></td><td>Exposes host preparation as a typed server tool.</td></tr><tr><td><a href="../../examples/bugbot/agent/lib/prepare-pr.ts"><code>agent/lib/prepare-pr.ts</code></a></td><td>Parses PR references, runs <code>gh</code> and <code>git</code>, and builds the evidence map.</td></tr><tr><td><a href="../../examples/bugbot/agent/channels/review.ts"><code>agent/channels/review.ts</code></a></td><td>Provides the loopback-only prepare-and-send HTTP route.</td></tr><tr><td><a href="../../examples/bugbot/agent/channels/slack.ts"><code>agent/channels/slack.ts</code></a></td><td>Extracts PR references and prepares evidence for mentions and direct messages.</td></tr><tr><td><a href="./../../examples/bugbot/agent/skills/pr-review.html"><code>agent/skills/pr-review.md</code></a></td><td>Sets finding limits, severities, and the machine-readable review format.</td></tr><tr><td><a href="../../examples/bugbot/evals/review/smoke.eval.ts"><code>evals/review/smoke.eval.ts</code></a></td><td>Seeds fake evidence and checks the review path without GitHub.</td></tr></tbody></table><p>There is no authored GitHub channel, MCP connection, subagent, schedule, hook, A/B experiment, approval, or custom storage.</p><h2 id="prepare-the-host" tabindex="-1">Prepare the host <a class="header-anchor" href="#prepare-the-host" aria-label="Permalink to "Prepare the host""></a></h2><p>You need:</p><ul><li>Node 22.13 or newer.</li><li>An agent-runtime credential for model turns and account-linked Slack.</li><li><code>gh</code> and <code>git</code> on <code>PATH</code>.</li><li><code>gh</code> access to the target PR.</li><li>Network access to GitHub and a writable temporary directory.</li></ul><p>The preparer can prefer a configured local checkout. Its <code>origin</code> must match the target repository. Otherwise the reviewer uses its bare cache. It never checks out the PR into the serve host's working tree.</p><h2 id="validate-the-surface" tabindex="-1">Validate the surface <a class="header-anchor" href="#validate-the-surface" aria-label="Permalink to "Validate the surface""></a></h2><div class="language-bash vp-adaptive-theme"><button title="Copy Code" class="copy"></button><span class="lang">bash</span><pre class="shiki shiki-themes github-light github-dark vp-code" tabindex="0"><code><span class="line"><span style="--shiki-light:#6F42C1;--shiki-dark:#B392F0;">agent-sdk</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> validate</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> --dir</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> examples/bugbot</span></span>
|
|
2
|
+
<span class="line"><span style="--shiki-light:#6F42C1;--shiki-dark:#B392F0;">agent-sdk</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> info</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> --dir</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> examples/bugbot</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> --json</span></span></code></pre></div><p>The manifest should show one server tool, one skill, and two authored channels.</p><h2 id="inspect-evidence-without-a-model-turn" tabindex="-1">Inspect evidence without a model turn <a class="header-anchor" href="#inspect-evidence-without-a-model-turn" aria-label="Permalink to "Inspect evidence without a model turn""></a></h2><p>Call the preparation tool directly:</p><div class="language-bash vp-adaptive-theme"><button title="Copy Code" class="copy"></button><span class="lang">bash</span><pre class="shiki shiki-themes github-light github-dark vp-code" tabindex="0"><code><span class="line"><span style="--shiki-light:#6F42C1;--shiki-dark:#B392F0;">agent-sdk</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> call</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> prepare_pr</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> \\</span></span>
|
|
3
3
|
<span class="line"><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> --dir</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> examples/bugbot</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> \\</span></span>
|
|
4
|
-
<span class="line"><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> --input</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> '{"pr":"https://github.com/owner/repo/pull/123"}'</span></span></code></pre></div><p>Direct tool calls use a scratch workspace removed after the call. <code>prepare_pr</code> detects this path and returns the complete file map in its result. In a model session, it writes the files and returns a smaller summary.</p><p>The evidence builder applies explicit limits:</p><table tabindex="0"><thead><tr><th>Evidence</th><th>Limit</th></tr></thead><tbody><tr><td>Post-change file</td><td>12,000 characters</td></tr><tr><td>One rule file</td><td>8,000 characters</td></tr><tr><td>Combined rules</td><td>12,000 characters</td></tr><tr><td>PR body in metadata</td><td>2,000 characters</td></tr></tbody></table><p>Large files remain visible in <code>diff.patch</code>. The manifest records which full files or rules were omitted.</p><p>The per-file limits aren't an aggregate context cap. Every changed file below 12,000 characters can be included. The diff command has a 12 MiB output buffer; a larger diff fails preparation instead of being truncated.</p><h2 id="run-the-http-review-path" tabindex="-1">Run the HTTP review path <a class="header-anchor" href="#run-the-http-review-path" aria-label="Permalink to "Run the HTTP review path""></a></h2><p>Start the server:</p><div class="language-bash vp-adaptive-theme"><button title="Copy Code" class="copy"></button><span class="lang">bash</span><pre class="shiki shiki-themes github-light github-dark vp-code" tabindex="0"><code><span class="line"><span style="--shiki-light:#6F42C1;--shiki-dark:#B392F0;">
|
|
4
|
+
<span class="line"><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> --input</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> '{"pr":"https://github.com/owner/repo/pull/123"}'</span></span></code></pre></div><p>Direct tool calls use a scratch workspace removed after the call. <code>prepare_pr</code> detects this path and returns the complete file map in its result. In a model session, it writes the files and returns a smaller summary.</p><p>The evidence builder applies explicit limits:</p><table tabindex="0"><thead><tr><th>Evidence</th><th>Limit</th></tr></thead><tbody><tr><td>Post-change file</td><td>12,000 characters</td></tr><tr><td>One rule file</td><td>8,000 characters</td></tr><tr><td>Combined rules</td><td>12,000 characters</td></tr><tr><td>PR body in metadata</td><td>2,000 characters</td></tr></tbody></table><p>Large files remain visible in <code>diff.patch</code>. The manifest records which full files or rules were omitted.</p><p>The per-file limits aren't an aggregate context cap. Every changed file below 12,000 characters can be included. The diff command has a 12 MiB output buffer; a larger diff fails preparation instead of being truncated.</p><h2 id="run-the-http-review-path" tabindex="-1">Run the HTTP review path <a class="header-anchor" href="#run-the-http-review-path" aria-label="Permalink to "Run the HTTP review path""></a></h2><p>Start the server:</p><div class="language-bash vp-adaptive-theme"><button title="Copy Code" class="copy"></button><span class="lang">bash</span><pre class="shiki shiki-themes github-light github-dark vp-code" tabindex="0"><code><span class="line"><span style="--shiki-light:#6F42C1;--shiki-dark:#B392F0;">agent-sdk</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> dev</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> examples/bugbot</span></span></code></pre></div><p>From another terminal:</p><div class="language-bash vp-adaptive-theme"><button title="Copy Code" class="copy"></button><span class="lang">bash</span><pre class="shiki shiki-themes github-light github-dark vp-code" tabindex="0"><code><span class="line"><span style="--shiki-light:#6F42C1;--shiki-dark:#B392F0;">curl</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> -s</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> -X</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> POST</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> \\</span></span>
|
|
5
5
|
<span class="line"><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> http://127.0.0.1:3000/bugbot/v1/channels/review/</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> \\</span></span>
|
|
6
6
|
<span class="line"><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> -H</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> 'content-type: application/json'</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> \\</span></span>
|
|
7
7
|
<span class="line"><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> -d</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> '{"pr":"https://github.com/owner/repo/pull/123"}'</span></span></code></pre></div><p>The route returns <code>status: "started"</code>, a continuation token, and session and playground URLs. Open the session URL to watch the model read the evidence and produce findings.</p><p>The channel declares <code>localDevStrict()</code>. Direct loopback callers can use it. Proxy-forwarding headers and non-loopback hosts are rejected.</p><p>Send a follow-up by passing the returned key:</p><div class="language-bash vp-adaptive-theme"><button title="Copy Code" class="copy"></button><span class="lang">bash</span><pre class="shiki shiki-themes github-light github-dark vp-code" tabindex="0"><code><span class="line"><span style="--shiki-light:#6F42C1;--shiki-dark:#B392F0;">curl</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> -s</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> -X</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> POST</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> \\</span></span>
|
|
8
8
|
<span class="line"><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> http://127.0.0.1:3000/bugbot/v1/channels/review/</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> \\</span></span>
|
|
9
9
|
<span class="line"><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> -H</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> 'content-type: application/json'</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> \\</span></span>
|
|
10
|
-
<span class="line"><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> -d</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> '{"pr":"owner/repo#123","key":"<continuation-token>"}'</span></span></code></pre></div><p>The follow-up resumes the session without fetching a new evidence tree.</p><h2 id="run-the-slack-path" tabindex="-1">Run the Slack path <a class="header-anchor" href="#run-the-slack-path" aria-label="Permalink to "Run the Slack path""></a></h2><p>The account-linked Slack channel handles review-bot mentions and direct messages:</p><blockquote><p>Review <a href="https://github.com/owner/repo/pull/123" target="_blank" rel="noreferrer">https://github.com/owner/repo/pull/123</a></p></blockquote><p>Slack handlers don't receive the channel <code>callTool</code> helper. This example calls the shared <code>preparePrReview</code> host function, then returns <code>workspaceFiles</code> in the Slack message preparation result. The model sees the same evidence and prompt as the HTTP path.</p><p>If a message contains no PR reference, the handler asks for one. Thread follow-ups keep the same session.</p><h2 id="see-how-the-skill-constrains-review" tabindex="-1">See how the skill constrains review <a class="header-anchor" href="#see-how-the-skill-constrains-review" aria-label="Permalink to "See how the skill constrains review""></a></h2><p><code>pr-review.md</code> tells the model to:</p><ul><li>read the manifest and unified diff first,</li><li>open at most one supporting file or rules file when a hunk is ambiguous,</li><li>avoid shell, network, <code>gh</code>, and <code>git</code>,</li><li>report no more than three findings,</li><li>keep each description under 120 words, and</li><li>emit the machine-readable review contract.</li></ul><p>The root instructions set the evidence boundary. The skill holds the reusable review procedure. Keeping those roles separate lets another agent reuse the same skill with different intake channels.</p><h2 id="run-the-fixture-backed-eval" tabindex="-1">Run the fixture-backed eval <a class="header-anchor" href="#run-the-fixture-backed-eval" aria-label="Permalink to "Run the fixture-backed eval""></a></h2><div class="language-bash vp-adaptive-theme"><button title="Copy Code" class="copy"></button><span class="lang">bash</span><pre class="shiki shiki-themes github-light github-dark vp-code" tabindex="0"><code><span class="line"><span style="--shiki-light:#6F42C1;--shiki-dark:#B392F0;">
|
|
11
|
-
<span class="line"><span style="--shiki-light:#6F42C1;--shiki-dark:#B392F0;">
|
|
10
|
+
<span class="line"><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> -d</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> '{"pr":"owner/repo#123","key":"<continuation-token>"}'</span></span></code></pre></div><p>The follow-up resumes the session without fetching a new evidence tree.</p><h2 id="run-the-slack-path" tabindex="-1">Run the Slack path <a class="header-anchor" href="#run-the-slack-path" aria-label="Permalink to "Run the Slack path""></a></h2><p>The account-linked Slack channel handles review-bot mentions and direct messages:</p><blockquote><p>Review <a href="https://github.com/owner/repo/pull/123" target="_blank" rel="noreferrer">https://github.com/owner/repo/pull/123</a></p></blockquote><p>Slack handlers don't receive the channel <code>callTool</code> helper. This example calls the shared <code>preparePrReview</code> host function, then returns <code>workspaceFiles</code> in the Slack message preparation result. The model sees the same evidence and prompt as the HTTP path.</p><p>If a message contains no PR reference, the handler asks for one. Thread follow-ups keep the same session.</p><h2 id="see-how-the-skill-constrains-review" tabindex="-1">See how the skill constrains review <a class="header-anchor" href="#see-how-the-skill-constrains-review" aria-label="Permalink to "See how the skill constrains review""></a></h2><p><code>pr-review.md</code> tells the model to:</p><ul><li>read the manifest and unified diff first,</li><li>open at most one supporting file or rules file when a hunk is ambiguous,</li><li>avoid shell, network, <code>gh</code>, and <code>git</code>,</li><li>report no more than three findings,</li><li>keep each description under 120 words, and</li><li>emit the machine-readable review contract.</li></ul><p>The root instructions set the evidence boundary. The skill holds the reusable review procedure. Keeping those roles separate lets another agent reuse the same skill with different intake channels.</p><h2 id="run-the-fixture-backed-eval" tabindex="-1">Run the fixture-backed eval <a class="header-anchor" href="#run-the-fixture-backed-eval" aria-label="Permalink to "Run the fixture-backed eval""></a></h2><div class="language-bash vp-adaptive-theme"><button title="Copy Code" class="copy"></button><span class="lang">bash</span><pre class="shiki shiki-themes github-light github-dark vp-code" tabindex="0"><code><span class="line"><span style="--shiki-light:#6F42C1;--shiki-dark:#B392F0;">agent-sdk</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> eval</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> --dir</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> examples/bugbot</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> --list</span></span>
|
|
11
|
+
<span class="line"><span style="--shiki-light:#6F42C1;--shiki-dark:#B392F0;">agent-sdk</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> eval</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> --dir</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> examples/bugbot</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> review/smoke</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> --json</span></span></code></pre></div><p>The eval constructs a <code>PreparedPrReview</code>, seeds its file map through <code>workspaceFiles</code>, and checks for at least one read call with no shell call. It doesn't assert which evidence file was read or whether the skill loaded. It accepts either a formatted review or a clean result.</p><p>This case tests review behavior without GitHub credentials or network data. Add fixtures with reachable bugs when you need stricter location and severity checks.</p><h2 id="keep-the-side-effect-boundary-clear" tabindex="-1">Keep the side-effect boundary clear <a class="header-anchor" href="#keep-the-side-effect-boundary-clear" aria-label="Permalink to "Keep the side-effect boundary clear""></a></h2><p>The reviewer makes no remote GitHub writes. It doesn't author a GitHub channel and doesn't call a review API. Host preparation does write session evidence and force-update <code>refs/pull/<N>/head</code> in either its bare cache or a matching local checkout when the commit is missing. Every result ends with a note saying no GitHub review was posted.</p><p>If you add publishing later, keep it in a separate tool. This preserves a read-only preparation and review path safe to run in evals.</p><h2 id="reuse-the-evidence-handoff" tabindex="-1">Reuse the evidence handoff <a class="header-anchor" href="#reuse-the-evidence-handoff" aria-label="Permalink to "Reuse the evidence handoff""></a></h2><p>Use host-prepared workspaces when:</p><ul><li>external APIs should stay off the model's tool surface,</li><li>context needs hard size limits,</li><li>the model should inspect a snapshot instead of a live checkout, or</li><li>several channels need the same preparation.</li></ul><p>Return <code>workspaceFiles</code> from direct host preparation, write into <code>ctx.workspaceDir</code> for mid-turn recovery, and encode the reading order in both the manifest and a skill.</p><h2 id="where-to-go-next" tabindex="-1">Where to go next <a class="header-anchor" href="#where-to-go-next" aria-label="Permalink to "Where to go next""></a></h2><ul><li><a href="./../guides/webhooks.html">Webhooks and custom channels</a></li><li><a href="./../reference/tools.html">Tools</a></li><li><a href="./../reference/skills.html">Skills</a></li><li><a href="./../guides/slack.html">Slack</a></li><li><a href="./../evals.html">Evals</a></li></ul>`,62)])])}const k=a(r,[["render",n]]);export{u as __pageData,k as default};
|