@cursor/july 0.1.34 → 0.1.36

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (208) hide show
  1. package/AGENTS.md +5 -9
  2. package/README.md +5 -9
  3. package/dist/channels/github/github-channel.d.ts +1 -1
  4. package/dist/channels/github/github-channel.js +1 -1
  5. package/dist/channels/github/index.d.ts +1 -1
  6. package/dist/channels/github/index.js +1 -1
  7. package/dist/channels/slack/index.d.ts +1 -1
  8. package/dist/channels/slack/index.js +1 -1
  9. package/dist/channels/slack/init.d.ts.map +1 -1
  10. package/dist/channels/slack/init.js +2 -1
  11. package/dist/docs/404.html +2 -2
  12. package/dist/docs/ab.html +8 -17
  13. package/dist/docs/assets/{ab.md.DAQoJ-up.js → ab.md.6cLOW7--.js} +4 -13
  14. package/dist/docs/assets/{ab.md.DAQoJ-up.lean.js → ab.md.6cLOW7--.lean.js} +1 -1
  15. package/dist/docs/assets/{app.FPupl4SP.js → app.DEcxy4oz.js} +1 -1
  16. package/dist/docs/assets/{building-with-agents.md.CnHqvYDd.js → building-with-agents.md.txrcGU2B.js} +2 -2
  17. package/dist/docs/assets/chunks/@localSearchIndexroot.ByFYcFly.js +1 -0
  18. package/dist/docs/assets/chunks/{VPLocalSearchBox.Cd182Cu0.js → VPLocalSearchBox.n1VOZcy3.js} +1 -1
  19. package/dist/docs/assets/chunks/{theme.BEM3Okcd.js → theme.BaF1MQ9c.js} +2 -2
  20. package/dist/docs/assets/{concepts.md.DFaQEFkA.js → concepts.md.CqOsxbMU.js} +1 -1
  21. package/dist/docs/assets/{deployment.md.9MYBuKM1.js → deployment.md.CuK5SNjN.js} +1 -1
  22. package/dist/docs/assets/{evals.md.BIUoVZ6X.js → evals.md.BQXI3rXy.js} +9 -15
  23. package/dist/docs/assets/{evals.md.BIUoVZ6X.lean.js → evals.md.BQXI3rXy.lean.js} +1 -1
  24. package/dist/docs/assets/{example-agents_approval-buddy.md.BhEfleVx.js → example-agents_approval-buddy.md.CIiZ9coo.js} +1 -1
  25. package/dist/docs/assets/{example-agents_benny.md.2Et1qa8f.js → example-agents_benny.md.l7JTmm8X.js} +1 -1
  26. package/dist/docs/assets/{example-agents_bugbot.md.ByUexi5i.js → example-agents_bugbot.md.Dp5JqHSQ.js} +2 -2
  27. package/dist/docs/assets/{example-agents_bugbot.md.ByUexi5i.lean.js → example-agents_bugbot.md.Dp5JqHSQ.lean.js} +1 -1
  28. package/dist/docs/assets/{example-agents_codebase-wiki.md.B4y-7ZVW.js → example-agents_codebase-wiki.md.D-lteFf0.js} +1 -1
  29. package/dist/docs/assets/{example-agents_codebase-wiki.md.B4y-7ZVW.lean.js → example-agents_codebase-wiki.md.D-lteFf0.lean.js} +1 -1
  30. package/dist/docs/assets/{example-agents_codeowners-review.md.D6ay4nvf.js → example-agents_codeowners-review.md.BU2ZXLf-.js} +1 -1
  31. package/dist/docs/assets/{example-agents_codeowners-review.md.D6ay4nvf.lean.js → example-agents_codeowners-review.md.BU2ZXLf-.lean.js} +1 -1
  32. package/dist/docs/assets/{example-agents_concierge.md.lL8rhYlj.js → example-agents_concierge.md.DA2al_NK.js} +2 -2
  33. package/dist/docs/assets/{example-agents_concierge.md.lL8rhYlj.lean.js → example-agents_concierge.md.DA2al_NK.lean.js} +1 -1
  34. package/dist/docs/assets/{example-agents_fsd.md.DfNKQTHz.js → example-agents_fsd.md.DPz9ezO4.js} +1 -1
  35. package/dist/docs/assets/{example-agents_knowledge-base.md.CzyZ2DCr.js → example-agents_knowledge-base.md.IneynQSR.js} +1 -1
  36. package/dist/docs/assets/{example-agents_oncall.md.wFFXXEyW.js → example-agents_oncall.md.ZE0n6ZFN.js} +1 -1
  37. package/dist/docs/assets/{example-agents_slack-agent.md.DvgvT4nn.js → example-agents_slack-agent.md.06jQXTAI.js} +1 -1
  38. package/dist/docs/assets/{example-agents_weather-agent.md.BADkPqxQ.js → example-agents_weather-agent.md.CrGZ0SqR.js} +3 -3
  39. package/dist/docs/assets/{example-agents_weather-agent.md.BADkPqxQ.lean.js → example-agents_weather-agent.md.CrGZ0SqR.lean.js} +1 -1
  40. package/dist/docs/assets/guides_cloud-runtime.md.V5igN4Sq.js +9 -0
  41. package/dist/docs/assets/guides_cloud-runtime.md.V5igN4Sq.lean.js +1 -0
  42. package/dist/docs/assets/{guides_webhooks.md.DiAwSR42.js → guides_webhooks.md.BERuBSJW.js} +1 -1
  43. package/dist/docs/assets/{hillclimbing.md.D9Y1_bYh.js → hillclimbing.md.yXqdlv2R.js} +1 -1
  44. package/dist/docs/assets/index.md.CmhptOmN.js +24 -0
  45. package/dist/docs/assets/{index.md.CZqbBJPB.lean.js → index.md.CmhptOmN.lean.js} +1 -1
  46. package/dist/docs/assets/{quickstart.md.TnEXYgYW.js → quickstart.md.C_b6ESpD.js} +7 -4
  47. package/dist/docs/assets/{quickstart.md.TnEXYgYW.lean.js → quickstart.md.C_b6ESpD.lean.js} +1 -1
  48. package/dist/docs/assets/{reference_agent-config.md.kuN6-OxK.js → reference_agent-config.md.CRmkoxd6.js} +6 -4
  49. package/dist/docs/assets/{reference_agent-config.md.kuN6-OxK.lean.js → reference_agent-config.md.CRmkoxd6.lean.js} +1 -1
  50. package/dist/docs/assets/reference_artifacts.md.BGG4bZo-.js +19 -0
  51. package/dist/docs/assets/reference_artifacts.md.BGG4bZo-.lean.js +1 -0
  52. package/dist/docs/assets/{reference_channels.md.CDhTRfUz.js → reference_channels.md.BIabFUAI.js} +2 -2
  53. package/dist/docs/assets/{reference_channels.md.CDhTRfUz.lean.js → reference_channels.md.BIabFUAI.lean.js} +1 -1
  54. package/dist/docs/assets/{reference_cli.md.sD-IUWjg.js → reference_cli.md.Byvrg8eu.js} +15 -9
  55. package/dist/docs/assets/{reference_cli.md.sD-IUWjg.lean.js → reference_cli.md.Byvrg8eu.lean.js} +1 -1
  56. package/dist/docs/assets/{reference_hooks.md.DyLVfE1O.js → reference_hooks.md.BGDw4VLm.js} +2 -2
  57. package/dist/docs/assets/{reference_hooks.md.DyLVfE1O.lean.js → reference_hooks.md.BGDw4VLm.lean.js} +1 -1
  58. package/dist/docs/assets/reference_http-api.md.DGrw_wOu.js +11 -0
  59. package/dist/docs/assets/reference_http-api.md.DGrw_wOu.lean.js +1 -0
  60. package/dist/docs/assets/{reference_project-layout.md.D8E6ZmHJ.js → reference_project-layout.md._XdeMahr.js} +2 -2
  61. package/dist/docs/assets/{reference_project-layout.md.D8E6ZmHJ.lean.js → reference_project-layout.md._XdeMahr.lean.js} +1 -1
  62. package/dist/docs/assets/{reference_sessions.md.C_ouF_uf.js → reference_sessions.md.DBVFi2Sx.js} +2 -2
  63. package/dist/docs/assets/{reference_subagents.md.zWAMNfi1.js → reference_subagents.md.DSrGLIuB.js} +2 -2
  64. package/dist/docs/assets/{reference_subagents.md.zWAMNfi1.lean.js → reference_subagents.md.DSrGLIuB.lean.js} +1 -1
  65. package/dist/docs/assets/{reference_tools.md.BswAQM41.js → reference_tools.md.lSrsTxYJ.js} +4 -4
  66. package/dist/docs/assets/{reference_tools.md.BswAQM41.lean.js → reference_tools.md.lSrsTxYJ.lean.js} +1 -1
  67. package/dist/docs/assets/scaffolding-agents.md.mkc3B_ZW.js +1 -0
  68. package/dist/docs/assets/{scaffolding-agents.md.Bsr9Pwzu.lean.js → scaffolding-agents.md.mkc3B_ZW.lean.js} +1 -1
  69. package/dist/docs/assets/{storage.md.xZoiGM58.js → storage.md.mQDtIULc.js} +3 -3
  70. package/dist/docs/assets/{storage.md.xZoiGM58.lean.js → storage.md.mQDtIULc.lean.js} +1 -1
  71. package/dist/docs/building-with-agents.html +6 -6
  72. package/dist/docs/concepts.html +5 -5
  73. package/dist/docs/deployment.html +6 -6
  74. package/dist/docs/evals.html +13 -19
  75. package/dist/docs/example-agents/approval-buddy.html +5 -5
  76. package/dist/docs/example-agents/benny.html +5 -5
  77. package/dist/docs/example-agents/bugbot.html +5 -5
  78. package/dist/docs/example-agents/codebase-wiki.html +5 -5
  79. package/dist/docs/example-agents/codeowners-review.html +5 -5
  80. package/dist/docs/example-agents/concierge.html +6 -6
  81. package/dist/docs/example-agents/fsd.html +5 -5
  82. package/dist/docs/example-agents/index.html +4 -4
  83. package/dist/docs/example-agents/knowledge-base.html +5 -5
  84. package/dist/docs/example-agents/oncall.html +5 -5
  85. package/dist/docs/example-agents/security-reviewer.html +4 -4
  86. package/dist/docs/example-agents/slack-agent.html +5 -5
  87. package/dist/docs/example-agents/weather-agent.html +6 -6
  88. package/dist/docs/guides/agent-to-agent.html +5 -5
  89. package/dist/docs/guides/cloud-runtime.html +6 -6
  90. package/dist/docs/guides/github.html +4 -4
  91. package/dist/docs/guides/human-in-the-loop.html +4 -4
  92. package/dist/docs/guides/mcp-oauth.html +5 -5
  93. package/dist/docs/guides/slack.html +4 -4
  94. package/dist/docs/guides/webhooks.html +6 -6
  95. package/dist/docs/hashmap.json +1 -1
  96. package/dist/docs/hillclimbing.html +6 -6
  97. package/dist/docs/index.html +11 -7
  98. package/dist/docs/quickstart.html +10 -7
  99. package/dist/docs/reference/agent-config.html +10 -8
  100. package/dist/docs/reference/artifacts.html +43 -0
  101. package/dist/docs/reference/channels.html +6 -6
  102. package/dist/docs/reference/cli.html +18 -12
  103. package/dist/docs/reference/connections.html +4 -4
  104. package/dist/docs/reference/hooks.html +6 -6
  105. package/dist/docs/reference/http-api.html +7 -7
  106. package/dist/docs/reference/instructions.html +4 -4
  107. package/dist/docs/reference/playground.html +4 -4
  108. package/dist/docs/reference/project-layout.html +6 -6
  109. package/dist/docs/reference/prompt.html +4 -4
  110. package/dist/docs/reference/schedules.html +4 -4
  111. package/dist/docs/reference/sessions.html +7 -7
  112. package/dist/docs/reference/skills.html +4 -4
  113. package/dist/docs/reference/subagents.html +6 -6
  114. package/dist/docs/reference/tools.html +7 -7
  115. package/dist/docs/scaffolding-agents.html +5 -5
  116. package/dist/docs/storage.html +6 -6
  117. package/dist/docs/troubleshooting.html +4 -4
  118. package/dist/files-backends/agent-store-presigned-url.d.ts.map +1 -1
  119. package/dist/files-backends/agent-store-presigned-url.js +15 -22
  120. package/dist/internal/cli-github.d.ts.map +1 -1
  121. package/dist/internal/cli-github.js +8 -7
  122. package/dist/internal/cli-slack.js +9 -9
  123. package/dist/internal/event-mapper.d.ts +3 -3
  124. package/dist/internal/event-mapper.d.ts.map +1 -1
  125. package/dist/internal/event-mapper.js +7 -4
  126. package/dist/internal/host-kv.d.ts +6 -2
  127. package/dist/internal/host-kv.d.ts.map +1 -1
  128. package/dist/internal/session-engine.d.ts.map +1 -1
  129. package/dist/internal/session-engine.js +15 -6
  130. package/dist/internal/storage-coordinator.d.ts +9 -1
  131. package/dist/internal/storage-coordinator.d.ts.map +1 -1
  132. package/dist/internal/storage-coordinator.js +7 -0
  133. package/dist/internal/storage-roles.d.ts +78 -0
  134. package/dist/internal/storage-roles.d.ts.map +1 -0
  135. package/dist/internal/storage-roles.js +24 -0
  136. package/dist/internal/workspace.d.ts +28 -0
  137. package/dist/internal/workspace.d.ts.map +1 -1
  138. package/dist/internal/workspace.js +57 -0
  139. package/dist/playground/assets/{index-CDDWw0YX.js → index-DOnKC85G.js} +40 -40
  140. package/dist/playground/assets/index-DoQjqj5w.css +1 -0
  141. package/dist/playground/index.html +2 -2
  142. package/docs/README.md +32 -13
  143. package/docs/ab.md +23 -36
  144. package/docs/building-with-agents.md +2 -2
  145. package/docs/concepts.md +3 -2
  146. package/docs/deployment.md +1 -1
  147. package/docs/evals.md +102 -33
  148. package/docs/example-agents/approval-buddy.md +2 -1
  149. package/docs/example-agents/benny.md +2 -0
  150. package/docs/example-agents/bugbot.md +3 -0
  151. package/docs/example-agents/codebase-wiki.md +2 -0
  152. package/docs/example-agents/codeowners-review.md +2 -0
  153. package/docs/example-agents/concierge.md +1 -0
  154. package/docs/example-agents/fsd.md +1 -0
  155. package/docs/example-agents/knowledge-base.md +2 -0
  156. package/docs/example-agents/oncall.md +2 -0
  157. package/docs/example-agents/slack-agent.md +1 -0
  158. package/docs/example-agents/weather-agent.md +9 -4
  159. package/docs/guides/cloud-runtime.md +11 -4
  160. package/docs/guides/webhooks.md +1 -1
  161. package/docs/hillclimbing.md +1 -1
  162. package/docs/quickstart.md +39 -7
  163. package/docs/reference/agent-config.md +74 -14
  164. package/docs/reference/artifacts.md +117 -0
  165. package/docs/reference/channels.md +45 -15
  166. package/docs/reference/cli.md +141 -20
  167. package/docs/reference/hooks.md +11 -4
  168. package/docs/reference/http-api.md +50 -4
  169. package/docs/reference/project-layout.md +6 -0
  170. package/docs/reference/sessions.md +5 -4
  171. package/docs/reference/subagents.md +5 -3
  172. package/docs/reference/tools.md +23 -7
  173. package/docs/scaffolding-agents.md +11 -2
  174. package/docs/storage.md +27 -2
  175. package/package.json +1 -1
  176. package/src/channels/github/github-channel.ts +1 -1
  177. package/src/channels/github/index.ts +1 -1
  178. package/src/channels/slack/index.ts +1 -1
  179. package/src/channels/slack/init.ts +2 -1
  180. package/src/files-backends/agent-store-presigned-url.ts +2 -1
  181. package/src/internal/cli-github.ts +8 -7
  182. package/src/internal/cli-slack.ts +9 -9
  183. package/src/internal/event-mapper.ts +9 -4
  184. package/src/internal/host-kv.ts +6 -2
  185. package/src/internal/session-engine.ts +20 -8
  186. package/src/internal/storage-coordinator.ts +15 -1
  187. package/src/internal/storage-roles.ts +86 -0
  188. package/src/internal/workspace.ts +66 -1
  189. package/dist/docs/assets/chunks/@localSearchIndexroot.WoYunhnT.js +0 -1
  190. package/dist/docs/assets/guides_cloud-runtime.md.CDJGvVC4.js +0 -9
  191. package/dist/docs/assets/guides_cloud-runtime.md.CDJGvVC4.lean.js +0 -1
  192. package/dist/docs/assets/index.md.CZqbBJPB.js +0 -20
  193. package/dist/docs/assets/reference_http-api.md.CfVM_ICa.js +0 -11
  194. package/dist/docs/assets/reference_http-api.md.CfVM_ICa.lean.js +0 -1
  195. package/dist/docs/assets/scaffolding-agents.md.Bsr9Pwzu.js +0 -1
  196. package/dist/playground/assets/index-MVuNTd8v.css +0 -1
  197. /package/dist/docs/assets/{building-with-agents.md.CnHqvYDd.lean.js → building-with-agents.md.txrcGU2B.lean.js} +0 -0
  198. /package/dist/docs/assets/{concepts.md.DFaQEFkA.lean.js → concepts.md.CqOsxbMU.lean.js} +0 -0
  199. /package/dist/docs/assets/{deployment.md.9MYBuKM1.lean.js → deployment.md.CuK5SNjN.lean.js} +0 -0
  200. /package/dist/docs/assets/{example-agents_approval-buddy.md.BhEfleVx.lean.js → example-agents_approval-buddy.md.CIiZ9coo.lean.js} +0 -0
  201. /package/dist/docs/assets/{example-agents_benny.md.2Et1qa8f.lean.js → example-agents_benny.md.l7JTmm8X.lean.js} +0 -0
  202. /package/dist/docs/assets/{example-agents_fsd.md.DfNKQTHz.lean.js → example-agents_fsd.md.DPz9ezO4.lean.js} +0 -0
  203. /package/dist/docs/assets/{example-agents_knowledge-base.md.CzyZ2DCr.lean.js → example-agents_knowledge-base.md.IneynQSR.lean.js} +0 -0
  204. /package/dist/docs/assets/{example-agents_oncall.md.wFFXXEyW.lean.js → example-agents_oncall.md.ZE0n6ZFN.lean.js} +0 -0
  205. /package/dist/docs/assets/{example-agents_slack-agent.md.DvgvT4nn.lean.js → example-agents_slack-agent.md.06jQXTAI.lean.js} +0 -0
  206. /package/dist/docs/assets/{guides_webhooks.md.DiAwSR42.lean.js → guides_webhooks.md.BERuBSJW.lean.js} +0 -0
  207. /package/dist/docs/assets/{hillclimbing.md.D9Y1_bYh.lean.js → hillclimbing.md.yXqdlv2R.lean.js} +0 -0
  208. /package/dist/docs/assets/{reference_sessions.md.C_ouF_uf.lean.js → reference_sessions.md.DBVFi2Sx.lean.js} +0 -0
@@ -1,4 +1,4 @@
1
- import{_ as i,c as a,o as e,ag as t}from"./chunks/framework.CAZyNGu9.js";const c=JSON.parse('{"title":"Evals","description":"Define repeatable checks with defineEval, run them with agent-sdk eval, and use them as regression checks.","frontmatter":{"title":"Evals","description":"Define repeatable checks with defineEval, run them with agent-sdk eval, and use them as regression checks."},"headers":[],"relativePath":"evals.md","filePath":"evals.md"}'),n={name:"evals.md"};function l(h,s,p,d,o,r){return e(),a("div",null,[...s[0]||(s[0]=[t(`<h1 id="evals" tabindex="-1">Evals <a class="header-anchor" href="#evals" aria-label="Permalink to &quot;Evals&quot;">​</a></h1><p>An eval is a repeatable check that runs your agent against a fixed input and gates the recorded trajectory: the run completed, the right tool ran, the reply has the right shape. Evals are how you know a prompt tweak helped, a refactor didn&#39;t regress the agent, and last month&#39;s fix is still holding.</p><p>Evals exercise the same surface your users hit. The runner starts (or targets) a real agent server, drives sessions over the public API, and grades what comes back. A passing eval means the agent started, accepted a message, and did what you asserted.</p><div class="note custom-block github-alert"><p class="custom-block-title">NOTE</p><p>Import paths here use <code>@cursor/july/evals</code>. On projects still using <code>@cursor/july</code>, swap the import and run <code>agent-serve eval</code>. See <a href="./README.html#run-the-cli">Run the CLI</a> for the full rename table.</p></div><h2 id="define-evals-with-defineeval" tabindex="-1">Define evals with <code>defineEval</code> <a class="header-anchor" href="#define-evals-with-defineeval" aria-label="Permalink to &quot;Define evals with \`defineEval\`&quot;">​</a></h2><p>The Agent SDK discovers evals under the project-root <code>evals/</code> directory, in <code>.eval.ts</code> or <code>.eval.js</code> files. That&#39;s a sibling of <code>agent/</code>, never inside it (<code>agent/evals/</code> is silently ignored). TypeScript is the normal authoring format.</p><p>The file path is the eval&#39;s identity, so you don&#39;t author an id. Directories group related evals: <code>evals/builds/api.eval.ts</code> becomes id <code>builds/api</code>. An <code>index</code> filename collapses to its directory, so <code>evals/builds/index.eval.ts</code> becomes <code>builds</code>.</p><p>An eval is a single <code>async test(t)</code>. You drive the agent with <code>t</code> and assert on the run with the same <code>t</code>:</p><div class="language-ts vp-adaptive-theme"><button title="Copy Code" class="copy"></button><span class="lang">ts</span><pre class="shiki shiki-themes github-light github-dark vp-code" tabindex="0"><code><span class="line"><span style="--shiki-light:#6A737D;--shiki-dark:#6A737D;">// evals/readiness.eval.ts</span></span>
1
+ import{_ as e,c as a,o as i,ag as t}from"./chunks/framework.CAZyNGu9.js";const k=JSON.parse('{"title":"Evals","description":"Define repeatable checks with defineEval, run them with agent-sdk eval, and use them as regression checks.","frontmatter":{"title":"Evals","description":"Define repeatable checks with defineEval, run them with agent-sdk eval, and use them as regression checks."},"headers":[],"relativePath":"evals.md","filePath":"evals.md"}'),n={name:"evals.md"};function l(d,s,h,o,r,p){return i(),a("div",null,[...s[0]||(s[0]=[t(`<h1 id="evals" tabindex="-1">Evals <a class="header-anchor" href="#evals" aria-label="Permalink to &quot;Evals&quot;">​</a></h1><p>An eval is a repeatable check that runs your agent against a fixed input and gates the recorded trajectory: the run completed, the right tool ran, the reply has the right shape. Evals are how you know a prompt tweak helped, a refactor didn&#39;t regress the agent, and last month&#39;s fix is still holding.</p><p>Evals exercise the same surface your users hit. The runner starts (or targets) a real agent server, drives sessions over the public API, and grades what comes back. A passing eval means the agent started, accepted a message, and did what you asserted.</p><div class="note custom-block github-alert"><p class="custom-block-title">NOTE</p><p>Import paths here use <code>@cursor/july/evals</code>. On projects still using <code>@anysphere/agent-serve</code>, swap the import and run <code>agent-serve eval</code>. See <a href="./README.html#run-the-cli">Run the CLI</a> for the full rename table.</p></div><h2 id="define-evals-with-defineeval" tabindex="-1">Define evals with <code>defineEval</code> <a class="header-anchor" href="#define-evals-with-defineeval" aria-label="Permalink to &quot;Define evals with \`defineEval\`&quot;">​</a></h2><p>The Agent SDK discovers evals under the project-root <code>evals/</code> directory, in <code>.eval.ts</code> or <code>.eval.js</code> files. That&#39;s a sibling of <code>agent/</code>, never inside it (<code>agent/evals/</code> is silently ignored). TypeScript is the normal authoring format.</p><p>The file path is the eval&#39;s identity, so you don&#39;t author an id. Directories group related evals: <code>evals/builds/api.eval.ts</code> becomes id <code>builds/api</code>. An <code>index</code> filename collapses to its directory, so <code>evals/builds/index.eval.ts</code> becomes <code>builds</code>.</p><p>An eval is a single <code>async test(t)</code>. You drive the agent with <code>t</code> and assert on the run with the same <code>t</code>:</p><div class="language-ts vp-adaptive-theme"><button title="Copy Code" class="copy"></button><span class="lang">ts</span><pre class="shiki shiki-themes github-light github-dark vp-code" tabindex="0"><code><span class="line"><span style="--shiki-light:#6A737D;--shiki-dark:#6A737D;">// evals/readiness.eval.ts</span></span>
2
2
  <span class="line"><span style="--shiki-light:#D73A49;--shiki-dark:#F97583;">import</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;"> { defineEval, includes } </span><span style="--shiki-light:#D73A49;--shiki-dark:#F97583;">from</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> &quot;@cursor/july/evals&quot;</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">;</span></span>
3
3
  <span class="line"></span>
4
4
  <span class="line"><span style="--shiki-light:#D73A49;--shiki-dark:#F97583;">export</span><span style="--shiki-light:#D73A49;--shiki-dark:#F97583;"> default</span><span style="--shiki-light:#6F42C1;--shiki-dark:#B392F0;"> defineEval</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">({</span></span>
@@ -40,31 +40,25 @@ import{_ as i,c as a,o as e,ag as t}from"./chunks/framework.CAZyNGu9.js";const c
40
40
  <span class="line"><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;"> },</span></span>
41
41
  <span class="line"><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;"> },</span></span>
42
42
  <span class="line"><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;"> ],</span></span>
43
- <span class="line"><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">});</span></span></code></pre></div><p>Case ids must be single path segments, unique within the file. Each case can set its own <code>description</code>, <code>tags</code>, <code>timeoutMs</code>, and <code>iterations</code>. A case-level value replaces the file-level value for that datapoint.</p><h3 id="iterations" tabindex="-1">Iterations <a class="header-anchor" href="#iterations" aria-label="Permalink to &quot;Iterations&quot;">​</a></h3><p><code>iterations</code> (file or case, default <code>1</code>) runs a datapoint repeatedly. Discovery expands <code>iterations: 3</code> on case <code>nyc</code> to runnable ids <code>weather/nyc/1</code>, <code>weather/nyc/2</code>, <code>weather/nyc/3</code> (filter prefix <code>weather/nyc</code> still selects all three). Each expanded case exposes <code>t.iteration</code> / <code>t.iterations</code> on the test context. Cap is 100.</p><p><code>maxConcurrency</code> counts <strong>authored datapoints</strong>, not expanded iterations: siblings <code>…/1</code>…<code>…/n</code> share one concurrency slot and run sequentially. A suite with 11 cases × 3 iterations and <code>maxConcurrency: 20</code> therefore has at most 11 cases in flight, not 33.</p><h2 id="configure-eval-runs" tabindex="-1">Configure eval runs <a class="header-anchor" href="#configure-eval-runs" aria-label="Permalink to &quot;Configure eval runs&quot;">​</a></h2><p>Each project with evals needs <code>evals/evals.config.ts</code> or <code>evals/evals.config.js</code>, and it must set <code>maxConcurrency</code>. Each case issues real model-provider requests, so concurrency is capped hard at 200. Existing projects use 20. Discovery with <code>eval --list</code> works without this file, but running a case does not.</p><div class="language-ts vp-adaptive-theme"><button title="Copy Code" class="copy"></button><span class="lang">ts</span><pre class="shiki shiki-themes github-light github-dark vp-code" tabindex="0"><code><span class="line"><span style="--shiki-light:#D73A49;--shiki-dark:#F97583;">import</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;"> {</span></span>
44
- <span class="line"><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;"> defineEvalConfig,</span></span>
45
- <span class="line"><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;"> persistEvalRunsToDir,</span></span>
46
- <span class="line"><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">} </span><span style="--shiki-light:#D73A49;--shiki-dark:#F97583;">from</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> &quot;@cursor/july/evals&quot;</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">;</span></span>
43
+ <span class="line"><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">});</span></span></code></pre></div><p>Case ids must be single path segments, unique within the file. Each case can set its own <code>description</code>, <code>tags</code>, <code>timeoutMs</code>, and <code>iterations</code>. A case-level value replaces the file-level value for that datapoint.</p><h3 id="iterations" tabindex="-1">Iterations <a class="header-anchor" href="#iterations" aria-label="Permalink to &quot;Iterations&quot;">​</a></h3><p><code>iterations</code> (file or case, default <code>1</code>) runs a datapoint repeatedly. Discovery expands <code>iterations: 3</code> on case <code>nyc</code> to runnable ids <code>weather/nyc/1</code>, <code>weather/nyc/2</code>, <code>weather/nyc/3</code> (filter prefix <code>weather/nyc</code> still selects all three). Each expanded case exposes <code>t.iteration</code> / <code>t.iterations</code> on the test context. Cap is 100.</p><p><code>maxConcurrency</code> counts <strong>authored datapoints</strong>, not expanded iterations: siblings <code>…/1</code>…<code>…/n</code> share one concurrency slot and run sequentially. A suite with 11 cases × 3 iterations and <code>maxConcurrency: 20</code> therefore has at most 11 cases in flight, not 33.</p><h2 id="configure-eval-runs" tabindex="-1">Configure eval runs <a class="header-anchor" href="#configure-eval-runs" aria-label="Permalink to &quot;Configure eval runs&quot;">​</a></h2><p>Each project with evals needs <code>evals/evals.config.ts</code> or <code>evals/evals.config.js</code>, and it must set <code>maxConcurrency</code>. Each case issues real model-provider requests, so concurrency is capped hard at 200. Existing projects use 20. Discovery with <code>eval --list</code> works without this file, but running a case does not.</p><div class="language-ts vp-adaptive-theme"><button title="Copy Code" class="copy"></button><span class="lang">ts</span><pre class="shiki shiki-themes github-light github-dark vp-code" tabindex="0"><code><span class="line"><span style="--shiki-light:#D73A49;--shiki-dark:#F97583;">import</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;"> { defineEvalConfig } </span><span style="--shiki-light:#D73A49;--shiki-dark:#F97583;">from</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> &quot;@cursor/july/evals&quot;</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">;</span></span>
47
44
  <span class="line"></span>
48
45
  <span class="line"><span style="--shiki-light:#D73A49;--shiki-dark:#F97583;">export</span><span style="--shiki-light:#D73A49;--shiki-dark:#F97583;"> default</span><span style="--shiki-light:#6F42C1;--shiki-dark:#B392F0;"> defineEvalConfig</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">({</span></span>
49
46
  <span class="line"><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;"> maxConcurrency: </span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;">20</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">, </span><span style="--shiki-light:#6A737D;--shiki-dark:#6A737D;">// required</span></span>
50
47
  <span class="line"><span style="--shiki-light:#6A737D;--shiki-dark:#6A737D;"> // timeoutMs: 180_000,</span><span style="--shiki-light:#6A737D;--shiki-dark:#6A737D;"> // optional project-wide default</span></span>
48
+ <span class="line"><span style="--shiki-light:#6A737D;--shiki-dark:#6A737D;"> // judge: { model: &quot;...&quot; },</span><span style="--shiki-light:#6A737D;--shiki-dark:#6A737D;"> // default judge model for t.judge.*</span></span>
49
+ <span class="line"><span style="--shiki-light:#6A737D;--shiki-dark:#6A737D;"> // reporters: [],</span><span style="--shiki-light:#6A737D;--shiki-dark:#6A737D;"> // destinations that observe every case</span></span>
51
50
  <span class="line"><span style="--shiki-light:#6A737D;--shiki-dark:#6A737D;"> // maxPlaygroundRuns: 50,</span><span style="--shiki-light:#6A737D;--shiki-dark:#6A737D;"> // playground /v1/dev/evals history only (default 20)</span></span>
52
- <span class="line"><span style="--shiki-light:#6A737D;--shiki-dark:#6A737D;"> //</span></span>
53
- <span class="line"><span style="--shiki-light:#6A737D;--shiki-dark:#6A737D;"> // Playground batches default to **process memory only** — they disappear</span></span>
54
- <span class="line"><span style="--shiki-light:#6A737D;--shiki-dark:#6A737D;"> // when \`serve\` exits. Opt into durable storage:</span></span>
55
- <span class="line"><span style="--shiki-light:#6A737D;--shiki-dark:#6A737D;"> // persistRuns: persistEvalRunsToDir(&quot;.agent-serve/eval-runs&quot;),</span></span>
56
- <span class="line"><span style="--shiki-light:#6A737D;--shiki-dark:#6A737D;"> // Or implement { load, save, delete } yourself (S3, DB, …).</span></span>
57
- <span class="line"><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">});</span></span></code></pre></div><p>The timeout order is case or file <code>timeoutMs</code>, CLI <code>--timeout-ms</code>, project config <code>timeoutMs</code>, then the 180-second runner default.</p><p>Playground batch retention is separate from case concurrency:</p><table tabindex="0"><thead><tr><th>Option</th><th>Default</th><th>Meaning</th></tr></thead><tbody><tr><td><code>maxPlaygroundRuns</code></td><td><code>20</code></td><td>Max batches in the playground / <code>/v1/dev/evals*</code> history (not CLI <code>eval</code>)</td></tr><tr><td><code>persistRuns</code></td><td>unset</td><td>Optional <code>{ load, save, delete }</code> so batches survive process restart (<code>delete</code> required for durable prune)</td></tr></tbody></table><p>Without <code>persistRuns</code>, navigating away and back still works while the same <code>serve</code> process is up; a restart clears history.</p><h2 id="drive-and-assert-with-t" tabindex="-1">Drive and assert with <code>t</code> <a class="header-anchor" href="#drive-and-assert-with-t" aria-label="Permalink to &quot;Drive and assert with \`t\`&quot;">​</a></h2><p><code>t</code> is both the driver and the assertion surface. You write ordinary control flow, sending turns and asserting inline.</p><p>Drive the agent with <code>t.send(message, options?)</code>. It runs one turn and waits for the session to park or fail. Multiple sends in one case share the session, which is how you write multi-turn evals. The return value contains the turn&#39;s <code>message</code>, <code>sessionId</code>, <code>events</code>, <code>toolCalls</code>, and <code>ok</code> state.</p><p>Read the full case state with <code>t.reply</code> (the last assistant text), <code>t.events</code> (every captured session event across turns), and <code>t.sessionId</code>.</p><p>Assert with the gates:</p><table tabindex="0"><thead><tr><th>Gate</th><th>Checks</th></tr></thead><tbody><tr><td><code>t.succeeded()</code></td><td>the captured trajectory has at least one turn and did not fail</td></tr><tr><td><code>t.calledTool(name)</code></td><td><code>name</code> appears in the captured tool calls</td></tr><tr><td><code>t.notCalledTool(name)</code></td><td><code>name</code> does not appear in the captured tool calls</td></tr><tr><td><code>t.messageIncludes(token)</code></td><td>the last assistant reply matches a string or <code>RegExp</code></td></tr><tr><td><code>t.check(value, expectation)</code></td><td>any value, against a builder</td></tr></tbody></table><p><code>calledTool</code> reads the recorded trajectory. A requested call counts even when its result has not arrived. To require a completed result, inspect <code>t.events</code> for an <code>action.result</code> event.</p><p>The expectation builders are <code>includes(string | RegExp)</code>, <code>equals(value)</code>, and <code>satisfies(predicate, label)</code>. <code>includes</code> stringifies its input, <code>equals</code> compares values deeply, and <code>satisfies</code> runs your predicate. <code>t.log(message)</code> records a debug line for the CLI and playground result.</p><p>Three <code>t.send</code> options apply on session create (first <code>t.send</code> only):</p><ul><li><code>workspaceFiles</code> — <code>{ path: contents }</code>, seeded into the local session workspace. Prefer this over machine-local paths.</li><li><code>workspaceDir</code> — absolute harness cwd (local runtime).</li><li><code>cloud</code> — per-session cloud options merged over the agent&#39;s static <code>cloud</code> config (repos / env / …). Use a pinned <code>repos</code> override to attach a fixture repo for cloud evals without putting it on the agent&#39;s default <code>cloud.repos</code>. Cloud ignores <code>workspaceFiles</code> seeds.</li></ul><div class="language-ts vp-adaptive-theme"><button title="Copy Code" class="copy"></button><span class="lang">ts</span><pre class="shiki shiki-themes github-light github-dark vp-code" tabindex="0"><code><span class="line"><span style="--shiki-light:#D73A49;--shiki-dark:#F97583;">const</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> toolResults</span><span style="--shiki-light:#D73A49;--shiki-dark:#F97583;"> =</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;"> t.events.</span><span style="--shiki-light:#6F42C1;--shiki-dark:#B392F0;">filter</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">((</span><span style="--shiki-light:#E36209;--shiki-dark:#FFAB70;">e</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">) </span><span style="--shiki-light:#D73A49;--shiki-dark:#F97583;">=&gt;</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;"> e.type </span><span style="--shiki-light:#D73A49;--shiki-dark:#F97583;">===</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> &quot;action.result&quot;</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">);</span></span>
51
+ <span class="line"><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">});</span></span></code></pre></div><p>The timeout order is case or file <code>timeoutMs</code>, CLI <code>--timeout-ms</code>, project config <code>timeoutMs</code>, then the 180-second runner default.</p><p>The optional fields:</p><table tabindex="0"><thead><tr><th>Option</th><th>Default</th><th>Meaning</th></tr></thead><tbody><tr><td><code>timeoutMs</code></td><td><code>180_000</code></td><td>Project-wide per-case timeout</td></tr><tr><td><code>judge</code></td><td>unset</td><td>Default judge model for <code>t.judge.*</code>; see <a href="#judge-free-form-output">Judge free-form output</a></td></tr><tr><td><code>reporters</code></td><td>unset</td><td>Destinations that observe every case; <code>--skip-report</code> suppresses them</td></tr><tr><td><code>maxPlaygroundRuns</code></td><td><code>20</code></td><td>Max batches in the playground / <code>/v1/dev/evals*</code> history (not CLI <code>eval</code>)</td></tr></tbody></table><p>Reporters come from <code>@cursor/july/evals/reporters</code>: <code>JUnit</code> writes a JUnit XML file for CI, <code>Artifacts</code> writes per-case files, and <code>combineReporters</code> merges several into one (<code>renderJUnitXml</code> renders the XML for a custom destination). A file or case can add its own <code>reporters</code> on top of the config list.</p><p>Playground batches live in process memory and disappear when <code>serve</code> exits. Navigating away and back still works while the process is up. To keep batches across restarts, declare an <code>evals</code> table in <code>agent/storage.ts</code>; see <a href="./storage.html#eval-and-ab-tables">Storage</a>.</p><h2 id="drive-and-assert-with-t" tabindex="-1">Drive and assert with <code>t</code> <a class="header-anchor" href="#drive-and-assert-with-t" aria-label="Permalink to &quot;Drive and assert with \`t\`&quot;">​</a></h2><p><code>t</code> is both the driver and the assertion surface. You write ordinary control flow, sending turns and asserting inline.</p><p>Drive the agent with <code>t.send(message, options?)</code>. It runs one turn and waits for the session to park or fail. Multiple sends in one case share the session, which is how you write multi-turn evals.</p><p>Each <code>t.send</code> resolves to a turn result with <code>message</code>, <code>sessionId</code>, <code>events</code>, <code>toolCalls</code>, <code>ok</code>, and <code>index</code>. The turn carries the same assertion vocabulary as <code>t</code>, scoped to that turn, so you can grade an intermediate turn before the next send overwrites <code>t.reply</code>. <code>turn.expectOk()</code> throws when the turn failed, for later steps that depend on it.</p><p>Read the full case state with <code>t.reply</code> (the last assistant text), <code>t.events</code> (every captured session event across turns), <code>t.turns</code> (settled turns, oldest first), and <code>t.sessionId</code>. <code>t.signal</code> aborts when the case hits its timeout; pass it to your own async work.</p><p>Assert with the gates:</p><table tabindex="0"><thead><tr><th>Gate</th><th>Checks</th></tr></thead><tbody><tr><td><code>t.succeeded()</code></td><td>the run did not fail and is not parked on an unanswered approval</td></tr><tr><td><code>t.parked()</code></td><td>the run cleanly parked on an unanswered approval request</td></tr><tr><td><code>t.messageIncludes(token)</code></td><td>the joined assistant text matches a string or <code>RegExp</code></td></tr><tr><td><code>t.calledTool(name, matcher?)</code></td><td>a matching call to <code>name</code> happened</td></tr><tr><td><code>t.notCalledTool(name)</code></td><td>no request for <code>name</code>, in any lifecycle state</td></tr><tr><td><code>t.loadedSkill(name)</code></td><td>the agent opened the skill&#39;s <code>SKILL.md</code> (read, grep, or shell <code>cat</code>)</td></tr><tr><td><code>t.toolOrder(names)</code></td><td>tool requests appear in this relative order (extra calls allowed)</td></tr><tr><td><code>t.usedNoTools()</code></td><td>no tool calls at all</td></tr><tr><td><code>t.maxToolCalls(max)</code></td><td>at most <code>max</code> tool calls</td></tr><tr><td><code>t.noFailedActions()</code></td><td>no tool call reported an error</td></tr><tr><td><code>t.calledSubagent(name, matcher?)</code></td><td>a matching subagent delegation happened</td></tr><tr><td><code>t.taggedArtifact(kind?, predicate?)</code></td><td>at least one <a href="./reference/artifacts.html">artifact</a> was tagged</td></tr><tr><td><code>t.event(type, matcher?)</code></td><td>at least one matching event of <code>type</code> occurred</td></tr><tr><td><code>t.notEvent(type, matcher?)</code></td><td>no matching event of <code>type</code> occurred</td></tr><tr><td><code>t.eventOrder(matchers)</code></td><td>matching event groups occur in this relative order</td></tr><tr><td><code>t.eventsSatisfy(label, predicate)</code></td><td>your predicate over the typed event stream</td></tr><tr><td><code>t.check(value, expectation)</code></td><td>any value, against a builder</td></tr><tr><td><code>t.score(name, value)</code></td><td>records a 0–1 score you computed; soft until you add a bar</td></tr><tr><td><code>t.requireToolCall(name, matcher?)</code></td><td>gates on a matching call and returns it, so later code can read its input and output</td></tr><tr><td><code>t.requireInputRequest(filter?)</code></td><td>gates on exactly one pending approval request and returns it</td></tr></tbody></table><p>Every gate returns a handle: <code>.soft()</code> demotes it to tracked-only, <code>.atLeast(0.7)</code> adds a soft score bar, and <code>.gate(0.8)</code> promotes a scored assertion into a hard gate.</p><p>With no matcher, <code>calledTool</code> is request-based: a requested call counts even when its result has not arrived. Pass <code>t.calledTool(&quot;inspect_pr&quot;, { status: &quot;completed&quot; })</code> to require the call to return. <code>input</code>, <code>output</code>, and <code>count</code> matcher fields accept a literal, a <code>RegExp</code>, or a predicate.</p><p>The expectation builders are <code>includes(string | RegExp)</code>, <code>equals(value)</code>, <code>matches(schema)</code>, <code>similarity(expected)</code>, and <code>satisfies(predicate, label)</code>. <code>includes</code> stringifies its input, <code>equals</code> compares values deeply, <code>matches</code> validates against a Standard Schema (or anything with <code>safeParse</code>, like Zod), <code>similarity</code> scores normalized text similarity, and <code>satisfies</code> runs your predicate. The plain function <code>normalizedSimilarity(actual, expected)</code> returns the same 0–1 score for use with <code>t.score</code>.</p><p>A few more context members shape a case: <code>t.require(value, expectation)</code> records a gate and stops the test body when it fails, without a duplicate execution error. <code>t.skip(reason)</code> ends the case as skipped (reported separately, never changes the exit code; call it before sending messages). <code>t.metric(name, value)</code> records a structured score for the playground case card. <code>t.log(message)</code> records a debug line for the CLI and playground result.</p><p>Three <code>t.send</code> options apply on session create (first <code>t.send</code> only):</p><ul><li><code>workspaceFiles</code> — <code>{ path: contents }</code>, seeded into the local session workspace. Prefer this over machine-local paths.</li><li><code>workspaceDir</code> — absolute harness cwd (local runtime).</li><li><code>cloud</code> — per-session cloud options merged over the agent&#39;s static <code>cloud</code> config (repos / env / …). Use a pinned <code>repos</code> override to attach a fixture repo for cloud evals without putting it on the agent&#39;s default <code>cloud.repos</code>. Cloud ignores <code>workspaceFiles</code> seeds.</li></ul><div class="language-ts vp-adaptive-theme"><button title="Copy Code" class="copy"></button><span class="lang">ts</span><pre class="shiki shiki-themes github-light github-dark vp-code" tabindex="0"><code><span class="line"><span style="--shiki-light:#D73A49;--shiki-dark:#F97583;">const</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> toolResults</span><span style="--shiki-light:#D73A49;--shiki-dark:#F97583;"> =</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;"> t.events.</span><span style="--shiki-light:#6F42C1;--shiki-dark:#B392F0;">filter</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">((</span><span style="--shiki-light:#E36209;--shiki-dark:#FFAB70;">e</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">) </span><span style="--shiki-light:#D73A49;--shiki-dark:#F97583;">=&gt;</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;"> e.type </span><span style="--shiki-light:#D73A49;--shiki-dark:#F97583;">===</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> &quot;action.result&quot;</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">);</span></span>
58
52
  <span class="line"><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">t.</span><span style="--shiki-light:#6F42C1;--shiki-dark:#B392F0;">check</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">(</span></span>
59
53
  <span class="line"><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;"> toolResults.</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;">length</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">,</span></span>
60
54
  <span class="line"><span style="--shiki-light:#6F42C1;--shiki-dark:#B392F0;"> satisfies</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">((</span><span style="--shiki-light:#E36209;--shiki-dark:#FFAB70;">n</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">) </span><span style="--shiki-light:#D73A49;--shiki-dark:#F97583;">=&gt;</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;"> (n </span><span style="--shiki-light:#D73A49;--shiki-dark:#F97583;">as</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> number</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">) </span><span style="--shiki-light:#D73A49;--shiki-dark:#F97583;">&lt;=</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> 4</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">, </span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;">&quot;at most 4 tool calls&quot;</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">)</span></span>
61
- <span class="line"><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">);</span></span></code></pre></div><p>A case with no explicit gates falls back to whether at least one turn completed successfully. Add <code>t.succeeded()</code> and behavior-specific gates anyway. They make the contract visible during review.</p><h2 id="run-evals-from-the-cli" tabindex="-1">Run evals from the CLI <a class="header-anchor" href="#run-evals-from-the-cli" aria-label="Permalink to &quot;Run evals from the CLI&quot;">​</a></h2><p>The <code>eval</code> command discovers, filters, and runs cases.</p><p>Run the CLI under Node 22.13 or newer. Do not use Bun. Its HTTP/2 client breaks tool-result streams and causes eval turns to fail.</p><div class="language-bash vp-adaptive-theme"><button title="Copy Code" class="copy"></button><span class="lang">bash</span><pre class="shiki shiki-themes github-light github-dark vp-code" tabindex="0"><code><span class="line"><span style="--shiki-light:#6F42C1;--shiki-dark:#B392F0;">agent-sdk</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> eval</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> --dir</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> .</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> --list</span><span style="--shiki-light:#6A737D;--shiki-dark:#6A737D;"> # discover only</span></span>
55
+ <span class="line"><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">);</span></span></code></pre></div><p>A case with no explicit gates falls back to whether at least one turn completed successfully. Add <code>t.succeeded()</code> and behavior-specific gates anyway. They make the contract visible during review.</p><h3 id="judge-free-form-output" tabindex="-1">Judge free-form output <a class="header-anchor" href="#judge-free-form-output" aria-label="Permalink to &quot;Judge free-form output&quot;">​</a></h3><p>When wording matters and no regex captures it, <code>t.judge</code> grades the reply with an LLM. The built-in graders are <code>factuality(expected)</code>, <code>summarizes(expected)</code>, <code>closedQA(criteria)</code>, and <code>sql(expected)</code>. Each scores <code>t.reply</code> by default; pass <code>{ on }</code> to grade another value.</p><div class="language-ts vp-adaptive-theme"><button title="Copy Code" class="copy"></button><span class="lang">ts</span><pre class="shiki shiki-themes github-light github-dark vp-code" tabindex="0"><code><span class="line"><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">t.judge.</span><span style="--shiki-light:#6F42C1;--shiki-dark:#B392F0;">factuality</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">(</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;">&quot;It is 54°F in NYC right now.&quot;</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">).</span><span style="--shiki-light:#6F42C1;--shiki-dark:#B392F0;">atLeast</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">(</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;">0.7</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">);</span></span></code></pre></div><p>Judge assertions are soft by default, so a judge never fails a build until you give it a bar with <code>.atLeast(0.7)</code> or promote it with <code>.gate(0.8)</code>. The judge model comes from <code>defineEvalConfig({ judge })</code>, <code>defineEval({ judge })</code>, a case-level <code>judge</code>, or a per-call <code>{ model }</code> override; the nearest one wins. For a domain-specific judge whose verdict is not a single score, <code>t.judge.model(prompt)</code> sends a raw prompt to the same model and returns the reply. You then record the parsed result with <code>t.score</code> or <code>t.check</code>.</p><h2 id="run-evals-from-the-cli" tabindex="-1">Run evals from the CLI <a class="header-anchor" href="#run-evals-from-the-cli" aria-label="Permalink to &quot;Run evals from the CLI&quot;">​</a></h2><p>The <code>eval</code> command discovers, filters, and runs cases.</p><p>Run the CLI under Node 22.13 or newer. Do not use Bun. Its HTTP/2 client breaks tool-result streams and causes eval turns to fail.</p><div class="language-bash vp-adaptive-theme"><button title="Copy Code" class="copy"></button><span class="lang">bash</span><pre class="shiki shiki-themes github-light github-dark vp-code" tabindex="0"><code><span class="line"><span style="--shiki-light:#6F42C1;--shiki-dark:#B392F0;">agent-sdk</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> eval</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> --dir</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> .</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> --list</span><span style="--shiki-light:#6A737D;--shiki-dark:#6A737D;"> # discover only</span></span>
62
56
  <span class="line"><span style="--shiki-light:#6F42C1;--shiki-dark:#B392F0;">agent-sdk</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> eval</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> --dir</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> .</span><span style="--shiki-light:#6A737D;--shiki-dark:#6A737D;"> # run all</span></span>
63
57
  <span class="line"><span style="--shiki-light:#6F42C1;--shiki-dark:#B392F0;">agent-sdk</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> eval</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> --dir</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> .</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> builds/checkout</span><span style="--shiki-light:#6A737D;--shiki-dark:#6A737D;"> # one datapoint</span></span>
64
58
  <span class="line"><span style="--shiki-light:#6F42C1;--shiki-dark:#B392F0;">agent-sdk</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> eval</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> --dir</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> .</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> builds</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> search</span><span style="--shiki-light:#6A737D;--shiki-dark:#6A737D;"> # several ids or prefixes</span></span>
65
59
  <span class="line"><span style="--shiki-light:#6F42C1;--shiki-dark:#B392F0;">agent-sdk</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> eval</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> --dir</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> .</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> --tag</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> smoke</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> --tag</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> pull-request</span><span style="--shiki-light:#6A737D;--shiki-dark:#6A737D;"> # any matching tag</span></span>
66
60
  <span class="line"><span style="--shiki-light:#6F42C1;--shiki-dark:#B392F0;">agent-sdk</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> eval</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> --dir</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> .</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> --json</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> --no-stream</span><span style="--shiki-light:#6A737D;--shiki-dark:#6A737D;"> # machine-readable results</span></span>
67
- <span class="line"><span style="--shiki-light:#6F42C1;--shiki-dark:#B392F0;">agent-sdk</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> eval</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> --dir</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> .</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> --verbose</span><span style="--shiki-light:#6A737D;--shiki-dark:#6A737D;"> # logs + reply snippets</span></span></code></pre></div><p>Id filters use OR semantics. Each filter selects an exact id and its descendants. For example, <code>builds</code> selects <code>builds</code>, <code>builds/checkout</code>, and every other case below that path. Repeated tags also use OR semantics. When you provide both ids and tags, a case must match both groups.</p><p><code>eval</code> boots an ephemeral server on port 0 with a temp state root outside the project, so cases don&#39;t inherit ambient monorepo rules and don&#39;t pollute <code>.agent-sdk/</code>. Point <code>--url</code> at a running server to eval a live agent instead:</p><div class="language-bash vp-adaptive-theme"><button title="Copy Code" class="copy"></button><span class="lang">bash</span><pre class="shiki shiki-themes github-light github-dark vp-code" tabindex="0"><code><span class="line"><span style="--shiki-light:#6F42C1;--shiki-dark:#B392F0;">agent-sdk</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> eval</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> --dir</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> .</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> \\</span></span>
61
+ <span class="line"><span style="--shiki-light:#6F42C1;--shiki-dark:#B392F0;">agent-sdk</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> eval</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> --dir</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> .</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> --verbose</span><span style="--shiki-light:#6A737D;--shiki-dark:#6A737D;"> # logs + reply snippets</span></span></code></pre></div><p>Id filters use OR semantics. Each filter selects an exact id and its descendants. For example, <code>builds</code> selects <code>builds</code>, <code>builds/checkout</code>, and every other case below that path. Repeated tags also use OR semantics. When you provide both ids and tags, a case must match both groups.</p><p><code>eval</code> boots an ephemeral server on port 0 with a temp state root outside the project, so cases don&#39;t inherit ambient monorepo rules and don&#39;t pollute <code>.agent-serve/</code>. Point <code>--url</code> at a running server to eval a live agent instead:</p><div class="language-bash vp-adaptive-theme"><button title="Copy Code" class="copy"></button><span class="lang">bash</span><pre class="shiki shiki-themes github-light github-dark vp-code" tabindex="0"><code><span class="line"><span style="--shiki-light:#6F42C1;--shiki-dark:#B392F0;">agent-sdk</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> eval</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> --dir</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> .</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> \\</span></span>
68
62
  <span class="line"><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> --url</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> http://127.0.0.1:3000/weather-agent</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> \\</span></span>
69
63
  <span class="line"><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> --bearer-token</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> &quot;</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">$AGENT_TOKEN</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;">&quot;</span></span></code></pre></div><p>The eval definitions still come from <code>--dir</code>; <code>--url</code> only changes the agent that receives the turns. For a locally mounted multi-agent directory, <code>--slug weather-agent</code> chooses the target. Use <code>--state-root</code> to keep ephemeral session state at a chosen path, <code>--timeout-ms</code> to override the project timeout, and <code>--no-stream</code> to keep live progress off stderr. A TTY streams turn progress by default. <code>--verbose</code> still writes <code>t.log</code> lines to stderr and adds reply snippets to text results.</p><p>Model turns need a Cursor credential from <code>agent-sdk login</code> or <code>CURSOR_API_KEY</code>.</p><p>The exit code is <code>0</code> when every selected case passes, <code>1</code> when any case fails, and <code>2</code> when no case matches. <code>--list</code> exits <code>0</code>, including when it finds no cases.</p><p>For a compact command index, see <a href="./reference/cli.html#eval">CLI: eval</a>.</p><h3 id="json-results" tabindex="-1">JSON results <a class="header-anchor" href="#json-results" aria-label="Permalink to &quot;JSON results&quot;">​</a></h3><p>Use <code>--json --no-stream</code> in scripts and CI. The top-level result carries the totals and one result per case:</p><div class="language-json vp-adaptive-theme"><button title="Copy Code" class="copy"></button><span class="lang">json</span><pre class="shiki shiki-themes github-light github-dark vp-code" tabindex="0"><code><span class="line"><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">{</span></span>
70
64
  <span class="line"><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> &quot;ok&quot;</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">: </span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;">true</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">,</span></span>
@@ -82,10 +76,10 @@ import{_ as i,c as a,o as e,ag as t}from"./chunks/framework.CAZyNGu9.js";const c
82
76
  <span class="line"><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> &quot;durationMs&quot;</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">: </span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;">12340</span></span>
83
77
  <span class="line"><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;"> }</span></span>
84
78
  <span class="line"><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;"> ]</span></span>
85
- <span class="line"><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">}</span></span></code></pre></div><p>Each case result can also include <code>description</code>, <code>finalText</code>, <code>tools</code>, <code>error</code>, and tool arguments or output. This shape lets CI report the failed assertion without parsing terminal text.</p><h2 id="run-evals-in-the-playground" tabindex="-1">Run evals in the playground <a class="header-anchor" href="#run-evals-in-the-playground" aria-label="Permalink to &quot;Run evals in the playground&quot;">​</a></h2><p>Start the server with <code>--dev</code>, open the playground, and choose <strong>Evals</strong>. You can run every case or one case, watch progress, and open the resulting session trace.</p><div class="language-bash vp-adaptive-theme"><button title="Copy Code" class="copy"></button><span class="lang">bash</span><pre class="shiki shiki-themes github-light github-dark vp-code" tabindex="0"><code><span class="line"><span style="--shiki-light:#6F42C1;--shiki-dark:#B392F0;">agent-sdk</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> serve</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> --dir</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> .</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> --dev</span></span></code></pre></div><p>Playground runs target the live server instead of an ephemeral one. Their sessions appear in the session list. One eval batch can run at a time. By default those batches are <strong>in-memory only</strong> (capped by <code>maxPlaygroundRuns</code>); set <code>persistRuns</code> in <code>evals.config.ts</code> if you need them after a serve restart — see <a href="#configure-eval-runs">Configure eval runs</a>.</p><p>The UI uses the playground eval routes (available without <code>--dev</code>): <code>GET /v1/dev/evals</code> lists datapoints and config (includes <code>maxPlaygroundRuns</code> / <code>durableRuns</code>), <code>GET /v1/dev/evals/runs</code> rehydrates recent batches after navigation, <code>POST /v1/dev/evals/runs</code> starts a batch (returns an <strong>Eval ID</strong> / <code>runId</code>), <code>GET /v1/dev/evals/runs/:runId</code> polls it, and <code>POST /v1/dev/evals/runs/:runId/cancel</code> cancels a running batch. See <a href="./reference/http-api.html#playground-eval-routes">Playground eval routes</a>. The start request returns <code>202</code> while cases run in the background. Poll until the snapshot status becomes <code>completed</code>, <code>failed</code>, or <code>cancelled</code>. Configuration errors appear on a failed snapshot.</p><p>On <code>--prod</code> / <code>--url</code>, the CLI prints the Eval ID as soon as the batch is accepted (and a Playground deep link with <code>?view=evals&amp;evalRunId=…</code>):</p><div class="language-bash vp-adaptive-theme"><button title="Copy Code" class="copy"></button><span class="lang">bash</span><pre class="shiki shiki-themes github-light github-dark vp-code" tabindex="0"><code><span class="line"><span style="--shiki-light:#6F42C1;--shiki-dark:#B392F0;">agent-sdk</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> eval</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> --prod</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> --slug</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> vulnerability-scanner</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> --tag</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> deepsec</span></span>
79
+ <span class="line"><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">}</span></span></code></pre></div><p>Each case result can also include <code>description</code>, <code>finalText</code>, <code>tools</code>, <code>error</code>, and tool arguments or output. This shape lets CI report the failed assertion without parsing terminal text.</p><h2 id="run-evals-in-the-playground" tabindex="-1">Run evals in the playground <a class="header-anchor" href="#run-evals-in-the-playground" aria-label="Permalink to &quot;Run evals in the playground&quot;">​</a></h2><p>Start the server with <code>--dev</code>, open the playground, and choose <strong>Evals</strong>. You can run every case or one case, watch progress, and open the resulting session trace.</p><div class="language-bash vp-adaptive-theme"><button title="Copy Code" class="copy"></button><span class="lang">bash</span><pre class="shiki shiki-themes github-light github-dark vp-code" tabindex="0"><code><span class="line"><span style="--shiki-light:#6F42C1;--shiki-dark:#B392F0;">agent-sdk</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> serve</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> --dir</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> .</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> --dev</span></span></code></pre></div><p>Playground runs target the live server instead of an ephemeral one. Their sessions appear in the session list. One eval batch can run at a time. By default those batches are <strong>in-memory only</strong> (capped by <code>maxPlaygroundRuns</code>); declare an <code>evals</code> table in <code>agent/storage.ts</code> if you need them after a serve restart — see <a href="./storage.html#eval-and-ab-tables">Storage</a>.</p><p>The UI uses the playground eval routes (available without <code>--dev</code>): <code>GET /v1/dev/evals</code> lists datapoints and config (includes <code>maxPlaygroundRuns</code> / <code>durableRuns</code>), <code>GET /v1/dev/evals/runs</code> rehydrates recent batches after navigation, <code>POST /v1/dev/evals/runs</code> starts a batch (returns an <strong>Eval ID</strong> / <code>runId</code>), <code>GET /v1/dev/evals/runs/:runId</code> polls it, and <code>POST /v1/dev/evals/runs/:runId/cancel</code> cancels a running batch. See <a href="./reference/http-api.html#playground-eval-routes">Playground eval routes</a>. The start request returns <code>202</code> while cases run in the background. Poll until the snapshot status becomes <code>completed</code>, <code>failed</code>, or <code>cancelled</code>. Configuration errors appear on a failed snapshot.</p><p>On <code>--prod</code> / <code>--url</code>, the CLI prints the Eval ID as soon as the batch is accepted (and a Playground deep link with <code>?view=evals&amp;evalRunId=…</code>):</p><div class="language-bash vp-adaptive-theme"><button title="Copy Code" class="copy"></button><span class="lang">bash</span><pre class="shiki shiki-themes github-light github-dark vp-code" tabindex="0"><code><span class="line"><span style="--shiki-light:#6F42C1;--shiki-dark:#B392F0;">agent-sdk</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> eval</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> --prod</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> --slug</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> vulnerability-scanner</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> --tag</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> deepsec</span></span>
86
80
  <span class="line"><span style="--shiki-light:#6A737D;--shiki-dark:#6A737D;"># Eval ID: evalrun_…</span></span>
87
81
  <span class="line"><span style="--shiki-light:#6A737D;--shiki-dark:#6A737D;"># Cancel: agent-sdk eval cancel evalrun_… --prod --slug vulnerability-scanner</span></span>
88
82
  <span class="line"><span style="--shiki-light:#6A737D;--shiki-dark:#6A737D;"># Playground: https://…/playground?view=evals&amp;evalRunId=evalrun_…</span></span>
89
83
  <span class="line"></span>
90
84
  <span class="line"><span style="--shiki-light:#6F42C1;--shiki-dark:#B392F0;">agent-sdk</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> eval</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> cancel</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> evalrun_…</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> --prod</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> --slug</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> vulnerability-scanner</span></span>
91
- <span class="line"><span style="--shiki-light:#6F42C1;--shiki-dark:#B392F0;">agent-sdk</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> eval</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> status</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> evalrun_…</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> --prod</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> --slug</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> vulnerability-scanner</span></span></code></pre></div><p>The Evals tab prefers the server’s in-flight batch (<code>activeRunId</code>) over a stale tab-local remembered id, so CLI / Slack kicks show up without an incognito window.</p><h2 id="what-good-cases-assert" tabindex="-1">What good cases assert <a class="header-anchor" href="#what-good-cases-assert" aria-label="Permalink to &quot;What good cases assert&quot;">​</a></h2><p>Gate decisions and shape, not prose. Model wording varies run to run. Tool choice, tool avoidance, and output structure are the stable contract.</p><ol><li><code>t.succeeded()</code>: always, first.</li><li>The tool decision: <code>calledTool</code> for the intended path, <code>notCalledTool</code> for the likely wrong alternative. The pair is stronger than either alone.</li><li>Output shape: a regex for the contract (<code>/ready|blocked/i</code>, a JSON marker, a findings-block fence), never exact sentences.</li><li>For structured output, parse <code>t.reply</code> and check fields with <code>satisfies</code> instead of substring-matching JSON.</li></ol><p>The common failure modes: asserting exact phrasing, packing more than about five gates into one case (split it), and cases that depend on live external state that drifts (pin the input; see fixtures).</p><h2 id="pick-fixtures-by-agent-type" tabindex="-1">Pick fixtures by agent type <a class="header-anchor" href="#pick-fixtures-by-agent-type" aria-label="Permalink to &quot;Pick fixtures by agent type&quot;">​</a></h2><p>The right fixture depends on the surface under test.</p><table tabindex="0"><thead><tr><th>Agent surface</th><th>Fixture</th></tr></thead><tbody><tr><td>Chat / domain assistant</td><td>A canonical prompt string, chosen once and frozen</td></tr><tr><td>Tool-heavy</td><td>Run <code>agent-sdk call &lt;tool&gt;</code> first to pin what the tool returns, then freeze the prompt that triggers it</td></tr><tr><td>GitHub webhook</td><td><code>agent-sdk github replay &lt;pr&gt; --events &#39;*&#39; --dry-run --out fixtures/github</code> snapshots real payloads for offline replay (<a href="./guides/github.html">GitHub guide</a>)</td></tr><tr><td>PR reviewer with host preparation</td><td>Diff, metadata, and gold labels pinned to commit SHAs; keep any live PR matrix small</td></tr><tr><td>Workspace-dependent</td><td><code>workspaceFiles</code> in <code>t.send</code> options, never developer-machine paths</td></tr></tbody></table><p>Tag the fast, reliably passing core <code>smoke</code> and run <code>--tag smoke</code> in the inner loop. Leave slow or flaky-prone cases untagged for explicit runs.</p><h3 id="materialize-api-backed-fixtures" tabindex="-1">Materialize API-backed fixtures <a class="header-anchor" href="#materialize-api-backed-fixtures" aria-label="Permalink to &quot;Materialize API-backed fixtures&quot;">​</a></h3><p>An input that only points at external data, such as a pull request URL, snapshot id, or pair of commit SHAs, is not self-contained. Fetch it once and commit the rendered fixture before you expand the suite.</p><ol><li>Save the diff, metadata, and labels under <code>fixtures/</code> at pinned revisions.</li><li>Seed those files with <code>workspaceFiles</code>, or read them from the fixture directory.</li><li>Assert decisions and output shape against the saved evidence.</li><li>Keep a small <code>smoke</code> subset for any remaining live pipeline checks.</li></ol><p><code>maxConcurrency</code> limits parallel datapoints. It does not limit model or API fan-out inside one datapoint. Materialized fixtures prevent a large suite from exhausting provider and GitHub rate limits. The <a href="./../skills/evals/SKILL.html">evals skill</a> has the full fixture workflow.</p><h2 id="keep-improvements-with-regression-evals" tabindex="-1">Keep improvements with regression evals <a class="header-anchor" href="#keep-improvements-with-regression-evals" aria-label="Permalink to &quot;Keep improvements with regression evals&quot;">​</a></h2><p>Every <a href="./hillclimbing.html">hillclimb</a> round that keeps a change must land an eval that would have failed before the change. If you can&#39;t express the improvement as a gate (a <code>calledTool</code> shift, a bounded <code>action.result</code> count, an output-shape regex), the improvement is unverified, and it&#39;ll regress silently.</p><p>The rule cuts the other way too: never weaken an existing gate to make a round pass. That&#39;s the freeze line moving, and it turns your regression suite into a list of checks that no longer protect anything.</p><h2 id="compare-variants-on-live-traffic" tabindex="-1">Compare variants on live traffic <a class="header-anchor" href="#compare-variants-on-live-traffic" aria-label="Permalink to &quot;Compare variants on live traffic&quot;">​</a></h2><p>Use <code>defineAB</code> to compare variant metrics on live sessions. It is not a test runner and has no <code>agent-sdk ab</code> command. Keep <code>defineEval</code> as the regression ratchet. Eval sessions do not enroll or change live metrics. See <a href="./ab.html">Live A/B metrics</a> for assignment, behavior, collection, and inspection.</p><h2 id="what-s-next" tabindex="-1">What&#39;s next <a class="header-anchor" href="#what-s-next" aria-label="Permalink to &quot;What&#39;s next&quot;">​</a></h2><p>Continue with these pages:</p><ul><li><a href="./ab.html">Live A/B metrics</a>: sticky variants and cumulative metrics on live sessions</li><li><a href="./hillclimbing.html">Hillclimbing</a>: the loop evals make trustworthy</li><li><a href="./building-with-agents.html">Building agents with agents</a>: have a coding agent write the first suite</li><li><a href="./guides/github.html">GitHub guide</a>: deterministic webhook fixtures with <code>github replay</code></li><li><a href="./reference/sessions.html">Sessions and streaming</a>: the events <code>t.events</code> contains</li></ul>`,77)])])}const g=i(n,[["render",l]]);export{c as __pageData,g as default};
85
+ <span class="line"><span style="--shiki-light:#6F42C1;--shiki-dark:#B392F0;">agent-sdk</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> eval</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> status</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> evalrun_…</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> --prod</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> --slug</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> vulnerability-scanner</span></span></code></pre></div><p>The Evals tab prefers the server’s in-flight batch (<code>activeRunId</code>) over a stale tab-local remembered id, so CLI / Slack kicks show up without an incognito window.</p><h2 id="what-good-cases-assert" tabindex="-1">What good cases assert <a class="header-anchor" href="#what-good-cases-assert" aria-label="Permalink to &quot;What good cases assert&quot;">​</a></h2><p>Gate decisions and shape, not prose. Model wording varies run to run. Tool choice, tool avoidance, and output structure are the stable contract.</p><ol><li><code>t.succeeded()</code>: always, first.</li><li>The tool decision: <code>calledTool</code> for the intended path, <code>notCalledTool</code> for the likely wrong alternative. The pair is stronger than either alone.</li><li>Output shape: a regex for the contract (<code>/ready|blocked/i</code>, a JSON marker, a findings-block fence), never exact sentences.</li><li>For structured output, parse <code>t.reply</code> and check fields with <code>satisfies</code> instead of substring-matching JSON.</li></ol><p>The common failure modes: asserting exact phrasing, packing more than about five gates into one case (split it), and cases that depend on live external state that drifts (pin the input; see fixtures).</p><h2 id="pick-fixtures-by-agent-type" tabindex="-1">Pick fixtures by agent type <a class="header-anchor" href="#pick-fixtures-by-agent-type" aria-label="Permalink to &quot;Pick fixtures by agent type&quot;">​</a></h2><p>The right fixture depends on the surface under test.</p><table tabindex="0"><thead><tr><th>Agent surface</th><th>Fixture</th></tr></thead><tbody><tr><td>Chat / domain assistant</td><td>A canonical prompt string, chosen once and frozen</td></tr><tr><td>Tool-heavy</td><td>Run <code>agent-sdk call &lt;tool&gt;</code> first to pin what the tool returns, then freeze the prompt that triggers it</td></tr><tr><td>GitHub webhook</td><td><code>agent-sdk github replay &lt;pr&gt; --events &#39;*&#39; --dry-run --out fixtures/github</code> snapshots real payloads for offline replay (<a href="./guides/github.html">GitHub guide</a>)</td></tr><tr><td>PR reviewer with host preparation</td><td>Diff, metadata, and gold labels pinned to commit SHAs; keep any live PR matrix small</td></tr><tr><td>Workspace-dependent</td><td><code>workspaceFiles</code> in <code>t.send</code> options, never developer-machine paths</td></tr></tbody></table><p>Tag the fast, reliably passing core <code>smoke</code> and run <code>--tag smoke</code> in the inner loop. Leave slow or flaky-prone cases untagged for explicit runs.</p><h3 id="materialize-api-backed-fixtures" tabindex="-1">Materialize API-backed fixtures <a class="header-anchor" href="#materialize-api-backed-fixtures" aria-label="Permalink to &quot;Materialize API-backed fixtures&quot;">​</a></h3><p>An input that only points at external data, such as a pull request URL, snapshot id, or pair of commit SHAs, is not self-contained. Fetch it once and commit the rendered fixture before you expand the suite.</p><ol><li>Save the diff, metadata, and labels under <code>fixtures/</code> at pinned revisions.</li><li>Seed those files with <code>workspaceFiles</code>, or read them from the fixture directory.</li><li>Assert decisions and output shape against the saved evidence.</li><li>Keep a small <code>smoke</code> subset for any remaining live pipeline checks.</li></ol><p>Read committed fixtures with <code>@cursor/july/evals/loaders</code>: <code>loadJson</code>, <code>loadJsonl</code>, and <code>loadYaml</code> resolve relative paths against the project root the runner discovered, not the cwd the CLI was invoked from (<code>resolveFixturePath</code> and <code>evalFixtureRoot</code> expose the same resolution for other file formats).</p><p><code>maxConcurrency</code> limits parallel datapoints. It does not limit model or API fan-out inside one datapoint. Materialized fixtures prevent a large suite from exhausting provider and GitHub rate limits. The <a href="./../skills/evals/SKILL.html">evals skill</a> has the full fixture workflow.</p><h2 id="keep-improvements-with-regression-evals" tabindex="-1">Keep improvements with regression evals <a class="header-anchor" href="#keep-improvements-with-regression-evals" aria-label="Permalink to &quot;Keep improvements with regression evals&quot;">​</a></h2><p>Every <a href="./hillclimbing.html">hillclimb</a> round that keeps a change must land an eval that would have failed before the change. If you can&#39;t express the improvement as a gate (a <code>calledTool</code> shift, a bounded <code>action.result</code> count, an output-shape regex), the improvement is unverified, and it&#39;ll regress silently.</p><p>The rule cuts the other way too: never weaken an existing gate to make a round pass. That&#39;s the freeze line moving, and it turns your regression suite into a list of checks that no longer protect anything.</p><h2 id="compare-variants-on-live-traffic" tabindex="-1">Compare variants on live traffic <a class="header-anchor" href="#compare-variants-on-live-traffic" aria-label="Permalink to &quot;Compare variants on live traffic&quot;">​</a></h2><p>Use <code>defineAB</code> to compare variant metrics on live sessions. It is not a test runner and has no <code>agent-sdk ab</code> command. Keep <code>defineEval</code> as the regression ratchet. Eval sessions do not enroll or change live metrics. See <a href="./ab.html">Live A/B metrics</a> for assignment, behavior, collection, and inspection.</p><h2 id="what-s-next" tabindex="-1">What&#39;s next <a class="header-anchor" href="#what-s-next" aria-label="Permalink to &quot;What&#39;s next&quot;">​</a></h2><p>Continue with these pages:</p><ul><li><a href="./ab.html">Live A/B metrics</a>: sticky variants and cumulative metrics on live sessions</li><li><a href="./hillclimbing.html">Hillclimbing</a>: the loop evals make trustworthy</li><li><a href="./building-with-agents.html">Building agents with agents</a>: have a coding agent write the first suite</li><li><a href="./guides/github.html">GitHub guide</a>: deterministic webhook fixtures with <code>github replay</code></li><li><a href="./reference/sessions.html">Sessions and streaming</a>: the events <code>t.events</code> contains</li></ul>`,86)])])}const g=e(n,[["render",l]]);export{k as __pageData,g as default};
@@ -1 +1 @@
1
- import{_ as i,c as a,o as e,ag as t}from"./chunks/framework.CAZyNGu9.js";const c=JSON.parse('{"title":"Evals","description":"Define repeatable checks with defineEval, run them with agent-sdk eval, and use them as regression checks.","frontmatter":{"title":"Evals","description":"Define repeatable checks with defineEval, run them with agent-sdk eval, and use them as regression checks."},"headers":[],"relativePath":"evals.md","filePath":"evals.md"}'),n={name:"evals.md"};function l(h,s,p,d,o,r){return e(),a("div",null,[...s[0]||(s[0]=[t("",77)])])}const g=i(n,[["render",l]]);export{c as __pageData,g as default};
1
+ import{_ as e,c as a,o as i,ag as t}from"./chunks/framework.CAZyNGu9.js";const k=JSON.parse('{"title":"Evals","description":"Define repeatable checks with defineEval, run them with agent-sdk eval, and use them as regression checks.","frontmatter":{"title":"Evals","description":"Define repeatable checks with defineEval, run them with agent-sdk eval, and use them as regression checks."},"headers":[],"relativePath":"evals.md","filePath":"evals.md"}'),n={name:"evals.md"};function l(d,s,h,o,r,p){return i(),a("div",null,[...s[0]||(s[0]=[t("",86)])])}const g=e(n,[["render",l]]);export{k as __pageData,g as default};
@@ -1,4 +1,4 @@
1
- import{_ as a,c as t,o as s,ag as i}from"./chunks/framework.CAZyNGu9.js";const u=JSON.parse('{"title":"Keep PR approval policy deterministic with Approval Buddy","description":"Separate code-owned eligibility from model-owned review, then connect GitHub, Slack, subagents, durable storage, and evals.","frontmatter":{"title":"Keep PR approval policy deterministic with Approval Buddy","description":"Separate code-owned eligibility from model-owned review, then connect GitHub, Slack, subagents, durable storage, and evals."},"headers":[],"relativePath":"example-agents/approval-buddy.md","filePath":"example-agents/approval-buddy.md"}'),o={name:"example-agents/approval-buddy.md"};function r(l,e,n,d,p,h){return s(),t("div",null,[...e[0]||(e[0]=[i(`<h1 id="keep-pr-approval-policy-deterministic-with-approval-buddy" tabindex="-1">Keep PR approval policy deterministic with Approval Buddy <a class="header-anchor" href="#keep-pr-approval-policy-deterministic-with-approval-buddy" aria-label="Permalink to &quot;Keep PR approval policy deterministic with Approval Buddy&quot;">​</a></h1><p>Approval Buddy approves eligible pull requests from a fixed roster and declines every other request. GitHub still blocks self-approval when the stamp identity authored the PR. Code decides eligibility. The model prepares evidence, runs two specialist reviews, and passes their findings to the approval tool without changing the policy decision.</p><p>Use this example when an agent can make a judgment inside a workflow, but authorization and the final side effect must stay in deterministic code.</p><p><a href="./../../examples/approval-buddy/">Browse the Approval Buddy source.</a></p><h2 id="keep-approval-policy-in-code" tabindex="-1">Keep approval policy in code <a class="header-anchor" href="#keep-approval-policy-in-code" aria-label="Permalink to &quot;Keep approval policy in code&quot;">​</a></h2><p>Approval Buddy draws three hard boundaries:</p><ul><li><code>prepare_review</code> and <code>approve_pr</code> re-read the live PR and apply the same eligibility rules.</li><li>Two subagents inspect prepared evidence, but their findings never grant or block approval.</li><li>Only <code>approve_pr</code> posts the GitHub review.</li></ul><p>A spoofed webhook, Slack message, or model claim can&#39;t add someone to the buddy roster. The mutating tool checks the source of truth immediately before it acts.</p><h2 id="follow-the-intended-stamp-flow" tabindex="-1">Follow the intended stamp flow <a class="header-anchor" href="#follow-the-intended-stamp-flow" aria-label="Permalink to &quot;Follow the intended stamp flow&quot;">​</a></h2><p>The root instructions ask the model to run this sequence for a qualifying PR:</p><ol><li>A non-draft <code>pull_request</code> event arrives with action <code>opened</code>, <code>reopened</code>, or <code>ready_for_review</code>.</li><li>The GitHub channel checks its repository allowlist and starts a session.</li><li><code>turn.started</code> posts a pending commit status.</li><li>The model calls <code>prepare_review</code>.</li><li>Host code fetches the live PR. It checks the author, open state, merged state, and draft state.</li><li>A qualifying PR gets <code>pr/MANIFEST.md</code>, <code>pr/meta.json</code>, and <code>pr/diff.patch</code> in the session workspace. Diffs above 2,000,000 characters are truncated and marked in metadata.</li><li>The model calls both review subagents through the built-in <code>task</code> tool.</li><li>It concatenates their contracted replies and calls <code>approve_pr</code>.</li><li><code>approve_pr</code> re-runs eligibility, posts an <code>APPROVE</code> review, and returns the outcome.</li><li>The channel posts a final commit status. A self-approval block also gets a short timeline comment because no approval review can appear.</li></ol><p>Ineligible PRs skip evidence and subagents. The model still calls <code>approve_pr</code> so the deterministic tool returns the formal decline reason.</p><p>Steps 4 through 9 are prompt-driven. The channel doesn&#39;t enforce tool order or prove both subagents ran, and <code>approve_pr</code> accepts missing findings. A failed turn clears the pending status with a green non-blocking result without approving the PR.</p><h2 id="map-the-framework-features" tabindex="-1">Map the framework features <a class="header-anchor" href="#map-the-framework-features" aria-label="Permalink to &quot;Map the framework features&quot;">​</a></h2><table tabindex="0"><thead><tr><th>Capability</th><th>Source</th><th>Role</th></tr></thead><tbody><tr><td>Root agent and policy prompt</td><td><a href="../../examples/approval-buddy/agent/agent.ts"><code>agent/agent.ts</code></a>, <a href="./../../examples/approval-buddy/agent/instructions.html"><code>agent/instructions.md</code></a></td><td>Configure the local agent and describe orchestration order.</td></tr><tr><td>GitHub channel</td><td><a href="../../examples/approval-buddy/agent/channels/github.ts"><code>agent/channels/github.ts</code></a></td><td>Filter wakes, lease GitHub access, and publish status events.</td></tr><tr><td>Slack channel</td><td><a href="../../examples/approval-buddy/agent/channels/slack.ts"><code>agent/channels/slack.ts</code></a></td><td>Accept approval-bot stamp and qualification requests.</td></tr><tr><td>Server tools</td><td><a href="./../../examples/approval-buddy/agent/tools/"><code>agent/tools/</code></a></td><td>Prepare evidence, approve, list buddies, and search GIFs.</td></tr><tr><td>Deterministic policy</td><td><a href="../../examples/approval-buddy/agent/lib/approve.ts"><code>agent/lib/approve.ts</code></a>, <a href="../../examples/approval-buddy/agent/lib/buddies.ts"><code>agent/lib/buddies.ts</code></a></td><td>Own the roster and live eligibility checks.</td></tr><tr><td>Review subagents</td><td><a href="./../../examples/approval-buddy/agent/subagents/"><code>agent/subagents/</code></a></td><td>Run deep audit and code-quality passes over the same evidence.</td></tr><tr><td>Storage</td><td><a href="../../examples/approval-buddy/agent/storage.ts"><code>agent/storage.ts</code></a></td><td>Persist sessions and events with <code>cursorHostedStorage</code> (Bugbot <code>agent_serve_*</code>).</td></tr><tr><td>Evals and unit tests</td><td><a href="./../../examples/approval-buddy/evals/"><code>evals/</code></a>, <a href="./../../examples/approval-buddy/agent/lib/"><code>agent/lib/</code></a></td><td>Protect routing, output contracts, policy, and GitHub behavior.</td></tr></tbody></table><p>There are no authored skills, MCP connections, schedules, reminders, hooks, A/B experiments, sandbox seeds, or tool approvals.</p><h2 id="prepare-credentials" tabindex="-1">Prepare credentials <a class="header-anchor" href="#prepare-credentials" aria-label="Permalink to &quot;Prepare credentials&quot;">​</a></h2><p>You need:</p><ul><li>Node 22.13 or newer.</li><li>An agent-runtime credential.</li><li>GitHub access to read PRs, post reviews, create commit statuses, and post the self-approval visibility comment.</li></ul><p>Optional GIF selection uses:</p><ul><li><code>GIPHY_API_KEY</code> or <code>APPROVAL_BUDDY_GIPHY_API_KEY</code>,</li><li><code>APPROVAL_BUDDY_STAMP_GIF</code>, or</li><li>severity-specific <code>APPROVAL_BUDDY_STAMP_GIF_&lt;LEVEL&gt;</code> variables.</li></ul><p>If you enable Giphy in a hosted copy, declare its secret and <code>api.giphy.com</code> egress.</p><h2 id="validate-without-approving-a-pr" tabindex="-1">Validate without approving a PR <a class="header-anchor" href="#validate-without-approving-a-pr" aria-label="Permalink to &quot;Validate without approving a PR&quot;">​</a></h2><div class="language-bash vp-adaptive-theme"><button title="Copy Code" class="copy"></button><span class="lang">bash</span><pre class="shiki shiki-themes github-light github-dark vp-code" tabindex="0"><code><span class="line"><span style="--shiki-light:#6F42C1;--shiki-dark:#B392F0;">agent-sdk</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> validate</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> --dir</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> examples/approval-buddy</span></span>
1
+ import{_ as a,c as t,o as s,ag as i}from"./chunks/framework.CAZyNGu9.js";const u=JSON.parse('{"title":"Keep PR approval policy deterministic with Approval Buddy","description":"Separate code-owned eligibility from model-owned review, then connect GitHub, Slack, subagents, durable storage, and evals.","frontmatter":{"title":"Keep PR approval policy deterministic with Approval Buddy","description":"Separate code-owned eligibility from model-owned review, then connect GitHub, Slack, subagents, durable storage, and evals."},"headers":[],"relativePath":"example-agents/approval-buddy.md","filePath":"example-agents/approval-buddy.md"}'),o={name:"example-agents/approval-buddy.md"};function r(l,e,n,d,p,h){return s(),t("div",null,[...e[0]||(e[0]=[i(`<h1 id="keep-pr-approval-policy-deterministic-with-approval-buddy" tabindex="-1">Keep PR approval policy deterministic with Approval Buddy <a class="header-anchor" href="#keep-pr-approval-policy-deterministic-with-approval-buddy" aria-label="Permalink to &quot;Keep PR approval policy deterministic with Approval Buddy&quot;">​</a></h1><p>Approval Buddy approves eligible pull requests from a fixed roster and declines every other request. GitHub still blocks self-approval when the stamp identity authored the PR. Code decides eligibility. The model prepares evidence, runs two specialist reviews, and passes their findings to the approval tool without changing the policy decision.</p><p>Use this example when an agent can make a judgment inside a workflow, but authorization and the final side effect must stay in deterministic code.</p><p><a href="./../../examples/approval-buddy/">Browse the Approval Buddy source.</a></p><h2 id="keep-approval-policy-in-code" tabindex="-1">Keep approval policy in code <a class="header-anchor" href="#keep-approval-policy-in-code" aria-label="Permalink to &quot;Keep approval policy in code&quot;">​</a></h2><p>Approval Buddy draws three hard boundaries:</p><ul><li><code>prepare_review</code> and <code>approve_pr</code> re-read the live PR and apply the same eligibility rules.</li><li>Two subagents inspect prepared evidence, but their findings never grant or block approval.</li><li>Only <code>approve_pr</code> posts the GitHub review.</li></ul><p>A spoofed webhook, Slack message, or model claim can&#39;t add someone to the buddy roster. The mutating tool checks the source of truth immediately before it acts.</p><h2 id="follow-the-intended-stamp-flow" tabindex="-1">Follow the intended stamp flow <a class="header-anchor" href="#follow-the-intended-stamp-flow" aria-label="Permalink to &quot;Follow the intended stamp flow&quot;">​</a></h2><p>The root instructions ask the model to run this sequence for a qualifying PR:</p><ol><li>A non-draft <code>pull_request</code> event arrives with action <code>opened</code>, <code>reopened</code>, or <code>ready_for_review</code>.</li><li>The GitHub channel checks its repository allowlist and starts a session.</li><li><code>turn.started</code> posts a pending commit status.</li><li>The model calls <code>prepare_review</code>.</li><li>Host code fetches the live PR. It checks the author, open state, merged state, and draft state.</li><li>A qualifying PR gets <code>pr/MANIFEST.md</code>, <code>pr/meta.json</code>, and <code>pr/diff.patch</code> in the session workspace. Diffs above 2,000,000 characters are truncated and marked in metadata.</li><li>The model calls both review subagents through the built-in <code>task</code> tool.</li><li>It concatenates their contracted replies and calls <code>approve_pr</code>.</li><li><code>approve_pr</code> re-runs eligibility, posts an <code>APPROVE</code> review, and returns the outcome.</li><li>The channel posts a final commit status. A self-approval block also gets a short timeline comment because no approval review can appear.</li></ol><p>Ineligible PRs skip evidence and subagents. The model still calls <code>approve_pr</code> so the deterministic tool returns the formal decline reason.</p><p>Steps 4 through 9 are prompt-driven. The channel doesn&#39;t enforce tool order or prove both subagents ran, and <code>approve_pr</code> accepts missing findings. A failed turn clears the pending status with a green non-blocking result without approving the PR.</p><h2 id="map-the-framework-features" tabindex="-1">Map the framework features <a class="header-anchor" href="#map-the-framework-features" aria-label="Permalink to &quot;Map the framework features&quot;">​</a></h2><table tabindex="0"><thead><tr><th>Capability</th><th>Source</th><th>Role</th></tr></thead><tbody><tr><td>Root agent and policy prompt</td><td><a href="../../examples/approval-buddy/agent/agent.ts"><code>agent/agent.ts</code></a>, <a href="./../../examples/approval-buddy/agent/instructions.html"><code>agent/instructions.md</code></a></td><td>Configure the local agent and describe orchestration order.</td></tr><tr><td>GitHub channel</td><td><a href="../../examples/approval-buddy/agent/channels/github.ts"><code>agent/channels/github.ts</code></a></td><td>Filter wakes, lease GitHub access, and publish status events.</td></tr><tr><td>Slack channel</td><td><a href="../../examples/approval-buddy/agent/channels/slack.ts"><code>agent/channels/slack.ts</code></a></td><td>Accept approval-bot stamp and qualification requests.</td></tr><tr><td>Server tools</td><td><a href="./../../examples/approval-buddy/agent/tools/"><code>agent/tools/</code></a></td><td>Prepare evidence, approve, list buddies, and search GIFs.</td></tr><tr><td>Deterministic policy</td><td><a href="../../examples/approval-buddy/agent/lib/approve.ts"><code>agent/lib/approve.ts</code></a>, <a href="../../examples/approval-buddy/agent/lib/buddies.ts"><code>agent/lib/buddies.ts</code></a></td><td>Own the roster and live eligibility checks.</td></tr><tr><td>Review subagents</td><td><a href="./../../examples/approval-buddy/agent/subagents/"><code>agent/subagents/</code></a></td><td>Run deep audit and code-quality passes over the same evidence.</td></tr><tr><td>Storage</td><td><a href="../../examples/approval-buddy/agent/storage.ts"><code>agent/storage.ts</code></a></td><td>Persist sessions and events with <code>cursorHostedStorage</code> (Bugbot <code>agent_serve_*</code>).</td></tr><tr><td>Live A/B experiment</td><td><a href="../../examples/approval-buddy/agent/ab.ts"><code>agent/ab.ts</code></a></td><td>Compare baseline responses with a concise, presentation-only treatment (<code>concise-results</code>).</td></tr><tr><td>Evals and unit tests</td><td><a href="./../../examples/approval-buddy/evals/"><code>evals/</code></a>, <a href="./../../examples/approval-buddy/agent/lib/"><code>agent/lib/</code></a></td><td>Protect routing, output contracts, policy, and GitHub behavior.</td></tr></tbody></table><p>There are no authored skills, MCP connections, schedules, reminders, hooks, sandbox seeds, or tool approvals.</p><h2 id="prepare-credentials" tabindex="-1">Prepare credentials <a class="header-anchor" href="#prepare-credentials" aria-label="Permalink to &quot;Prepare credentials&quot;">​</a></h2><p>You need:</p><ul><li>Node 22.13 or newer.</li><li>An agent-runtime credential.</li><li>GitHub access to read PRs, post reviews, create commit statuses, and post the self-approval visibility comment.</li></ul><p>Optional GIF selection uses:</p><ul><li><code>GIPHY_API_KEY</code> or <code>APPROVAL_BUDDY_GIPHY_API_KEY</code>,</li><li><code>APPROVAL_BUDDY_STAMP_GIF</code>, or</li><li>severity-specific <code>APPROVAL_BUDDY_STAMP_GIF_&lt;LEVEL&gt;</code> variables.</li></ul><p>If you enable Giphy in a hosted copy, declare its secret and <code>api.giphy.com</code> egress.</p><h2 id="validate-without-approving-a-pr" tabindex="-1">Validate without approving a PR <a class="header-anchor" href="#validate-without-approving-a-pr" aria-label="Permalink to &quot;Validate without approving a PR&quot;">​</a></h2><div class="language-bash vp-adaptive-theme"><button title="Copy Code" class="copy"></button><span class="lang">bash</span><pre class="shiki shiki-themes github-light github-dark vp-code" tabindex="0"><code><span class="line"><span style="--shiki-light:#6F42C1;--shiki-dark:#B392F0;">agent-sdk</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> validate</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> --dir</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> examples/approval-buddy</span></span>
2
2
  <span class="line"><span style="--shiki-light:#6F42C1;--shiki-dark:#B392F0;">agent-sdk</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> info</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> --dir</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> examples/approval-buddy</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> --json</span></span></code></pre></div><p>List the deterministic roster:</p><div class="language-bash vp-adaptive-theme"><button title="Copy Code" class="copy"></button><span class="lang">bash</span><pre class="shiki shiki-themes github-light github-dark vp-code" tabindex="0"><code><span class="line"><span style="--shiki-light:#6F42C1;--shiki-dark:#B392F0;">agent-sdk</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> call</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> list_buddies</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> \\</span></span>
3
3
  <span class="line"><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> --dir</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> examples/approval-buddy</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> \\</span></span>
4
4
  <span class="line"><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> --input</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> &#39;{}&#39;</span></span></code></pre></div><p>Set a known merged PR, then run the read-only precheck:</p><div class="language-bash vp-adaptive-theme"><button title="Copy Code" class="copy"></button><span class="lang">bash</span><pre class="shiki shiki-themes github-light github-dark vp-code" tabindex="0"><code><span class="line"><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">MERGED_PR_URL</span><span style="--shiki-light:#D73A49;--shiki-dark:#F97583;">=</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;">https://github.com/your-org/your-repo/pull/123</span></span>
@@ -1,4 +1,4 @@
1
- import{_ as a,c as t,o as s,ag as i}from"./chunks/framework.CAZyNGu9.js";const k=JSON.parse('{"title":"Route Slack work through repository playbooks","description":"Combine account-linked chat, allowlisted Socket Mode channel watching, inherited repository skills, and a custom local workspace.","frontmatter":{"title":"Route Slack work through repository playbooks","description":"Combine account-linked chat, allowlisted Socket Mode channel watching, inherited repository skills, and a custom local workspace."},"headers":[],"relativePath":"example-agents/benny.md","filePath":"example-agents/benny.md"}'),o={name:"example-agents/benny.md"};function n(l,e,r,h,d,p){return s(),t("div",null,[...e[0]||(e[0]=[i(`<h1 id="route-slack-work-through-repository-playbooks" tabindex="-1">Route Slack work through repository playbooks <a class="header-anchor" href="#route-slack-work-through-repository-playbooks" aria-label="Permalink to &quot;Route Slack work through repository playbooks&quot;">​</a></h1><p>This agent is a Slack teammate for a product team. Mentions and direct messages reach it through an account-linked transport. New top-level posts in an allowlisted issue channel reach it through a dedicated Slack app, even without a mention. The agent then selects a repository playbook for triage, reproduction, fixes, reviews, on-call work, or design critique.</p><p>Use this example when Slack is the intake surface and your durable procedures already live as repository skills.</p><p><a href="./../../examples/benny/">Browse the current playbook-router source.</a></p><h2 id="combine-two-slack-transports-with-repo-skills" tabindex="-1">Combine two Slack transports with repo skills <a class="header-anchor" href="#combine-two-slack-transports-with-repo-skills" aria-label="Permalink to &quot;Combine two Slack transports with repo skills&quot;">​</a></h2><p>The playbook router uniquely combines three decisions:</p><ul><li>Two Slack transports serve different engagement modes.</li><li><code>local.cwd</code> keeps session workspaces inside the monorepo.</li><li>Instructions route work to inherited repository playbooks instead of authored <code>agent/skills/</code>.</li></ul><p>The result is a thin agent project over a mature procedure library.</p><h2 id="follow-an-issue-report" tabindex="-1">Follow an issue report <a class="header-anchor" href="#follow-an-issue-report" aria-label="Permalink to &quot;Follow an issue report&quot;">​</a></h2><ol><li>A teammate creates a top-level post in the allowlisted issue channel.</li><li>The dedicated Socket Mode channel accepts the allowlisted channel.</li><li>A 15-second debounce lets edits settle. Deleting the post during that window cancels the dispatch.</li><li>The Agent SDK creates a thread-scoped session and sends the report to the playbook router.</li><li>The instructions select the matching triage playbook.</li><li>The harness finds the repository root, opens the inherited playbook, and follows its procedure.</li><li>The agent posts only in the source thread and reports the evidence it gathered.</li></ol><p>Mentions and direct messages follow the same agent instructions. They don&#39;t need the watched-channel path.</p><h2 id="map-the-playbook-router-files" tabindex="-1">Map the playbook router files <a class="header-anchor" href="#map-the-playbook-router-files" aria-label="Permalink to &quot;Map the playbook router files&quot;">​</a></h2><table tabindex="0"><thead><tr><th>File</th><th>Purpose</th></tr></thead><tbody><tr><td><a href="../../examples/benny/agent/agent.ts"><code>agent/agent.ts</code></a></td><td>Names the agent, selects its model, and keeps the harness under <code>.agent-serve/harness</code>.</td></tr><tr><td><a href="./../../examples/benny/agent/instructions.html"><code>agent/instructions.md</code></a></td><td>Defines engagement rules, evidence policy, and the playbook routing map.</td></tr><tr><td><a href="../../examples/benny/agent/channels/slack.ts"><code>agent/channels/slack.ts</code></a></td><td>Handles account-linked mentions and direct messages.</td></tr><tr><td><a href="../../examples/benny/agent/channels/slack-app.ts"><code>agent/channels/slack-app.ts</code></a></td><td>Runs the dedicated app and watches one allowlisted channel.</td></tr><tr><td><a href="../../examples/benny/evals/smoke.eval.ts"><code>evals/smoke.eval.ts</code></a></td><td>Checks the agent identity and expected triage route.</td></tr></tbody></table><p>The playbook router authors no tools, MCP connections, subagents, schedules, hooks, A/B experiments, or sandbox seeds.</p><h2 id="see-why-local-cwd-matters" tabindex="-1">See why <code>local.cwd</code> matters <a class="header-anchor" href="#see-why-local-cwd-matters" aria-label="Permalink to &quot;See why \`local.cwd\` matters&quot;">​</a></h2><p>The Agent SDK normally keeps an ephemeral <code>run</code> or <code>eval</code> workspace outside a large monorepo. This prevents ancestor instruction and repository-rule files from leaking into an unrelated agent.</p><p>The playbook router needs the opposite. Its procedures live at the repository root, so <code>agent.ts</code> sets:</p><div class="language-ts vp-adaptive-theme"><button title="Copy Code" class="copy"></button><span class="lang">ts</span><pre class="shiki shiki-themes github-light github-dark vp-code" tabindex="0"><code><span class="line"><span style="--shiki-light:#6F42C1;--shiki-dark:#B392F0;">local</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">: {</span></span>
1
+ import{_ as a,c as t,o as s,ag as i}from"./chunks/framework.CAZyNGu9.js";const k=JSON.parse('{"title":"Route Slack work through repository playbooks","description":"Combine account-linked chat, allowlisted Socket Mode channel watching, inherited repository skills, and a custom local workspace.","frontmatter":{"title":"Route Slack work through repository playbooks","description":"Combine account-linked chat, allowlisted Socket Mode channel watching, inherited repository skills, and a custom local workspace."},"headers":[],"relativePath":"example-agents/benny.md","filePath":"example-agents/benny.md"}'),o={name:"example-agents/benny.md"};function n(l,e,r,h,d,p){return s(),t("div",null,[...e[0]||(e[0]=[i(`<h1 id="route-slack-work-through-repository-playbooks" tabindex="-1">Route Slack work through repository playbooks <a class="header-anchor" href="#route-slack-work-through-repository-playbooks" aria-label="Permalink to &quot;Route Slack work through repository playbooks&quot;">​</a></h1><p>This agent is a Slack teammate for a product team. Mentions and direct messages reach it through an account-linked transport. New top-level posts in an allowlisted issue channel reach it through a dedicated Slack app, even without a mention. The agent then selects a repository playbook for triage, reproduction, fixes, reviews, on-call work, or design critique.</p><p>Use this example when Slack is the intake surface and your durable procedures already live as repository skills.</p><p><a href="./../../examples/benny/">Browse the current playbook-router source.</a></p><h2 id="combine-two-slack-transports-with-repo-skills" tabindex="-1">Combine two Slack transports with repo skills <a class="header-anchor" href="#combine-two-slack-transports-with-repo-skills" aria-label="Permalink to &quot;Combine two Slack transports with repo skills&quot;">​</a></h2><p>The playbook router uniquely combines three decisions:</p><ul><li>Two Slack transports serve different engagement modes.</li><li><code>local.cwd</code> keeps session workspaces inside the monorepo.</li><li>Instructions route work to inherited repository playbooks instead of authored <code>agent/skills/</code>.</li></ul><p>The result is a thin agent project over a mature procedure library.</p><h2 id="follow-an-issue-report" tabindex="-1">Follow an issue report <a class="header-anchor" href="#follow-an-issue-report" aria-label="Permalink to &quot;Follow an issue report&quot;">​</a></h2><ol><li>A teammate creates a top-level post in the allowlisted issue channel.</li><li>The dedicated Socket Mode channel accepts the allowlisted channel.</li><li>A 15-second debounce lets edits settle. Deleting the post during that window cancels the dispatch.</li><li>The Agent SDK creates a thread-scoped session and sends the report to the playbook router.</li><li>The instructions select the matching triage playbook.</li><li>The harness finds the repository root, opens the inherited playbook, and follows its procedure.</li><li>The agent posts only in the source thread and reports the evidence it gathered.</li></ol><p>Mentions and direct messages follow the same agent instructions. They don&#39;t need the watched-channel path.</p><h2 id="map-the-playbook-router-files" tabindex="-1">Map the playbook router files <a class="header-anchor" href="#map-the-playbook-router-files" aria-label="Permalink to &quot;Map the playbook router files&quot;">​</a></h2><table tabindex="0"><thead><tr><th>File</th><th>Purpose</th></tr></thead><tbody><tr><td><a href="../../examples/benny/agent/agent.ts"><code>agent/agent.ts</code></a></td><td>Names the agent, selects its model, and keeps the harness under <code>.agent-serve/harness</code>.</td></tr><tr><td><a href="./../../examples/benny/agent/instructions.html"><code>agent/instructions.md</code></a></td><td>Defines engagement rules, evidence policy, and the playbook routing map.</td></tr><tr><td><a href="../../examples/benny/agent/channels/slack.ts"><code>agent/channels/slack.ts</code></a></td><td>Handles account-linked mentions and direct messages.</td></tr><tr><td><a href="../../examples/benny/agent/channels/slack-app.ts"><code>agent/channels/slack-app.ts</code></a></td><td>Runs the dedicated app and watches one allowlisted channel.</td></tr><tr><td><a href="../../examples/benny/agent/storage.ts"><code>agent/storage.ts</code></a></td><td>Persists sessions and events with <code>cursorHostedStorage</code>.</td></tr><tr><td><a href="../../examples/benny/evals/evals.config.ts"><code>evals/evals.config.ts</code></a></td><td>Caps eval run concurrency.</td></tr><tr><td><a href="../../examples/benny/evals/smoke.eval.ts"><code>evals/smoke.eval.ts</code></a></td><td>Checks the agent identity and expected triage route.</td></tr></tbody></table><p>The playbook router authors no tools, MCP connections, subagents, schedules, hooks, A/B experiments, or sandbox seeds.</p><h2 id="see-why-local-cwd-matters" tabindex="-1">See why <code>local.cwd</code> matters <a class="header-anchor" href="#see-why-local-cwd-matters" aria-label="Permalink to &quot;See why \`local.cwd\` matters&quot;">​</a></h2><p>The Agent SDK normally keeps an ephemeral <code>run</code> or <code>eval</code> workspace outside a large monorepo. This prevents ancestor instruction and repository-rule files from leaking into an unrelated agent.</p><p>The playbook router needs the opposite. Its procedures live at the repository root, so <code>agent.ts</code> sets:</p><div class="language-ts vp-adaptive-theme"><button title="Copy Code" class="copy"></button><span class="lang">ts</span><pre class="shiki shiki-themes github-light github-dark vp-code" tabindex="0"><code><span class="line"><span style="--shiki-light:#6F42C1;--shiki-dark:#B392F0;">local</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">: {</span></span>
2
2
  <span class="line"><span style="--shiki-light:#6F42C1;--shiki-dark:#B392F0;"> cwd</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">: </span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;">&quot;.agent-serve/harness&quot;</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">,</span></span>
3
3
  <span class="line"><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">}</span></span></code></pre></div><p>Each harness workspace lands under <code>examples/benny/.agent-serve/harness/&lt;sessionId&gt;</code>. Walking up the directory tree reaches the host repository and its inherited playbook directory.</p><p>Those playbooks are inherited context. <code>agent-sdk info</code> reports zero authored skills for the agent. Copying this project into another repository removes its main procedures unless you copy or replace the skill library too.</p><h2 id="connect-both-slack-paths" tabindex="-1">Connect both Slack paths <a class="header-anchor" href="#connect-both-slack-paths" aria-label="Permalink to &quot;Connect both Slack paths&quot;">​</a></h2><p>The account-linked path needs an agent-runtime login and a connected Slack account:</p><div class="language-bash vp-adaptive-theme"><button title="Copy Code" class="copy"></button><span class="lang">bash</span><pre class="shiki shiki-themes github-light github-dark vp-code" tabindex="0"><code><span class="line"><span style="--shiki-light:#6F42C1;--shiki-dark:#B392F0;">agent-sdk</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> login</span></span>
4
4
  <span class="line"><span style="--shiki-light:#6F42C1;--shiki-dark:#B392F0;">agent-sdk</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> whoami</span></span></code></pre></div><p>It routes explicit mentions without a dedicated Slack token on the host.</p><p>For the watched-channel path, configure a dedicated Socket Mode app with:</p><ul><li>subscribe to <code>message.channels</code> and <code>message.groups</code>,</li><li>have an App-Level Token with <code>connections:write</code>, and</li><li>be a member of the watched channel.</li></ul><p>Run <code>agent-sdk slack setup</code> for the guided app workflow. Generate the project manifest with <code>--channel-posts</code> when you create a new copy, then validate the configured channel prefix with <code>agent-sdk slack doctor</code>.</p><p>Missing dedicated-app tokens leave that channel idle. They don&#39;t stop the account-linked channel.</p><h2 id="validate-and-start-the-server" tabindex="-1">Validate and start the server <a class="header-anchor" href="#validate-and-start-the-server" aria-label="Permalink to &quot;Validate and start the server&quot;">​</a></h2><div class="language-bash vp-adaptive-theme"><button title="Copy Code" class="copy"></button><span class="lang">bash</span><pre class="shiki shiki-themes github-light github-dark vp-code" tabindex="0"><code><span class="line"><span style="--shiki-light:#6F42C1;--shiki-dark:#B392F0;">agent-sdk</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> validate</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> --dir</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> examples/benny</span></span>
@@ -1,4 +1,4 @@
1
- import{_ as a,c as t,o as s,ag as i}from"./chunks/framework.CAZyNGu9.js";const u=JSON.parse('{"title":"Review prepared pull-request evidence","description":"Fetch a PR on the host, seed a trimmed diff-first workspace, and run a GitHub-read-only review through HTTP, Slack, or an eval.","frontmatter":{"title":"Review prepared pull-request evidence","description":"Fetch a PR on the host, seed a trimmed diff-first workspace, and run a GitHub-read-only review through HTTP, Slack, or an eval."},"headers":[],"relativePath":"example-agents/bugbot.md","filePath":"example-agents/bugbot.md"}'),r={name:"example-agents/bugbot.md"};function n(l,e,o,h,d,p){return s(),t("div",null,[...e[0]||(e[0]=[i(`<h1 id="review-prepared-pull-request-evidence" tabindex="-1">Review prepared pull-request evidence <a class="header-anchor" href="#review-prepared-pull-request-evidence" aria-label="Permalink to &quot;Review prepared pull-request evidence&quot;">​</a></h1><p>This GitHub-read-only reviewer uses host code to fetch the PR with <code>gh</code> and <code>git</code>, builds a trimmed <code>pr/</code> evidence tree, then hands that tree to the model. The model reads the diff, loads a review skill, and returns at most three high-confidence findings.</p><p>Use this example when the host should control evidence collection and the model shouldn&#39;t browse or mutate the source repository.</p><p><a href="./../../examples/bugbot/">Browse the current reviewer source.</a></p><h2 id="separate-evidence-preparation-from-review" tabindex="-1">Separate evidence preparation from review <a class="header-anchor" href="#separate-evidence-preparation-from-review" aria-label="Permalink to &quot;Separate evidence preparation from review&quot;">​</a></h2><p>The reviewer separates preparation from judgment:</p><ul><li>Host code owns GitHub and Git access.</li><li>A server tool turns untrusted PR input into bounded workspace files.</li><li>A custom channel seeds those files before the model starts.</li><li>An on-demand skill defines the review procedure and output contract.</li><li>The model returns chat text. No path posts a GitHub review.</li></ul><p>This architecture gives the model a purpose-built evidence package instead of a checkout.</p><h2 id="follow-a-review" tabindex="-1">Follow a review <a class="header-anchor" href="#follow-a-review" aria-label="Permalink to &quot;Follow a review&quot;">​</a></h2><p>The custom HTTP path runs this sequence:</p><ol><li><code>POST /v1/channels/review/</code> receives a PR reference.</li><li>The handler calls <code>prepare_pr</code> without a model turn.</li><li>Host code reads PR metadata and the unified diff.</li><li>It reuses a matching checkout, force-fetching the PR ref there when the commit is missing. Without a matching checkout, it uses a temporary bare cache.</li><li>It creates <code>pr/MANIFEST.md</code>, <code>pr/meta.json</code>, <code>pr/diff.patch</code>, and selected small files and rules.</li><li><code>send({ workspaceFiles })</code> creates the model session with that evidence.</li><li>The model reads the manifest and diff, then loads <code>pr-review</code>.</li><li>The channel returns session and playground URLs while the review streams.</li></ol><p>If a normal chat starts without evidence, the model can call <code>prepare_pr</code> mid-turn. That form writes the same files into the active session workspace.</p><h2 id="map-the-evidence-review-files" tabindex="-1">Map the evidence-review files <a class="header-anchor" href="#map-the-evidence-review-files" aria-label="Permalink to &quot;Map the evidence-review files&quot;">​</a></h2><table tabindex="0"><thead><tr><th>File</th><th>Purpose</th></tr></thead><tbody><tr><td><a href="../../examples/bugbot/agent/agent.ts"><code>agent/agent.ts</code></a></td><td>Selects the local runtime and model.</td></tr><tr><td><a href="./../../examples/bugbot/agent/instructions.html"><code>agent/instructions.md</code></a></td><td>Requires diff-first review and confines model work to <code>pr/</code>.</td></tr><tr><td><a href="../../examples/bugbot/agent/tools/prepare_pr.ts"><code>agent/tools/prepare_pr.ts</code></a></td><td>Exposes host preparation as a typed server tool.</td></tr><tr><td><a href="../../examples/bugbot/agent/lib/prepare-pr.ts"><code>agent/lib/prepare-pr.ts</code></a></td><td>Parses PR references, runs <code>gh</code> and <code>git</code>, and builds the evidence map.</td></tr><tr><td><a href="../../examples/bugbot/agent/channels/review.ts"><code>agent/channels/review.ts</code></a></td><td>Provides the loopback-only prepare-and-send HTTP route.</td></tr><tr><td><a href="../../examples/bugbot/agent/channels/slack.ts"><code>agent/channels/slack.ts</code></a></td><td>Extracts PR references and prepares evidence for mentions and direct messages.</td></tr><tr><td><a href="./../../examples/bugbot/agent/skills/pr-review.html"><code>agent/skills/pr-review.md</code></a></td><td>Sets finding limits, severities, and the machine-readable review format.</td></tr><tr><td><a href="../../examples/bugbot/evals/review/smoke.eval.ts"><code>evals/review/smoke.eval.ts</code></a></td><td>Seeds fake evidence and checks the review path without GitHub.</td></tr></tbody></table><p>There is no authored GitHub channel, MCP connection, subagent, schedule, hook, A/B experiment, approval, or custom storage.</p><h2 id="prepare-the-host" tabindex="-1">Prepare the host <a class="header-anchor" href="#prepare-the-host" aria-label="Permalink to &quot;Prepare the host&quot;">​</a></h2><p>You need:</p><ul><li>Node 22.13 or newer.</li><li>An agent-runtime credential for model turns and account-linked Slack.</li><li><code>gh</code> and <code>git</code> on <code>PATH</code>.</li><li><code>gh</code> access to the target PR.</li><li>Network access to GitHub and a writable temporary directory.</li></ul><p>The preparer can prefer a configured local checkout. Its <code>origin</code> must match the target repository. Otherwise the reviewer uses its bare cache. It never checks out the PR into the serve host&#39;s working tree.</p><h2 id="validate-the-surface" tabindex="-1">Validate the surface <a class="header-anchor" href="#validate-the-surface" aria-label="Permalink to &quot;Validate the surface&quot;">​</a></h2><div class="language-bash vp-adaptive-theme"><button title="Copy Code" class="copy"></button><span class="lang">bash</span><pre class="shiki shiki-themes github-light github-dark vp-code" tabindex="0"><code><span class="line"><span style="--shiki-light:#6F42C1;--shiki-dark:#B392F0;">agent-sdk</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> validate</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> --dir</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> examples/bugbot</span></span>
1
+ import{_ as t,c as a,o as s,ag as i}from"./chunks/framework.CAZyNGu9.js";const u=JSON.parse('{"title":"Review prepared pull-request evidence","description":"Fetch a PR on the host, seed a trimmed diff-first workspace, and run a GitHub-read-only review through HTTP, Slack, or an eval.","frontmatter":{"title":"Review prepared pull-request evidence","description":"Fetch a PR on the host, seed a trimmed diff-first workspace, and run a GitHub-read-only review through HTTP, Slack, or an eval."},"headers":[],"relativePath":"example-agents/bugbot.md","filePath":"example-agents/bugbot.md"}'),r={name:"example-agents/bugbot.md"};function n(l,e,o,h,d,p){return s(),a("div",null,[...e[0]||(e[0]=[i(`<h1 id="review-prepared-pull-request-evidence" tabindex="-1">Review prepared pull-request evidence <a class="header-anchor" href="#review-prepared-pull-request-evidence" aria-label="Permalink to &quot;Review prepared pull-request evidence&quot;">​</a></h1><p>This GitHub-read-only reviewer uses host code to fetch the PR with <code>gh</code> and <code>git</code>, builds a trimmed <code>pr/</code> evidence tree, then hands that tree to the model. The model reads the diff, loads a review skill, and returns at most three high-confidence findings.</p><p>Use this example when the host should control evidence collection and the model shouldn&#39;t browse or mutate the source repository.</p><p><a href="./../../examples/bugbot/">Browse the current reviewer source.</a></p><h2 id="separate-evidence-preparation-from-review" tabindex="-1">Separate evidence preparation from review <a class="header-anchor" href="#separate-evidence-preparation-from-review" aria-label="Permalink to &quot;Separate evidence preparation from review&quot;">​</a></h2><p>The reviewer separates preparation from judgment:</p><ul><li>Host code owns GitHub and Git access.</li><li>A server tool turns untrusted PR input into bounded workspace files.</li><li>A custom channel seeds those files before the model starts.</li><li>An on-demand skill defines the review procedure and output contract.</li><li>The model returns chat text. No path posts a GitHub review.</li></ul><p>This architecture gives the model a purpose-built evidence package instead of a checkout.</p><h2 id="follow-a-review" tabindex="-1">Follow a review <a class="header-anchor" href="#follow-a-review" aria-label="Permalink to &quot;Follow a review&quot;">​</a></h2><p>The custom HTTP path runs this sequence:</p><ol><li><code>POST /v1/channels/review/</code> receives a PR reference.</li><li>The handler calls <code>prepare_pr</code> without a model turn.</li><li>Host code reads PR metadata and the unified diff.</li><li>It reuses a matching checkout, force-fetching the PR ref there when the commit is missing. Without a matching checkout, it uses a temporary bare cache.</li><li>It creates <code>pr/MANIFEST.md</code>, <code>pr/meta.json</code>, <code>pr/diff.patch</code>, and selected small files and rules.</li><li><code>send({ workspaceFiles })</code> creates the model session with that evidence.</li><li>The model reads the manifest and diff, then loads <code>pr-review</code>.</li><li>The channel returns session and playground URLs while the review streams.</li></ol><p>If a normal chat starts without evidence, the model can call <code>prepare_pr</code> mid-turn. That form writes the same files into the active session workspace.</p><h2 id="map-the-evidence-review-files" tabindex="-1">Map the evidence-review files <a class="header-anchor" href="#map-the-evidence-review-files" aria-label="Permalink to &quot;Map the evidence-review files&quot;">​</a></h2><table tabindex="0"><thead><tr><th>File</th><th>Purpose</th></tr></thead><tbody><tr><td><a href="../../examples/bugbot/agent/agent.ts"><code>agent/agent.ts</code></a></td><td>Selects the local runtime and model.</td></tr><tr><td><a href="./../../examples/bugbot/agent/instructions.html"><code>agent/instructions.md</code></a></td><td>Requires diff-first review and confines model work to <code>pr/</code>.</td></tr><tr><td><a href="../../examples/bugbot/agent/tools/prepare_pr.ts"><code>agent/tools/prepare_pr.ts</code></a></td><td>Exposes host preparation as a typed server tool.</td></tr><tr><td><a href="../../examples/bugbot/agent/lib/prepare-pr.ts"><code>agent/lib/prepare-pr.ts</code></a></td><td>Parses PR references, runs <code>gh</code> and <code>git</code>, and builds the evidence map.</td></tr><tr><td><a href="../../examples/bugbot/agent/channels/review.ts"><code>agent/channels/review.ts</code></a></td><td>Provides the loopback-only prepare-and-send HTTP route.</td></tr><tr><td><a href="../../examples/bugbot/agent/channels/slack.ts"><code>agent/channels/slack.ts</code></a></td><td>Extracts PR references and prepares evidence for mentions and direct messages.</td></tr><tr><td><a href="./../../examples/bugbot/agent/skills/pr-review.html"><code>agent/skills/pr-review.md</code></a></td><td>Sets finding limits, severities, and the machine-readable review format.</td></tr><tr><td><a href="../../examples/bugbot/agent/lib/log.ts"><code>agent/lib/log.ts</code></a></td><td>Writes timing logs for the host tools to stderr.</td></tr><tr><td><a href="../../examples/bugbot/agent/storage.ts"><code>agent/storage.ts</code></a></td><td>Persists sessions and events with <code>cursorHostedStorage</code>.</td></tr><tr><td><a href="../../examples/bugbot/evals/evals.config.ts"><code>evals/evals.config.ts</code></a></td><td>Caps eval run concurrency.</td></tr><tr><td><a href="../../examples/bugbot/evals/review/smoke.eval.ts"><code>evals/review/smoke.eval.ts</code></a></td><td>Seeds fake evidence and checks the review path without GitHub.</td></tr></tbody></table><p>There is no authored GitHub channel, MCP connection, subagent, schedule, hook, A/B experiment, approval, or custom storage.</p><h2 id="prepare-the-host" tabindex="-1">Prepare the host <a class="header-anchor" href="#prepare-the-host" aria-label="Permalink to &quot;Prepare the host&quot;">​</a></h2><p>You need:</p><ul><li>Node 22.13 or newer.</li><li>An agent-runtime credential for model turns and account-linked Slack.</li><li><code>gh</code> and <code>git</code> on <code>PATH</code>.</li><li><code>gh</code> access to the target PR.</li><li>Network access to GitHub and a writable temporary directory.</li></ul><p>The preparer can prefer a configured local checkout. Its <code>origin</code> must match the target repository. Otherwise the reviewer uses its bare cache. It never checks out the PR into the serve host&#39;s working tree.</p><h2 id="validate-the-surface" tabindex="-1">Validate the surface <a class="header-anchor" href="#validate-the-surface" aria-label="Permalink to &quot;Validate the surface&quot;">​</a></h2><div class="language-bash vp-adaptive-theme"><button title="Copy Code" class="copy"></button><span class="lang">bash</span><pre class="shiki shiki-themes github-light github-dark vp-code" tabindex="0"><code><span class="line"><span style="--shiki-light:#6F42C1;--shiki-dark:#B392F0;">agent-sdk</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> validate</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> --dir</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> examples/bugbot</span></span>
2
2
  <span class="line"><span style="--shiki-light:#6F42C1;--shiki-dark:#B392F0;">agent-sdk</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> info</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> --dir</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> examples/bugbot</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> --json</span></span></code></pre></div><p>The manifest should show one server tool, one skill, and two authored channels.</p><h2 id="inspect-evidence-without-a-model-turn" tabindex="-1">Inspect evidence without a model turn <a class="header-anchor" href="#inspect-evidence-without-a-model-turn" aria-label="Permalink to &quot;Inspect evidence without a model turn&quot;">​</a></h2><p>Call the preparation tool directly:</p><div class="language-bash vp-adaptive-theme"><button title="Copy Code" class="copy"></button><span class="lang">bash</span><pre class="shiki shiki-themes github-light github-dark vp-code" tabindex="0"><code><span class="line"><span style="--shiki-light:#6F42C1;--shiki-dark:#B392F0;">agent-sdk</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> call</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> prepare_pr</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> \\</span></span>
3
3
  <span class="line"><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> --dir</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> examples/bugbot</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> \\</span></span>
4
4
  <span class="line"><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> --input</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> &#39;{&quot;pr&quot;:&quot;https://github.com/owner/repo/pull/123&quot;}&#39;</span></span></code></pre></div><p>Direct tool calls use a scratch workspace removed after the call. <code>prepare_pr</code> detects this path and returns the complete file map in its result. In a model session, it writes the files and returns a smaller summary.</p><p>The evidence builder applies explicit limits:</p><table tabindex="0"><thead><tr><th>Evidence</th><th>Limit</th></tr></thead><tbody><tr><td>Post-change file</td><td>12,000 characters</td></tr><tr><td>One rule file</td><td>8,000 characters</td></tr><tr><td>Combined rules</td><td>12,000 characters</td></tr><tr><td>PR body in metadata</td><td>2,000 characters</td></tr></tbody></table><p>Large files remain visible in <code>diff.patch</code>. The manifest records which full files or rules were omitted.</p><p>The per-file limits aren&#39;t an aggregate context cap. Every changed file below 12,000 characters can be included. The diff command has a 12 MiB output buffer; a larger diff fails preparation instead of being truncated.</p><h2 id="run-the-http-review-path" tabindex="-1">Run the HTTP review path <a class="header-anchor" href="#run-the-http-review-path" aria-label="Permalink to &quot;Run the HTTP review path&quot;">​</a></h2><p>Start the server:</p><div class="language-bash vp-adaptive-theme"><button title="Copy Code" class="copy"></button><span class="lang">bash</span><pre class="shiki shiki-themes github-light github-dark vp-code" tabindex="0"><code><span class="line"><span style="--shiki-light:#6F42C1;--shiki-dark:#B392F0;">agent-sdk</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> dev</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> examples/bugbot</span></span></code></pre></div><p>From another terminal:</p><div class="language-bash vp-adaptive-theme"><button title="Copy Code" class="copy"></button><span class="lang">bash</span><pre class="shiki shiki-themes github-light github-dark vp-code" tabindex="0"><code><span class="line"><span style="--shiki-light:#6F42C1;--shiki-dark:#B392F0;">curl</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> -s</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> -X</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> POST</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> \\</span></span>
@@ -8,4 +8,4 @@ import{_ as a,c as t,o as s,ag as i}from"./chunks/framework.CAZyNGu9.js";const u
8
8
  <span class="line"><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> http://127.0.0.1:3000/bugbot/v1/channels/review/</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> \\</span></span>
9
9
  <span class="line"><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> -H</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> &#39;content-type: application/json&#39;</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> \\</span></span>
10
10
  <span class="line"><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> -d</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> &#39;{&quot;pr&quot;:&quot;owner/repo#123&quot;,&quot;key&quot;:&quot;&lt;continuation-token&gt;&quot;}&#39;</span></span></code></pre></div><p>The follow-up resumes the session without fetching a new evidence tree.</p><h2 id="run-the-slack-path" tabindex="-1">Run the Slack path <a class="header-anchor" href="#run-the-slack-path" aria-label="Permalink to &quot;Run the Slack path&quot;">​</a></h2><p>The account-linked Slack channel handles review-bot mentions and direct messages:</p><blockquote><p>Review <a href="https://github.com/owner/repo/pull/123" target="_blank" rel="noreferrer">https://github.com/owner/repo/pull/123</a></p></blockquote><p>Slack handlers don&#39;t receive the channel <code>callTool</code> helper. This example calls the shared <code>preparePrReview</code> host function, then returns <code>workspaceFiles</code> in the Slack message preparation result. The model sees the same evidence and prompt as the HTTP path.</p><p>If a message contains no PR reference, the handler asks for one. Thread follow-ups keep the same session.</p><h2 id="see-how-the-skill-constrains-review" tabindex="-1">See how the skill constrains review <a class="header-anchor" href="#see-how-the-skill-constrains-review" aria-label="Permalink to &quot;See how the skill constrains review&quot;">​</a></h2><p><code>pr-review.md</code> tells the model to:</p><ul><li>read the manifest and unified diff first,</li><li>open at most one supporting file or rules file when a hunk is ambiguous,</li><li>avoid shell, network, <code>gh</code>, and <code>git</code>,</li><li>report no more than three findings,</li><li>keep each description under 120 words, and</li><li>emit the machine-readable review contract.</li></ul><p>The root instructions set the evidence boundary. The skill holds the reusable review procedure. Keeping those roles separate lets another agent reuse the same skill with different intake channels.</p><h2 id="run-the-fixture-backed-eval" tabindex="-1">Run the fixture-backed eval <a class="header-anchor" href="#run-the-fixture-backed-eval" aria-label="Permalink to &quot;Run the fixture-backed eval&quot;">​</a></h2><div class="language-bash vp-adaptive-theme"><button title="Copy Code" class="copy"></button><span class="lang">bash</span><pre class="shiki shiki-themes github-light github-dark vp-code" tabindex="0"><code><span class="line"><span style="--shiki-light:#6F42C1;--shiki-dark:#B392F0;">agent-sdk</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> eval</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> --dir</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> examples/bugbot</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> --list</span></span>
11
- <span class="line"><span style="--shiki-light:#6F42C1;--shiki-dark:#B392F0;">agent-sdk</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> eval</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> --dir</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> examples/bugbot</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> review/smoke</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> --json</span></span></code></pre></div><p>The eval constructs a <code>PreparedPrReview</code>, seeds its file map through <code>workspaceFiles</code>, and checks for at least one read call with no shell call. It doesn&#39;t assert which evidence file was read or whether the skill loaded. It accepts either a formatted review or a clean result.</p><p>This case tests review behavior without GitHub credentials or network data. Add fixtures with reachable bugs when you need stricter location and severity checks.</p><h2 id="keep-the-side-effect-boundary-clear" tabindex="-1">Keep the side-effect boundary clear <a class="header-anchor" href="#keep-the-side-effect-boundary-clear" aria-label="Permalink to &quot;Keep the side-effect boundary clear&quot;">​</a></h2><p>The reviewer makes no remote GitHub writes. It doesn&#39;t author a GitHub channel and doesn&#39;t call a review API. Host preparation does write session evidence and force-update <code>refs/pull/&lt;N&gt;/head</code> in either its bare cache or a matching local checkout when the commit is missing. Every result ends with a note saying no GitHub review was posted.</p><p>If you add publishing later, keep it in a separate tool. This preserves a read-only preparation and review path safe to run in evals.</p><h2 id="reuse-the-evidence-handoff" tabindex="-1">Reuse the evidence handoff <a class="header-anchor" href="#reuse-the-evidence-handoff" aria-label="Permalink to &quot;Reuse the evidence handoff&quot;">​</a></h2><p>Use host-prepared workspaces when:</p><ul><li>external APIs should stay off the model&#39;s tool surface,</li><li>context needs hard size limits,</li><li>the model should inspect a snapshot instead of a live checkout, or</li><li>several channels need the same preparation.</li></ul><p>Return <code>workspaceFiles</code> from direct host preparation, write into <code>ctx.workspaceDir</code> for mid-turn recovery, and encode the reading order in both the manifest and a skill.</p><h2 id="where-to-go-next" tabindex="-1">Where to go next <a class="header-anchor" href="#where-to-go-next" aria-label="Permalink to &quot;Where to go next&quot;">​</a></h2><ul><li><a href="./../guides/webhooks.html">Webhooks and custom channels</a></li><li><a href="./../reference/tools.html">Tools</a></li><li><a href="./../reference/skills.html">Skills</a></li><li><a href="./../guides/slack.html">Slack</a></li><li><a href="./../evals.html">Evals</a></li></ul>`,62)])])}const k=a(r,[["render",n]]);export{u as __pageData,k as default};
11
+ <span class="line"><span style="--shiki-light:#6F42C1;--shiki-dark:#B392F0;">agent-sdk</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> eval</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> --dir</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> examples/bugbot</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> review/smoke</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> --json</span></span></code></pre></div><p>The eval constructs a <code>PreparedPrReview</code>, seeds its file map through <code>workspaceFiles</code>, and checks for at least one read call with no shell call. It doesn&#39;t assert which evidence file was read or whether the skill loaded. It accepts either a formatted review or a clean result.</p><p>This case tests review behavior without GitHub credentials or network data. Add fixtures with reachable bugs when you need stricter location and severity checks.</p><h2 id="keep-the-side-effect-boundary-clear" tabindex="-1">Keep the side-effect boundary clear <a class="header-anchor" href="#keep-the-side-effect-boundary-clear" aria-label="Permalink to &quot;Keep the side-effect boundary clear&quot;">​</a></h2><p>The reviewer makes no remote GitHub writes. It doesn&#39;t author a GitHub channel and doesn&#39;t call a review API. Host preparation does write session evidence and force-update <code>refs/pull/&lt;N&gt;/head</code> in either its bare cache or a matching local checkout when the commit is missing. Every result ends with a note saying no GitHub review was posted.</p><p>If you add publishing later, keep it in a separate tool. This preserves a read-only preparation and review path safe to run in evals.</p><h2 id="reuse-the-evidence-handoff" tabindex="-1">Reuse the evidence handoff <a class="header-anchor" href="#reuse-the-evidence-handoff" aria-label="Permalink to &quot;Reuse the evidence handoff&quot;">​</a></h2><p>Use host-prepared workspaces when:</p><ul><li>external APIs should stay off the model&#39;s tool surface,</li><li>context needs hard size limits,</li><li>the model should inspect a snapshot instead of a live checkout, or</li><li>several channels need the same preparation.</li></ul><p>Return <code>workspaceFiles</code> from direct host preparation, write into <code>ctx.workspaceDir</code> for mid-turn recovery, and encode the reading order in both the manifest and a skill.</p><h2 id="where-to-go-next" tabindex="-1">Where to go next <a class="header-anchor" href="#where-to-go-next" aria-label="Permalink to &quot;Where to go next&quot;">​</a></h2><ul><li><a href="./../guides/webhooks.html">Webhooks and custom channels</a></li><li><a href="./../reference/tools.html">Tools</a></li><li><a href="./../reference/skills.html">Skills</a></li><li><a href="./../guides/slack.html">Slack</a></li><li><a href="./../evals.html">Evals</a></li></ul>`,62)])])}const k=t(r,[["render",n]]);export{u as __pageData,k as default};
@@ -1 +1 @@
1
- import{_ as a,c as t,o as s,ag as i}from"./chunks/framework.CAZyNGu9.js";const u=JSON.parse('{"title":"Review prepared pull-request evidence","description":"Fetch a PR on the host, seed a trimmed diff-first workspace, and run a GitHub-read-only review through HTTP, Slack, or an eval.","frontmatter":{"title":"Review prepared pull-request evidence","description":"Fetch a PR on the host, seed a trimmed diff-first workspace, and run a GitHub-read-only review through HTTP, Slack, or an eval."},"headers":[],"relativePath":"example-agents/bugbot.md","filePath":"example-agents/bugbot.md"}'),r={name:"example-agents/bugbot.md"};function n(l,e,o,h,d,p){return s(),t("div",null,[...e[0]||(e[0]=[i("",62)])])}const k=a(r,[["render",n]]);export{u as __pageData,k as default};
1
+ import{_ as t,c as a,o as s,ag as i}from"./chunks/framework.CAZyNGu9.js";const u=JSON.parse('{"title":"Review prepared pull-request evidence","description":"Fetch a PR on the host, seed a trimmed diff-first workspace, and run a GitHub-read-only review through HTTP, Slack, or an eval.","frontmatter":{"title":"Review prepared pull-request evidence","description":"Fetch a PR on the host, seed a trimmed diff-first workspace, and run a GitHub-read-only review through HTTP, Slack, or an eval."},"headers":[],"relativePath":"example-agents/bugbot.md","filePath":"example-agents/bugbot.md"}'),r={name:"example-agents/bugbot.md"};function n(l,e,o,h,d,p){return s(),a("div",null,[...e[0]||(e[0]=[i("",62)])])}const k=t(r,[["render",n]]);export{u as __pageData,k as default};
@@ -1,4 +1,4 @@
1
- import{_ as s,c as a,o as t,ag as i}from"./chunks/framework.CAZyNGu9.js";const g=JSON.parse('{"title":"Build a feature wiki from merged pull requests","description":"Ingest every merged PR into per-feature wiki pages, consolidate with a daily digest schedule, and answer codebase questions with page citations.","frontmatter":{"title":"Build a feature wiki from merged pull requests","description":"Ingest every merged PR into per-feature wiki pages, consolidate with a daily digest schedule, and answer codebase questions with page citations."},"headers":[],"relativePath":"example-agents/codebase-wiki.md","filePath":"example-agents/codebase-wiki.md"}'),n={name:"example-agents/codebase-wiki.md"};function d(o,e,l,h,r,p){return t(),a("div",null,[...e[0]||(e[0]=[i(`<h1 id="build-a-feature-wiki-from-merged-pull-requests" tabindex="-1">Build a feature wiki from merged pull requests <a class="header-anchor" href="#build-a-feature-wiki-from-merged-pull-requests" aria-label="Permalink to &quot;Build a feature wiki from merged pull requests&quot;">​</a></h1><p>Codebase wiki keeps a living, feature-organized wiki of a repository. The GitHub channel acknowledges every closed pull request instantly, fetches a compact digest on the host, and spends a model turn only on merged PRs. The turn maps the change onto feature pages; a daily schedule writes a digest of what changed and rebuilds the index. Chat sessions answer codebase questions from the wiki with page citations.</p><p>Use this project when documentation should accumulate from merges instead of being regenerated from scratch. Use <a href="./knowledge-base.html">Knowledge base</a> when people should curate organizational context through conversation.</p><p><a href="./../../examples/codebase-wiki/">Browse the codebase wiki source.</a></p><h2 id="treat-prs-as-evidence-and-features-as-pages" tabindex="-1">Treat PRs as evidence and features as pages <a class="header-anchor" href="#treat-prs-as-evidence-and-features-as-pages" aria-label="Permalink to &quot;Treat PRs as evidence and features as pages&quot;">​</a></h2><p>The wiki refuses to become a merge log:</p><ul><li>The page tree is rigid: <code>index</code>, <code>features/&lt;slug&gt;</code>, and <code>digests/&lt;yyyy-mm-dd&gt;</code>. The store rejects anything else, so the wiki can&#39;t sprawl.</li><li>The <code>feature-mapping</code> skill requires a <code>wiki_search</code> before every write. A PR updates the page that owns its feature; a new page needs a genuinely new feature; chores change nothing.</li><li>Every touched page gets a dated changelog entry citing the PR number, so each fact traces back to a merge.</li></ul><p>The wiki itself is markdown on the serve host, in <code>.agent-serve/wiki/</code> by default with a <code>CODEBASE_WIKI_DIR</code> override. Sessions are disposable; the wiki is the durable state.</p><h2 id="follow-a-merged-pr" tabindex="-1">Follow a merged PR <a class="header-anchor" href="#follow-a-merged-pr" aria-label="Permalink to &quot;Follow a merged PR&quot;">​</a></h2><ol><li>GitHub delivers <code>pull_request</code> with action <code>closed</code>. The channel returns a task acknowledgement immediately.</li><li>The task fetches the digest with the host <code>gh</code> CLI: title, body, labels, changed files, and a bounded diff excerpt. No checkout.</li><li>The webhook payload can&#39;t say whether the PR merged, so the host checks <code>mergedAt</code> and skips abandoned PRs without a model turn.</li><li>For merged PRs, the task starts the turn with <code>pr/DIGEST.md</code> seeded through <code>workspaceFiles</code> and a <code>pr:&lt;owner/repo#N&gt;</code> continuation token, so redeliveries resume instead of double-ingesting.</li><li>The model follows <code>feature-mapping</code>: search, update or create feature pages, add changelog entries, and refresh <code>index</code> when pages were added.</li></ol><p>In chat, &quot;ingest PR #123&quot; runs the same flow through the <code>ingest_pr</code> tool, which writes the digest into the active session workspace.</p><h2 id="map-the-wiki-files" tabindex="-1">Map the wiki files <a class="header-anchor" href="#map-the-wiki-files" aria-label="Permalink to &quot;Map the wiki files&quot;">​</a></h2><table tabindex="0"><thead><tr><th>File</th><th>Purpose</th></tr></thead><tbody><tr><td><a href="../../examples/codebase-wiki/agent/agent.ts"><code>agent/agent.ts</code></a></td><td>Selects the cloud runtime and model.</td></tr><tr><td><a href="./../../examples/codebase-wiki/agent/instructions.html"><code>agent/instructions.md</code></a></td><td>Splits the job into merge ingestion and wiki-cited Q&amp;A.</td></tr><tr><td><a href="../../examples/codebase-wiki/agent/lib/wiki-store.ts"><code>agent/lib/wiki-store.ts</code></a></td><td>Enforces the rigid page tree and owns reads, writes, and search.</td></tr><tr><td><a href="../../examples/codebase-wiki/agent/lib/pr-digest.ts"><code>agent/lib/pr-digest.ts</code></a></td><td>Fetches PR metadata and diff, and formats <code>pr/DIGEST.md</code>.</td></tr><tr><td><a href="../../examples/codebase-wiki/agent/tools/ingest_pr.ts"><code>agent/tools/ingest_pr.ts</code></a></td><td>Exposes host digest preparation for chat-driven backfills.</td></tr><tr><td><a href="../../examples/codebase-wiki/agent/tools/wiki_read.ts"><code>agent/tools/wiki_read.ts</code></a>, <a href="../../examples/codebase-wiki/agent/tools/wiki_search.ts"><code>wiki_search.ts</code></a>, <a href="../../examples/codebase-wiki/agent/tools/wiki_write.ts"><code>wiki_write.ts</code></a></td><td>Read, search, and rewrite wiki pages.</td></tr><tr><td><a href="./../../examples/codebase-wiki/agent/skills/feature-mapping.html"><code>agent/skills/feature-mapping.md</code></a></td><td>Maps changes onto features and fixes the page and changelog shape.</td></tr><tr><td><a href="./../../examples/codebase-wiki/agent/schedules/daily-digest.html"><code>agent/schedules/daily-digest.md</code></a></td><td>Writes <code>digests/&lt;date&gt;</code>, rebuilds the index, and flags stale pages.</td></tr><tr><td><a href="../../examples/codebase-wiki/agent/channels/github.ts"><code>agent/channels/github.ts</code></a></td><td>Acknowledges closed PRs and starts merged-only ingest turns.</td></tr><tr><td><a href="../../examples/codebase-wiki/evals/ingest.eval.ts"><code>evals/ingest.eval.ts</code></a></td><td>Gates ingest decisions against the wiki filesystem.</td></tr></tbody></table><p>There is no MCP connection, subagent, hook, A/B experiment, or custom storage.</p><h2 id="prepare-credentials-and-services" tabindex="-1">Prepare credentials and services <a class="header-anchor" href="#prepare-credentials-and-services" aria-label="Permalink to &quot;Prepare credentials and services&quot;">​</a></h2><p>You need:</p><ul><li>Node 22.13 or newer.</li><li>An agent-runtime credential for model turns.</li><li><code>gh</code> on <code>PATH</code> with read access to the PRs you ingest.</li></ul><p>The channel verifies webhook signatures when <code>GITHUB_WEBHOOK_SECRET</code> is set and narrows repositories with <code>CODEBASE_WIKI_REPOS=owner/repo,owner/other</code>. The agent never writes to GitHub. Its only side effects are wiki files on the serve host.</p><h2 id="validate-the-surface" tabindex="-1">Validate the surface <a class="header-anchor" href="#validate-the-surface" aria-label="Permalink to &quot;Validate the surface&quot;">​</a></h2><div class="language-bash vp-adaptive-theme"><button title="Copy Code" class="copy"></button><span class="lang">bash</span><pre class="shiki shiki-themes github-light github-dark vp-code" tabindex="0"><code><span class="line"><span style="--shiki-light:#6F42C1;--shiki-dark:#B392F0;">agent-sdk</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> validate</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> --dir</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> examples/codebase-wiki</span></span>
1
+ import{_ as s,c as a,o as t,ag as i}from"./chunks/framework.CAZyNGu9.js";const g=JSON.parse('{"title":"Build a feature wiki from merged pull requests","description":"Ingest every merged PR into per-feature wiki pages, consolidate with a daily digest schedule, and answer codebase questions with page citations.","frontmatter":{"title":"Build a feature wiki from merged pull requests","description":"Ingest every merged PR into per-feature wiki pages, consolidate with a daily digest schedule, and answer codebase questions with page citations."},"headers":[],"relativePath":"example-agents/codebase-wiki.md","filePath":"example-agents/codebase-wiki.md"}'),n={name:"example-agents/codebase-wiki.md"};function d(o,e,l,r,h,c){return t(),a("div",null,[...e[0]||(e[0]=[i(`<h1 id="build-a-feature-wiki-from-merged-pull-requests" tabindex="-1">Build a feature wiki from merged pull requests <a class="header-anchor" href="#build-a-feature-wiki-from-merged-pull-requests" aria-label="Permalink to &quot;Build a feature wiki from merged pull requests&quot;">​</a></h1><p>Codebase wiki keeps a living, feature-organized wiki of a repository. The GitHub channel acknowledges every closed pull request instantly, fetches a compact digest on the host, and spends a model turn only on merged PRs. The turn maps the change onto feature pages; a daily schedule writes a digest of what changed and rebuilds the index. Chat sessions answer codebase questions from the wiki with page citations.</p><p>Use this project when documentation should accumulate from merges instead of being regenerated from scratch. Use <a href="./knowledge-base.html">Knowledge base</a> when people should curate organizational context through conversation.</p><p><a href="./../../examples/codebase-wiki/">Browse the codebase wiki source.</a></p><h2 id="treat-prs-as-evidence-and-features-as-pages" tabindex="-1">Treat PRs as evidence and features as pages <a class="header-anchor" href="#treat-prs-as-evidence-and-features-as-pages" aria-label="Permalink to &quot;Treat PRs as evidence and features as pages&quot;">​</a></h2><p>The wiki refuses to become a merge log:</p><ul><li>The page tree is rigid: <code>index</code>, <code>features/&lt;slug&gt;</code>, and <code>digests/&lt;yyyy-mm-dd&gt;</code>. The store rejects anything else, so the wiki can&#39;t sprawl.</li><li>The <code>feature-mapping</code> skill requires a <code>wiki_search</code> before every write. A PR updates the page that owns its feature; a new page needs a genuinely new feature; chores change nothing.</li><li>Every touched page gets a dated changelog entry citing the PR number, so each fact traces back to a merge.</li></ul><p>The wiki itself is markdown on the serve host, in <code>.agent-serve/wiki/</code> by default with a <code>CODEBASE_WIKI_DIR</code> override. Sessions are disposable; the wiki is the durable state.</p><h2 id="follow-a-merged-pr" tabindex="-1">Follow a merged PR <a class="header-anchor" href="#follow-a-merged-pr" aria-label="Permalink to &quot;Follow a merged PR&quot;">​</a></h2><ol><li>GitHub delivers <code>pull_request</code> with action <code>closed</code>. The channel returns a task acknowledgement immediately.</li><li>The task fetches the digest with the host <code>gh</code> CLI: title, body, labels, changed files, and a bounded diff excerpt. No checkout.</li><li>The webhook payload can&#39;t say whether the PR merged, so the host checks <code>mergedAt</code> and skips abandoned PRs without a model turn.</li><li>For merged PRs, the task starts the turn with <code>pr/DIGEST.md</code> seeded through <code>workspaceFiles</code> and a <code>pr:&lt;owner/repo#N&gt;</code> continuation token, so redeliveries resume instead of double-ingesting.</li><li>The model follows <code>feature-mapping</code>: search, update or create feature pages, add changelog entries, and refresh <code>index</code> when pages were added.</li></ol><p>In chat, &quot;ingest PR #123&quot; runs the same flow through the <code>ingest_pr</code> tool, which writes the digest into the active session workspace.</p><h2 id="map-the-wiki-files" tabindex="-1">Map the wiki files <a class="header-anchor" href="#map-the-wiki-files" aria-label="Permalink to &quot;Map the wiki files&quot;">​</a></h2><table tabindex="0"><thead><tr><th>File</th><th>Purpose</th></tr></thead><tbody><tr><td><a href="../../examples/codebase-wiki/agent/agent.ts"><code>agent/agent.ts</code></a></td><td>Selects the cloud runtime and model.</td></tr><tr><td><a href="./../../examples/codebase-wiki/agent/instructions.html"><code>agent/instructions.md</code></a></td><td>Splits the job into merge ingestion and wiki-cited Q&amp;A.</td></tr><tr><td><a href="../../examples/codebase-wiki/agent/lib/wiki-store.ts"><code>agent/lib/wiki-store.ts</code></a></td><td>Enforces the rigid page tree and owns reads, writes, and search.</td></tr><tr><td><a href="../../examples/codebase-wiki/agent/lib/pr-digest.ts"><code>agent/lib/pr-digest.ts</code></a></td><td>Fetches PR metadata and diff, and formats <code>pr/DIGEST.md</code>.</td></tr><tr><td><a href="../../examples/codebase-wiki/agent/tools/ingest_pr.ts"><code>agent/tools/ingest_pr.ts</code></a></td><td>Exposes host digest preparation for chat-driven backfills.</td></tr><tr><td><a href="../../examples/codebase-wiki/agent/tools/wiki_read.ts"><code>agent/tools/wiki_read.ts</code></a>, <a href="../../examples/codebase-wiki/agent/tools/wiki_search.ts"><code>wiki_search.ts</code></a>, <a href="../../examples/codebase-wiki/agent/tools/wiki_write.ts"><code>wiki_write.ts</code></a></td><td>Read, search, and rewrite wiki pages.</td></tr><tr><td><a href="./../../examples/codebase-wiki/agent/skills/feature-mapping.html"><code>agent/skills/feature-mapping.md</code></a></td><td>Maps changes onto features and fixes the page and changelog shape.</td></tr><tr><td><a href="./../../examples/codebase-wiki/agent/schedules/daily-digest.html"><code>agent/schedules/daily-digest.md</code></a></td><td>Writes <code>digests/&lt;date&gt;</code>, rebuilds the index, and flags stale pages.</td></tr><tr><td><a href="../../examples/codebase-wiki/agent/channels/github.ts"><code>agent/channels/github.ts</code></a></td><td>Acknowledges closed PRs and starts merged-only ingest turns.</td></tr><tr><td><a href="../../examples/codebase-wiki/agent/storage.ts"><code>agent/storage.ts</code></a></td><td>Persists sessions and events with <code>cursorHostedStorage</code>.</td></tr><tr><td><a href="../../examples/codebase-wiki/evals/evals.config.ts"><code>evals/evals.config.ts</code></a></td><td>Caps eval run concurrency.</td></tr><tr><td><a href="../../examples/codebase-wiki/evals/ingest.eval.ts"><code>evals/ingest.eval.ts</code></a></td><td>Gates ingest decisions against the wiki filesystem.</td></tr></tbody></table><p>There is no MCP connection, subagent, hook, A/B experiment, or custom storage.</p><h2 id="prepare-credentials-and-services" tabindex="-1">Prepare credentials and services <a class="header-anchor" href="#prepare-credentials-and-services" aria-label="Permalink to &quot;Prepare credentials and services&quot;">​</a></h2><p>You need:</p><ul><li>Node 22.13 or newer.</li><li>An agent-runtime credential for model turns.</li><li><code>gh</code> on <code>PATH</code> with read access to the PRs you ingest.</li></ul><p>The channel verifies webhook signatures when <code>GITHUB_WEBHOOK_SECRET</code> is set and narrows repositories with <code>CODEBASE_WIKI_REPOS=owner/repo,owner/other</code>. The agent never writes to GitHub. Its only side effects are wiki files on the serve host.</p><h2 id="validate-the-surface" tabindex="-1">Validate the surface <a class="header-anchor" href="#validate-the-surface" aria-label="Permalink to &quot;Validate the surface&quot;">​</a></h2><div class="language-bash vp-adaptive-theme"><button title="Copy Code" class="copy"></button><span class="lang">bash</span><pre class="shiki shiki-themes github-light github-dark vp-code" tabindex="0"><code><span class="line"><span style="--shiki-light:#6F42C1;--shiki-dark:#B392F0;">agent-sdk</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> validate</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> --dir</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> examples/codebase-wiki</span></span>
2
2
  <span class="line"><span style="--shiki-light:#6F42C1;--shiki-dark:#B392F0;">agent-sdk</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> info</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> --dir</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> examples/codebase-wiki</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> --json</span></span></code></pre></div><p>The manifest should report four server tools, one skill, one schedule, and the authored GitHub channel.</p><h2 id="ingest-without-webhook-plumbing" tabindex="-1">Ingest without webhook plumbing <a class="header-anchor" href="#ingest-without-webhook-plumbing" aria-label="Permalink to &quot;Ingest without webhook plumbing&quot;">​</a></h2><p>Replay a real merged PR as a closed delivery:</p><div class="language-bash vp-adaptive-theme"><button title="Copy Code" class="copy"></button><span class="lang">bash</span><pre class="shiki shiki-themes github-light github-dark vp-code" tabindex="0"><code><span class="line"><span style="--shiki-light:#6F42C1;--shiki-dark:#B392F0;">agent-sdk</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> dev</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> examples/codebase-wiki</span></span>
3
3
  <span class="line"></span>
4
4
  <span class="line"><span style="--shiki-light:#6F42C1;--shiki-dark:#B392F0;">agent-sdk</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> github</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> replay</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> https://github.com/owner/repo/pull/123</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> \\</span></span>
@@ -1 +1 @@
1
- import{_ as s,c as a,o as t,ag as i}from"./chunks/framework.CAZyNGu9.js";const g=JSON.parse('{"title":"Build a feature wiki from merged pull requests","description":"Ingest every merged PR into per-feature wiki pages, consolidate with a daily digest schedule, and answer codebase questions with page citations.","frontmatter":{"title":"Build a feature wiki from merged pull requests","description":"Ingest every merged PR into per-feature wiki pages, consolidate with a daily digest schedule, and answer codebase questions with page citations."},"headers":[],"relativePath":"example-agents/codebase-wiki.md","filePath":"example-agents/codebase-wiki.md"}'),n={name:"example-agents/codebase-wiki.md"};function d(o,e,l,h,r,p){return t(),a("div",null,[...e[0]||(e[0]=[i("",41)])])}const k=s(n,[["render",d]]);export{g as __pageData,k as default};
1
+ import{_ as s,c as a,o as t,ag as i}from"./chunks/framework.CAZyNGu9.js";const g=JSON.parse('{"title":"Build a feature wiki from merged pull requests","description":"Ingest every merged PR into per-feature wiki pages, consolidate with a daily digest schedule, and answer codebase questions with page citations.","frontmatter":{"title":"Build a feature wiki from merged pull requests","description":"Ingest every merged PR into per-feature wiki pages, consolidate with a daily digest schedule, and answer codebase questions with page citations."},"headers":[],"relativePath":"example-agents/codebase-wiki.md","filePath":"example-agents/codebase-wiki.md"}'),n={name:"example-agents/codebase-wiki.md"};function d(o,e,l,r,h,c){return t(),a("div",null,[...e[0]||(e[0]=[i("",41)])])}const k=s(n,[["render",d]]);export{g as __pageData,k as default};