@cursor/july 0.1.112 → 0.1.114

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (305) hide show
  1. package/README.md +4 -0
  2. package/dist/bin/agent-serve.js +2 -3
  3. package/dist/channels/checks.d.ts +10 -0
  4. package/dist/channels/checks.d.ts.map +1 -1
  5. package/dist/channels/origin/checks.d.ts +1 -1
  6. package/dist/channels/origin/checks.d.ts.map +1 -1
  7. package/dist/channels/slack/dispatch.d.ts.map +1 -1
  8. package/dist/channels/slack/dispatch.js +6 -2
  9. package/dist/docs/404.html +2 -2
  10. package/dist/docs/assets/{app.DxTdhphC.js → app.BqkJwOZ-.js} +4 -4
  11. package/dist/docs/assets/chunks/@localSearchIndexroot.BnSgidYE.js +1 -0
  12. package/dist/docs/assets/chunks/{VPLocalSearchBox.CR3KTF0X.js → VPLocalSearchBox.BJAi2KiV.js} +1 -1
  13. package/dist/docs/assets/chunks/{arc.CVVqBOdS.js → arc.BZpXTgvV.js} +1 -1
  14. package/dist/docs/assets/chunks/{architectureDiagram-Q4EWVU46.CJHGP4ki.js → architectureDiagram-Q4EWVU46.WYI-7F-Y.js} +1 -1
  15. package/dist/docs/assets/chunks/{baseUniq.r7UVVRBP.js → baseUniq.CZaUPpg0.js} +1 -1
  16. package/dist/docs/assets/chunks/{blockDiagram-DXYQGD6D.DKmMaTre.js → blockDiagram-DXYQGD6D.D6UES2pD.js} +1 -1
  17. package/dist/docs/assets/chunks/{c4Diagram-AHTNJAMY.DDJsntUO.js → c4Diagram-AHTNJAMY.cwebIe4i.js} +1 -1
  18. package/dist/docs/assets/chunks/channel.DdM5EfNW.js +1 -0
  19. package/dist/docs/assets/chunks/{chunk-4BX2VUAB.BK2rKt6W.js → chunk-4BX2VUAB.fVyFnjxg.js} +1 -1
  20. package/dist/docs/assets/chunks/{chunk-4TB4RGXK.DRLV8RnF.js → chunk-4TB4RGXK.BanufG1c.js} +1 -1
  21. package/dist/docs/assets/chunks/{chunk-55IACEB6.DaKjxtb7.js → chunk-55IACEB6.VaSMz5-2.js} +1 -1
  22. package/dist/docs/assets/chunks/{chunk-EDXVE4YY.C5sPCIT1.js → chunk-EDXVE4YY.CN2diZOM.js} +1 -1
  23. package/dist/docs/assets/chunks/{chunk-FMBD7UC4.CSGWyNTB.js → chunk-FMBD7UC4.g4ivypu3.js} +1 -1
  24. package/dist/docs/assets/chunks/{chunk-OYMX7WX6.D5tK9XEr.js → chunk-OYMX7WX6.GZXKn9JJ.js} +1 -1
  25. package/dist/docs/assets/chunks/{chunk-QZHKN3VN.BeZGd1UZ.js → chunk-QZHKN3VN.itXxJZCd.js} +1 -1
  26. package/dist/docs/assets/chunks/{chunk-YZCP3GAM.U_tfWwQR.js → chunk-YZCP3GAM.-rw2GfvX.js} +1 -1
  27. package/dist/docs/assets/chunks/classDiagram-6PBFFD2Q.CjfGHeg2.js +1 -0
  28. package/dist/docs/assets/chunks/classDiagram-v2-HSJHXN6E.CjfGHeg2.js +1 -0
  29. package/dist/docs/assets/chunks/clone.wSOICb_f.js +1 -0
  30. package/dist/docs/assets/chunks/{cose-bilkent-S5V4N54A.DVeRXIb6.js → cose-bilkent-S5V4N54A.CmaI5br0.js} +1 -1
  31. package/dist/docs/assets/chunks/{dagre-KV5264BT.BpKJAeRZ.js → dagre-KV5264BT.4wY9S4Kt.js} +1 -1
  32. package/dist/docs/assets/chunks/{diagram-5BDNPKRD.BQOtrd1Z.js → diagram-5BDNPKRD.Pc3c0u9W.js} +1 -1
  33. package/dist/docs/assets/chunks/{diagram-G4DWMVQ6.CSDAhjPI.js → diagram-G4DWMVQ6.CYrWz-nj.js} +1 -1
  34. package/dist/docs/assets/chunks/{diagram-MMDJMWI5.Dpztst2S.js → diagram-MMDJMWI5.Bgj5hukb.js} +1 -1
  35. package/dist/docs/assets/chunks/{diagram-TYMM5635.qJHRizHR.js → diagram-TYMM5635.DGMEXalS.js} +1 -1
  36. package/dist/docs/assets/chunks/{erDiagram-SMLLAGMA.vbDotH3l.js → erDiagram-SMLLAGMA.GepTV9Im.js} +1 -1
  37. package/dist/docs/assets/chunks/{flowDiagram-DWJPFMVM.CqS_ZQr4.js → flowDiagram-DWJPFMVM.DVKywg3j.js} +1 -1
  38. package/dist/docs/assets/chunks/{ganttDiagram-T4ZO3ILL.DTLdR4pN.js → ganttDiagram-T4ZO3ILL.C7qt9Mlo.js} +1 -1
  39. package/dist/docs/assets/chunks/{gitGraphDiagram-UUTBAWPF.D04lnbnr.js → gitGraphDiagram-UUTBAWPF.U30_r82P.js} +1 -1
  40. package/dist/docs/assets/chunks/{graph.BlfqLJsM.js → graph.CyyMyAWv.js} +1 -1
  41. package/dist/docs/assets/chunks/{infoDiagram-42DDH7IO.tAooImWA.js → infoDiagram-42DDH7IO.Dn9ACW3y.js} +1 -1
  42. package/dist/docs/assets/chunks/{ishikawaDiagram-UXIWVN3A.ClUsVqnJ.js → ishikawaDiagram-UXIWVN3A.DlIdIGOA.js} +1 -1
  43. package/dist/docs/assets/chunks/{journeyDiagram-VCZTEJTY.C3tUgyCg.js → journeyDiagram-VCZTEJTY.DZj4vy4E.js} +1 -1
  44. package/dist/docs/assets/chunks/{kanban-definition-6JOO6SKY.CtD9-QCe.js → kanban-definition-6JOO6SKY.Dl63eMUV.js} +1 -1
  45. package/dist/docs/assets/chunks/{layout.D38U-LnT.js → layout.BLHZLWPH.js} +1 -1
  46. package/dist/docs/assets/chunks/{linear.BJmssyhN.js → linear.aXKGKaNw.js} +1 -1
  47. package/dist/docs/assets/chunks/{min.DNgXoouU.js → min.zWnFcpcc.js} +1 -1
  48. package/dist/docs/assets/chunks/{mindmap-definition-QFDTVHPH.Dcp6cxeu.js → mindmap-definition-QFDTVHPH.Qs4MQBea.js} +1 -1
  49. package/dist/docs/assets/chunks/{pieDiagram-DEJITSTG.CLDw6zIs.js → pieDiagram-DEJITSTG.BmPHgsk7.js} +1 -1
  50. package/dist/docs/assets/chunks/{quadrantDiagram-34T5L4WZ.CYaeeY4c.js → quadrantDiagram-34T5L4WZ.D5MQ3gwA.js} +1 -1
  51. package/dist/docs/assets/chunks/{requirementDiagram-MS252O5E.gMYuRpq2.js → requirementDiagram-MS252O5E.CkdUFrO7.js} +1 -1
  52. package/dist/docs/assets/chunks/{sankeyDiagram-XADWPNL6.CZqyHFbc.js → sankeyDiagram-XADWPNL6.KZrljrAV.js} +1 -1
  53. package/dist/docs/assets/chunks/{sequenceDiagram-FGHM5R23.BTsCjUDN.js → sequenceDiagram-FGHM5R23.XMoEW-Lx.js} +1 -1
  54. package/dist/docs/assets/chunks/{stateDiagram-FHFEXIEX.CftT9mLJ.js → stateDiagram-FHFEXIEX.BmTzePLj.js} +1 -1
  55. package/dist/docs/assets/chunks/stateDiagram-v2-QKLJ7IA2.Cu5X28zZ.js +1 -0
  56. package/dist/docs/assets/chunks/{theme.B_7J9ZsV.js → theme.BfQzpxsg.js} +2 -2
  57. package/dist/docs/assets/chunks/{timeline-definition-GMOUNBTQ.DbU3WUNw.js → timeline-definition-GMOUNBTQ.Dug0oamp.js} +1 -1
  58. package/dist/docs/assets/chunks/{vennDiagram-DHZGUBPP.ixsq-q2u.js → vennDiagram-DHZGUBPP.BOTHrEFu.js} +1 -1
  59. package/dist/docs/assets/chunks/wardley-RL74JXVD.DXy2i1LS.js +162 -0
  60. package/dist/docs/assets/chunks/{wardleyDiagram-NUSXRM2D.C0ewvgbp.js → wardleyDiagram-NUSXRM2D.CoXKdfi6.js} +1 -1
  61. package/dist/docs/assets/chunks/{xychartDiagram-5P7HB3ND.nAEhF4bO.js → xychartDiagram-5P7HB3ND.DXoSCjAW.js} +1 -1
  62. package/dist/docs/assets/{deployment.md.D2jQZuFx.js → deployment.md.D2YX7u_I.js} +1 -1
  63. package/dist/docs/assets/{evals.md.BYvfZ-PO.js → evals.md.D3Y3Aixt.js} +2 -2
  64. package/dist/docs/assets/{evals.md.BYvfZ-PO.lean.js → evals.md.D3Y3Aixt.lean.js} +1 -1
  65. package/dist/docs/assets/guides_agent-to-agent.md.C6kPY8nu.js +41 -0
  66. package/dist/docs/assets/guides_agent-to-agent.md.C6kPY8nu.lean.js +1 -0
  67. package/dist/docs/assets/{guides_cloud-agents.md.Cp1O3u-X.js → guides_cloud-agents.md.BPJqTZjT.js} +1 -1
  68. package/dist/docs/assets/{guides_grokbot-agents.md.CMhZNdEU.js → guides_grokbot-agents.md.CzV715v8.js} +1 -1
  69. package/dist/docs/assets/guides_hooks.md.BT9GLwEp.js +50 -0
  70. package/dist/docs/assets/guides_hooks.md.BT9GLwEp.lean.js +1 -0
  71. package/dist/docs/assets/guides_jev.md.DeSCqMaO.js +151 -0
  72. package/dist/docs/assets/guides_jev.md.DeSCqMaO.lean.js +1 -0
  73. package/dist/docs/assets/reference_agent-config.md.BRxAlnRy.js +36 -0
  74. package/dist/docs/assets/{reference_agent-config.md.DGPyw7ms.lean.js → reference_agent-config.md.BRxAlnRy.lean.js} +1 -1
  75. package/dist/docs/assets/reference_artifacts.md.KRX0sAdt.js +18 -0
  76. package/dist/docs/assets/reference_artifacts.md.KRX0sAdt.lean.js +1 -0
  77. package/dist/docs/assets/reference_channels.md.DZr14vm7.js +23 -0
  78. package/dist/docs/assets/reference_channels.md.DZr14vm7.lean.js +1 -0
  79. package/dist/docs/assets/{reference_connections.md.CmyrlXfY.js → reference_connections.md.DJGUCxrr.js} +18 -30
  80. package/dist/docs/assets/{reference_connections.md.CmyrlXfY.lean.js → reference_connections.md.DJGUCxrr.lean.js} +1 -1
  81. package/dist/docs/assets/{reference_evals.md.DNJzM_yf.js → reference_evals.md.C6umwNC6.js} +6 -7
  82. package/dist/docs/assets/reference_evals.md.C6umwNC6.lean.js +1 -0
  83. package/dist/docs/assets/{reference_extensions.md.Ceq-qT8d.js → reference_extensions.md.DbNYu-DP.js} +3 -3
  84. package/dist/docs/assets/{reference_extensions.md.Ceq-qT8d.lean.js → reference_extensions.md.DbNYu-DP.lean.js} +1 -1
  85. package/dist/docs/assets/reference_hooks.md.BfOkhTU0.js +45 -0
  86. package/dist/docs/assets/{reference_hooks.md.B7uzNENk.lean.js → reference_hooks.md.BfOkhTU0.lean.js} +1 -1
  87. package/dist/docs/assets/reference_http-api.md.DdwtBeCj.js +11 -0
  88. package/dist/docs/assets/{reference_http-api.md.CduHavZ2.lean.js → reference_http-api.md.DdwtBeCj.lean.js} +1 -1
  89. package/dist/docs/assets/reference_instructions.md.B2mcIzT6.js +14 -0
  90. package/dist/docs/assets/reference_instructions.md.B2mcIzT6.lean.js +1 -0
  91. package/dist/docs/assets/reference_playground.md.CyrQD_n3.js +1 -0
  92. package/dist/docs/assets/reference_playground.md.CyrQD_n3.lean.js +1 -0
  93. package/dist/docs/assets/reference_project-layout.md.BEMzxAkq.js +19 -0
  94. package/dist/docs/assets/{reference_project-layout.md.BGhgpy9V.lean.js → reference_project-layout.md.BEMzxAkq.lean.js} +1 -1
  95. package/dist/docs/assets/reference_prompt.md.BFrqjHFL.js +9 -0
  96. package/dist/docs/assets/reference_prompt.md.BFrqjHFL.lean.js +1 -0
  97. package/dist/docs/assets/reference_schedules.md.BB9N3tRR.js +47 -0
  98. package/dist/docs/assets/reference_schedules.md.BB9N3tRR.lean.js +1 -0
  99. package/dist/docs/assets/reference_sessions.md.BBp-GIt-.js +1 -0
  100. package/dist/docs/assets/{reference_sessions.md.1_6Vyv7x.lean.js → reference_sessions.md.BBp-GIt-.lean.js} +1 -1
  101. package/dist/docs/assets/reference_skills.md.BVmi3UJ_.js +15 -0
  102. package/dist/docs/assets/{reference_skills.md.DjQkRefx.lean.js → reference_skills.md.BVmi3UJ_.lean.js} +1 -1
  103. package/dist/docs/assets/reference_subagents.md.DRoRy2Uj.js +10 -0
  104. package/dist/docs/assets/{reference_subagents.md.BHsSMMyO.lean.js → reference_subagents.md.DRoRy2Uj.lean.js} +1 -1
  105. package/dist/docs/assets/{reference_tools.md.BYzUTeVA.js → reference_tools.md.CgocLDX1.js} +10 -7
  106. package/dist/docs/assets/{reference_tools.md.BYzUTeVA.lean.js → reference_tools.md.CgocLDX1.lean.js} +1 -1
  107. package/dist/docs/assets/troubleshooting.md.HY95rCCz.js +1 -0
  108. package/dist/docs/building-with-agents.html +35 -35
  109. package/dist/docs/deployment.html +37 -37
  110. package/dist/docs/deployment.md +1 -1
  111. package/dist/docs/evals.html +36 -36
  112. package/dist/docs/evals.md +3 -0
  113. package/dist/docs/guides/agent-to-agent.html +65 -54
  114. package/dist/docs/guides/agent-to-agent.md +73 -69
  115. package/dist/docs/guides/bitbucket.html +35 -35
  116. package/dist/docs/guides/cloud-agents.html +36 -36
  117. package/dist/docs/guides/cloud-agents.md +1 -1
  118. package/dist/docs/guides/convert-automation.html +35 -35
  119. package/dist/docs/guides/github.html +35 -35
  120. package/dist/docs/guides/gitlab.html +35 -35
  121. package/dist/docs/guides/grokbot-agents.html +37 -37
  122. package/dist/docs/guides/grokbot-agents.md +1 -1
  123. package/dist/docs/guides/hooks.html +109 -0
  124. package/dist/docs/guides/hooks.md +111 -0
  125. package/dist/docs/guides/improve.html +36 -36
  126. package/dist/docs/guides/jev.html +210 -0
  127. package/dist/docs/guides/jev.md +291 -0
  128. package/dist/docs/guides/mcp-oauth.html +36 -36
  129. package/dist/docs/guides/opentelemetry.html +35 -35
  130. package/dist/docs/guides/slack.html +35 -35
  131. package/dist/docs/guides/webhooks.html +35 -35
  132. package/dist/docs/hashmap.json +1 -1
  133. package/dist/docs/hillclimbing.html +35 -35
  134. package/dist/docs/index.html +35 -35
  135. package/dist/docs/llms-full.txt +1338 -1282
  136. package/dist/docs/llms.txt +9 -7
  137. package/dist/docs/quickstart.html +35 -35
  138. package/dist/docs/reference/agent-config.html +42 -46
  139. package/dist/docs/reference/agent-config.md +48 -81
  140. package/dist/docs/reference/artifacts.html +39 -40
  141. package/dist/docs/reference/artifacts.md +71 -70
  142. package/dist/docs/reference/channels.html +41 -61
  143. package/dist/docs/reference/channels.md +134 -201
  144. package/dist/docs/reference/cli.html +35 -35
  145. package/dist/docs/reference/connections.html +54 -66
  146. package/dist/docs/reference/connections.md +98 -134
  147. package/dist/docs/reference/evals.html +42 -43
  148. package/dist/docs/reference/evals.md +42 -50
  149. package/dist/docs/reference/extensions.html +39 -39
  150. package/dist/docs/reference/extensions.md +10 -13
  151. package/dist/docs/reference/hooks.html +39 -67
  152. package/dist/docs/reference/hooks.md +72 -146
  153. package/dist/docs/reference/http-api.html +39 -39
  154. package/dist/docs/reference/http-api.md +137 -161
  155. package/dist/docs/reference/instructions.html +39 -39
  156. package/dist/docs/reference/instructions.md +21 -36
  157. package/dist/docs/reference/playground.html +36 -36
  158. package/dist/docs/reference/playground.md +26 -43
  159. package/dist/docs/reference/project-layout.html +38 -38
  160. package/dist/docs/reference/project-layout.md +12 -17
  161. package/dist/docs/reference/prompt.html +42 -42
  162. package/dist/docs/reference/prompt.md +18 -13
  163. package/dist/docs/reference/schedules.html +56 -91
  164. package/dist/docs/reference/schedules.md +52 -99
  165. package/dist/docs/reference/sessions.html +36 -36
  166. package/dist/docs/reference/sessions.md +36 -40
  167. package/dist/docs/reference/skills.html +38 -38
  168. package/dist/docs/reference/skills.md +15 -26
  169. package/dist/docs/reference/subagents.html +38 -38
  170. package/dist/docs/reference/subagents.md +21 -31
  171. package/dist/docs/reference/tools.html +45 -42
  172. package/dist/docs/reference/tools.md +49 -64
  173. package/dist/docs/templates/agentic-owners.html +35 -35
  174. package/dist/docs/templates/pr-autofixer.html +35 -35
  175. package/dist/docs/templates/security-reviewer.html +35 -35
  176. package/dist/docs/templates/thermo-quality-review.html +35 -35
  177. package/dist/docs/templates/thermo-review.html +35 -35
  178. package/dist/docs/templates/triage.html +35 -35
  179. package/dist/docs/troubleshooting.html +36 -36
  180. package/dist/docs/troubleshooting.md +1 -1
  181. package/dist/extensions/jev/extension.d.ts +43 -0
  182. package/dist/extensions/jev/extension.d.ts.map +1 -0
  183. package/dist/extensions/jev/extension.js +47 -0
  184. package/dist/extensions/jev/lib/evaluate.d.ts +101 -0
  185. package/dist/extensions/jev/lib/evaluate.d.ts.map +1 -0
  186. package/dist/extensions/jev/lib/evaluate.js +167 -0
  187. package/dist/extensions/jev/skills/gated-write.md +25 -0
  188. package/dist/extensions/jev/skills/questions.md +33 -0
  189. package/dist/extensions/jev/tools/evaluate.d.ts +4 -0
  190. package/dist/extensions/jev/tools/evaluate.d.ts.map +1 -0
  191. package/dist/extensions/jev/tools/evaluate.js +88 -0
  192. package/dist/extensions.d.ts +1 -1
  193. package/dist/extensions.d.ts.map +1 -1
  194. package/dist/extensions.js +2 -0
  195. package/dist/internal/advertise-tools.d.ts.map +1 -1
  196. package/dist/internal/advertise-tools.js +6 -0
  197. package/dist/internal/discovery/connections.d.ts.map +1 -1
  198. package/dist/internal/discovery/connections.js +18 -0
  199. package/dist/internal/discovery/extensions.d.ts.map +1 -1
  200. package/dist/internal/discovery/extensions.js +8 -4
  201. package/dist/internal/discovery/info.d.ts.map +1 -1
  202. package/dist/internal/discovery/info.js +1 -0
  203. package/dist/internal/hosted-delivery-protocol.d.ts +3 -0
  204. package/dist/internal/hosted-delivery-protocol.d.ts.map +1 -1
  205. package/dist/internal/hosted-delivery-protocol.js +1 -0
  206. package/dist/internal/hosted-delivery.d.ts.map +1 -1
  207. package/dist/internal/hosted-delivery.js +15 -25
  208. package/dist/internal/hosted-execution-diag.d.ts +12 -4
  209. package/dist/internal/hosted-execution-diag.d.ts.map +1 -1
  210. package/dist/internal/hosted-execution-diag.js +26 -4
  211. package/dist/internal/hosted-execution-flush.d.ts +1 -0
  212. package/dist/internal/hosted-execution-flush.d.ts.map +1 -1
  213. package/dist/internal/hosted-execution-flush.js +4 -2
  214. package/dist/internal/server.d.ts.map +1 -1
  215. package/dist/internal/server.js +17 -9
  216. package/dist/internal/session-engine.d.ts +4 -1
  217. package/dist/internal/session-engine.d.ts.map +1 -1
  218. package/dist/internal/session-engine.js +28 -4
  219. package/dist/playground/assets/index-C61EWMBK.css +1 -0
  220. package/dist/playground/assets/{index-B1c1LeIf.js → index-CrMWlgUU.js} +43 -43
  221. package/dist/playground/index.html +2 -2
  222. package/dist/types.d.ts +23 -3
  223. package/dist/types.d.ts.map +1 -1
  224. package/docs/deployment.md +1 -1
  225. package/docs/evals.md +3 -0
  226. package/docs/guides/agent-to-agent.md +74 -70
  227. package/docs/guides/cloud-agents.md +1 -1
  228. package/docs/guides/grokbot-agents.md +1 -1
  229. package/docs/guides/hooks.md +116 -0
  230. package/docs/guides/jev.md +296 -0
  231. package/docs/reference/agent-config.md +48 -81
  232. package/docs/reference/artifacts.md +72 -71
  233. package/docs/reference/channels.md +135 -202
  234. package/docs/reference/connections.md +99 -135
  235. package/docs/reference/evals.md +43 -51
  236. package/docs/reference/extensions.md +10 -13
  237. package/docs/reference/hooks.md +72 -146
  238. package/docs/reference/http-api.md +137 -161
  239. package/docs/reference/instructions.md +22 -37
  240. package/docs/reference/playground.md +26 -43
  241. package/docs/reference/project-layout.md +12 -17
  242. package/docs/reference/prompt.md +20 -15
  243. package/docs/reference/schedules.md +52 -99
  244. package/docs/reference/sessions.md +36 -40
  245. package/docs/reference/skills.md +15 -26
  246. package/docs/reference/subagents.md +21 -31
  247. package/docs/reference/tools.md +49 -64
  248. package/docs/troubleshooting.md +1 -1
  249. package/package.json +8 -1
  250. package/src/bin/agent-serve.ts +2 -3
  251. package/src/channels/checks.ts +8 -0
  252. package/src/channels/origin/checks.ts +3 -1
  253. package/src/channels/slack/dispatch.ts +7 -2
  254. package/src/extensions/jev/extension.ts +95 -0
  255. package/src/extensions/jev/lib/evaluate.ts +289 -0
  256. package/src/extensions/jev/skills/gated-write.md +25 -0
  257. package/src/extensions/jev/skills/questions.md +33 -0
  258. package/src/extensions/jev/tools/evaluate.ts +90 -0
  259. package/src/extensions.ts +2 -0
  260. package/src/internal/advertise-tools.ts +6 -0
  261. package/src/internal/discovery/connections.ts +21 -0
  262. package/src/internal/discovery/extensions.ts +12 -4
  263. package/src/internal/discovery/info.ts +1 -0
  264. package/src/internal/hosted-delivery-protocol.ts +4 -0
  265. package/src/internal/hosted-delivery.ts +15 -0
  266. package/src/internal/hosted-execution-diag.ts +33 -4
  267. package/src/internal/hosted-execution-flush.ts +4 -0
  268. package/src/internal/server.ts +26 -12
  269. package/src/internal/session-engine.ts +30 -4
  270. package/src/types.ts +24 -3
  271. package/dist/docs/assets/chunks/@localSearchIndexroot.QmjDU6Jh.js +0 -1
  272. package/dist/docs/assets/chunks/channel.BjpoSbz_.js +0 -1
  273. package/dist/docs/assets/chunks/classDiagram-6PBFFD2Q.BgxOlMHw.js +0 -1
  274. package/dist/docs/assets/chunks/classDiagram-v2-HSJHXN6E.BgxOlMHw.js +0 -1
  275. package/dist/docs/assets/chunks/clone.DRuGBKZC.js +0 -1
  276. package/dist/docs/assets/chunks/stateDiagram-v2-QKLJ7IA2.-43J68xB.js +0 -1
  277. package/dist/docs/assets/chunks/wardley-RL74JXVD.WRXz-Dux.js +0 -162
  278. package/dist/docs/assets/guides_agent-to-agent.md.8oDTfu-E.js +0 -30
  279. package/dist/docs/assets/guides_agent-to-agent.md.8oDTfu-E.lean.js +0 -1
  280. package/dist/docs/assets/reference_agent-config.md.DGPyw7ms.js +0 -40
  281. package/dist/docs/assets/reference_artifacts.md.Bu_4HmsD.js +0 -19
  282. package/dist/docs/assets/reference_artifacts.md.Bu_4HmsD.lean.js +0 -1
  283. package/dist/docs/assets/reference_channels.md.nFWbzAic.js +0 -43
  284. package/dist/docs/assets/reference_channels.md.nFWbzAic.lean.js +0 -1
  285. package/dist/docs/assets/reference_evals.md.DNJzM_yf.lean.js +0 -1
  286. package/dist/docs/assets/reference_hooks.md.B7uzNENk.js +0 -73
  287. package/dist/docs/assets/reference_http-api.md.CduHavZ2.js +0 -11
  288. package/dist/docs/assets/reference_instructions.md.CU1My5My.js +0 -14
  289. package/dist/docs/assets/reference_instructions.md.CU1My5My.lean.js +0 -1
  290. package/dist/docs/assets/reference_playground.md.Ch2d0Iqi.js +0 -1
  291. package/dist/docs/assets/reference_playground.md.Ch2d0Iqi.lean.js +0 -1
  292. package/dist/docs/assets/reference_project-layout.md.BGhgpy9V.js +0 -19
  293. package/dist/docs/assets/reference_prompt.md.Ccp0R53H.js +0 -1
  294. package/dist/docs/assets/reference_prompt.md.Ccp0R53H.lean.js +0 -1
  295. package/dist/docs/assets/reference_schedules.md.B2Nm6FaD.js +0 -82
  296. package/dist/docs/assets/reference_schedules.md.B2Nm6FaD.lean.js +0 -1
  297. package/dist/docs/assets/reference_sessions.md.1_6Vyv7x.js +0 -1
  298. package/dist/docs/assets/reference_skills.md.DjQkRefx.js +0 -15
  299. package/dist/docs/assets/reference_subagents.md.BHsSMMyO.js +0 -10
  300. package/dist/docs/assets/troubleshooting.md.mnfFG2Em.js +0 -1
  301. package/dist/playground/assets/index-CK2LX3iD.css +0 -1
  302. /package/dist/docs/assets/{deployment.md.D2jQZuFx.lean.js → deployment.md.D2YX7u_I.lean.js} +0 -0
  303. /package/dist/docs/assets/{guides_cloud-agents.md.Cp1O3u-X.lean.js → guides_cloud-agents.md.BPJqTZjT.lean.js} +0 -0
  304. /package/dist/docs/assets/{guides_grokbot-agents.md.CMhZNdEU.lean.js → guides_grokbot-agents.md.CzV715v8.lean.js} +0 -0
  305. /package/dist/docs/assets/{troubleshooting.md.mnfFG2Em.lean.js → troubleshooting.md.HY95rCCz.lean.js} +0 -0
@@ -331,7 +331,7 @@ per-caller and channel-specific auth.
331
331
 
332
332
  Use `agent-sdk logs --prod` for runtime output,
333
333
  [OpenTelemetry](/docs/guides/opentelemetry.md) for traces and metrics, and
334
- [session traces](/docs/reference/sessions.md#how-do-i-inspect-a-saved-event-stream)
334
+ [session traces](/docs/reference/sessions.md#inspect-a-saved-event-stream)
335
335
  for one conversation.
336
336
 
337
337
  ## Related
@@ -366,6 +366,9 @@ This smoke case asks for a PR verdict, requires the read-only inspection
366
366
  tool, and fails if the agent tries to approve. The CLI reports all three
367
367
  decisions together instead of stopping at the first miss.
368
368
 
369
+ Use the [Jev extension](/docs/guides/jev.md) when host code needs a typed
370
+ choice, score, or boolean before it writes.
371
+
369
372
  ```ts
370
373
  // evals/readiness.eval.ts
371
374
  import { defineEval, includes } from "@cursor/july/evals";
@@ -606,32 +609,21 @@ moves the freeze line instead of proving the change.
606
609
 
607
610
  Source: /docs/guides/agent-to-agent.md
608
611
 
609
- # Agent-to-agent
612
+ # Peer agents
610
613
 
611
614
  A peer connection lets one agent ask another agent on the same host for
612
615
  help. The specialist keeps its own instructions, tools, and session,
613
616
  while the caller decides when to delegate and returns the answer to the
614
- user.
615
-
616
- ## How do I wire two agents?
617
+ user. Co-host both projects under one `agent-sdk` process.
617
618
 
618
- Put both projects under one directory, then point the concierge at the
619
- specialist's mount slug. When a user asks about weather, the concierge
620
- calls the peer's `ask` tool, the weather agent runs in its own context,
621
- and the concierge returns that answer.
619
+ ## Delegate a question to a specialist
622
620
 
623
- ```text
624
- agents/
625
- concierge/
626
- agent/instructions.md
627
- agent/mcp-connections/weather.ts
628
- weather-agent/
629
- agent/agent.ts
630
- agent/instructions.md
631
- ```
621
+ Ask the concierge about the weather and it hands the question to the
622
+ specialist. The weather agent answers in its own context, then the
623
+ concierge returns that answer to you.
632
624
 
633
625
  ```ts
634
- // agents/concierge/agent/mcp-connections/weather.ts
626
+ // agent/mcp-connections/weather.ts
635
627
  import { defineConnection } from "@cursor/july/connections";
636
628
 
637
629
  export default defineConnection({
@@ -644,7 +636,7 @@ Give the concierge a narrow routing rule:
644
636
 
645
637
  ```md
646
638
  When a request needs current weather, ask the `weather` peer. Return its
647
- answer without delegating the same request again.
639
+ answer.
648
640
  ```
649
641
 
650
642
  ```bash
@@ -653,58 +645,69 @@ agent-sdk chat --url http://127.0.0.1:3000/concierge \
653
645
  --message "What's the weather in Paris right now?"
654
646
  ```
655
647
 
656
- The connection filename becomes the MCP server name. The `agent` value
657
- is the specialist's mount slug, normally its directory name.
648
+ The connection filename is the MCP server name the concierge calls, and
649
+ `agent` is the specialist's mount slug, normally its directory name.
658
650
 
659
- ## Continue the specialist's conversation
651
+ ## Continue the specialist's session
660
652
 
661
- The first `ask` returns a peer `sessionId`. Pass that ID into the next
662
- `ask` when the specialist should remember its earlier work instead of
663
- starting with fresh context.
653
+ Ask a follow-up about the same forecast and the specialist picks up
654
+ where it left off, with its own instructions and tool history still in
655
+ place. The first `ask` returns `{ status, sessionId, reply }` as text on
656
+ the `callTool` result. Parse `sessionId` and pass it on the next `ask`
657
+ so the peer remembers its earlier work instead of starting fresh.
664
658
 
665
- ```json
666
- {
667
- "message": "How does that compare with tomorrow?",
668
- "sessionId": "<session-id>"
659
+ ```ts
660
+ const first = await ctx.host.mcp.callTool("weather", "ask", {
661
+ message: "What's the weather in Paris right now?",
662
+ });
663
+ const firstText = first.content.find((part) => part.type === "text");
664
+ if (firstText?.text === undefined) {
665
+ throw new Error("weather ask returned no text payload");
669
666
  }
670
- ```
671
-
672
- The peer owns this session. Follow-ups resume its instructions and tool
673
- history without merging them into the concierge's conversation.
667
+ const { sessionId } = JSON.parse(firstText.text) as { sessionId: string };
674
668
 
675
- ## Wait for long-running specialist work
669
+ await ctx.host.mcp.callTool("weather", "ask", {
670
+ message: "How does that compare with tomorrow?",
671
+ sessionId,
672
+ });
673
+ ```
676
674
 
677
- `ask` waits for a bounded time. If the specialist is still working, it
678
- returns `status: "running"` with the session ID, allowing the caller to
679
- do other work and check again later.
675
+ If the specialist is still working, `ask` returns `status: "running"`
676
+ with the session ID, and you can do other work before calling `check`.
680
677
 
681
- ```json
682
- {
683
- "sessionId": "<session-id>",
684
- "waitSeconds": 20
685
- }
678
+ ```ts
679
+ await ctx.host.mcp.callTool("weather", "check", {
680
+ sessionId,
681
+ waitSeconds: 20,
682
+ });
686
683
  ```
687
684
 
688
- Call the peer's `check` tool with that input. It returns the final reply
689
- when the turn settles, or another running status when more time is
690
- needed.
685
+ The peer owns this session, so follow-ups continue the specialist's
686
+ context.
691
687
 
692
688
  ## Call a deterministic peer tool
693
689
 
694
- When the specialist has a server tool and no second model turn should
695
- make a decision, call its `call_tool` endpoint through the peer
696
- connection. This example runs `get_forecast` in the weather agent's
697
- workspace and returns its structured result to the concierge tool.
690
+ You need a forecast for Paris, and the weather agent already has
691
+ `get_forecast`. Host code calls that tool in the specialist's workspace
692
+ and the forecast comes back to the concierge, with no second model turn.
698
693
 
699
694
  ```ts
700
695
  const result = await ctx.host.mcp.callTool("weather", "call_tool", {
701
696
  toolName: "get_forecast",
702
697
  input: { city: "Paris" },
703
698
  });
699
+ const part = result.content.find((item) => item.type === "text");
700
+ if (part?.text === undefined) {
701
+ return { ok: false, forecast: null };
702
+ }
703
+ const payload = JSON.parse(part.text) as {
704
+ isError?: boolean;
705
+ result?: unknown;
706
+ };
704
707
 
705
708
  return {
706
- ok: result.isError !== true,
707
- forecast: result.structuredContent ?? null,
709
+ ok: result.isError !== true && payload.isError !== true,
710
+ forecast: payload.result ?? null,
708
711
  };
709
712
  ```
710
713
 
@@ -723,27 +726,31 @@ complete agent you would also run, inspect, or expose on its own.
723
726
  | Tools and connections | Inherits the parent project | Owns its project surface |
724
727
  | Reachability | Parent only | Other co-hosted agents and MCP clients |
725
728
 
726
- ## Peer sessions
729
+ ## Affinity
727
730
 
728
- Calls through `ask` create sessions on the peer's `mcp` channel. They
729
- use the caller's authenticated principal and appear in the peer's
730
- playground, so the same authorization boundary applies to starts,
731
- follow-ups, and checks.
731
+ Those turns show up in the peer's playground. Only the session owner
732
+ can continue or check them.
732
733
 
733
- There is no automatic recursion guard between peers. Give each caller a
734
- one-way routing rule so agent A cannot delegate the same work to B and
735
- receive it back from B.
734
+ ## Practices
736
735
 
737
- ## Run peers on one host
736
+ - Give each caller a one-way routing rule, and don't configure reciprocal
737
+ routes for the same request.
738
738
 
739
- Peers require the default multi-agent layout and one shared `serve`
740
- process. Unknown slugs and self-references fail at startup.
741
- Cursor-managed hosting deploys one agent per slug, so use subagents
742
- there instead of peer connections.
739
+ - Prefer `ask` when the specialist should reason. Use `call_tool` when
740
+ host code already knows which server tool to run.
743
741
 
744
- For a self-hosted cloud-runtime caller, expose the shared process with
745
- `--public-url` and protect it with `--bearer-token`. The
746
- [Deployment guide](/docs/deployment.md) owns that hosting setup.
742
+ - Cursor-managed hosting deploys one agent per slug. Use
743
+ [subagents](/docs/reference/subagents.md) there instead of peer
744
+ connections.
745
+
746
+ ## Other hosts
747
+
748
+ Peers require the default multi-agent layout; unknown slugs and
749
+ self-references fail at startup.
750
+
751
+ If a cloud-runtime turn needs a self-hosted peer, set `--public-url` to
752
+ the shared host's reachable URL and protect it with `--bearer-token`.
753
+ The [Deployment guide](/docs/deployment.md) owns that hosting setup.
747
754
 
748
755
  ## Related
749
756
 
@@ -1056,7 +1063,7 @@ agent, keep talking while it works, inspect the result, and steer or
1056
1063
  stop the same branch.
1057
1064
 
1058
1065
  This is different from choosing
1059
- [`runtime: "cloud"`](/docs/reference/agent-config.md#choose-a-runtime).
1066
+ [`runtime: "cloud"`](/docs/reference/agent-config.md#runtime).
1060
1067
  That setting moves this agent's turns to a cloud VM; this extension lets
1061
1068
  the agent delegate separate coding tasks.
1062
1069
 
@@ -2153,11 +2160,127 @@ directory and disable that tool. See
2153
2160
  overlays
2154
2161
  - [Tool approvals](/docs/reference/tools.md#gate-a-tool-on-human-approval):
2155
2162
  approve asks
2156
- - [Agent config](/docs/reference/agent-config.md#choose-a-runtime): local
2163
+ - [Agent config](/docs/reference/agent-config.md#runtime): local
2157
2164
  and hosted runtimes
2158
2165
 
2159
2166
  ---
2160
2167
 
2168
+ Source: /docs/guides/hooks.md
2169
+
2170
+ # Hooks
2171
+
2172
+ Hooks observe session events after they are recorded and run side effects such
2173
+ as updating metrics or sending alerts. They never change the turn, prompt, or
2174
+ reply. Author them under `agent/hooks/` with `defineHook` from
2175
+ `@cursor/july/hooks`.
2176
+
2177
+ ## Meter token usage
2178
+
2179
+ With [OpenTelemetry](/docs/guides/opentelemetry.md) configured, this hook adds a live
2180
+ turn's reported input and output tokens to your counters. Eval runs stay quiet,
2181
+ so the dashboard reflects live traffic instead of the test suite.
2182
+
2183
+ ```ts
2184
+ // agent/hooks/usage.ts
2185
+ import { defineHook } from "@cursor/july/hooks";
2186
+
2187
+ export default defineHook({
2188
+ events: {
2189
+ async "turn.completed"(event, ctx) {
2190
+ if (ctx.session.purpose === "eval") {
2191
+ return;
2192
+ }
2193
+ if (event.data.usage === undefined) {
2194
+ return;
2195
+ }
2196
+
2197
+ const { inputTokens, outputTokens } = event.data.usage;
2198
+ ctx.host.otel.increment("acme.tokens.input", inputTokens);
2199
+ ctx.host.otel.increment("acme.tokens.output", outputTokens);
2200
+ },
2201
+ },
2202
+ });
2203
+ ```
2204
+
2205
+ ## Alert on failure
2206
+
2207
+ Set `PAGER_WEBHOOK_URL` to your pager's webhook. When a live turn fails, this
2208
+ hook posts the agent, session, and failure message there. Eval runs and
2209
+ interrupted turns stay quiet, so tests and preemptions do not page anyone.
2210
+
2211
+ ```ts
2212
+ // agent/hooks/page-on-failure.ts
2213
+ import { defineHook } from "@cursor/july/hooks";
2214
+
2215
+ export default defineHook({
2216
+ events: {
2217
+ async "turn.failed"(event, ctx) {
2218
+ if (ctx.session.purpose === "eval") {
2219
+ return;
2220
+ }
2221
+ if (event.data.message === "turn interrupted") {
2222
+ return;
2223
+ }
2224
+
2225
+ const pagerUrl = process.env.PAGER_WEBHOOK_URL;
2226
+ if (pagerUrl === undefined) {
2227
+ return;
2228
+ }
2229
+
2230
+ await fetch(pagerUrl, {
2231
+ method: "POST",
2232
+ headers: { "content-type": "application/json" },
2233
+ body: JSON.stringify({
2234
+ agent: ctx.agent.name,
2235
+ session: ctx.session.id,
2236
+ channel: ctx.channel.id,
2237
+ message: event.data.message,
2238
+ }),
2239
+ signal: AbortSignal.timeout(5_000),
2240
+ });
2241
+ },
2242
+ },
2243
+ });
2244
+ ```
2245
+
2246
+ ## Events / when hooks run
2247
+
2248
+ Use event names from the
2249
+ [session event vocabulary](/docs/reference/sessions.md#stream-events). A hook
2250
+ receives each matching event after it is recorded, and the model does not wait
2251
+ for the handler.
2252
+
2253
+ Within one session, handlers run one at a time. A slow handler delays later
2254
+ handlers for that session, but it does not delay the model or handlers for
2255
+ other sessions. Hooks also fire for evals, so check
2256
+ `ctx.session.purpose === "eval"` before metering or paging. A restart does not
2257
+ replay recorded events into hooks.
2258
+
2259
+ ## When not to use a hook
2260
+
2261
+ | Want | Use instead |
2262
+ | --- | --- |
2263
+ | Add context before the model | `instructions.md`, skills, or `workspaceFiles` |
2264
+ | Deliver to Slack or a PR | Channel [`events`](/docs/reference/channels.md#events) or packs |
2265
+ | Block or approve a tool | [`needsApproval`](/docs/reference/tools.md#gate-a-tool-on-human-approval) |
2266
+ | Gate final assistant text | `defineResult` |
2267
+ | Gate behavior | [Evals](/docs/evals.md) |
2268
+
2269
+ [Cursor Agent hooks](https://cursor.com/docs/agent/hooks) in
2270
+ `.cursor/hooks.json` are a different product. They can observe, block, or
2271
+ modify the local agent loop.
2272
+
2273
+ ## Related
2274
+
2275
+ - [Hooks reference](/docs/reference/hooks.md): payloads, context, and discovery
2276
+ - [Sessions: stream events](/docs/reference/sessions.md#stream-events): event
2277
+ vocabulary and payload sequence
2278
+ - [OpenTelemetry](/docs/guides/opentelemetry.md): export traces and custom metrics
2279
+ - [Channels: events](/docs/reference/channels.md#events): deliver replies back to
2280
+ Slack, source control, or another surface
2281
+
2282
+ ---
2283
+
2161
2284
  Source: /docs/guides/improve.md
2162
2285
 
2163
2286
  # Self-improvement
@@ -2307,6 +2430,302 @@ replace them.
2307
2430
 
2308
2431
  ---
2309
2432
 
2433
+ Source: /docs/guides/jev.md
2434
+
2435
+ # Use Jev in review tools
2436
+
2437
+ Use Jev when a review workflow needs a typed decision instead of prose.
2438
+ A review agent can filter speculative findings, classify pull request
2439
+ risk, choose a reviewer, or decide whether a merge needs a documentation
2440
+ follow-up. Jev returns a choice, score, or probability; your TypeScript
2441
+ decides what happens next.
2442
+
2443
+ Set `TYPESAFE_API_KEY`. The default model is `jev-latest`.
2444
+
2445
+ ```ts
2446
+ // agent/extensions/jev.ts
2447
+ import jev from "@cursor/july/extensions/jev";
2448
+
2449
+ export default jev();
2450
+ ```
2451
+
2452
+ ## Start only the review turns you need
2453
+
2454
+ Call Jev from a channel hook before a model turn starts. When a pull
2455
+ request opens or becomes ready for review, this hook sends its title,
2456
+ body, labels, and filenames to Jev to decide whether the change needs
2457
+ security review. A result that clears the threshold starts the review
2458
+ turn; otherwise, the hook returns `null`, so the agent doesn't run or
2459
+ post to GitHub.
2460
+
2461
+ ```ts
2462
+ // agent/channels/github.ts
2463
+ import {
2464
+ defaultGitHubAuth,
2465
+ githubChannel,
2466
+ } from "@cursor/july/channels/github";
2467
+ import { above, decide } from "@cursor/july/extensions/jev";
2468
+
2469
+ export default githubChannel({
2470
+ botName: "security-reviewer",
2471
+ cursorAccount: { repos: ["acme/checkout"] },
2472
+ onPullRequest: async (ctx, pr) => {
2473
+ if (pr.action !== "opened" && pr.action !== "ready_for_review") {
2474
+ return null;
2475
+ }
2476
+
2477
+ const octokit = await ctx.github.getOctokit();
2478
+ const [{ data }, files] = await Promise.all([
2479
+ octokit.rest.pulls.get({
2480
+ owner: ctx.repository.owner,
2481
+ repo: ctx.repository.name,
2482
+ pull_number: pr.number,
2483
+ }),
2484
+ octokit.paginate(octokit.rest.pulls.listFiles, {
2485
+ owner: ctx.repository.owner,
2486
+ repo: ctx.repository.name,
2487
+ pull_number: pr.number,
2488
+ }),
2489
+ ]);
2490
+ const answers = await decide({
2491
+ state: {
2492
+ title: data.title,
2493
+ body: data.body,
2494
+ labels: data.labels.map(label => label.name),
2495
+ files: files.map(file => file.filename),
2496
+ },
2497
+ questions: {
2498
+ review: {
2499
+ type: "boolean",
2500
+ instructions:
2501
+ "Does this change need security review? Answer yes for auth, permissions, secrets, request parsing, or external inputs.",
2502
+ },
2503
+ },
2504
+ });
2505
+
2506
+ if (!above(answers.review, 0.8)) {
2507
+ return null;
2508
+ }
2509
+ // `auth` starts a model turn running as the pull request sender.
2510
+ return { auth: defaultGitHubAuth(ctx) };
2511
+ },
2512
+ });
2513
+ ```
2514
+
2515
+ Jev receives only the pull request metadata shown here, not the diff.
2516
+
2517
+ ## Filter findings before you post them
2518
+
2519
+ Let the chat model draft a finding, then ask Jev whether the finding is
2520
+ a real bug in the new code. Below your threshold, the tool returns and
2521
+ the author never sees the draft. Above it, the finding becomes a review
2522
+ comment.
2523
+
2524
+ ```ts
2525
+ // agent/tools/post_finding.ts
2526
+ import { parseGitHubPrContinuationKey } from "@cursor/july/channels/github";
2527
+ import { above, decide } from "@cursor/july/extensions/jev";
2528
+ import { defineTool } from "@cursor/july/tools";
2529
+ import { z } from "zod";
2530
+
2531
+ export default defineTool({
2532
+ description:
2533
+ "Post one security finding on this session's pull request. Call once. Hold when it is not a real bug.",
2534
+ inputSchema: z.object({
2535
+ title: z.string(),
2536
+ summary: z.string().describe("What the pull request changes."),
2537
+ draft: z.string().describe("The finding to post, one or two sentences."),
2538
+ }),
2539
+ async execute({ title, summary, draft }, ctx) {
2540
+ if (ctx.session.purpose === "eval") {
2541
+ return { posted: false, reason: "eval" };
2542
+ }
2543
+
2544
+ const ref = parseGitHubPrContinuationKey(ctx.session.continuationKey ?? "");
2545
+ if (ref === undefined) {
2546
+ throw new Error("post_finding requires a GitHub pull request session");
2547
+ }
2548
+
2549
+ const answers = await decide({
2550
+ state: { title, summary, draft },
2551
+ questions: {
2552
+ real: {
2553
+ type: "boolean",
2554
+ instructions:
2555
+ "Is the draft an exploitable bug in the new code, not a style note or a hypothetical?",
2556
+ },
2557
+ },
2558
+ });
2559
+
2560
+ if (!above(answers.real, 0.85)) {
2561
+ return { posted: false, reason: "clean" };
2562
+ }
2563
+
2564
+ const octokit = await ctx.host.github.getOctokit();
2565
+ await octokit.rest.pulls.createReview({
2566
+ owner: ref.owner,
2567
+ repo: ref.repo,
2568
+ pull_number: ref.number,
2569
+ event: "COMMENT",
2570
+ body: draft,
2571
+ });
2572
+ return {
2573
+ posted: true,
2574
+ pr: `${ref.owner}/${ref.repo}#${ref.number}`,
2575
+ };
2576
+ },
2577
+ });
2578
+ ```
2579
+
2580
+ ## Approve changes by risk tier
2581
+
2582
+ You can use the same pattern for Agentic Owners. Ask Jev to put the
2583
+ pull request in a closed set of risk tiers. Approve only a confident
2584
+ `very-low` or `low`; send everything else to a person.
2585
+
2586
+ ```ts
2587
+ import { decide, needsHuman } from "@cursor/july/extensions/jev";
2588
+
2589
+ const answers = await decide({
2590
+ state: { title, summary },
2591
+ questions: {
2592
+ risk: {
2593
+ type: "choice",
2594
+ instructions: "What risk tier is this pull request?",
2595
+ criteria: {
2596
+ "very-low": "docs, formatting, or a mechanical rename",
2597
+ low: "a local change with tests and no new trust boundary",
2598
+ medium: "auth, billing, or a behavior change callers depend on",
2599
+ high: "a likely exploit, data loss, or a broken public contract",
2600
+ },
2601
+ },
2602
+ },
2603
+ });
2604
+
2605
+ const tier = answers.risk.choice;
2606
+ if (needsHuman(answers.risk) || tier === "medium" || tier === "high") {
2607
+ return { verdict: "hold", tier };
2608
+ }
2609
+ return { verdict: "approve", tier };
2610
+ ```
2611
+
2612
+ Your GitHub tool resolves the pull request from `ctx.session` and posts
2613
+ that verdict. The model does not choose the repository, pull request, or
2614
+ approval event.
2615
+
2616
+ ## Open documentation follow-ups selectively
2617
+
2618
+ After a pull request merges, a code-wiki agent can ask whether the
2619
+ change introduced a durable fact that belongs in the docs. A low
2620
+ probability skips the follow-up, while a high probability opens a
2621
+ documentation pull request.
2622
+
2623
+ ```ts
2624
+ import { above, decide } from "@cursor/july/extensions/jev";
2625
+ import { openDocsPullRequest } from "../lib/wiki";
2626
+
2627
+ const answers = await decide({
2628
+ state: { title, summary },
2629
+ questions: {
2630
+ updateDocs: {
2631
+ type: "boolean",
2632
+ instructions:
2633
+ "Does this merge change a durable contract that the project docs should explain?",
2634
+ },
2635
+ },
2636
+ });
2637
+
2638
+ if (!above(answers.updateDocs, 0.8)) {
2639
+ return { action: "skip" };
2640
+ }
2641
+ return openDocsPullRequest({ title, summary });
2642
+ ```
2643
+
2644
+ ## Write tools with Jev
2645
+
2646
+ Call `decide` from a channel hook, server tool, or router when you want
2647
+ the answer map directly. Use `evaluate` when you want `{ answers }`.
2648
+
2649
+ Use a boolean for a yes-or-no gate, a choice for a closed set such as
2650
+ risk tiers or owners, and a score for an ordered rubric. `above` returns
2651
+ `true` when a boolean probability or score meets the threshold.
2652
+ `needsHuman` returns `true` when a boolean probability or the selected
2653
+ choice's probability falls below the confidence threshold.
2654
+
2655
+ Keep the questions atomic and combine them in TypeScript. For example,
2656
+ ask separately whether a finding is real, whether its impact is
2657
+ user-visible, and whether the changed line is new. Your code owns the
2658
+ rule that decides whether all three are enough to post.
2659
+
2660
+ ## Let the agent ask Jev
2661
+
2662
+ Mounting the extension adds a read-only harness tool named
2663
+ `<namespace>__evaluate`. With the `agent/extensions/jev.ts` mount shown
2664
+ earlier, the model sees `jev__evaluate`.
2665
+
2666
+ The tool accepts one state and a list of boolean, choice, or score
2667
+ questions. It returns `{ answers }` and never posts, approves, or opens
2668
+ a pull request. Use it when the agent needs the result during the turn.
2669
+ Use `decide` inside a project tool when the answer and the write belong
2670
+ in one operation.
2671
+
2672
+ ## Skills included with the extension
2673
+
2674
+ The extension adds two skills by default:
2675
+
2676
+ - `jev__questions` teaches the model how to structure atomic questions,
2677
+ choose a question type, and read the answers.
2678
+
2679
+ - `jev__gated-write` teaches the model to put `decide` and the write in
2680
+ one server tool, hold on low confidence, and skip writes during evals.
2681
+
2682
+ The model sees each skill's description and loads the full procedure
2683
+ when it applies.
2684
+
2685
+ ## Choose what to mount
2686
+
2687
+ Both contribution groups are on by default. Turn off the harness tool
2688
+ when Jev should only run inside tools you wrote. Turn off the skills
2689
+ when your agent already has its own Jev instructions.
2690
+
2691
+ ```ts
2692
+ // agent/extensions/jev.ts
2693
+ import jev from "@cursor/july/extensions/jev";
2694
+
2695
+ export default jev({
2696
+ harnessTools: false,
2697
+ skills: true,
2698
+ });
2699
+ ```
2700
+
2701
+ `harnessTools: false` removes `jev__evaluate` from discovery.
2702
+ `skills: false` removes both Jev skills. These switches do not remove
2703
+ the exported helpers, so project tools can still import `decide`,
2704
+ `above`, and `needsHuman`.
2705
+
2706
+ ## Practices
2707
+
2708
+ - Pass the pull request title, a short summary, and the draft finding.
2709
+ Don't send the checkout.
2710
+
2711
+ - Calibrate `above` and `needsHuman` on pull requests you have already
2712
+ labeled. `needsHuman` defaults to `0.7`.
2713
+
2714
+ ## Related
2715
+
2716
+ - [Agentic owners](/docs/templates/agentic-owners.md): a risk tier, then
2717
+ the host approves or asks for reviewers
2718
+ - [Security reviewer](/docs/templates/security-reviewer.md): a finding, or
2719
+ no comment
2720
+ - [Thermo review](/docs/templates/thermo-review.md): bugs and breakage,
2721
+ posted the same way
2722
+ - [Code wiki](/docs/reference/cli.md#init): documentation
2723
+ follow-ups after a merge
2724
+ - [Evals](/docs/evals.md): checks on the full turn
2725
+ - [GitHub agents](/docs/guides/github.md): how the review gets onto the pull request
2726
+
2727
+ ---
2728
+
2310
2729
  Source: /docs/guides/mcp-oauth.md
2311
2730
 
2312
2731
  # Connect a private MCP server with OAuth
@@ -3671,11 +4090,11 @@ coding agent extend the project for you
3671
4090
 
3672
4091
  Source: /docs/reference/agent-config.md
3673
4092
 
3674
- # Agent config (`agent/agent.ts`)
4093
+ # Agent config
3675
4094
 
3676
- `agent/agent.ts` default-exports `defineAgent(config)`: which model runs
3677
- the agent, where turns execute, and runtime-specific defaults.
3678
- Everything is optional on the root agent.
4095
+ `agent/agent.ts` default-exports `defineAgent(config)`, which sets the
4096
+ model, execution runtime, and runtime-specific defaults. Every root
4097
+ config field is optional.
3679
4098
 
3680
4099
  ```ts
3681
4100
  import { defineAgent } from "@cursor/july";
@@ -3695,9 +4114,7 @@ export default defineAgent({
3695
4114
  });
3696
4115
  ```
3697
4116
 
3698
- ## Fields on `defineAgent`
3699
-
3700
- `defineAgent` accepts these fields.
4117
+ ## Agent fields
3701
4118
 
3702
4119
  | Field | Type | Meaning |
3703
4120
  | --- | --- | --- |
@@ -3706,14 +4123,14 @@ export default defineAgent({
3706
4123
  | `description` | string | What the agent is for. Required on subagents; the parent model reads it to decide when to delegate. Documentation-only on the root. |
3707
4124
  | `instructions` | string | Inline instructions. Prefer `instructions.md`; this exists for subagents and generated configs. |
3708
4125
  | `runtime` | `"local"` or `"cloud"` | Where turns execute. Default `"local"`. |
3709
- | `cloud` | object | Cloud agent defaults: repos, env, envVars, forwarded to the Cursor SDK. Used when `runtime` is `"cloud"`, and as the base merged under per-session `cloud` send options. |
4126
+ | `cloud` | object | Cloud agent defaults: repos, env, envVars. Used when `runtime` is `"cloud"`, and as the base merged under per-session `cloud` send options. |
3710
4127
  | `local` | `{ cwd?, workspaceDir?, sandbox? }` | Local harness defaults; ignored for cloud turns. See [Local options](#local-options). |
3711
- | `hosting` | `{ egressDomains?, secretNames? }` | Managed-hosting declarations read by `agent-sdk deploy`: the pod's egress allowlist and the secret names the agent expects. Ignored by local serving. |
3712
- | `concurrency` | `{ maxRunningTurns? }` | Engine-wide turn admission limit. See [Concurrency](#concurrency). |
4128
+ | `hosting` | `{ egressDomains?, secretNames? }` | `agent-sdk deploy` declarations for allowed egress domains and expected secret names. Ignored by local serving. |
4129
+ | `concurrency` | `{ maxRunningTurns? }` | Agent-wide turn admission limit. See [Concurrency](#concurrency). |
3713
4130
  | `builtinTools` | `{ reminders? }` | Framework-provided model-facing tools, opted in per capability. See [Built-in tools](#built-in-tools). |
3714
- | `tools` | `ToolName[]` | Allowlist of built-in harness tools offered to the model. Unset = the model's full standard toolset. See [Allowlist built-in harness tools](#allowlist-built-in-harness-tools). |
4131
+ | `tools` | `ToolName[]` | Allowlist of built-in harness tools offered to the model. Unset = the model's full standard toolset. See [Harness tools](#harness-tools). |
3715
4132
 
3716
- ## Choose a model
4133
+ ## Model
3717
4134
 
3718
4135
  `model` is a Cursor model id string, or `{ id, params }`. Effort and
3719
4136
  speed are params, not id suffixes. The SDK rejects suffix-style ids
@@ -3735,24 +4152,17 @@ A plain string works when you don't need params:
3735
4152
  model: "composer-2.5",
3736
4153
  ```
3737
4154
 
3738
- ## Choose a runtime
4155
+ ## Runtime
3739
4156
 
3740
- `runtime: "local"` (the default) runs turns on the Cursor SDK harness on
3741
- this machine. Server tools, skills, sandbox seeds, and tool approvals
3742
- all apply.
4157
+ `runtime: "local"` (the default) runs turns on this machine. Server
4158
+ tools, skills, sandbox seeds, and tool approvals all apply.
3743
4159
 
3744
- `runtime: "cloud"` runs turns on Cursor cloud agents (`bc-…` ids). Pass
3745
- a `cloud` block with the repos the VM carries. Server tools stay
3746
- reachable over authenticated HTTP MCP back to the serve host when
3747
- `--public-url` or `--cloud-tools-url` is set (omitted with a warning
3748
- otherwise), and instructions and agent-tool catalogs are prepended to
3749
- the first prompt, because the local session workspace is not the cloud
3750
- VM.
4160
+ `runtime: "cloud"` runs turns on Cursor cloud agents. Pass a `cloud`
4161
+ block with the repositories the VM needs. See [Tools](/docs/reference/tools.md) for
4162
+ server- and agent-tool behavior on cloud turns.
3751
4163
 
3752
- `validate` warns when `runtime: "cloud"` is combined with agent tools
3753
- (they are described on the first prompt instead of written to the VM),
3754
- when skills or sandbox seeds are present (they sync onto an Agent Store
3755
- rather than the session workspace), and when the `cloud` block is
4164
+ `validate` warns when `runtime: "cloud"` is combined with agent tools,
4165
+ when skills or sandbox seeds are present, and when the `cloud` block is
3756
4166
  missing.
3757
4167
 
3758
4168
  ## Local options
@@ -3761,18 +4171,15 @@ missing.
3761
4171
 
3762
4172
  `local.workspaceDir` points every session at one shared harness cwd,
3763
4173
  for agents that work inside an existing checkout. It takes precedence
3764
- over `cwd`, and a per-send `workspaceDir` still wins over both. The SDK
3765
- keys its local executor (rules, skills, MCP, ignore mappings) on the
3766
- harness cwd, so a shared directory resolves the workspace once per
3767
- serve process instead of once per session. The trade: sessions share a
3768
- working tree, so a file one turn writes is visible to the next.
4174
+ over `cwd`, and a per-send `workspaceDir` still wins over both.
4175
+ Sessions share a working tree, so a file one turn writes is visible to
4176
+ the next.
3769
4177
 
3770
4178
  `local.sandbox` runs the harness inside Cursor's local sandbox. It's
3771
- off by default, matching the SDK: shell then auto-approves and inherits
3772
- the serve process environment, including any credentials the host
3773
- holds. Turn it on for agents whose turns read untrusted input (webhook
3774
- payloads, PR diffs, inbound chat); it's a real tool boundary rather
3775
- than a prompt-level one.
4179
+ off by default: shell then auto-approves and inherits the serve process
4180
+ environment, including any credentials the host holds. Turn it on for
4181
+ agents whose turns read untrusted input (webhook payloads, PR diffs,
4182
+ inbound chat); it's a tool boundary, not a prompt-level one.
3776
4183
 
3777
4184
  ### Local cwd
3778
4185
 
@@ -3788,13 +4195,12 @@ does not leak rules, skills, or MCP servers into the turn. A standalone git
3788
4195
  root keeps the in-project session workspace. Point `cwd` at a checkout only
3789
4196
  when the agent should inherit that tree.
3790
4197
 
3791
- ## Allowlist built-in harness tools
4198
+ ## Harness tools
3792
4199
 
3793
4200
  Use `tools` to limit which built-in Cursor harness tools the model can
3794
4201
  call. Omit it to keep the standard toolset. When you set it, the model
3795
4202
  gets only the tools you list. An empty list disables all native
3796
- built-in tools. Because this field is an allowlist, new platform tools
3797
- stay disabled until you add them.
4203
+ built-in tools. New platform tools stay disabled until you add them.
3798
4204
 
3799
4205
  ```ts
3800
4206
  export default defineAgent({
@@ -3805,12 +4211,13 @@ export default defineAgent({
3805
4211
  });
3806
4212
  ```
3807
4213
 
3808
- The Agent SDK always adds `"mcp"` to a configured allowlist. Authored
3809
- server tools in `agent/tools/` use MCP to reach the model. MCP can also
3810
- expose declared connections and servers from the harness directory's
3811
- ambient `.cursor` config. To exclude a checkout's MCP servers, point
3812
- `local.cwd` outside the checkout. See [Local cwd](#local-cwd).
3813
- `local.sandbox` makes MCP tool calls fail closed.
4214
+ The Agent SDK always adds `"mcp"` to a configured allowlist, because
4215
+ authored server tools in `agent/tools/` reach the model over MCP. MCP
4216
+ can also expose declared connections and servers from the harness
4217
+ directory's ambient `.cursor` config. To exclude a checkout's MCP
4218
+ servers, point `local.cwd` outside the checkout. See
4219
+ [Local cwd](#local-cwd). `local.sandbox` makes MCP tool calls fail
4220
+ closed.
3814
4221
 
3815
4222
  Use the SDK's public tool names, including `"shell"`, `"read"`,
3816
4223
  `"edit"`, `"grep"`, `"glob"`, `"ls"`, and `"task"`. Unknown names
@@ -3827,8 +4234,7 @@ Two names have broader effects:
3827
4234
  Tool allowlists work only with the local runtime. A
3828
4235
  `runtime: "cloud"` agent that sets `tools` fails at serve startup.
3829
4236
  The Agent SDK also refuses per-send cloud sessions from a hybrid agent
3830
- with an allowlist. It won't run those sessions with unrestricted tool
3831
- access.
4237
+ with an allowlist.
3832
4238
 
3833
4239
  The allowlist controls which tools the model can call. It does not
3834
4240
  isolate the serve host. For agents that process untrusted input, also
@@ -3836,10 +4242,10 @@ set `local: { sandbox: true }`.
3836
4242
 
3837
4243
  ## Cloud options
3838
4244
 
3839
- Cloud agent defaults forwarded to the Cursor SDK: `repos` (each
3840
- `{ url, startingRef? }`), environment selection, `envVars`, and the
3841
- rest. A local agent uses the same block as the base config when a
3842
- channel opens a cloud-attached session per send (the `cloud` option on
4245
+ The `cloud` block sets default repositories (each
4246
+ `{ url, startingRef? }`), environment selection, and `envVars`. A local
4247
+ agent uses the same block as the base config when a channel opens a
4248
+ cloud-attached session per send (the `cloud` option on
3843
4249
  [`send`](/docs/reference/channels.md#handler-arguments)).
3844
4250
 
3845
4251
  ## Concurrency
@@ -3862,9 +4268,8 @@ export default defineAgent({
3862
4268
  ## Built-in tools
3863
4269
 
3864
4270
  `builtinTools` opts into framework-provided model-facing tools. Each
3865
- enabled capability materializes as ordinary server tools at discovery
3866
- time, so turns, direct calls, `info`, and the playground treat them
3867
- like authored tools. Authored tools with the same name win, with a
4271
+ enabled capability shows up as ordinary server tools, so turns, direct
4272
+ calls, `info`, and the playground treat them like authored tools. Authored tools with the same name win, with a
3868
4273
  warning, and like all server tools they run on the local runtime.
3869
4274
 
3870
4275
  `builtinTools: { reminders: true }` adds three tools bound to the
@@ -3873,22 +4278,6 @@ current conversation over `host.reminders`: `reminders_create`,
3873
4278
  continuation key can't arm reminders. See
3874
4279
  [Schedules and reminders](/docs/reference/schedules.md#reminders).
3875
4280
 
3876
- ## Generate instructions
3877
-
3878
- When the system prompt must be computed, author `agent/instructions.ts`
3879
- instead of markdown:
3880
-
3881
- ```ts
3882
- import { defineInstructions } from "@cursor/july";
3883
-
3884
- export default defineInstructions({
3885
- markdown: `You are the on-call assistant for ${process.env.TEAM_NAME}.`,
3886
- });
3887
- ```
3888
-
3889
- The directory form and the runtime mapping are in
3890
- [Instructions](/docs/reference/instructions.md).
3891
-
3892
4281
  ## Serve programmatically
3893
4282
 
3894
4283
  `serve(dirOrProject, options)` embeds the server in your own process:
@@ -3898,7 +4287,7 @@ import { serve } from "@cursor/july";
3898
4287
 
3899
4288
  const handle = await serve("./my-agent", {
3900
4289
  port: 3000,
3901
- apiKey: process.env.CURSOR_API_KEY, // optional; see credential order
4290
+ apiKey: process.env.CURSOR_API_KEY,
3902
4291
  });
3903
4292
  console.log(`listening on ${handle.url}`);
3904
4293
  // handle.callTool(...), handle.dispatchSchedule("heartbeat"),
@@ -3907,20 +4296,17 @@ console.log(`listening on ${handle.url}`);
3907
4296
 
3908
4297
  Host settings match the documented [CLI](/docs/reference/cli.md) `serve` flags.
3909
4298
  `serve()` also accepts `discovery` (project-loading options) and
3910
- `mode: "single" | "multi"`. The Cursor credential resolves in one order
3911
- everywhere: explicit `apiKey`, then `CURSOR_API_KEY`, then
3912
- `CURSOR_API_KEY_FILE` (hosted default `/run/cursor/secrets/CURSOR_API_KEY`
3913
- when unset), then `CURSOR_SERVICE_ACCOUNT_KEY`, then the key stored by
3914
- `agent-sdk login`. On a host that has both the service-account key and a
3915
- bind file, the file principal wins.
3916
-
3917
- ## What's next
4299
+ `mode: "single" | "multi"`. Pass `apiKey` or use the same Cursor
4300
+ credential as the CLI: `CURSOR_API_KEY`, `CURSOR_API_KEY_FILE`,
4301
+ `CURSOR_SERVICE_ACCOUNT_KEY`, or `agent-sdk login`.
3918
4302
 
3919
- Continue with these pages:
4303
+ ## Related
3920
4304
 
3921
4305
  - [Instructions](/docs/reference/instructions.md): the required half of a minimal
3922
4306
  agent
3923
4307
  - [CLI](/docs/reference/cli.md): the `serve` flags `serve()` accepts
4308
+ - [Sessions](/docs/reference/sessions.md): workspaces, identifiers, and turn admission
4309
+ - [Schedules](/docs/reference/schedules.md): reminder tools opted in here
3924
4310
 
3925
4311
  ---
3926
4312
 
@@ -3928,13 +4314,12 @@ Source: /docs/reference/artifacts.md
3928
4314
 
3929
4315
  # Artifacts
3930
4316
 
3931
- An artifact marks a durable output the agent produced: a reviewed PR
3932
- URL, a generated report, a decision record. Sessions come and go;
3933
- artifacts persist across them, capped and listable, so the people
3934
- supervising an agent see what it shipped without replaying event
3935
- streams.
4317
+ Artifacts are durable outputs such as reviewed pull requests, reports,
4318
+ or decision records. An artifact kind defines the data it accepts, and
4319
+ the artifacts API tags or updates records by key. Records persist across
4320
+ sessions and can be listed, streamed, or downloaded.
3936
4321
 
3937
- ## Declare kinds
4322
+ ## Artifact kinds
3938
4323
 
3939
4324
  Author `agent/artifacts.ts` with `defineArtifacts` from
3940
4325
  `@cursor/july/artifacts`:
@@ -3955,26 +4340,21 @@ export default defineArtifacts({
3955
4340
  });
3956
4341
  ```
3957
4342
 
3958
- `defineArtifacts` accepts three fields. `kinds` declares the artifact
3959
- kinds: with kinds declared, `tag` accepts only these; with none, any
3960
- kind string is accepted freeform. Each kind's `description` says what it
3961
- holds and doubles as the model-facing prompt for `tag_artifact`. An
3962
- optional Zod `schema` validates payloads before they persist (the parsed
3963
- output is stored, so defaults and coercions apply). `agentTool` exposes
3964
- the model-facing `tag_artifact` tool generated from the kinds registry;
3965
- it requires at least one declared kind. `max` is the retention cap,
3966
- default 1000: on insert past the cap, the oldest-updated artifact is
3967
- evicted.
4343
+ | Option | Contract |
4344
+ | --- | --- |
4345
+ | `kinds` | Map of accepted kind names to a non-empty `description` and optional Zod `schema` |
4346
+ | `agentTool` | Expose `tag_artifact` to the model; requires at least one declared kind |
4347
+ | `max` | Positive retention cap; defaults to `1000` and evicts the oldest-updated record |
4348
+
4349
+ With declared kinds, `tag` rejects any other kind. With no registry, it
4350
+ accepts free-form kind names and defaults an omitted kind to
4351
+ `"artifact"`. A kind's schema validates `data`, and the parsed value is
4352
+ stored, including schema defaults and coercions.
3968
4353
 
3969
- ## Tag from host code
4354
+ ## Tag artifacts from host code
3970
4355
 
3971
- Every handler surface carries `ctx.artifacts` (or `args.artifacts`),
3972
- an `ArtifactsApi` with `tag` and `list`: tools, hooks, channel route
3973
- handlers and `onStart`, schedule `run` handlers, and reminder `run`
3974
- handlers. Tool and hook facades are session-bound, so `tag` auto-fills
3975
- the `sessionId` (and `turnId` when known). Channel, schedule, and
3976
- reminder facades are unbound; pass `sessionId` in the tag input to
3977
- attribute one.
4356
+ Tools, hooks, channel handlers, channel `onStart`, schedules, and
4357
+ reminders receive an `ArtifactsApi` with `tag` and `list`.
3978
4358
 
3979
4359
  ```ts
3980
4360
  await ctx.artifacts.tag({
@@ -3985,59 +4365,66 @@ await ctx.artifacts.tag({
3985
4365
  });
3986
4366
  ```
3987
4367
 
3988
- `key` is the upsert handle: tagging the same key again replaces the
3989
- record instead of creating a new one, so re-reviewing a PR updates one
3990
- row. A `contents` payload (string or bytes, with an optional
3991
- `contentType`) attaches a file or blob served at
3992
- `GET /v1/artifacts/:id/content`; re-tagging a keyed artifact without
3993
- `contents` keeps the existing payload.
4368
+ | Tag field | Contract |
4369
+ | --- | --- |
4370
+ | `data` | Required JSON data, validated when the kind has a schema |
4371
+ | `kind` | Declared or free-form kind |
4372
+ | `key` | Agent-wide upsert key; the same key updates one record across kinds |
4373
+ | `title` | Optional display title |
4374
+ | `contents` | String or bytes served by the content route |
4375
+ | `contentType` | MIME type for `contents` |
4376
+ | `sessionId`, `turnId` | Attribute the artifact to a session or turn |
4377
+ | `source` | `"host"` or `"model"`; defaults to `"host"` |
4378
+
4379
+ Tool, hook, result, and channel-event contexts are session-bound, so
4380
+ they fill `sessionId` and the current `turnId`. Channel routes,
4381
+ `onStart`, schedules, and reminders receive an unbound facade; pass
4382
+ `sessionId` to attribute an artifact.
4383
+
4384
+ Re-tagging a key without `contents` keeps its file or blob when the
4385
+ `sessionId` stays the same. Rebinding the key to another session without
4386
+ new contents removes the previous payload. `tag` returns the stored
4387
+ `ArtifactRecord`.
4388
+
4389
+ | Record field | Contract |
4390
+ | --- | --- |
4391
+ | `id` | Stable ID derived from `key`, or a generated ID when no key is set |
4392
+ | `kind`, `data` | Validated kind and JSON payload |
4393
+ | `key`, `title` | Optional upsert key and display title |
4394
+ | `content` | Optional `{ size, contentType? }` metadata |
4395
+ | `sessionId`, `turnId` | Optional session attribution |
4396
+ | `source` | `"host"` or `"model"` |
4397
+ | `createdAt`, `updatedAt` | ISO-8601 timestamps |
3994
4398
 
3995
- ## Let the model tag
4399
+ ## Expose `tag_artifact` to the model
3996
4400
 
3997
4401
  With `agentTool: true`, the `tag_artifact` server tool materializes from
3998
- the kinds registry. Its description tells the model to tag notable
3999
- outputs and lists each kind with its description, and its input schema
4000
- is a discriminated union over the declared kinds, so a schema'd kind is
4001
- validated exactly like a host-side tag. An authored tool named
4002
- `tag_artifact` shadows the built-in, with a warning.
4402
+ the kinds registry. Its input accepts the declared kinds and validates
4403
+ their data with the same schemas as host-side tagging. The kind
4404
+ descriptions tell the model which output each one represents.
4003
4405
 
4004
- ## Observe and list
4406
+ An authored tool named `tag_artifact` takes precedence over the generated
4407
+ tool.
4005
4408
 
4006
- Tagging emits an `artifact.tagged` event on the attributed session's
4007
- stream, carrying the record: `id`, `kind`, `key`, `title`, `data`, and
4008
- `source` (`"host"` for host code, `"model"` for `tag_artifact`). Hooks,
4009
- channel `events`, and evals see it like any other
4010
- [stream event](/docs/reference/sessions.md#which-events-can-i-stream).
4409
+ ## List artifacts
4011
4410
 
4012
- Over HTTP:
4411
+ `list({ kind?, sessionId? })` returns matching records newest-updated
4412
+ first. HTTP callers can list records and download content through the
4413
+ [artifact routes](/docs/reference/http-api.md#list-and-download-artifacts).
4013
4414
 
4014
- ```bash
4015
- curl 'http://127.0.0.1:3000/<slug>/v1/artifacts?kind=reviewed-pr&limit=20'
4016
- curl 'http://127.0.0.1:3000/<slug>/v1/artifacts/<id>/content'
4017
- ```
4415
+ ## Stream artifact tags
4018
4416
 
4019
- `GET /v1/artifacts` returns records newest-updated first, filterable by
4020
- `kind` and `sessionId`. Session ownership applies, same as
4021
- `/v1/sessions`. The playground renders tagged artifacts too.
4417
+ Tagging an artifact with a `sessionId` emits `artifact.tagged` on that
4418
+ session. Its event data contains `id`, `kind`, `key`, `title`, `data`,
4419
+ and `source`. See [Stream events](/docs/reference/sessions.md#stream-events) for the
4420
+ event envelope.
4022
4421
 
4023
- ## Gate evals on tagging
4024
-
4025
- `t.taggedArtifact(kind?, predicate?)` gates an eval on at least one
4026
- artifact tagged during the test turn, optionally of one kind and
4027
- matching a predicate over the record:
4028
-
4029
- ```ts
4030
- t.taggedArtifact("reviewed-pr", (record) => record.source === "model");
4031
- ```
4032
-
4033
- ## What's next
4034
-
4035
- Continue with these pages:
4422
+ ## Related
4036
4423
 
4037
- - [Sessions and streaming](/docs/reference/sessions.md): the `artifact.tagged` event
4038
- in the full vocabulary
4039
- - [Tools](/docs/reference/tools.md): the `ctx` that carries `artifacts`
4040
- - [Evals](/docs/evals.md): the assertions `taggedArtifact` sits beside
4424
+ - [Sessions](/docs/reference/sessions.md)
4425
+ - [Tools](/docs/reference/tools.md)
4426
+ - [Evals](/docs/reference/evals.md)
4427
+ - [HTTP API](/docs/reference/http-api.md)
4041
4428
 
4042
4429
  ---
4043
4430
 
@@ -4045,24 +4432,21 @@ Source: /docs/reference/channels.md
4045
4432
 
4046
4433
  # Channels
4047
4434
 
4048
- A channel is the surface an agent lives on. The built-in HTTP session
4049
- channel is always mounted. Custom channels declare their own routes
4050
- under `/v1/channels/<id>`. The Slack and GitHub packs are prebuilt
4051
- channels with platform transports. GitLab and Bitbucket packs cover their
4052
- hosted and self-managed products. This page is the authoring reference;
4053
- for the walkthrough, see the [Webhooks guide](/docs/guides/webhooks.md).
4435
+ A channel connects an external surface to agent sessions. A file at
4436
+ `agent/channels/<id>.ts` defines channel `<id>` and mounts its routes
4437
+ under `/v1/channels/<id>`. The built-in HTTP session channel is always
4438
+ available alongside any custom or prebuilt channels.
4054
4439
 
4055
4440
  ## Built-in HTTP channel
4056
4441
 
4057
- It's always mounted, under `/<slug>` in the default multi-agent layout:
4058
- session create, follow-up, stream, stop, the sessions list, approvals,
4059
- deterministic tool calls, health, and info. For the route-by-route
4060
- contract, see the [HTTP API reference](/docs/reference/http-api.md).
4442
+ The built-in channel serves the session, approval, tool, discovery, and
4443
+ health routes. In a multi-agent host, each agent's routes sit under its
4444
+ slug. See the [HTTP API](/docs/reference/http-api.md) for request and response
4445
+ contracts.
4061
4446
 
4062
4447
  ## Define a custom channel
4063
4448
 
4064
- Author `agent/channels/<id>.ts`. The filename is the channel id and the
4065
- route prefix:
4449
+ Use `defineChannel` with one or more typed routes:
4066
4450
 
4067
4451
  ```ts
4068
4452
  import { defineChannel, POST } from "@cursor/july/channels";
@@ -4075,237 +4459,173 @@ export default defineChannel({
4075
4459
  bodySchema: z.object({
4076
4460
  message: z.string(),
4077
4461
  prUrl: z.string().url(),
4078
- thread: z.string().optional(),
4079
4462
  }),
4080
- handler: async (_req, { callTool, send, body }) => {
4081
- const prepared = await callTool("inspect_pr", {
4082
- prUrl: body.prUrl,
4083
- });
4084
- if (prepared.isError) {
4085
- return Response.json(
4086
- { error: prepared.errorMessage ?? "Could not inspect pull request" },
4087
- { status: 502 }
4088
- );
4089
- }
4090
-
4463
+ handler: async (_request, { send, body }) => {
4091
4464
  const session = await send(
4092
- `${body.message}\n\nRead pr.json before answering.`,
4465
+ `${body.message}\n\nPull request: ${body.prUrl}`,
4093
4466
  {
4094
- continuationToken: body.thread ?? `pr:${body.prUrl}`,
4095
- workspaceFiles: {
4096
- "pr.json": JSON.stringify(prepared.result, null, 2) ?? "null",
4097
- },
4467
+ continuationToken: `pr:${body.prUrl}`,
4098
4468
  }
4099
4469
  );
4100
4470
  return Response.json({ sessionId: session.id });
4101
4471
  },
4102
4472
  }),
4103
4473
  ],
4104
- events: {
4105
- "message.completed"(event, channel, ctx) {
4106
- // deliver the reply to the surface that owns this channel
4107
- },
4108
- },
4109
- // auth: [...], state: {...}, onStart(...), onStop(...)
4110
4474
  });
4111
4475
  ```
4112
4476
 
4113
- This example assumes `agent/tools/inspect_pr.ts` exists. The handler
4114
- calls it before the model turn, so every review starts with validated PR
4115
- data. It also derives a stable conversation key from the PR URL and
4116
- writes the tool result to `pr.json`. Instructions can ask the model to
4117
- inspect a PR, but host code guarantees it.
4477
+ The route above is
4478
+ `POST /v1/channels/<id>/review`. Calls for the same pull request reuse
4479
+ one conversation because they pass the same continuation token.
4118
4480
 
4119
4481
  ## Route verbs and schemas
4120
4482
 
4121
- `GET`, `POST`, `PUT`, `PATCH`, and `DELETE` helpers build routes. Their
4122
- schemas are Zod, enforced at compile time:
4483
+ Route paths must begin with `/`. The method helper determines which Zod
4484
+ schemas the route accepts:
4123
4485
 
4124
- | Verb | Required schema |
4125
- | ------------------------ | ------------------------------------- |
4126
- | `GET` | `querySchema` |
4127
- | `POST` / `PUT` / `PATCH` | `bodySchema` (optional `querySchema`) |
4128
- | `DELETE` | both optional |
4486
+ | Helper | Schema contract |
4487
+ | --- | --- |
4488
+ | `GET` | `querySchema` is required |
4489
+ | `POST`, `PUT`, `PATCH` | `bodySchema` is required; `querySchema` is optional |
4490
+ | `DELETE` | Both schemas are optional |
4129
4491
 
4130
- Plain JSON Schema objects won't type-check; use `z.object({})` or
4131
- `z.unknown()` for intentionally open surfaces. The host validates before
4132
- the handler runs. Handlers receive typed `args.body` and `args.query`,
4133
- and empty POST bodies are coerced to `{}` first. Declared schemas are
4134
- projected on `GET /v1/info`, which powers the playground's **Try**
4135
- buttons and composer **slash commands**.
4492
+ Plain JSON Schema objects don't type-check. Use `z.object({})` or
4493
+ `z.unknown()` for an open surface. The host validates the body and query
4494
+ before calling the handler, returning `400` on failure; an empty body is
4495
+ read as `{}`. Declared schemas also appear in `GET /v1/info` for
4496
+ playground requests and slash commands.
4136
4497
 
4137
4498
  ## Handler arguments
4138
4499
 
4139
- Handlers receive the Fetch `Request` and an args object:
4140
-
4141
- | Member | What it is |
4142
- | ----------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
4143
- | `send(message, options?)` | Run a model turn on this channel; returns the session handle (options below) |
4144
- | `getSession(sessionId)` | Look up an existing session on this channel |
4145
- | `receive(channelDefinition, input)` | Hand off to another channel (schedules use this) |
4146
- | `callTool(name, input, options?)` | Deterministic server-tool call ([Tools](/docs/reference/tools.md#call-a-tool-without-a-model-turn)) |
4147
- | `body`, `query`, `params` | Validated payloads and `:param` path segments |
4148
- | `auth` | The `AuthContext` resolved by this route's auth chain |
4149
- | `requestIp` | The TCP peer address |
4150
- | `host` | Shared services: `host.mcp`, `host.github`, `host.slack`, `host.kv`, `host.files`, `host.reminders` |
4151
- | `waitUntil(promise)` | Background work that outlives the response |
4152
- | `sessionUrls(request, sessionId)` | Absolute playground + trace URLs for a session on this mount |
4153
- | `artifacts` | Unbound [artifacts](/docs/reference/artifacts.md) facade; pass `sessionId` in `tag` input to attribute one |
4154
-
4155
- `send` options: `continuationToken` (the conversation key),
4156
- `admission` (`"preempt"` interrupts a busy session, the default;
4157
- `"coalesce"` enqueues behind the running turn, the
4158
- [Slack policy](/docs/reference/sessions.md#what-happens-when-i-send-a-follow-up)),
4159
- `workspaceFiles`, `workspaceDir`, `cloud` (attach cloud repos for this
4160
- session), `auth` (defaults to the request principal), `state` (starting
4161
- channel state for new sessions), `title` (session display title), and
4162
- `purpose` (`"eval"` marks the session as regression traffic).
4500
+ Each handler receives the Fetch `Request` and a typed arguments object.
4501
+
4502
+ | Member | Contract |
4503
+ | --- | --- |
4504
+ | `send(message, options?)` | Start or resume a session on this channel |
4505
+ | `getSession(sessionId)` | Return this channel's session, or `null` |
4506
+ | `receive(channel, input)` | Hand work to another channel |
4507
+ | `callTool(name, input, options?)` | [Call a server tool](/docs/reference/tools.md#call-a-tool-without-a-model-turn) without a model turn |
4508
+ | `body`, `query`, `params` | Validated inputs and `:param` path segments |
4509
+ | `auth` | The `AuthContext` returned by the route's auth chain |
4510
+ | `requestIp` | The TCP peer address, or `null` |
4511
+ | `host` | Shared MCP, provider, storage, telemetry, and reminder services |
4512
+ | `waitUntil(promise)` | Track work after the response returns |
4513
+ | `sessionUrls(request, sessionId)` | Build absolute playground and trace URLs for this mount |
4514
+ | `artifacts` | List or tag [artifacts](/docs/reference/artifacts.md); pass `sessionId` when attributing one |
4515
+
4516
+ ### `send` options
4517
+
4518
+ | Option | Contract |
4519
+ | --- | --- |
4520
+ | `continuationToken` | Resume the session with this channel-local key, or create one when the key is new |
4521
+ | `admission` | `"preempt"` interrupts a busy turn; `"coalesce"` queues behind it. The default is `"preempt"` |
4522
+ | `workspaceFiles` | Add relative files for the next turn |
4523
+ | `workspaceDir` | Use an absolute local working directory |
4524
+ | `cloud` | Override cloud session options when creating a session |
4525
+ | `auth` | Set the session principal; defaults to the request principal |
4526
+ | `state` | Set starting channel state for a new session |
4527
+ | `title` | Set the display title for a new session |
4528
+ | `purpose` | Use `"eval"` to mark a new session as regression traffic |
4529
+ | `dryRun` | Run read tools and stub write tools for a new session |
4530
+ | `asOf` | Freeze a new session at an ISO-8601 instant with a timezone |
4531
+
4532
+ `send` returns a `ChannelSession`. Its `id` identifies the session,
4533
+ `continuationToken` contains its current channel key, and `isNew` says
4534
+ whether this call created it. A coalesced call also returns
4535
+ `coalesced: true`.
4163
4536
 
4164
4537
  ## Events
4165
4538
 
4166
4539
  The `events` map subscribes the channel to stream events for the
4167
4540
  sessions it owns. Keys are event types from the
4168
- [event vocabulary](/docs/reference/sessions.md#which-events-can-i-stream), or `"*"`.
4169
- Handlers receive `(event, channel, ctx)`, where `channel.state` is the
4170
- per-session adapter state, `ctx.session` is the session info, and
4171
- `ctx.host` is the shared host services, bound to that session as in a
4172
- [hook](/docs/reference/hooks.md#handler-context). This is where a channel delivers
4173
- replies back to its surface.
4174
-
4175
- ## State and lifecycle
4176
-
4177
- `state` declares the starting per-session adapter state (JSON), persisted
4178
- on the session record. Routes and event handlers read and mutate it
4179
- through `channel.state`. `onStart(args)` runs when the channel mounts.
4180
- `onStop()` runs when the server stops.
4181
-
4182
- `onStart` receives the route helpers (`send`, `getSession`, `receive`,
4183
- `callTool`, `host`, `waitUntil`, `artifacts`, `logger`) plus helpers
4184
- for long-lived transports:
4185
-
4186
- - `emitAssistantMessage(sessionId, text)` appends an assistant message
4187
- without a model turn, for host tasks that already produced the final
4188
- text.
4189
- - `hasContinuationSession(token)` and `isContinuationBusy(token)`
4190
- report whether a continuation token has a live session and whether a
4191
- turn is in flight on it.
4192
- - `interruptContinuation(token)` stops the in-flight turn and clears
4193
- coalesced follow-ups queued behind it.
4194
- - `resolveApproval(sessionId, callId, decision, auth, options?)`
4195
- approves or denies a parked tool call, how Slack Block Kit buttons
4196
- unblock a turn without the HTTP approvals route.
4541
+ [event vocabulary](/docs/reference/sessions.md#stream-events), or `"*"`. Each handler
4542
+ receives `(event, channel, ctx)`, including the session's
4543
+ `channel.state`, session info, and shared host services. Use these
4544
+ handlers to deliver progress and replies to the surface that owns the
4545
+ channel.
4197
4546
 
4198
- ## Auth policies
4547
+ ## Session state
4548
+
4549
+ `state` on the channel definition supplies starting JSON for each new
4550
+ session. A route can instead pass `state` to `send`; event handlers read
4551
+ and update the active value through `channel.state`.
4552
+
4553
+ ## Start and stop a channel
4199
4554
 
4200
- Every route runs an auth-policy chain: the channel's `auth` array, or
4201
- `[localDevStrict()]` when unset. A policy is a function
4202
- `(request, info) => AuthContext | null` (async allowed); the first
4203
- non-null wins, and a request no policy admits gets `401`.
4555
+ `onStart(args)` runs when the channel mounts, and `onStop()` runs when
4556
+ the host stops. `onStart` receives the route helpers plus these
4557
+ transport controls:
4204
4558
 
4205
- | Policy | Admits |
4206
- | ------------------- | ---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
4207
- | `localDevStrict()` | Direct loopback callers with no proxy-forwarding headers (`X-Forwarded-For`, `X-Real-IP`, `Forwarded`, and `X-Forwarded-Host` are all rejected, so tunnels and same-host reverse proxies don't silently re-expose the route), plus a loopback `Host` header, which rejects DNS-rebinding callers that reach 127.0.0.1 with a remote hostname. |
4208
- | `localDev()` | Like `localDevStrict()` but without the `Host` check. An explicit, weaker opt-in. |
4209
- | `loopbackOnly()` | A loopback TCP peer, ignoring forwarding headers; for dev relays that legitimately carry them, like `gh webhook forward`. |
4210
- | `bearerAuth(token)` | `Authorization: Bearer <token>`, compared in constant time. Also accepts a verifier function mapping a presented token to an `AuthContext`. |
4211
- | `allowAll()` | Everyone, as an `anonymous` principal. Only for surfaces protected upstream (an HMAC-verified webhook) or intentionally public. |
4212
- | `publicEndpoint()` | Everyone on this custom channel. Managed hosting also serves the channel without an alias token. Use it only when the handler verifies the provider signature. |
4559
+ | Helper | Contract |
4560
+ | --- | --- |
4561
+ | `emitAssistantMessage(sessionId, text)` | Append final assistant text without starting a model turn |
4562
+ | `hasContinuationSession(token)` | Check whether a token maps to a session |
4563
+ | `isContinuationBusy(token)` | Check whether a turn is active for a token |
4564
+ | `interruptContinuation(token)` | Stop the active turn and clear queued coalesced follow-ups |
4565
+ | `resolveApproval(...)` | Approve or deny a parked tool call |
4566
+
4567
+ ## Auth policies
4213
4568
 
4214
- `publicEndpoint()` applies only to custom channel routes. It does not open
4215
- the built-in session or tool API.
4569
+ Every route runs the channel's `auth` array. The default is
4570
+ `[localDevStrict()]`. Policies may be asynchronous; the first one to
4571
+ return an `AuthContext` admits the request, and an all-null result
4572
+ returns `401`.
4573
+
4574
+ | Policy | Admits |
4575
+ | --- | --- |
4576
+ | `localDevStrict()` | Direct loopback requests with a loopback hostname and no proxy-forwarding headers |
4577
+ | `localDev()` | Direct loopback requests with no proxy-forwarding headers, without checking the hostname |
4578
+ | `loopbackOnly()` | Any loopback TCP peer, including local relays that carry forwarding headers |
4579
+ | `bearerAuth(tokenOrVerify)` | A matching bearer token, or a token accepted by the verifier |
4580
+ | `sharedSecretAuth({ header, secret })` | A header matching the named environment or deployment secret |
4581
+ | `hmacSignatureAuth({ header, secret, prefix? })` | A hex HMAC-SHA256 signature over the raw body using the named secret |
4582
+ | `allowAll()` | Every caller as an anonymous principal |
4583
+ | `publicEndpoint()` | Every caller on this custom channel; managed hosting also exposes the route without an alias token |
4584
+
4585
+ Use `allowAll()` only for an intentionally public surface or one
4586
+ protected upstream. With `publicEndpoint()`, the handler must verify the
4587
+ provider's signature. The policy applies only to custom channel routes;
4588
+ it doesn't open the built-in session or tool API.
4216
4589
 
4217
4590
  The resolved `AuthContext` (`{ authenticator, principalId,
4218
4591
  principalType, attributes? }`) becomes the request principal. Sessions
4219
- bind to the principal that created them, and follow-up, stream, and list
4220
- routes enforce ownership (`403` otherwise).
4592
+ belong to the principal that created them, and other principals receive
4593
+ `403` on owned routes.
4221
4594
 
4222
4595
  Server flags interact with authored auth: `--bearer-token` swaps the
4223
4596
  default `localDevStrict()` for `bearerAuth(...)` on channels that don't
4224
4597
  author their own chain, and `--allow-anonymous` swaps it for
4225
- `allowAll()`. Authored `auth` arrays always win over both. A channel that
4226
- declares `[localDevStrict()]` stays loopback-only even on an
4227
- `--allow-anonymous` host.
4228
-
4229
- ## First class channels
4230
-
4231
- **Slack** (`@cursor/july/channels/slack`): Socket Mode
4232
- transport, streaming replies, engagement rules, approval cards, and a
4233
- default block on Slack Connect / guest / other-workspace senders. Author
4234
- `agent/channels/slack.ts` with `slackChannel()`. Guide:
4235
- [Slack](/docs/guides/slack.md).
4236
-
4237
- **GitHub** (`@cursor/july/channels/github`): webhook dispatch
4238
- with signature verification, per-event hooks returning `{ auth }` (a
4239
- model turn), `{ task }` (host work), or `null`, and CLI tooling for
4240
- replay and live forwarding. Author `agent/channels/github.ts` with
4241
- `githubChannel()`. Opt-in `progress.commitStatus` and `progress.banner`
4242
- converge a merge-box check and sticky PR comment from default stream
4243
- events. Supports GitHub.com and GitHub Enterprise Server. Guide:
4244
- [GitHub](/docs/guides/github.md).
4245
-
4246
- **GitLab** (`@cursor/july/channels/gitlab`): verified project hooks for
4247
- merge requests, notes, pipelines, pushes, and custom event types. Supports
4248
- GitLab.com and self-managed GitLab. Author `agent/channels/gitlab.ts` with
4249
- `gitlabChannel()`. Guide: [GitLab](/docs/guides/gitlab.md).
4250
-
4251
- **Bitbucket** (`@cursor/july/channels/bitbucket`): verified repository hooks
4252
- for pull requests, comments, pushes, and custom event types. Supports
4253
- Bitbucket Cloud and Bitbucket Data Center through one normalized hook API.
4254
- Author `agent/channels/bitbucket.ts` with `bitbucketChannel()`. Guide:
4255
- [Bitbucket](/docs/guides/bitbucket.md).
4256
-
4257
- **Deployments** (`@cursor/july/channels/deployments`): pull deploy
4258
- events. Declare `events` and handle each one in `onEvent`. Each event
4259
- carries `deploySourceUri` and `deployVersion`. Author
4260
- `agent/channels/deployments.ts` with `deploymentsChannel()`.
4261
-
4262
- On Cursor-managed hosting, omit `deploySourceUris`. The deployment's
4263
- watched repositories bind the event scope automatically. Name sources
4264
- to narrow the scope or to drive the self-hosted pull relay. Subscribe
4265
- per deploy source with `deploySourceUris` and narrow with `environments`
4266
- or `events`. Each entry must match `Deployment.deploy_source_uri` as
4267
- your deployment writer records it. Matching is case-insensitive but
4268
- otherwise literal. It uses the host credential. A restart resumes
4269
- rather than dropping events. An empty `deploySourceUris` list mounts the
4270
- channel but starts no pull, so an env-configured agent stays inert until
4271
- its deploy sources are set.
4272
-
4273
- **Change Monitors** (`@cursor/july/channels/change-monitors`): Change
4274
- Monitor Checkpoint events. The channel publishes Factory
4275
- `checkpoint.created` for every Checkpoint create. The agent filters the
4276
- result (for example to the `issues` arm). The payload contains the
4277
- full Checkpoint resource. This channel is scoped to Change Monitors,
4278
- not generic Factory Checkpoints. There is no repository filter or
4279
- resource filter. Author `agent/channels/change-monitors.ts` with
4280
- `changeMonitorsChannel()`. It uses the host credential.
4281
-
4282
- **Issues** (`@cursor/july/channels/issues`): Factory issue events.
4283
- The channel publishes `issue.created` for every Issue create. The
4284
- agent filters if it needs a subset. The payload contains the full
4285
- Issue resource. There is no repository filter or resource filter.
4286
- Author `agent/channels/issues.ts` with `issuesChannel()`. It uses the
4287
- host credential.
4598
+ `allowAll()`. An authored `auth` array always takes precedence.
4288
4599
 
4289
- For other platforms like Discord or Teams, use the authored
4290
- `defineChannel` webhook form.
4600
+ ## Prebuilt channels
4291
4601
 
4292
- ## Continuation semantics
4602
+ | Import | Factory | Surface |
4603
+ | --- | --- | --- |
4604
+ | `@cursor/july/channels/slack` | `slackChannel()` | Slack messages, threads, streaming replies, and approvals. See [Slack](/docs/guides/slack.md) |
4605
+ | `@cursor/july/channels/github` | `githubChannel()` | GitHub and GitHub Enterprise webhooks. See [GitHub](/docs/guides/github.md) |
4606
+ | `@cursor/july/channels/gitlab` | `gitlabChannel()` | GitLab.com and self-managed GitLab hooks. See [GitLab](/docs/guides/gitlab.md) |
4607
+ | `@cursor/july/channels/bitbucket` | `bitbucketChannel()` | Bitbucket Cloud and Data Center hooks. See [Bitbucket](/docs/guides/bitbucket.md) |
4608
+ | `@cursor/july/channels/deployments` | `deploymentsChannel()` | Deployment events filtered by source, environment, or event name |
4609
+ | `@cursor/july/channels/change-monitors` | `changeMonitorsChannel()` | Change Monitor `checkpoint.created` events |
4610
+ | `@cursor/july/channels/issues` | `issuesChannel()` | Factory `issue.created` events |
4293
4611
 
4294
- Channels own their continuation-token format. The built-in HTTP channel
4295
- mints opaque rotating tokens, Slack uses `channelId:threadTs`, and PR
4296
- automations use keys like `pr:owner/repo#N`. Same token, same durable
4297
- session; one active continuation per session; the HTTP channel returns
4298
- `409` for stale tokens. For the full session model, see
4299
- [Sessions](/docs/reference/sessions.md).
4612
+ For other platforms like Discord or Teams, use the authored
4613
+ `defineChannel` route form.
4300
4614
 
4301
- ## What's next
4615
+ ## Continuation tokens
4616
+
4617
+ Each channel defines its continuation-token format. The same token
4618
+ resumes the same conversation; the built-in HTTP channel rotates its
4619
+ opaque token after every accepted follow-up and returns `409` for a
4620
+ stale token. See [Session identifiers](/docs/reference/sessions.md#session-identifiers)
4621
+ for the full contract.
4302
4622
 
4303
- Continue with these pages:
4623
+ ## Related
4304
4624
 
4305
- - [Webhooks guide](/docs/guides/webhooks.md): the same API, walked through
4306
- - [HTTP API](/docs/reference/http-api.md): session, discovery, and channel routes
4307
- - [Sessions and streaming](/docs/reference/sessions.md): the events channels
4308
- subscribe to
4625
+ - [Webhooks](/docs/guides/webhooks.md)
4626
+ - [HTTP API](/docs/reference/http-api.md)
4627
+ - [Sessions](/docs/reference/sessions.md)
4628
+ - [Hooks](/docs/reference/hooks.md)
4309
4629
 
4310
4630
  ---
4311
4631
 
@@ -5281,19 +5601,18 @@ These environment variables affect the CLI and its channel packs.
5281
5601
 
5282
5602
  Source: /docs/reference/connections.md
5283
5603
 
5284
- # MCP Connections
5604
+ # MCP connections
5285
5605
 
5286
- An MCP connection gives the agent tools from an MCP server. One file per
5287
- server under `agent/mcp-connections/`, and the filename becomes the server
5288
- name the model sees. An MCP connection default-exports `defineConnection`
5289
- from `@cursor/july/connections`, and the transport comes in
5290
- four shapes: remote HTTP, local stdio, the signed-in Cursor account's
5291
- connectors, and peer agents on the same host.
5606
+ An MCP connection gives the agent tools from an MCP server. Define one
5607
+ file per server under `agent/mcp-connections/`; the filename becomes the
5608
+ server name. Default-export `defineConnection` from
5609
+ `@cursor/july/connections`. The transport is remote HTTP, local stdio,
5610
+ the signed-in Cursor account's connectors, or a peer agent on the same
5611
+ host.
5292
5612
 
5293
- Put a server in `agent/host-connections/` when host tools should call it
5294
- and the model should not. Same `defineConnection` shape. `agent-sdk mcp
5295
- oauth` still works. The playground and the turn's MCP servers never see
5296
- those files.
5613
+ Put a server in `agent/host-connections/` when only host tools should
5614
+ call it. Host connections use the same `defineConnection` shape and
5615
+ support `agent-sdk mcp oauth`; the model and playground don't see them.
5297
5616
 
5298
5617
  ## Remote MCP server
5299
5618
 
@@ -5315,8 +5634,7 @@ Tokens come from env vars. Never hardcode them in the file.
5315
5634
  For servers that speak OAuth, set `oauth: true` and authorize with the
5316
5635
  CLI or mid-run Connect. Tokens live in `mcp-auth.json` under the CLI
5317
5636
  config directory. `--store` copies them onto the deployment as
5318
- `MCP_OAUTH_<NAME>_*` secrets. Hosted Connect lets the current
5319
- process retry.
5637
+ `MCP_OAUTH_<NAME>_*` secrets.
5320
5638
 
5321
5639
  ```ts
5322
5640
  export default defineConnection({
@@ -5328,24 +5646,17 @@ export default defineConnection({
5328
5646
  ```bash
5329
5647
  agent-sdk mcp oauth inventory # browser PKCE → local mcp-auth.json
5330
5648
  agent-sdk mcp oauth inventory --store # also upsert deployment secrets
5331
- # Hosted Connect retries this process. Self-hosted stays file-only.
5332
5649
  ```
5333
5650
 
5334
5651
  Full walkthrough: [Host MCP OAuth](/docs/guides/mcp-oauth.md). Companion
5335
5652
  skill: [`skills/mcp-auth/SKILL.md`](https://github.com/cursor/cursor/blob/main/packages/agent-serve/skills/mcp-auth/SKILL.md).
5336
5653
 
5337
- Account MCP (`cursorAccount: true`) is the right choice for connectors
5338
- already linked in the Cursor dashboard. Omit `servers` (or pass `"*"`)
5339
- to forward every connected connector. If the model should call those
5340
- tools by name on local turns, set `advertiseTools: true`.
5654
+ ## Per-session auth
5341
5655
 
5342
- ## Per-session auth (`auth`)
5343
-
5344
- For http/sse connections whose credential depends on **who the session is
5345
- for** (a multi-tenant agent asserting the tenant it is acting for),
5346
- declare an `auth` callback instead of static headers. It runs host-side
5347
- at turn-build time with the session's `SessionInfo` and returns headers
5348
- merged over the static ones:
5656
+ For http/sse connections whose credential depends on who the session is
5657
+ for, declare an `auth` callback instead of static headers. It runs
5658
+ host-side with the session's `SessionInfo` and returns headers merged
5659
+ over the static ones:
5349
5660
 
5350
5661
  ```ts
5351
5662
  export default defineConnection({
@@ -5353,39 +5664,31 @@ export default defineConnection({
5353
5664
  auth: async (session) => ({
5354
5665
  headers: { Authorization: `Bearer ${await grantFor(session)}` },
5355
5666
  }),
5356
- advertiseTools: true, // optional — named tools instead of meta-tools
5667
+ advertiseTools: true,
5357
5668
  });
5358
5669
  ```
5359
5670
 
5360
- The callback is evaluated on **every local turn**, including reminder
5361
- fires and post-restart follow-ups, so the identity always comes from the
5362
- session itself, never from state parked in memory. The model never sees a
5363
- tenant parameter and can never choose the tenant. A callback that throws
5364
- fails the turn: a turn never silently runs without the connection's
5365
- identity. Local runtime only; cloud turns are refused. `host.mcp` calls
5366
- from server tools keep the static headers only. Not combinable with
5367
- `oauth: true`; the host OAuth provider owns the Authorization header.
5671
+ The identity always comes from the session itself, never from a tenant
5672
+ parameter the model could invent. A callback that throws fails the
5673
+ turn; a cloud turn with `auth` is refused instead of running without
5674
+ that identity. `host.mcp` calls from server tools keep the static
5675
+ headers only. You can't combine `auth` with `oauth: true`.
5368
5676
 
5369
5677
  Derive the identity from durable session facts: `session.auth`,
5370
- `session.id`, or your channel's own session state. Do **not** key it off
5371
- `session.continuationKey`: the HTTP channel rotates the continuation key
5372
- after every accepted follow-up, so a tenant mapping keyed on it silently
5373
- breaks mid-conversation. (Channels that mint stable, parseable tokens by
5374
- design are the exception.)
5375
-
5376
- `auth` works attached or advertised. Advertised connections open
5377
- per-operation clients with the evaluated headers. Attached connections
5378
- ride the turn's SDK `mcpServers`, passed on **every send** rather than
5379
- pinned on the cached per-session agent handle, so a rotated credential is
5380
- live on the very next turn. A stateful stdio server cannot share a
5381
- process with an attached `auth` connection. Advertise the auth
5382
- connection instead.
5383
-
5384
- ## Advertise a connection's tools by name (`advertiseTools`) {#advertise-tools}
5385
-
5386
- Set `advertiseTools: true` when the model should call an MCP server's tools
5387
- by name. The Agent SDK preserves each tool's name, description, input and
5388
- output schemas, and MCP annotations.
5678
+ `session.id`, or your channel's own session state. Do not key it off
5679
+ `session.continuationKey`. The HTTP channel rotates that key after every
5680
+ accepted follow-up, so a tenant mapping keyed on it breaks
5681
+ mid-conversation.
5682
+
5683
+ `auth` works attached or advertised. Advertise the auth connection
5684
+ instead of attaching it next to a stateful stdio server.
5685
+
5686
+ ## Advertise tools {#advertise-tools}
5687
+
5688
+ Set `advertiseTools: true` when the model should call an MCP server's
5689
+ tools by name. The Agent SDK preserves each tool's description, input
5690
+ and output schemas, and MCP annotations. Exposed names normalize invalid
5691
+ characters and add numeric suffixes to avoid collisions.
5389
5692
 
5390
5693
  ```ts
5391
5694
  export default defineConnection({
@@ -5395,20 +5698,22 @@ export default defineConnection({
5395
5698
  });
5396
5699
  ```
5397
5700
 
5398
- A listing failure, invalid tool name, or name collision fails the turn.
5399
- Advertised tools follow the same runtime support as server tools. They cannot
5400
- be called through the direct tool API.
5701
+ A listing or authentication failure fails the turn by default. Set
5702
+ `optional: true` to omit an unavailable connection instead. Advertised
5703
+ tools follow the same runtime support as server tools, and [direct tool
5704
+ calls](/docs/reference/tools.md#call-a-tool-without-a-model-turn) use the same names.
5401
5705
 
5402
- In a dry-run session, MCP tools marked read-only run normally. Tools marked
5403
- as writes are stubbed. Tools without effect annotations are unavailable.
5706
+ In a dry-run session, MCP tools marked read-only run normally. Tools
5707
+ marked as writes are stubbed. Tools without effect annotations are
5708
+ unavailable.
5404
5709
 
5405
- ## Restrict which tools a connection serves
5710
+ ## Filter connection tools
5406
5711
 
5407
5712
  Use `tools` the same way you allowlist harness tools on the agent. When
5408
5713
  set, the connection serves only those names. Use `disallowedTools` to
5409
- drop names instead. The two combine as deny-wins, same as the Cursor
5410
- SDK. The model and `host.mcp` only see what remains. On a model-visible
5411
- connection, set `advertiseTools: true` so the raw server is not attached.
5714
+ drop names instead. The two combine as deny-wins. The model and
5715
+ `host.mcp` only see what remains. A model-visible filter requires
5716
+ `advertiseTools: true`.
5412
5717
 
5413
5718
  ```ts
5414
5719
  export default defineConnection({
@@ -5416,7 +5721,9 @@ export default defineConnection({
5416
5721
  advertiseTools: true,
5417
5722
  tools: ["search_skus", "get_stock"],
5418
5723
  });
5724
+ ```
5419
5725
 
5726
+ ```ts
5420
5727
  export default defineConnection({
5421
5728
  url: "https://mcp.example.com/inventory",
5422
5729
  advertiseTools: true,
@@ -5425,9 +5732,9 @@ export default defineConnection({
5425
5732
  ```
5426
5733
 
5427
5734
  Names are the server's `tools/list` names. Unknown names are omitted. A
5428
- filter that matches nothing on the server fails the turn. Combine with
5429
- `effects: "read"` to keep only the listed tools the server classifies as
5430
- reads.
5735
+ filter that matches nothing on the server fails the turn unless
5736
+ `optional: true`. Combine with `effects: "read"` to keep only the listed
5737
+ tools the server classifies as reads.
5431
5738
 
5432
5739
  A list of names is an allowlist. An object of handlers authors TypeScript
5433
5740
  tools. On `host-connections/`, a name list restricts `host.mcp` without
@@ -5445,12 +5752,8 @@ export default defineConnection({
5445
5752
  });
5446
5753
  ```
5447
5754
 
5448
- This suits small purpose-built servers, like a `units` converter
5449
- shipped next to the agent.
5450
-
5451
- To run TypeScript in the agent environment (including a repo-less cloud
5452
- VM), author the tools on the connection instead. The Agent SDK packages
5453
- them as stdio MCP. You write `execute`. The Agent SDK speaks the protocol.
5755
+ To run TypeScript in the agent environment, including a repo-less cloud
5756
+ VM, author the tools on the connection instead. You write `execute`:
5454
5757
 
5455
5758
  ```ts
5456
5759
  export default defineConnection({
@@ -5468,12 +5771,12 @@ export default defineConnection({
5468
5771
  ## Cursor account MCP connection
5469
5772
 
5470
5773
  `{ cursorAccount: true }` forwards the MCP connectors the signed-in
5471
- Cursor account already authorized (dashboard → MCP): Linear, Notion,
5472
- Slack, and the rest. You don't configure tokens. Every tool runs on the
5774
+ Cursor account already authorized (dashboard → MCP), such as Linear,
5775
+ Notion, and Slack. You don't configure tokens. Every tool runs on the
5473
5776
  Cursor backend with the account's stored OAuth credentials, so raw
5474
- tokens never reach the serve host, session workspaces, or traces.
5777
+ tokens never reach the serving host, session workspaces, or traces.
5475
5778
 
5476
- By default the agent gets **every** connected HTTP/SSE connector on the
5779
+ By default the agent gets every connected HTTP/SSE connector on the
5477
5780
  account. Pass `servers: "*"` (or `["*"]`) for the same all-connectors
5478
5781
  behavior in an explicit form. Pass a name list when you want a smaller
5479
5782
  set.
@@ -5484,12 +5787,9 @@ export default defineConnection({
5484
5787
  cursorAccount: true,
5485
5788
  advertiseTools: true,
5486
5789
  });
5487
- // same, spelled out:
5488
- export default defineConnection({
5489
- cursorAccount: true,
5490
- servers: "*",
5491
- advertiseTools: true,
5492
- });
5790
+ ```
5791
+
5792
+ ```ts
5493
5793
  // only Linear:
5494
5794
  export default defineConnection({
5495
5795
  cursorAccount: true,
@@ -5500,17 +5800,15 @@ export default defineConnection({
5500
5800
 
5501
5801
  Name the file `account.ts`. `cursor.ts` collides with the IDE `cursor`
5502
5802
  MCP namespace. `advertiseTools: true` puts connector tools on local
5503
- turns by name. Without it they sit behind harness meta-tools.
5803
+ turns by name.
5504
5804
 
5505
5805
  The host must be signed in (`agent-sdk login`, `CURSOR_API_KEY`, or
5506
- `CURSOR_SERVICE_ACCOUNT_KEY`).
5507
- `serve` fails fast at startup otherwise, and logs each connector's live
5508
- status (`connected`, `needsAuth`, `error`) as it starts.
5806
+ `CURSOR_SERVICE_ACCOUNT_KEY`). `serve` fails at startup otherwise.
5509
5807
 
5510
5808
  Filtered account connections work on managed cloud deployments. A
5511
- self-hosted cloud agent with a concrete `servers` list needs `--public-url`.
5512
- Serve fails instead of ignoring the filter. Use a `{ command }` connection
5513
- for stdio servers.
5809
+ self-hosted cloud agent with a concrete `servers` list needs
5810
+ `--public-url`. Serve fails instead of ignoring the filter. Use a
5811
+ `{ command }` connection for stdio servers.
5514
5812
 
5515
5813
  > [!CAUTION]
5516
5814
  > Whoever can talk to the agent can drive these connectors, because they
@@ -5520,11 +5818,11 @@ for stdio servers.
5520
5818
  > for example an SSO proxy or the hosted alias token). Prefer
5521
5819
  > `--bearer-token` on shared hosts.
5522
5820
 
5523
- ## Peer MCP connection
5821
+ ## Peer MCP connection {#peer-mcp-connection}
5524
5822
 
5525
5823
  `{ agent: "<slug>" }` addresses another agent mounted on the same serve
5526
- host. The model gets the peer's `ask` and `check` (and `call_tool`)
5527
- tools and can delegate work to it:
5824
+ host. The model gets the peer's `ask` and `check` tools, plus
5825
+ `call_tool` when the peer has server tools, and can delegate work to it:
5528
5826
 
5529
5827
  ```ts
5530
5828
  export default defineConnection({
@@ -5534,65 +5832,53 @@ export default defineConnection({
5534
5832
  ```
5535
5833
 
5536
5834
  Unknown slugs and self-references fail `serve` at startup. Walkthrough:
5537
- [Agent-to-agent](/docs/guides/agent-to-agent.md#how-do-i-wire-two-agents).
5538
-
5539
- ## Every model-visible MCP connection is available in three places
5540
-
5541
- A file under `agent/mcp-connections/` serves three consumers. Host
5542
- connections skip the first one.
5543
-
5544
- 1. **Cursor agent:** Attached connections ride SDK `mcpServers` behind
5545
- harness MCP meta-tools. Set `advertiseTools: true` so local turns see
5546
- named tools.
5547
- 2. **Server tools:** Deterministic host code composes MCP calls
5548
- through `ctx.host.mcp`:
5549
-
5550
- ```ts
5551
- export default defineTool({
5552
- description: "Search Linear issues.",
5553
- inputSchema: z.object({ query: z.string() }),
5554
- async execute({ query }, ctx) {
5555
- return ctx.host.mcp.callTool("linear", "list_issues", { query });
5556
- },
5557
- });
5558
- ```
5559
-
5560
- 3. **Channel and schedule handlers:** Webhooks hit MCP servers with no
5561
- model turn at all, through `args.host.mcp`:
5562
-
5563
- ```ts
5564
- POST("/sync", {
5565
- bodySchema: z.object({}),
5566
- handler: async (_req, { host }) => {
5567
- const result = await host.mcp.callTool("linear", "list_issues", {});
5568
- return Response.json(result);
5569
- },
5570
- });
5571
- ```
5572
-
5573
- The host registry is small: `host.mcp.names()` lists MCP connection names,
5574
- and `listTools(name)` / `callTool(name, tool, args)` open the client
5575
- lazily on first use.
5835
+ [Peer agents](/docs/guides/agent-to-agent.md#delegate-a-question-to-a-specialist).
5576
5836
 
5577
- ## What's next
5837
+ ## Call MCP from host code
5838
+
5839
+ A file under `agent/mcp-connections/` is available to the Cursor agent,
5840
+ to server tools through `ctx.host.mcp`, and to channel and schedule
5841
+ handlers through `args.host.mcp`. Host connections skip the model.
5842
+
5843
+ ```ts
5844
+ // agent/tools/search_linear.ts
5845
+ import { defineTool } from "@cursor/july/tools";
5846
+ import { z } from "zod";
5578
5847
 
5579
- Continue with these pages:
5848
+ export default defineTool({
5849
+ description: "Search Linear issues.",
5850
+ inputSchema: z.object({ query: z.string() }),
5851
+ async execute({ query }, ctx) {
5852
+ return ctx.host.mcp.callTool("linear", "list_issues", { query });
5853
+ },
5854
+ });
5855
+ ```
5856
+
5857
+ `host.mcp.names()` lists connection names.
5858
+ `listTools(name)` / `callTool(name, tool, args)` call into a named
5859
+ connection.
5860
+
5861
+ ## Related
5580
5862
 
5581
5863
  - [Host MCP OAuth](/docs/guides/mcp-oauth.md): `mcp oauth`, Connect, `--store`
5582
5864
  - [Tools](/docs/reference/tools.md): authored tools that wrap MCP connections
5583
5865
  - [Webhooks](/docs/guides/webhooks.md): calling MCP connections from handlers
5866
+ - [Peer agents](/docs/guides/agent-to-agent.md): when a specialist is its
5867
+ own agent
5584
5868
 
5585
5869
  ---
5586
5870
 
5587
5871
  Source: /docs/reference/evals.md
5588
5872
 
5589
- # Evals reference
5873
+ # Evals
5590
5874
 
5591
- This page is the complete authoring and runner contract for
5592
- `@cursor/july/evals`. Start with the [Evals guide](/docs/evals.md) for the
5593
- workflow and first regression case.
5875
+ Evals run fixed cases against an agent and record whether its turns,
5876
+ tools, events, and output meet a contract. Files under the project-root
5877
+ `evals/` directory define cases with `@cursor/july/evals`; the runner
5878
+ discovers their IDs, executes them, and reports every assertion. See the
5879
+ [Evals guide](/docs/evals.md) for the regression workflow.
5594
5880
 
5595
- ## Discovery and case IDs
5881
+ ## Eval discovery and case IDs
5596
5882
 
5597
5883
  Eval files live under the project-root `evals/` directory and end in
5598
5884
  `.eval.ts` or `.eval.js`. The path under that directory becomes the
@@ -5631,15 +5917,15 @@ Those cases are `prs/checkout` and `prs/search` when the file is
5631
5917
  `evals/prs.eval.ts`. Case IDs must be unique single path segments.
5632
5918
 
5633
5919
  A file can also export an array of `defineEval` calls. The runner names
5634
- them with zero-padded indexes such as `sql/0000`. Use named `cases` for
5635
- handwritten scenarios and arrays for loaded datasets.
5920
+ them with zero-padded indexes such as `sql/0000`. Named `cases` give
5921
+ handwritten scenarios stable IDs; arrays fit loaded datasets.
5636
5922
 
5637
5923
  `iterations` repeats one datapoint from 1 to 100 times. For
5638
5924
  `iterations: 3`, `weather/nyc` expands to `weather/nyc/1`,
5639
5925
  `weather/nyc/2`, and `weather/nyc/3`; selecting `weather/nyc` runs all
5640
5926
  three.
5641
5927
 
5642
- ## Configuration
5928
+ ## Eval configuration
5643
5929
 
5644
5930
  Every running suite needs `evals/evals.config.ts` with
5645
5931
  `maxConcurrency`:
@@ -5660,38 +5946,39 @@ export default defineEvalConfig({
5660
5946
  | `timeoutMs` | Per-case timeout; case/file, CLI, then config precedence |
5661
5947
  | `judge` | Default model for `t.judge` |
5662
5948
  | `reporters` | Objects notified as cases and runs complete |
5663
- | `maxPlaygroundRuns` | Number of server-side batches kept in playground history |
5949
+ | `maxPlaygroundRuns` | Number of server-side batches kept in playground history; defaults to 20 |
5664
5950
 
5665
5951
  A case can override `description`, `tags`, `timeoutMs`, `iterations`,
5666
5952
  `judge`, `reporters`, and `metadata`. Case metadata merges over
5667
5953
  file-level metadata; case reporters add to the file's reporters.
5668
5954
 
5669
- ## Drive turns with `t.send`
5955
+ ## Send turns in a case
5670
5956
 
5671
5957
  `await t.send(message, options?)` runs one turn and waits until it
5672
5958
  finishes, fails, or parks for approval. Several sends in one test share
5673
5959
  the session.
5674
5960
 
5675
5961
  The returned turn exposes its assistant `message`, `sessionId`,
5676
- `events`, ordered `toolCalls`, `ok`, and turn `index`. Assertions on the
5677
- turn inspect only that turn; assertions on `t` inspect the whole run.
5678
- Useful run values include `t.reply`, `t.events`, `t.turns`,
5679
- `t.sessionId`, and the timeout `t.signal`.
5962
+ `events`, ordered `toolCalls`, `ok`, and one-based `index`. Assertions
5963
+ on the turn inspect only that turn; assertions on `t` inspect the whole
5964
+ case. Run values include `t.reply`, `t.events`, `t.turns`,
5965
+ `t.sessionId`, `t.iteration`, `t.iterations`, and the timeout
5966
+ `t.signal`.
5680
5967
 
5681
5968
  These options apply on the first send because they shape the session:
5682
5969
 
5683
5970
  | Option | Contract |
5684
5971
  | --- | --- |
5685
5972
  | `workspaceFiles` | Relative path-to-content map seeded into the session workspace |
5686
- | `workspaceDir` | Absolute local harness working directory |
5973
+ | `workspaceDir` | Absolute working directory for a new local session |
5687
5974
  | `cloud` | Cloud session options merged over the agent's static cloud config |
5688
5975
 
5689
- Use `turn.expectOk()` when later test steps depend on that turn
5690
- succeeding.
5976
+ Use `turn.expectOk()` when later test steps depend on the turn
5977
+ succeeding. It throws when the turn failed.
5691
5978
 
5692
5979
  ## Trajectory assertions
5693
5980
 
5694
- Assertions record failures and let the test continue, so one case
5981
+ Assertions record failures without stopping the test, so one case
5695
5982
  reports every violated contract.
5696
5983
 
5697
5984
  | Assertion | Checks |
@@ -5814,53 +6101,43 @@ export default rows.map(row =>
5814
6101
  );
5815
6102
  ```
5816
6103
 
5817
- Materialize API-backed evidence before running a large suite. Commit
5818
- the exact payload, diff, or metadata revision and seed it with
5819
- `workspaceFiles`.
5820
-
5821
6104
  ## Reporters and results
5822
6105
 
5823
6106
  Built-in reporters include `JUnit({ filePath, suiteName? })` and
5824
6107
  `Artifacts({ dir })`. Custom reporters can expose `onRunStart`,
5825
- `onEvalComplete`, and `onRunComplete`. Reporter errors are logged and
5826
- do not change the eval verdict.
6108
+ `onEvalComplete`, and `onRunComplete`.
5827
6109
 
5828
- JSON output contains run totals plus one result per case. A result can
5829
- include the case ID, verdict, assertions, session ID, inputs, tool
5830
- calls, metrics, logs, duration, metadata, tags, final text, and error or
5831
- skip details.
6110
+ JSON output contains run totals and one result per case. Results include
6111
+ the case ID, verdict, assertions, session ID, inputs, final text, tools,
6112
+ tool calls, metrics, logs, duration, metadata, tags, and any error or
6113
+ skip reason.
5832
6114
 
5833
- ## CLI selection and artifacts
6115
+ ## Select cases
5834
6116
 
5835
6117
  ```bash
5836
6118
  agent-sdk eval --dir . --list
5837
6119
  agent-sdk eval --dir .
5838
6120
  agent-sdk eval --dir . builds/checkout
5839
6121
  agent-sdk eval --dir . --tag smoke
5840
- agent-sdk eval --dir . --verbose
5841
6122
  ```
5842
6123
 
5843
6124
  ID filters match an exact ID and its descendants. Repeated IDs use OR;
5844
6125
  repeated tags use OR. When both are present, a case must match both
5845
6126
  groups.
5846
6127
 
5847
- The local runner starts an ephemeral server. Default artifacts go under
5848
- `evals/<stamp>/` in the project state directory and include
6128
+ ## Inspect run artifacts
6129
+
6130
+ Default run artifacts go under `evals/<stamp>/` in the project state
6131
+ directory and include
5849
6132
  `summary.json`, `results.jsonl`, and `evals/<case-id>.json`.
5850
6133
  `--artifacts <dir>` chooses another destination; `--no-artifacts`
5851
6134
  disables them. This location is independent of `--state-root`.
5852
6135
 
5853
- Model and judge turns resolve credentials in this order:
5854
- `CURSOR_API_KEY`, `CURSOR_API_KEY_FILE`,
5855
- `CURSOR_SERVICE_ACCOUNT_KEY`, then `agent-sdk login`. Listing cases
5856
- needs no credential.
5857
-
5858
- ## Playground and hosted runs
6136
+ ## Run evals against a server
5859
6137
 
5860
- The playground **Evals** view starts batches on its running server and
5861
- keeps their sessions in the normal session list. `--url` targets a
5862
- named running server; `--prod --slug <slug>` targets a hosted
5863
- deployment.
6138
+ The playground **Evals** view starts batches on its running server.
6139
+ `--url` targets a named running server, and `--prod --slug <slug>`
6140
+ targets a hosted deployment.
5864
6141
 
5865
6142
  ```bash
5866
6143
  agent-sdk eval --prod --slug vulnerability-scanner --tag smoke
@@ -5868,17 +6145,16 @@ agent-sdk eval status <eval-id> --prod --slug vulnerability-scanner
5868
6145
  agent-sdk eval cancel <eval-id> --prod --slug vulnerability-scanner
5869
6146
  ```
5870
6147
 
5871
- Server-side batches use the target server's discovered evals and store
5872
- results in its playground history. Local-only reporter and concurrency
5873
- flags do not apply to `--url` or `--prod` runs.
6148
+ Server-side batches use the target's discovered cases and return an eval
6149
+ ID for `status` or `cancel`. Local reporter and concurrency flags don't
6150
+ apply to `--url` or `--prod` runs.
5874
6151
 
5875
6152
  ## Related
5876
6153
 
5877
- - [Evals guide](/docs/evals.md): author and run a regression workflow
5878
- - [CLI reference](/docs/reference/cli.md#eval): every runner flag
5879
- - [Sessions](/docs/reference/sessions.md): event vocabulary used by trajectory checks
5880
- - [Artifacts](/docs/reference/artifacts.md): durable outputs asserted by
5881
- `taggedArtifact`
6154
+ - [Evals guide](/docs/evals.md)
6155
+ - [CLI reference](/docs/reference/cli.md#eval)
6156
+ - [Sessions](/docs/reference/sessions.md)
6157
+ - [Artifacts](/docs/reference/artifacts.md)
5882
6158
 
5883
6159
  ---
5884
6160
 
@@ -5916,10 +6192,10 @@ The extension validates its configuration while the project loads.
5916
6192
  Missing or mistyped settings fail `agent-sdk validate` before a turn
5917
6193
  can call the contributed code.
5918
6194
 
5919
- Extension instructions are appended to the agent's prompt. The root
5920
- agent still needs its own `agent/instructions.md`.
6195
+ Extension instructions are appended to the root agent's instructions.
6196
+ The root agent still needs its own `agent/instructions.md`.
5921
6197
 
5922
- ## See what the extension added
6198
+ ## Inspect a mount
5923
6199
 
5924
6200
  Run discovery after mounting or upgrading an extension:
5925
6201
 
@@ -6007,18 +6283,14 @@ server. OAuth-backed plugin servers use
6007
6283
  [Host MCP OAuth](/docs/guides/mcp-oauth.md); transport and tool-filter
6008
6284
  details live in [MCP connections](/docs/reference/connections.md).
6009
6285
 
6010
- ## What an extension can contain
6011
-
6012
- Share capabilities that can safely join another agent under a
6013
- namespace:
6286
+ ## Extension contents
6014
6287
 
6015
- - Instructions, tools, skills, MCP and host connections
6016
- - Hooks, channels, schedules, and subagents
6017
- - Artifact kinds and local workspace seed files
6288
+ | Ships under the namespace | Ignored |
6289
+ | --- | --- |
6290
+ | Instructions, tools, skills, MCP and host connections | `agent.ts`, storage, OpenTelemetry |
6291
+ | Hooks, channels, schedules, and subagents | Playground code, custom sandbox backends |
6292
+ | Artifact kinds and workspace seed files | Nested extensions |
6018
6293
 
6019
- Agent-wide configuration does not compose through a mount.
6020
- `agent.ts`, storage, OpenTelemetry, playground code, custom sandbox
6021
- backends, and nested extensions are ignored in an extension package;
6022
6294
  `agent-sdk validate` reports unsupported paths.
6023
6295
 
6024
6296
  ## Build an extension
@@ -6073,6 +6345,7 @@ namespace-neutral because the consumer chooses the final prefix.
6073
6345
  - [Cloud agents](/docs/guides/cloud-agents.md): delegate repository work
6074
6346
  through an extension
6075
6347
  - [Grok Bot agents](/docs/guides/grokbot-agents.md): consult named bots
6348
+ - [Jev](/docs/guides/jev.md): typed answers, then gated writes
6076
6349
  - [Self-improvement](/docs/guides/improve.md): propose source changes
6077
6350
  through pull requests
6078
6351
  - [Project layout](/docs/reference/project-layout.md): contribution slots
@@ -6084,18 +6357,16 @@ Source: /docs/reference/hooks.md
6084
6357
 
6085
6358
  # Hooks
6086
6359
 
6087
- A hook subscribes to the session event stream and runs a side effect
6088
- after each event is recorded: an audit line, a metric, a copy of the
6089
- transcript in your own store, or derived state for later turns. Hooks
6090
- run in the serving process for every session of the agent, on local and
6091
- cloud runtime turns alike.
6360
+ A hook subscribes to selected session events and can run a side effect
6361
+ after each matching event is recorded: an audit line, a metric, a
6362
+ transcript copy, or derived state. Hooks run for every session of the
6363
+ agent, on local and cloud turns alike. They cannot change the turn, the
6364
+ prompt, or the reply; a handler that throws is logged and skipped.
6092
6365
 
6093
- Hooks observe. They can't change the turn, the prompt, or the reply, and
6094
- a handler that throws is logged and skipped. Treat the event as
6095
- read-only; later subscribers see the same object. That makes hooks safe
6096
- to add to a production agent, and the wrong tool for anything that must
6097
- happen before the model runs or must fail a turn; see
6098
- [When not to use a hook](#when-not-to-use-a-hook).
6366
+ Treat the event as read-only. Later subscribers see the same object.
6367
+ That makes hooks safe to add to a production agent, and the wrong
6368
+ surface for anything that must run before the model or must fail a
6369
+ turn; see [Hook boundaries](#hook-boundaries).
6099
6370
 
6100
6371
  `defineHook` is unrelated to
6101
6372
  [Cursor Agent hooks](https://cursor.com/docs/agent/hooks), the
@@ -6144,95 +6415,54 @@ later sessions to read. Delete the file to opt out.
6144
6415
  ## Events and payloads
6145
6416
 
6146
6417
  Keys are event types from the
6147
- [event vocabulary](/docs/reference/sessions.md#which-events-can-i-stream), or `"*"`
6418
+ [event vocabulary](/docs/reference/sessions.md#stream-events), or `"*"`
6148
6419
  for every event. A typed key narrows `event.data`; a `"*"` handler
6149
6420
  receives the union, so switch on `event.type`. Every event carries the
6150
6421
  stream envelope `{ type, index, sessionId, turnId?, at, data }`, with
6151
- `turnId` set on turn-scoped events.
6152
-
6153
- The payloads hooks read most often:
6422
+ `turnId` set on turn-scoped events. Skip `*.appended` deltas when you
6423
+ want the final text; `message.completed` already has it.
6154
6424
 
6155
6425
  | Event | `event.data` |
6156
6426
  | ------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
6157
6427
  | `message.received` | `{ text }` |
6158
6428
  | `turn.completed` | `{ result?, usage?, cost? }`. `usage` has `inputTokens`, `outputTokens`, `cacheReadTokens`, `cacheWriteTokens`, and optional `reasoningTokens`. `cost` has `totalUsd` and the `model` it was priced against |
6159
- | `turn.failed` | `{ message }` |
6160
- | `actions.requested` | `{ calls: [{ callId, toolName, args? }] }`. A call with `parentCallId` belongs to a subagent |
6161
- | `action.result` | `{ callId, toolName, output?, isError, stubbed? }`. `stubbed` means a dry-run session answered a write without running it |
6429
+ | `turn.failed` | `{ message, status? }`. `status` is `"error"` or `"cancelled"` when present |
6430
+ | `actions.requested` | `{ calls: [{ callId, toolName, args?, parentCallId? }], parentCallId? }`. `parentCallId` marks subagent work |
6431
+ | `action.result` | `{ callId, toolName, output?, isError, stubbed?, parentCallId? }`. `stubbed` means a dry-run session answered a write without running it |
6162
6432
 
6163
6433
  The types are `SessionEvent`, `SessionEventType`, and `HookContext`,
6164
6434
  exported from `@cursor/july`.
6165
6435
 
6166
6436
  ## Handler context
6167
6437
 
6168
- | Member | What it is |
6169
- | --------------------------------------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
6170
- | `ctx.session` | Read-only session info: `id`, `channelId`, `mode` (`chat` or `task`), `purpose` (`live` or `eval`), `auth`, plus `title` and `sdkAgentId` when set |
6171
- | `ctx.agent` | `{ name }` of the agent the event belongs to |
6172
- | `ctx.channel` | `{ id, continuationToken }`. The token is `null` when the session can't take follow-ups |
6173
- | `ctx.host.kv` | Durable JSON, shared by every session of the agent; the storage backend decides whether it survives a hosted replace. Prefix keys with `ctx.session.id` for per-session state |
6174
- | `ctx.host.files` | Durable files, bound to this session. Pass `{ scope: "deployment" }` for agent-wide files |
6175
- | `ctx.host.otel` | Counters, histograms, and tags, attributed to this session |
6176
- | `ctx.host.mcp`, `ctx.host.github`, `ctx.host.slack` | The same shared clients tools get |
6177
- | `ctx.host.reminders` | Per-session [reminders](/docs/reference/schedules.md), the same API tools get |
6178
- | `ctx.artifacts` | Session-bound [artifacts](/docs/reference/artifacts.md) facade: `tag` fills in `sessionId` and `turnId` |
6179
- | `ctx.stateRoot` | Absolute path of the local state root. It resets when a hosted deployment is replaced; keep derived state in `kv` or `files` |
6180
-
6181
- ## When hooks run
6182
-
6183
- A hook runs after the event is durably recorded. It never delays the
6184
- model turn and never sees an event that wasn't recorded.
6185
-
6186
- Within one session, events dispatch in order, one at a time: the
6187
- channel's `events` handlers first, then each hook in discovery order.
6188
- Sessions don't wait on each other.
6189
-
6190
- Two consequences:
6191
-
6192
- - A slow handler holds up the next event's handlers for that session,
6193
- not the model. Keep handlers short and queue anything slow.
6194
- - Hooks fire for eval sessions too. Check
6195
- `ctx.session.purpose === "eval"` before metering or paging.
6196
-
6197
- Each event reaches a hook at most once. A restart doesn't replay the log
6198
- into hooks, so a mirror needs no dedupe, and the event log rather than
6199
- the hook's copy is the source of truth.
6200
-
6201
- ## Hooks, channel events, or evals?
6202
-
6203
- All of them consume the same stream, for different jobs:
6204
-
6205
- | | Hooks | Channel `events` | Evals |
6206
- | ------------------ | ----------------------------------------------- | ----------------------------------------------------------------- | --------------------------------- |
6207
- | Scope | every session of the agent | sessions the channel owns | one test turn |
6208
- | Job | observe: audit, metrics, mirrors, derived state | deliver: replies back to the channel's surface | assert: gates over the trajectory |
6209
- | Context | `ctx.host`, `ctx.artifacts`, session info | `channel.state`, `setContinuationToken`, `ctx.host`, session info | the `t` assertion helpers |
6210
- | Can affect the run | no | yes, it owns the surface | n/a |
6211
- | Authored at | `agent/hooks/*.ts` | channel config | `evals/**/*.eval.ts` |
6212
-
6213
- ## When not to use a hook
6214
-
6215
- | You want to | Use instead |
6216
- | ------------------------------------------------------------------------------------- | ---------------------------------------------------------------------------------------------------------------- |
6217
- | Add context before the model runs | The channel's `send` message and `workspaceFiles`, `instructions.md`, skills, or `sandbox/workspace/` seed files |
6218
- | Reply on Slack, comment on a PR, or post any other delivery | The channel's `events` map, or the Slack and GitHub packs |
6219
- | Show PR progress (merge-box check, sticky banner) | `githubChannel({ progress: { commitStatus, banner } })`; see the [PR autofixer](/docs/templates/pr-autofixer.md) |
6220
- | Block, approve, or rewrite a tool call | [`needsApproval`](/docs/reference/tools.md#gate-a-tool-on-human-approval) on the tool |
6221
- | Act on the final assistant text, reject it for a same-turn repair, or fail a bad turn | `defineResult` |
6222
- | Gate a change on behavior | [Evals](/docs/evals.md) |
6223
-
6224
- ## Patterns
6225
-
6226
- Usage metering is the [authoring example](#author-a-hook). Three more:
6227
-
6228
- ### Alert on failure
6438
+ | Member | What it is |
6439
+ | --------------------------------------------------- | ---------------------------------------------------------------------------------------------------------------------------------------------------------- |
6440
+ | `ctx.session` | Read-only session info: `id`, `channelId`, `mode` (`chat` or `task`), `purpose` (`live` or `eval`), `auth`, plus `title` and `sdkAgentId` when set |
6441
+ | `ctx.agent` | `{ name }` of the agent the event belongs to |
6442
+ | `ctx.channel` | `{ id, continuationToken }`. The token is `null` when the session can't take follow-ups |
6443
+ | `ctx.host.kv` | Durable JSON, shared by every session of the agent. Prefix keys with `ctx.session.id` for per-session state |
6444
+ | `ctx.host.files` | Durable files, bound to this session. Pass `{ scope: "deployment" }` for agent-wide files |
6445
+ | `ctx.host.otel` | Counters, histograms, and tags, attributed to this session |
6446
+ | `ctx.host.mcp`, `ctx.host.github`, `ctx.host.slack` | The same shared clients tools get |
6447
+ | `ctx.host.reminders` | Per-session [reminders](/docs/reference/schedules.md#reminders), the same API tools get |
6448
+ | `ctx.artifacts` | Session-bound [artifacts](/docs/reference/artifacts.md) facade: `tag` fills in `sessionId` and `turnId` |
6449
+ | `ctx.stateRoot` | Absolute path of the local state root. It resets when a hosted deployment is replaced; keep derived state in `kv` or `files` |
6450
+
6451
+ ## Hook dispatch
6452
+
6453
+ A hook runs after the event is recorded. It never delays the model and
6454
+ never sees an event that wasn't recorded.
6455
+
6456
+ | Rule | What happens |
6457
+ | --- | --- |
6458
+ | Same session | Events dispatch one at a time |
6459
+ | Eval sessions | The same stream fires; skip metering or paging when `ctx.session.purpose === "eval"` |
6460
+ | Host restart | Recorded events are not replayed into hooks, so a mirror needs no dedupe |
6461
+ | Slow handler | Holds the next event's handlers on that session, not the model. Keep handlers short and queue anything slow |
6229
6462
 
6230
- `turn.failed` carries the message, and `ctx.session.id` points at the
6231
- trace. Skip interrupted turns; those are preemptions, not failures. Read
6232
- secrets inside the handler: hosted deployments bind them after the
6233
- process starts, so a module-scope read stays empty. Give the call a
6234
- timeout, since a stalled request holds up later handlers on that
6235
- session.
6463
+ Read hosted secrets inside the handler, not at module scope. Give
6464
+ outbound calls a timeout; a stalled request holds later handlers on that
6465
+ session. Skip cancelled turns when paging; they record interrupted work.
6236
6466
 
6237
6467
  ```ts
6238
6468
  // agent/hooks/page-on-failure.ts
@@ -6245,7 +6475,7 @@ export default defineHook({
6245
6475
  if (
6246
6476
  pagerUrl === undefined ||
6247
6477
  ctx.session.purpose === "eval" ||
6248
- event.data.message === "turn interrupted"
6478
+ event.data.status === "cancelled"
6249
6479
  ) {
6250
6480
  return;
6251
6481
  }
@@ -6265,78 +6495,47 @@ export default defineHook({
6265
6495
  });
6266
6496
  ```
6267
6497
 
6268
- ### Mirror the transcript
6269
-
6270
- Subscribe to `"*"` and write one file per event, skipping the
6271
- `*.appended` deltas: they arrive per token, and `message.completed`
6272
- carries the final text. Session scope keeps transcripts apart without a
6273
- session id in the path. The mirror holds reasoning text and raw tool
6274
- arguments and outputs, so pick the store accordingly, and write to your
6275
- own store instead when you need cross-session queries.
6276
-
6277
- ```ts
6278
- // agent/hooks/mirror.ts
6279
- import { defineHook } from "@cursor/july/hooks";
6280
-
6281
- export default defineHook({
6282
- events: {
6283
- async "*"(event, ctx) {
6284
- if (event.type.endsWith(".appended")) {
6285
- return;
6286
- }
6287
- const name = String(event.index).padStart(6, "0");
6288
- await ctx.host.files.write(
6289
- `transcript/${name}.json`,
6290
- JSON.stringify(event)
6291
- );
6292
- },
6293
- },
6294
- });
6295
- ```
6498
+ ## Hooks vs channels vs evals
6296
6499
 
6297
- ### Keep derived state across a replace
6500
+ All three consume the same stream, for different jobs:
6298
6501
 
6299
- Write it to `ctx.host.kv` under a session-prefixed key; a tool reads it
6300
- back with `ctx.host.kv.get`.
6502
+ | | Hooks | Channel `events` | Evals |
6503
+ | ------------------ | ----------------------------------------------- | ----------------------------------------------------------------- | --------------------------------- |
6504
+ | Scope | every session of the agent | sessions the channel owns | one test turn |
6505
+ | Job | observe: audit, metrics, mirrors, derived state | deliver: replies back to the channel's surface | assert: gates over the trajectory |
6506
+ | Context | `ctx.host`, `ctx.artifacts`, session info | `channel.state`, `setContinuationToken`, `ctx.host`, session info | the `t` assertion helpers |
6507
+ | Can affect the run | no | yes, it owns the surface | n/a |
6508
+ | Authored at | `agent/hooks/*.ts` | channel config | `evals/**/*.eval.ts` |
6301
6509
 
6302
- ```ts
6303
- // agent/hooks/last-result.ts
6304
- import { defineHook } from "@cursor/july/hooks";
6510
+ ## Hook boundaries
6305
6511
 
6306
- export default defineHook({
6307
- events: {
6308
- async "turn.completed"(event, ctx) {
6309
- await ctx.host.kv.put(`last-result/${ctx.session.id}`, {
6310
- at: event.at,
6311
- result: event.data.result ?? null,
6312
- });
6313
- },
6314
- },
6315
- });
6316
- ```
6512
+ | You want to | Use instead |
6513
+ | ------------------------------------------------------------------------------------- | ---------------------------------------------------------------------------------------------------------------- |
6514
+ | Add context before the model runs | The channel's `send` message and `workspaceFiles`, `instructions.md`, skills, or `sandbox/workspace/` seed files |
6515
+ | Reply on Slack, comment on a PR, or post any other delivery | The channel's `events` map, or the Slack and GitHub packs |
6516
+ | Show PR progress (merge-box check, sticky banner) | `githubChannel({ progress: { commitStatus, banner } })`; see the [PR autofixer](/docs/templates/pr-autofixer.md) |
6517
+ | Block, approve, or rewrite a tool call | [`needsApproval`](/docs/reference/tools.md#gate-a-tool-on-human-approval) on the tool |
6518
+ | Act on the final assistant text, reject it for a same-turn repair, or fail a bad turn | `defineResult` |
6519
+ | Gate a change on behavior | [Evals](/docs/evals.md) |
6317
6520
 
6318
- ## Test and debug a hook
6521
+ ## Test a hook
6319
6522
 
6320
6523
  A hook definition is a plain object, so a unit test calls
6321
6524
  `hook.events["turn.completed"]` directly with an event and a stub
6322
6525
  `HookContext`. Discovery skips `*.test.ts`, so the test can live next to
6323
6526
  the hook.
6324
6527
 
6325
- At runtime:
6326
-
6327
- - `agent-sdk validate --dir .` reports discovery errors and the
6328
- empty-handlers warning.
6329
- - `agent-sdk info --dir . --json` lists the loaded hooks under
6330
- `agents[].hooks`.
6331
- - Send a turn with `agent-sdk dev` or `agent-sdk run --dir . --message "…"`
6332
- and watch the serve log for
6333
- `hook "<name>" handler for <event> threw: …`. `run` prints that log on
6334
- stderr. On hosting, read it with [`agent-sdk logs`](/docs/reference/cli.md#logs).
6528
+ | Command | What it reports |
6529
+ | --- | --- |
6530
+ | `agent-sdk validate --dir .` | Discovery errors and the empty-handlers warning |
6531
+ | `agent-sdk info --dir . --json` | Loaded hooks under `agents[].hooks` |
6532
+ | `agent-sdk run --dir . --message "…"` | Serve log on stderr, including `hook "<name>" handler for <event> threw: …` |
6335
6533
 
6336
- ## What's next
6534
+ On hosting, read the same log with [`agent-sdk logs`](/docs/reference/cli.md#logs).
6337
6535
 
6338
- Continue with these pages:
6536
+ ## Related
6339
6537
 
6538
+ - [Hooks guide](/docs/guides/hooks.md): meter usage and page on failure
6340
6539
  - [Sessions and streaming](/docs/reference/sessions.md): the event vocabulary hooks observe
6341
6540
  - [OpenTelemetry](/docs/guides/opentelemetry.md): OTLP traces and metrics
6342
6541
  from the same event stream
@@ -6348,34 +6547,33 @@ Continue with these pages:
6348
6547
 
6349
6548
  Source: /docs/reference/http-api.md
6350
6549
 
6351
- # HTTP API reference
6550
+ # HTTP API
6551
+
6552
+ Agent SDK hosts expose one public HTTP surface. In the default
6553
+ multi-agent layout, each agent uses `/<slug>/v1/*`; `--mode single`
6554
+ serves the same routes at `/v1/*`. Routes use the agent's HTTP auth
6555
+ chain, and session-owned resources return `403` to another principal.
6352
6556
 
6353
- Agent SDK hosts expose the same public HTTP surface. In the default
6354
- multi-agent layout each agent is namespaced under its slug
6355
- (`/<slug>/v1/session`, `/<slug>/playground`), with host-level routes at
6356
- the root. With `--mode single`, one agent serves the same surface
6357
- unslugged (`/v1/*`).
6557
+ Unless a section says otherwise, the default auth policy is
6558
+ `localDevStrict()`. `--bearer-token` replaces it with bearer auth, and
6559
+ `--allow-anonymous` replaces it with anonymous access. Built-in JSON
6560
+ routes use `{ ok: false, error: "<code>", message? }` for errors; MCP
6561
+ uses JSON-RPC, and custom channel handlers define their own responses.
6358
6562
 
6359
- Unless noted otherwise, routes run the agent's HTTP auth chain: the
6360
- default is `localDevStrict()` (loopback only), replaced by `bearerAuth` under
6361
- `--bearer-token` or `allowAll()` under `--allow-anonymous`. Session
6362
- routes also require the caller to be the session's owner (`403`
6363
- otherwise). Errors return JSON
6364
- `{ ok: false, error: "<code>", message? }` with a matching HTTP status.
6563
+ ## Host routes
6365
6564
 
6366
- ## Host-level routes (multi-agent mode)
6565
+ These routes live at the host root in multi-agent mode. The index routes
6566
+ exist only when the playground is enabled.
6367
6567
 
6368
- These routes live at the host root, above any agent. The two index
6369
- routes exist only while the playground is enabled (`--no-playground`
6370
- removes them) and run no auth. The documentation site is mounted in
6371
- both layouts and removed by `--no-docs`.
6568
+ | Route | Contract |
6569
+ | --- | --- |
6570
+ | `GET /` | HTML index of mounted agents; no auth |
6571
+ | `GET /v1/agents` | JSON index of mounted agents; no auth |
6572
+ | `GET /docs`, `GET /docs/*` | Documentation site in either layout; no auth |
6573
+ | `GET /v1/health` | Host liveness; no auth |
6372
6574
 
6373
- | Route | What it does |
6374
- | -------------------------- | ---------------------------------------------------------------------------------------------------------------------------- |
6375
- | `GET /` | A web index of every mounted agent, linking to playgrounds (playground only) |
6376
- | `GET /v1/agents` | The JSON index of mounted agents (playground only, no auth) |
6377
- | `GET /docs`, `GET /docs/*` | This documentation, served as a static site (both layouts, no auth) |
6378
- | `GET /v1/health` | Host-level liveness, no auth |
6575
+ `--no-playground` removes the two index routes. `--no-docs` removes the
6576
+ documentation site.
6379
6577
 
6380
6578
  ## Start a session
6381
6579
 
@@ -6389,18 +6587,18 @@ curl -X POST http://127.0.0.1:3000/<slug>/v1/session \
6389
6587
  # "playgroundUrl":"…?sessionId=ses_…","traceUrl":"…/v1/session/ses_…/events"}
6390
6588
  ```
6391
6589
 
6392
- The response returns as soon as the message is accepted; follow the
6393
- stream for progress. The continuation token is the follow-up credential,
6394
- and `playgroundUrl` deep-links the session in the playground.
6590
+ Once accepted, the response returns `sessionId` for inspection and
6591
+ `continuationToken` for follow-ups; follow the stream for progress.
6395
6592
 
6396
- | Body field | Meaning |
6593
+ | Body field | Contract |
6397
6594
  | --- | --- |
6398
6595
  | `message` | Required user message |
6399
- | `title` | Session title |
6596
+ | `title` | Display title |
6400
6597
  | `dryRun` | Run read tools and stub write tools |
6401
- | `asOf` | ISO-8601 instant with a timezone, frozen at create; the prompt states it, `ctx.now()` returns it, and tool calls with relative, later-than-`asOf`, or omitted schema-declared time bounds are refused. `400` when unusable |
6402
- | `workspaceFiles` | UTF-8 files written into the session workspace |
6598
+ | `asOf` | ISO-8601 instant with a timezone; sets `ctx.now()` and rejects omitted, relative, or later declared tool time arguments. Invalid values return `400` |
6599
+ | `workspaceFiles` | Relative files added to the session workspace |
6403
6600
  | `cloud` | Per-session cloud options merged over the agent defaults |
6601
+ | `purpose` | Use `"eval"` to mark regression traffic |
6404
6602
 
6405
6603
  ## Send a follow-up
6406
6604
 
@@ -6412,15 +6610,15 @@ curl -X POST http://127.0.0.1:3000/<slug>/v1/session/ses_… \
6412
6610
  -d '{"continuationToken":"http:…","message":"Make it shorter."}'
6413
6611
  ```
6414
6612
 
6415
- Works for any chat session, including ones created by custom channels.
6416
- Each accepted follow-up rotates the token, and the response carries the
6417
- new one. Sending to a busy session interrupts the in-flight turn, waits
6418
- for it to settle, then sends.
6613
+ The route accepts any chat session, including one created by a custom
6614
+ channel. Each accepted follow-up rotates the continuation token and
6615
+ returns the replacement. A message sent to a busy session interrupts
6616
+ the active turn before starting.
6419
6617
 
6420
- Expect `409` on a stale token or a task session. Task sessions do not accept
6421
- follow-ups. Expect `403` when the caller is not the session owner.
6618
+ The route returns `409` for a stale token or task session, and `403`
6619
+ when the caller doesn't own the session.
6422
6620
 
6423
- ## Stream a session
6621
+ ## Stream or replay session events
6424
6622
 
6425
6623
  `GET /v1/session/:sessionId/stream` is the live NDJSON feed.
6426
6624
 
@@ -6428,44 +6626,36 @@ follow-ups. Expect `403` when the caller is not the session owner.
6428
6626
  curl -N 'http://127.0.0.1:3000/<slug>/v1/session/ses_…/stream?startIndex=0'
6429
6627
  ```
6430
6628
 
6431
- One NDJSON event per line, from `startIndex`, then following live. The
6432
- default is `0`: omitting the parameter replays the entire recorded
6433
- stream before following. Pass the last index you've seen plus one to
6434
- resume without duplicates. The stream is durable and reconnectable. For
6435
- the vocabulary, see
6436
- [Sessions](/docs/reference/sessions.md#which-events-can-i-stream).
6629
+ The route replays one event per line from `startIndex`, then follows new
6630
+ events. The default `0` replays the full stream. To reconnect without
6631
+ duplicates, pass the last index you received plus one.
6437
6632
 
6438
6633
  `GET /v1/session/:sessionId/events` returns a one-shot NDJSON dump.
6439
- Pass `?format=json` for `{ sessionId, events, playgroundUrl }`.
6634
+ Pass `?format=json` for `{ sessionId, events, playgroundUrl }`. See
6635
+ [Stream events](/docs/reference/sessions.md#stream-events) for the event vocabulary.
6440
6636
 
6441
- ## Stop and list
6637
+ ## Manage sessions
6442
6638
 
6443
- `POST /v1/session/:sessionId/stop` interrupts the in-flight turn without
6444
- sending a new message. `GET /v1/sessions` lists sessions owned by the
6445
- calling principal. Under `serve --dev` on loopback it includes all
6446
- sessions, which is how webhook and schedule sessions show up in the
6447
- playground.
6448
-
6449
- ## Session cost
6450
-
6451
- `GET /v1/session/:sessionId/cost` returns the session's cost report:
6452
- per-turn token usage and the engine's estimated cost, folded from
6453
- `turn.completed` events. It runs the same owner check as the other
6454
- session routes and returns `404` for an unknown session. The
6455
- [`agent-sdk cost`](/docs/reference/cli.md#cost) command reports the same data.
6639
+ | Route | Contract |
6640
+ | --- | --- |
6641
+ | `POST /v1/session/:sessionId/stop` | Interrupt the active turn without sending another message |
6642
+ | `GET /v1/sessions` | List sessions owned by the caller |
6643
+ | `GET /v1/session/:sessionId/cost` | Return per-turn token usage and estimated cost |
6456
6644
 
6457
- ## Approvals
6645
+ On loopback under `serve --dev`, the session list includes every
6646
+ principal so webhook and schedule sessions appear in the playground.
6647
+ The cost route returns `404` for an unknown session.
6458
6648
 
6459
- Two routes list and resolve parked tool calls.
6649
+ ## Resolve tool approvals
6460
6650
 
6461
- | Route | What it does |
6462
- | ----------------------------------------------- | -------------------------------------------------------------- |
6463
- | `GET /v1/session/:sessionId/approvals` | Pending human-in-the-loop tool approvals |
6464
- | `POST /v1/session/:sessionId/approvals/:callId` | Resolve one: `{"decision":"approve"}` or `{"decision":"deny"}` |
6651
+ | Route | Contract |
6652
+ | --- | --- |
6653
+ | `GET /v1/session/:sessionId/approvals` | List pending tool approvals |
6654
+ | `POST /v1/session/:sessionId/approvals/:callId` | Resolve one with `{"decision":"approve"}` or `{"decision":"deny"}` |
6465
6655
 
6466
6656
  For the lifecycle, see [Gate a tool on human approval](/docs/reference/tools.md#gate-a-tool-on-human-approval).
6467
6657
 
6468
- ## Call a tool directly
6658
+ ## Call a tool without a model turn
6469
6659
 
6470
6660
  `POST /v1/tools/:toolName` runs a server tool with no model turn.
6471
6661
 
@@ -6477,133 +6667,118 @@ curl -X POST http://127.0.0.1:3000/<slug>/v1/tools/inspect_pr \
6477
6667
  # "isError":false,"result":{…},"durationMs":12}
6478
6668
  ```
6479
6669
 
6480
- It runs an authored server tool in-process: schema-validated, no model
6481
- turn. An optional `"sessionId"` in the body runs it inside an existing
6482
- session and records it on that session's stream (`409 session_busy`
6483
- for a write-effect call while a turn runs; reads run alongside the
6484
- turn). Agent-execution tools are rejected with `400`, and
6485
- unknown tools with `404` and the list of available names. For the
6486
- semantics, see [Tools](/docs/reference/tools.md#call-a-tool-without-a-model-turn).
6670
+ | Body field | Contract |
6671
+ | --- | --- |
6672
+ | `input` | Tool input; defaults to `{}` |
6673
+ | `sessionId` | Bind the call to an existing session |
6674
+ | `continuationToken` | Bind the call by its wire continuation token; mutually exclusive with `sessionId` |
6487
6675
 
6488
- An optional `"continuationToken"` (`<channelId>:<key>`, as
6489
- `/v1/sessions` lists it; mutually exclusive with `sessionId`) addresses
6490
- the session by continuation token instead; malformed tokens are
6491
- rejected with `400 invalid_continuation_token`. For the semantics, see
6492
- [Tools](/docs/reference/tools.md#call-a-tool-without-a-model-turn).
6676
+ Omit both identifiers for an unbound call. Agent-execution tools return
6677
+ `400`, and an unknown name returns `404` with the available names. See
6678
+ [Tools](/docs/reference/tools.md#call-a-tool-without-a-model-turn) for session
6679
+ binding, busy-session rules, and error codes.
6493
6680
 
6494
- ## Discovery
6681
+ ## Discovery routes
6495
6682
 
6496
6683
  These read-only routes describe the running agent.
6497
6684
 
6498
- | Route | What it does |
6499
- | ---------------------- | ---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
6500
- | `GET /v1/info` | The discovered surface: model, tools, skills, MCP connections, subagents, channels and routes (with schemas), schedules, hooks, diagnostics |
6501
- | `GET /v1/tools` | The live tool catalog: authored server tools plus advertised MCP passthroughs under model-facing names, as light `{ name, title?, source? }` entries. `session` / `continuationToken` query parameters bind the listing to a session identity (advertised inventories can be tenant-scoped); a connection whose listing fails is skipped and reported in `connectionErrors` |
6502
- | `GET /v1/tools/:name` | One catalog tool's full description: description, execution, `needsApproval`, `effect`, input and output schemas, source connection. Same session binding as the listing; unknown names get `404` with the available names |
6503
- | `GET /v1/health` | Per-agent liveness, no auth |
6504
- | `GET /v1/logs?after=N` | Recent server log lines, with a polling cursor |
6685
+ | Route | Contract |
6686
+ | --- | --- |
6687
+ | `GET /v1/info` | Return the discovered model, tools, skills, connections, subagents, channels, schedules, hooks, and diagnostics |
6688
+ | `GET /v1/tools` | Return live server tools and advertised MCP tools as `{ name, title?, source? }`; bind tenant-scoped listings with `session` or `continuationToken` |
6689
+ | `GET /v1/tools/:name` | Return one tool's description, execution mode, approval and effect rules, schemas, and source |
6690
+ | `GET /v1/health` | Return per-agent liveness; no auth |
6691
+ | `GET /v1/logs?after=N` | Return recent server logs and the next polling cursor |
6505
6692
 
6506
- ## Artifacts
6693
+ An unknown tool name returns `404` with available names. If an MCP
6694
+ connection can't list its tools, the catalog skips that connection and
6695
+ includes it in `connectionErrors`.
6507
6696
 
6508
- Two routes read durable artifacts tagged by `ctx.artifacts` or
6509
- `tag_artifact`. See [Artifacts](/docs/reference/artifacts.md).
6697
+ ## List and download artifacts
6510
6698
 
6511
- | Route | What it does |
6512
- | -------------------------------- | ------------------------------------------------------------------------------------------------------ |
6513
- | `GET /v1/artifacts` | List artifacts as `{ artifacts }`, newest-updated first. Filter with `?kind=`, `?sessionId=`, and `?limit=` (a positive integer) |
6514
- | `GET /v1/artifacts/:id/content` | Download one artifact's file or blob payload. Served as an attachment, never rendered inline; `404` when the artifact is unknown or carries no content |
6699
+ | Route | Contract |
6700
+ | --- | --- |
6701
+ | `GET /v1/artifacts` | Return `{ artifacts }`, newest-updated first; filter with `kind`, `sessionId`, and a positive `limit` |
6702
+ | `GET /v1/artifacts/:id/content` | Download an artifact's file or blob as an attachment; return `404` when no content exists |
6515
6703
 
6516
- Session ownership applies the same way as `GET /v1/sessions`: under
6517
- `serve --dev` on loopback (or `--allow-anonymous`) the list spans all
6518
- principals, while bearer or custom channel auth keeps strict
6519
- per-principal isolation.
6704
+ The list follows the same ownership rules as `GET /v1/sessions`. See
6705
+ [Artifacts](/docs/reference/artifacts.md) for tagging and record fields.
6520
6706
 
6521
- ## Custom channel routes
6707
+ ## Call custom channel routes
6522
6708
 
6523
- Authored routes mount under `/v1/channels/<id>` with the methods, paths,
6524
- and Zod schemas the channel declared (a
6525
- `POST /<slug>/v1/channels/drive` route, say). Bodies are validated before
6526
- handlers run (`400` on schema violations), and each channel's auth chain
6527
- applies. The GitHub channel verifies `X-Hub-Signature-256` when a
6528
- secret is configured. See [Channels](/docs/reference/channels.md).
6709
+ Authored routes mount under `/v1/channels/<id>` with their declared
6710
+ methods and paths. The host validates their Zod body and query schemas
6711
+ before calling the handler, returning `400` on failure. Each channel's
6712
+ auth chain applies. See [Channels](/docs/reference/channels.md).
6529
6713
 
6530
6714
  ## MCP endpoint
6531
6715
 
6532
- `/v1/mcp` serves the Model Context Protocol over streamable HTTP
6533
- (stateless; POST carries the protocol, and GET/DELETE return
6534
- spec-compliant 405s). The tools are `ask` (delegate a message, bounded
6535
- waits), `check` (poll a running session), and `call_tool` (deterministic
6536
- server-tool passthrough, present when the agent has server tools). The
6537
- route runs the same auth chain as the session API. Peer wiring:
6538
- [MCP connections](/docs/reference/connections.md#peer-mcp-connection).
6716
+ Both MCP routes use stateless streamable HTTP. Send protocol requests
6717
+ with `POST`; `GET` and `DELETE` return `405`.
6718
+
6719
+ | Route | Tools and auth |
6720
+ | --- | --- |
6721
+ | `/v1/mcp` | `ask`, `check`, and `call_tool` when server tools exist; uses the session API auth chain |
6722
+ | `/v1/mcp/tools` | Deterministic server tools only; uses the CLI-level loopback, bearer, or anonymous auth chain |
6539
6723
 
6540
- `/v1/mcp/tools` is a second stateless MCP endpoint exposing only the
6541
- agent's deterministic server tools. Hosted cloud turns call back into
6542
- it through the URL configured by `serve --cloud-tools-url`. Unlike
6543
- `/v1/mcp`, it runs the CLI-level auth chain (loopback, bearer, or
6544
- anonymous), not any authored channel auth.
6724
+ See [MCP connections](/docs/reference/connections.md#peer-mcp-connection) to connect
6725
+ one agent to another.
6545
6726
 
6546
6727
  ## Playground eval routes
6547
6728
 
6548
- The playground Evals tab and `agent-sdk eval --prod` / `--url` use these:
6729
+ The playground Evals tab and remote eval commands use these routes:
6549
6730
 
6550
- | Route | What it does |
6551
- | ------------------------------- | ---------------------------------------------------------------------------------------------------------- |
6552
- | `GET /v1/dev/evals` | List discovered eval datapoints and project config |
6553
- | `GET /v1/dev/evals/runs` | List recent run snapshots, newest first |
6554
- | `POST /v1/dev/evals/runs` | Start an eval run (`{filterIds?, tags?}`); `202` with a snapshot (`runId` is the Eval ID), `404` when nothing matches, `409` when one is running |
6555
- | `GET /v1/dev/evals/runs/:runId` | Poll a run's progress |
6556
- | `POST /v1/dev/evals/runs/:runId/cancel` | Cancel a running batch; `200` with snapshot, `404` unknown, `409` when not running |
6731
+ | Route | Contract |
6732
+ | --- | --- |
6733
+ | `GET /v1/dev/evals` | List discovered cases and eval config |
6734
+ | `GET /v1/dev/evals/runs` | List recent run snapshots, newest first |
6735
+ | `POST /v1/dev/evals/runs` | Start `{ filterIds?, tags?, timeoutMs?, verbose? }`; return `202`, `404` for no match, or `409` while another run is active |
6736
+ | `GET /v1/dev/evals/runs/:runId` | Return progress and the final snapshot |
6737
+ | `POST /v1/dev/evals/runs/:runId/cancel` | Cancel an active run; return `404` when unknown or `409` when no longer running |
6557
6738
 
6558
- Eval runs are asynchronous. Poll the run route for case progress and
6559
- the final `completed` or `failed` status. Batch errors appear on the
6560
- snapshot returned by the poll. Entries within `filterIds` and `tags`
6561
- use OR semantics. When both fields are present, a case must match one
6562
- entry from each field. Listed runs persist across restarts when
6563
- durable storage is configured. Otherwise they are
6564
- process-memory only.
6739
+ Poll the run route until its status is `completed`, `failed`, or
6740
+ `cancelled`. Entries within `filterIds` and `tags` use OR semantics;
6741
+ when both fields are present, a case must match each group.
6565
6742
 
6566
6743
  ## Dev-mode routes
6567
6744
 
6568
6745
  These routes exist only under `serve --dev`.
6569
6746
 
6570
- | Route | What it does |
6571
- | ------------------------------------ | ---------------------------------------------------------------------------------------------------------- |
6747
+ | Route | Contract |
6748
+ | --- | --- |
6572
6749
  | `POST /v1/dev/schedules/:scheduleId` | Dispatch a schedule by hand, exactly once. Returns `{scheduleId, sessionIds}` |
6573
- | `GET /v1/dev/reminders` | List reminders |
6574
- | `POST /v1/dev/reminders/:reminderId` | Fire a reminder by hand |
6750
+ | `GET /v1/dev/reminders` | List reminders |
6751
+ | `POST /v1/dev/reminders/:reminderId` | Fire a reminder by hand |
6575
6752
 
6576
6753
  Schedules and reminders never fire automatically in dev mode. These
6577
- routes are the only way they run, which keeps iteration deterministic.
6754
+ routes run them manually.
6578
6755
 
6579
6756
  ## Playground assets
6580
6757
 
6581
6758
  `GET /playground` and `GET /playground/assets/:file` serve the
6582
- playground (omitted with `--no-playground`). It calls the JSON API
6583
- above and has no privileged surface.
6759
+ playground. `--no-playground` removes both routes.
6584
6760
 
6585
6761
  ## Status codes
6586
6762
 
6587
- Error responses use a small, consistent set of status codes.
6763
+ Built-in routes use these common status codes.
6588
6764
 
6589
- | Code | Meaning here |
6590
- | ----- | -------------------------------------------------------------------------------------------------------------------------- |
6591
- | `400` | Schema-invalid body or query, agent-execution tool called on the host, malformed request |
6592
- | `401` | No auth policy admitted the request |
6593
- | `403` | Authenticated, but not the session owner |
6594
- | `404` | Unknown session, tool, schedule, reminder, or eval run; no eval datapoints match a run request |
6595
- | `405` | Wrong method (GET on the MCP endpoint, say) |
6596
- | `409` | Stale continuation token, a busy session-bound tool call, a non-followable task session, or an eval run already in progress |
6597
- | `202` | Accepted for background work (GitHub `{ task }` hooks, eval runs) |
6598
-
6599
- ## What's next
6765
+ | Code | Meaning |
6766
+ | --- | --- |
6767
+ | `400` | Invalid body, query, tool input, or request shape |
6768
+ | `401` | No auth policy admitted the request |
6769
+ | `403` | The caller is authenticated but doesn't own the resource |
6770
+ | `404` | A named resource doesn't exist, or an eval selection matches no cases |
6771
+ | `405` | The route doesn't accept this method |
6772
+ | `409` | The resource state rejects the request, such as a stale token or busy write |
6773
+ | `202` | The request was accepted for asynchronous work |
6774
+ | `500` | A write-effect tool can't create its session workspace (`workspace_unavailable`) |
6600
6775
 
6601
- Continue with these pages:
6776
+ ## Related
6602
6777
 
6603
- - [Sessions and streaming](/docs/reference/sessions.md): the handles and events these
6604
- routes traffic in
6605
- - [Channels](/docs/reference/channels.md): authoring your own routes
6606
- - [Deployment](/docs/deployment.md): auth on real hosts
6778
+ - [Sessions](/docs/reference/sessions.md)
6779
+ - [Channels](/docs/reference/channels.md)
6780
+ - [Tools](/docs/reference/tools.md)
6781
+ - [Deployment](/docs/deployment.md)
6607
6782
 
6608
6783
  ---
6609
6784
 
@@ -6611,19 +6786,17 @@ Source: /docs/reference/instructions.md
6611
6786
 
6612
6787
  # Instructions
6613
6788
 
6614
- `agent/instructions.md` is the always-on system prompt. It's the one
6615
- piece of prose the model sees on every turn. It's required on the root
6616
- agent; subagents may inline `instructions` in their `agent.ts` instead.
6789
+ Agent instructions form the always-on system prompt and reach the model
6790
+ on every turn. A root agent requires them; a subagent may inline
6791
+ `instructions` in `agent.ts` instead.
6617
6792
 
6618
6793
  ## Authoring forms
6619
6794
 
6620
- Three forms cover every case.
6621
-
6622
- | Form | Reach for it when |
6795
+ | Form | Use it when |
6623
6796
  | --- | --- |
6624
- | `agent/instructions.md` | Plain Markdown for most agents. |
6625
- | `agent/instructions.ts` | Generated prompts. Default-export `defineInstructions({ markdown })` or a plain string. |
6626
- | `agent/instructions/` directory | A long prompt split across files, composed in filename order. |
6797
+ | `agent/instructions.md` | Plain Markdown for most agents |
6798
+ | `agent/instructions.ts` | Generated prompts. Default-export `defineInstructions({ markdown })` or a plain string |
6799
+ | `agent/instructions/` directory | A long prompt split across files, composed in filename order |
6627
6800
 
6628
6801
  ```ts
6629
6802
  // agent/instructions.ts
@@ -6634,22 +6807,17 @@ export default defineInstructions({
6634
6807
  });
6635
6808
  ```
6636
6809
 
6637
- ## How instructions reach the model
6810
+ ## Delivery
6638
6811
 
6639
- On the local runtime, instructions land in the session workspace as
6640
- `AGENTS.md`, and the harness loads them natively. On the cloud runtime,
6641
- they're prepended to the first prompt, because the cloud VM doesn't
6642
- share the local session workspace.
6812
+ Local and cloud turns receive the composed instructions. They aren't
6813
+ written into the session workspace. Parent directories can still
6814
+ contribute ambient `AGENTS.md` and `.cursor` settings; [Agent config:
6815
+ local cwd](/docs/reference/agent-config.md#local-cwd) covers how to control that.
6643
6816
 
6644
- The local workspace is a real Cursor project directory, so the harness
6645
- may also load ambient `AGENTS.md` and `.cursor` config from ancestor
6646
- directories. [Agent config → Local cwd](/docs/reference/agent-config.md#local-cwd)
6647
- covers controlling that.
6817
+ ## Contents
6648
6818
 
6649
- ## What to put in instructions
6650
-
6651
- Keep them a few lines: identity, when to use which tool, output shape.
6652
- The [quickstart PR approver](/docs/quickstart.md) is the pattern:
6819
+ Keep them a few lines: identity, when to use which tool, and the output
6820
+ shape. The [quickstart PR approver](/docs/quickstart.md) is the pattern:
6653
6821
 
6654
6822
  ```md
6655
6823
  # PR approver
@@ -6663,21 +6831,13 @@ You review GitHub pull requests. Be specific and brief.
6663
6831
  End with one sentence: the verdict and why.
6664
6832
  ```
6665
6833
 
6666
- - Name the tools and the decision rule ("use X before answering about
6667
- Y"), not general encouragement.
6668
- - State the output contract: length, format, fences. That contract is
6669
- what your [evals](/docs/evals.md) gate.
6670
- - Move procedures to [skills](/docs/reference/skills.md). A multi-step workflow the
6671
- model only sometimes needs belongs in `agent/skills/`, where it loads
6672
- on demand and keeps the always-on prompt small.
6673
-
6674
- Instructions are the third lever in the
6675
- [hillclimbing loop](/docs/hillclimbing.md), after host preparation and evidence
6676
- shape. If a fixture keeps failing, look there before rewriting prose.
6677
-
6678
- ## What's next
6834
+ Name the tools and the decision rule ("use X before answering about
6835
+ Y"), not general encouragement. State the output contract, including
6836
+ length, format, and fences, so your [evals](/docs/evals.md) can gate it.
6837
+ Put multi-step workflows the model only sometimes needs in
6838
+ `agent/skills/`; they load on demand and keep the always-on prompt small.
6679
6839
 
6680
- Continue with these pages:
6840
+ ## Related
6681
6841
 
6682
6842
  - [Skills](/docs/reference/skills.md): procedures the model loads only when relevant
6683
6843
  - [Agent config](/docs/reference/agent-config.md): the file next to this one
@@ -6690,55 +6850,38 @@ Source: /docs/reference/playground.md
6690
6850
 
6691
6851
  # Playground
6692
6852
 
6693
- Every served agent ships with a web playground at
6853
+ `agent-sdk serve` enables a web playground at
6694
6854
  `http://127.0.0.1:3000/<slug>/playground` (or `/playground` in single
6695
- mode). Anything you can do there you can also do with curl.
6696
-
6697
- ## What it does
6698
-
6699
- Use the playground to chat, try channel routes, and inspect sessions.
6700
-
6701
- - **Chat** with the agent. Text and reasoning stream live, and tool
6702
- calls appear inline with their arguments, output, and error state.
6703
- - **Slash commands**: custom channel routes become composer commands
6704
- (a `drive` route becomes `/drive <pr-url>`), with `/help` and
6705
- autocomplete.
6706
- - **Try** any channel route from the Agent surface. The modal remembers
6707
- your last body per endpoint and has Copy curl, and a successful Try
6708
- opens the created session.
6709
- - **Sessions**: browse the sessions you own (chat, custom-channel,
6710
- schedule tasks) and replay their event streams. In `--dev` on
6711
- loopback, or with `--allow-anonymous`, the list includes every
6712
- principal. Search by session ID to filter the list, or press Enter
6713
- to open an ID directly. "Open trace" renders a saved event stream.
6714
- - **Approvals**: parked `needsApproval` tool calls render Approve /
6715
- Deny buttons.
6716
- - **Evals**: list and run filesystem evals from the browser (backed by
6717
- `/v1/dev/evals`). Schedule hand-dispatch still requires `--dev`.
6718
- - **The surface**: inspect the discovered tools, skills, subagents, MCP
6719
- connections, channels, and hooks.
6720
- - **Custom tool chips**: drop `agent/playground/tools/<toolName>.tsx` to
6721
- change how that tool renders. Chips compile from the agent tree;
6722
- an [extension](/docs/reference/extensions.md) cannot contribute them.
6723
- - **Raw events pane**: flip it on to inspect the event stream.
6724
- - **Logs tab**: recent server log lines, polled from `GET /v1/logs`.
6725
-
6726
- In multi-agent mode each agent has its own playground at
6727
- `/<slug>/playground`, and `/` is an index of them all.
6728
-
6729
- ## Share it beyond localhost
6855
+ mode). In multi-agent mode, each agent has its own playground, and `/`
6856
+ lists them all.
6857
+
6858
+ ## Playground surfaces
6859
+
6860
+ | Surface | What you can do |
6861
+ | --- | --- |
6862
+ | Chat | Talk to the agent. Text and reasoning stream live, and tool calls appear inline with their arguments, output, and error state |
6863
+ | Slash commands | Custom channel routes without path parameters become composer commands; GitHub and Slack ingress routes are excluded. A `drive` route becomes `/drive <pr-url>`, with `/help` and autocomplete |
6864
+ | Try | Invoke a custom channel route from the Agent tab. The modal remembers your last body per endpoint, copies curl, and opens a session when the route creates one |
6865
+ | Runs | Browse the sessions you own and eval runs. Search by title or identifier; open sessions in Chat, Trace, or Raw and eval runs in Evals |
6866
+ | Approvals | Parked `needsApproval` tool calls render Approve / Deny buttons |
6867
+ | Evals | List and run filesystem evals from the browser |
6868
+ | Agent | Inspect the discovered tools, skills, subagents, MCP connections, channels, and hooks |
6869
+ | Raw | Inspect the selected session's event stream |
6870
+ | Logs | Recent server log lines, polled from `GET /v1/logs` |
6871
+
6872
+ In `--dev` on loopback, or with `--allow-anonymous`, the session list
6873
+ includes every principal.
6874
+
6875
+ ## Remote access
6730
6876
 
6731
6877
  The default `localDevStrict()` auth admits direct loopback calls only
6732
6878
  and rejects proxy-forwarding headers, so a tunnel or LAN address won't
6733
- work
6734
- until you pass `--bearer-token <secret>` (or
6879
+ work until you pass `--bearer-token <secret>` (or
6735
6880
  `serve(dir, { authToken })`). Open the playground on the remote device
6736
- and paste the token into the token field in the navbar. `--allow-anonymous` is the
6737
- demo-only alternative for trusted networks.
6738
-
6739
- ## What's next
6881
+ and paste the token into the token field in the navbar.
6882
+ `--allow-anonymous` is the demo-only alternative for trusted networks.
6740
6883
 
6741
- Continue with these pages:
6884
+ ## Related
6742
6885
 
6743
6886
  - [HTTP API](/docs/reference/http-api.md): the HTTP surface the playground uses
6744
6887
  - [Sessions and streaming](/docs/reference/sessions.md): the streams it renders
@@ -6775,7 +6918,7 @@ to the directory name. When serving multiple agents, the slug is the
6775
6918
  directory name and must match `[A-Za-z0-9][A-Za-z0-9_-]*` (and not the
6776
6919
  reserved `v1`, `playground`, or `docs` segments).
6777
6920
 
6778
- ## Project overview
6921
+ ## Project tree
6779
6922
 
6780
6923
  Most projects start with this shape.
6781
6924
 
@@ -6800,8 +6943,8 @@ my-agent/
6800
6943
  ```
6801
6944
 
6802
6945
  Evals live in `evals/` at the project root, a sibling of `agent/`, never
6803
- inside it. `agent/evals/` is silently ignored. See
6804
- [Evals](/docs/evals.md).
6946
+ inside it. `agent/evals/` is ignored, and `validate` warns about it.
6947
+ See [Evals](/docs/evals.md).
6805
6948
 
6806
6949
  ## Folder reference
6807
6950
 
@@ -6819,30 +6962,26 @@ Each path maps to a capability and a reference page.
6819
6962
  | `agent/extensions/<ns>.ts` or `agent/extensions/<ns>/` | A mounted extension or Cursor plugin; its contributions become `<ns>__<name>` | [Extensions](/docs/reference/extensions.md) |
6820
6963
  | `agent/channels/*.ts` | HTTP surfaces beyond the built-in session API; `slack.ts` and `github.ts` use the platform packs | [Channels](/docs/reference/channels.md) |
6821
6964
  | `agent/hooks/*.ts` | Observe-only event subscribers, never fatal | [Hooks](/docs/reference/hooks.md) |
6822
- | `agent/otel.ts` | Factory-only OTLP authoring (`defineOtel`). Public path is env / `serve({ otel })`. | [OpenTelemetry](/docs/guides/opentelemetry.md) |
6823
- | `agent/storage.ts` | `defineStorage` backend for the durable `host.kv` / `host.files` APIs | None |
6824
6965
  | `agent/artifacts.ts` | `defineArtifacts` kinds, the `tag_artifact` opt-in, and retention | [Artifacts](/docs/reference/artifacts.md) |
6825
6966
  | `agent/result.ts` | `defineResult` host `commit` on the final assistant text (`throw` or `ctx.reject`) | None |
6826
6967
  | `agent/schedules/*` | Cron-driven runs (UTC, 5-field; never auto-fire under `--dev`) | [Schedules](/docs/reference/schedules.md) |
6827
- | `agent/sandbox/workspace/**` | Seed files copied into each local session workspace | [Sessions](/docs/reference/sessions.md#what-goes-into-a-local-session-workspace) |
6828
- | `agent/playground/` | Custom playground tool chips | [Playground](/docs/reference/playground.md) |
6968
+ | `agent/reminders/*.ts` | Named reminder handlers that stay armed across restarts | [Schedules](/docs/reference/schedules.md#reminders) |
6969
+ | `agent/sandbox/workspace/**` | Seed files copied into each local session workspace | [Sessions](/docs/reference/sessions.md#local-session-workspace) |
6829
6970
  | `agent/lib/` | Import-only shared code, never discovered | None |
6830
6971
  | `evals/evals.config.ts` | Shared eval settings (e.g. `maxConcurrency`); required when evals exist | [Evals](/docs/evals.md) |
6831
6972
  | `evals/**/*.eval.ts` | Filesystem evals; case id = path under `evals/` | [Evals](/docs/evals.md) |
6832
6973
 
6833
- `agent/lib/` is the only place for shared code. Everything else under
6834
- `agent/` is discovery surface. A stray `.ts` file in one of these
6835
- folders is treated as a definition.
6974
+ Use `agent/lib/` for shared imports. A `.ts` file in one of the listed
6975
+ discovery folders loads as a definition; unrecognized directories
6976
+ produce a validation warning.
6836
6977
 
6837
- ## Why didn't the Agent SDK discover my file?
6978
+ ## Inspect discovery
6838
6979
 
6839
6980
  Run `agent-sdk validate --dir .` and `agent-sdk info --dir .`.
6840
6981
  `validate` prints diagnostics, and `serve` refuses to start on
6841
6982
  error-severity ones. Warnings, such as cloud runtime combined with
6842
6983
  local-only capabilities, print but don't block. `info` lists the discovered
6843
- surface, so a missing tool or channel shows up immediately. From
6844
- there, check the folder reference: the file is usually in the wrong directory
6845
- or has the wrong extension.
6984
+ surface, so a missing tool or channel shows up immediately.
6846
6985
 
6847
6986
  ```bash
6848
6987
  agent-sdk validate --dir . # diagnostics; non-zero exit on errors
@@ -6850,51 +6989,50 @@ agent-sdk info --dir . # human-readable surface
6850
6989
  agent-sdk info --dir . --json # machine-readable project info (same shape as GET /v1/info)
6851
6990
  ```
6852
6991
 
6853
- ## What's next
6854
-
6855
- Continue with these pages:
6992
+ ## Related
6856
6993
 
6857
6994
  - [Agent config](/docs/reference/agent-config.md): the runtime config at the root
6858
6995
  - [Tools](/docs/reference/tools.md): add typed actions under `agent/tools/`
6996
+ - [CLI](/docs/reference/cli.md): commands that discover this tree
6859
6997
 
6860
6998
  ---
6861
6999
 
6862
7000
  Source: /docs/reference/prompt.md
6863
7001
 
6864
- # `prompt`
7002
+ # Prompt strings
6865
7003
 
6866
- Authoring helper for long strings that live next to indented TypeScript:
6867
- tool descriptions, reminder `prompt` fields, GitHub channel `context`,
6868
- and error messages.
7004
+ The `prompt` template tag keeps multi-line strings aligned with the
7005
+ surrounding TypeScript while returning dedented text. Use it for tool
7006
+ descriptions, reminder prompts, channel context, and errors. Import it
7007
+ from the package root or the dedicated entrypoint:
6869
7008
 
6870
7009
  ```ts
6871
7010
  import { prompt } from "@cursor/july";
6872
7011
  // or: import { prompt } from "@cursor/july/prompt";
6873
7012
  ```
6874
7013
 
6875
- ## `prompt\`…\``
7014
+ ## Dedented strings
6876
7015
 
6877
- Returns a single dedented string. Common leading whitespace is stripped;
6878
- a leading newline after the opening backtick is dropped so the usual
6879
- multiline form stays readable in source.
7016
+ `prompt` returns one string. It removes the common leading whitespace
7017
+ and one newline immediately after the opening backtick.
6880
7018
 
6881
7019
  ```ts
6882
7020
  throw new Error(prompt`
6883
- It is outside business hours (MonFri 9am5pm ET).
7021
+ It is outside business hours (Mon-Fri 9am-5pm ET).
6884
7022
  Use request_author_approval, or pass approval=human_request.
6885
7023
  `);
6886
7024
  ```
6887
7025
 
6888
7026
  Blank lines inside the body are preserved. Relative indentation after the
6889
- common prefix is kept (handy for nested bullet lists).
7027
+ common prefix is preserved, so nested lists keep their shape.
6890
7028
 
6891
7029
  When interpolating multi-line values (for example a list of services), give
6892
7030
  those lines the same indent as the `prompt` body so dedent stays consistent.
6893
7031
 
6894
- ## `prompt.lines\`…\``
7032
+ ## Line arrays
6895
7033
 
6896
- Same dedent rules, but returns `string[]`, one entry per line. Use this
6897
- where an API wants separate lines (for example GitHub channel `context`):
7034
+ `prompt.lines` applies the same rules and returns `string[]`, with one
7035
+ entry per line. Use it when an API accepts separate context lines:
6898
7036
 
6899
7037
  ```ts
6900
7038
  context: prompt.lines`
@@ -6904,13 +7042,18 @@ context: prompt.lines`
6904
7042
  `
6905
7043
  ```
6906
7044
 
7045
+ ## Related
7046
+
7047
+ - [Tools](/docs/reference/tools.md)
7048
+ - [Channels](/docs/reference/channels.md)
7049
+
6907
7050
  ---
6908
7051
 
6909
7052
  Source: /docs/reference/schedules.md
6910
7053
 
6911
7054
  # Schedules and reminders
6912
7055
 
6913
- Two ways an agent acts without an inbound message. A schedule is
7056
+ An agent can act without an inbound message in two ways. A schedule is
6914
7057
  deploy-time cron: "every weekday at 09:00, summarize open incidents." A
6915
7058
  reminder is a runtime wake bound to one conversation: "re-check this
6916
7059
  PR's CI in two hours." Schedules live in the filesystem; reminders are
@@ -6921,10 +7064,9 @@ created by running code.
6921
7064
  Cron expressions are standard 5-field, evaluated in UTC with minute
6922
7065
  granularity.
6923
7066
 
6924
- ### Schedules in Markdown
7067
+ ### Markdown schedules
6925
7068
 
6926
- A plain markdown file with `cron:` frontmatter is a fire-and-forget
6927
- task:
7069
+ A markdown file with `cron:` frontmatter is a fire-and-forget task:
6928
7070
 
6929
7071
  ```md
6930
7072
  ---
@@ -6938,10 +7080,10 @@ Each firing starts a task-mode session: the body is the prompt, the
6938
7080
  session runs to `session.completed` or `session.failed`, and it isn't
6939
7081
  followable.
6940
7082
 
6941
- ### Schedules as handlers
7083
+ ### Schedule handlers
6942
7084
 
6943
- `defineSchedule` with a `run` handler gives you full control, most
6944
- usefully to hand the work into a channel so its delivery events fire:
7085
+ Use `defineSchedule` with a `run` handler to call tools or hand work into
7086
+ a channel so its delivery events fire:
6945
7087
 
6946
7088
  ```ts
6947
7089
  import { defineSchedule } from "@cursor/july/schedules";
@@ -6950,7 +7092,6 @@ import webhook from "../channels/webhook.js";
6950
7092
  export default defineSchedule({
6951
7093
  cron: "*/30 * * * *",
6952
7094
  async run({ receive, waitUntil, appAuth, host }) {
6953
- // optional: await host.mcp.callTool("units", "celsius_to_fahrenheit", { value: 0 });
6954
7095
  waitUntil(
6955
7096
  receive(webhook, {
6956
7097
  message:
@@ -6969,14 +7110,12 @@ schedule-scoped principal for work the agent does on its own behalf),
6969
7110
  and `host` (shared services: `host.mcp`, `host.github`, `host.slack`,
6970
7111
  `host.reminders`).
6971
7112
 
6972
- ### Dispatch and dev mode
7113
+ ### Dispatch a schedule
6973
7114
 
6974
- In production (`agent-sdk serve`), schedules fire on their cron
6975
- cadence. Disable them with `--no-schedules`. There's no cross-host
6976
- coordination, so run them in exactly one process per project.
6977
-
6978
- In dev (`serve --dev`), schedules never fire automatically. Dispatch one
6979
- by hand, exactly once, through the same path production uses:
7115
+ | Mode | What fires |
7116
+ | --- | --- |
7117
+ | Production `agent-sdk serve` | Cron cadence. `--no-schedules` disables them. Run them in exactly one process per project |
7118
+ | `serve --dev` | Nothing automatic. Dispatch by hand through the same path production uses |
6980
7119
 
6981
7120
  ```bash
6982
7121
  curl -X POST http://127.0.0.1:3000/<slug>/v1/dev/schedules/heartbeat
@@ -6988,104 +7127,67 @@ The playground can dispatch schedules in dev mode too, and
6988
7127
 
6989
7128
  ## Reminders
6990
7129
 
6991
- A reminder is created at runtime and bound to a channel continuation.
6992
- When it fires, it wakes that conversation. Recurring reminders behave
6993
- like `setInterval`, one-shots like `setTimeout`, and both are durable on
6994
- disk.
7130
+ A reminder is created at runtime and bound to a channel continuation. A
7131
+ prompt reminder wakes that conversation when it fires; a handler reminder
7132
+ runs host code, which can call `followup` to wake it. Recurring reminders
7133
+ behave like `setInterval`, and one-shots behave like `setTimeout`. Prompt
7134
+ and named-handler reminders stay armed across restarts.
6995
7135
 
6996
7136
  ```ts
6997
7137
  await handle.createReminder({
6998
7138
  purpose: "ci_recheck",
6999
7139
  channelId: "drive",
7000
- continuationToken: "pr:owner/repo#1",
7001
- delay: "2h", // or a cron / explicit schedule
7140
+ continuationToken: "pr:acme/checkout#42",
7141
+ delay: "2h",
7002
7142
  prompt: "Re-check CI. Only act if still failing.",
7003
7143
  until: "Cancel once CI is green or the PR is merged.",
7004
7144
  });
7005
7145
  ```
7006
7146
 
7007
- Use `run` when host code should decide what happens on each tick:
7147
+ When host code must decide what happens on each tick, author a named
7148
+ handler and pass its default export with serializable `args`:
7008
7149
 
7009
7150
  ```ts
7010
- await handle.createReminder({
7011
- purpose: "ci_recheck",
7012
- channelId: "drive",
7013
- continuationToken: "pr:owner/repo#1",
7014
- every: "30m",
7015
- async run({ fireCount, followup }) {
7016
- if (fireCount >= 3) {
7017
- return { action: "stop" };
7018
- }
7151
+ // agent/reminders/ci_recheck.ts
7152
+ import { defineReminder } from "@cursor/july/reminders";
7019
7153
 
7154
+ export default defineReminder({
7155
+ async run({ args, followup }) {
7020
7156
  await followup({
7021
- message: "Re-check CI and report only if the status changed.",
7157
+ message: `Re-check CI for ${String(args.prUrl)}. Report only if the status changed.`,
7022
7158
  });
7023
7159
  return { action: "delivered" };
7024
7160
  },
7025
7161
  });
7026
7162
  ```
7027
7163
 
7028
- This reminder wakes the conversation three times, then stops itself.
7029
-
7030
- The same API is `host.reminders` on channel handlers, tools, and
7031
- schedule runs. An agent can even be given a tool that creates its own
7032
- reminders.
7033
-
7034
- For example, create `agent/tools/remind_me.ts`:
7035
-
7036
7164
  ```ts
7037
- import { defineTool } from "@cursor/july/tools";
7038
- import { z } from "zod";
7165
+ import ciRecheck from "./agent/reminders/ci_recheck.js";
7039
7166
 
7040
- export default defineTool({
7041
- description: "Schedule a one-time reminder in this conversation.",
7042
- inputSchema: z.object({
7043
- delay: z
7044
- .string()
7045
- .describe("When to wake the conversation, such as 20m or 2h"),
7046
- prompt: z
7047
- .string()
7048
- .min(1)
7049
- .describe("What the agent should do when it wakes"),
7050
- }),
7051
- async execute({ delay, prompt }, ctx) {
7052
- const reminders = ctx.host.reminders;
7053
- if (reminders === undefined) {
7054
- throw new Error("Reminders are disabled on this host.");
7055
- }
7056
-
7057
- const continuationToken = ctx.session.continuationKey;
7058
- if (continuationToken == null) {
7059
- throw new Error("This session cannot receive reminder follow-ups.");
7060
- }
7061
-
7062
- const reminder = await reminders.create({
7063
- purpose: "user_follow_up",
7064
- channelId: ctx.session.channelId,
7065
- continuationToken,
7066
- delay,
7067
- prompt,
7068
- });
7069
-
7070
- return {
7071
- reminderId: reminder.id,
7072
- nextFireAt: reminder.nextFireAt,
7073
- };
7074
- },
7167
+ await handle.createReminder({
7168
+ purpose: "ci_recheck",
7169
+ channelId: "drive",
7170
+ continuationToken: "pr:acme/checkout#42",
7171
+ every: "30m",
7172
+ handler: ciRecheck,
7173
+ args: { prUrl: "https://github.com/acme/checkout/pull/42" },
7075
7174
  });
7076
7175
  ```
7077
7176
 
7078
- The tool binds the reminder to the current channel conversation. When
7079
- the delay expires, the prompt returns to the same session as a follow-up.
7177
+ Use an anonymous `run` handler only for legacy projects that can re-arm
7178
+ it after a restart. New projects should use a named handler.
7080
7179
 
7081
- Reminders fire in one of two styles. The **prompt form** (above) sends
7082
- `prompt` into the session, with `until` stating the standing
7083
- cancellation condition for the model to honor. The **run form** passes a
7084
- `run` handler instead: it returns `stop`, `skip`, or `delivered` per
7085
- tick. That's silent host-side policy with no model turn. Run handlers are
7086
- in-memory, so after a restart those reminders are disarmed
7087
- (`handler_lost_on_restart`); re-arm them from the code path that created
7088
- them, or prefer the prompt form.
7180
+ | Form | What it does | After a restart |
7181
+ | --- | --- | --- |
7182
+ | Prompt | Sends `prompt` into the session. `until` is the standing cancellation condition for the model | Stays armed |
7183
+ | Named handler | Runs a discovered handler with serializable `args`; it starts a model turn only if the handler calls `followup` | Stays armed |
7184
+ | Anonymous `run` | Host handler returns `stop`, `skip`, or `delivered` per tick; it starts a model turn only if it calls `followup` | Must be re-armed |
7185
+
7186
+ The same API is `host.reminders` on channel handlers, tools, and
7187
+ schedule runs. `builtinTools: { reminders: true }` adds
7188
+ `reminders_create`, `reminders_list`, and `reminders_cancel` on the
7189
+ current conversation; see
7190
+ [Agent config: built-in tools](/docs/reference/agent-config.md#built-in-tools).
7089
7191
 
7090
7192
  `--dev` does not auto-fire reminders. Dispatch one by hand:
7091
7193
 
@@ -7096,27 +7198,21 @@ curl -X POST http://127.0.0.1:3000/<slug>/v1/dev/reminders/<id> # fire one
7096
7198
 
7097
7199
  `handle.dispatchReminder(id)` is the programmatic equivalent.
7098
7200
 
7099
- Two habits worth copying: cancel reminders when their subject
7100
- dies (say, cancel PR-scoped reminders on `pull_request.closed`), and keep
7101
- wake prompts generic. A plain "re-check the PR" works better than
7102
- replaying stale payload details, because the agent re-reads the live
7103
- state when it wakes.
7104
-
7105
- ## Schedule or reminder?
7201
+ Cancel reminders when their subject dies, for example on
7202
+ `pull_request.closed`. Keep wake prompts generic so the agent re-reads
7203
+ live state instead of replaying a stale payload.
7106
7204
 
7107
- The split comes down to scope and timing.
7205
+ ## Schedules vs reminders
7108
7206
 
7109
7207
  | | Schedule | Reminder |
7110
7208
  | -------- | ---------------------------------------------------------- | ----------------------------------------------- |
7111
7209
  | Defined | at deploy time, `agent/schedules/*` | at runtime, `createReminder` / `host.reminders` |
7112
7210
  | Scope | global to the agent | one channel continuation (one conversation) |
7113
7211
  | Session | starts a new task session (or hands off through `receive`) | wakes an existing conversation |
7114
- | Cadence | cron (UTC) | delay, cron, or explicit schedule |
7212
+ | Cadence | cron (UTC) | `every`, `cron`, `delay`, or `at` |
7115
7213
  | Dev mode | manual dispatch only | manual dispatch only (timers off) |
7116
7214
 
7117
- ## What's next
7118
-
7119
- Continue with these pages:
7215
+ ## Related
7120
7216
 
7121
7217
  - [Channels](/docs/reference/channels.md): `receive` and the delivery events
7122
7218
  - [GitHub guide](/docs/guides/github.md): reminders in a real webhook loop
@@ -7130,9 +7226,12 @@ Source: /docs/reference/sessions.md
7130
7226
  # Sessions, events, and streaming
7131
7227
 
7132
7228
  A session keeps one conversation, its workspace, and an append-only
7133
- record of every message and tool call.
7229
+ record of every message and tool call. Continue it with a continuation
7230
+ token, inspect it with a session ID, and follow progress on the NDJSON
7231
+ stream. Chat sessions wait for follow-ups; task sessions run once and
7232
+ stop.
7134
7233
 
7135
- ## What does a session contain?
7234
+ ## Session contents
7136
7235
 
7137
7236
  Each session combines:
7138
7237
 
@@ -7140,12 +7239,12 @@ Each session combines:
7140
7239
  - A conversation the caller can continue
7141
7240
  - A workspace for local turns
7142
7241
  - An NDJSON event stream
7143
- - Runtime state needed to resume after a server restart
7242
+ - State that lets the conversation resume after a restart
7144
7243
 
7145
7244
  Sessions belong to the principal that created them. Follow-up, stream,
7146
7245
  and list routes return `403` when another caller tries to access one.
7147
7246
 
7148
- ## Which session identifier should I use?
7247
+ ## Session identifiers
7149
7248
 
7150
7249
  Sessions have two identifiers because conversation routing and
7151
7250
  inspection are different jobs.
@@ -7163,7 +7262,7 @@ accepted follow-up. Reusing a stale HTTP token returns `409`.
7163
7262
  Use the continuation token to keep talking. Use the session ID to
7164
7263
  observe or manage the stored session.
7165
7264
 
7166
- ## Which session modes are available?
7265
+ ## Session modes
7167
7266
 
7168
7267
  | Mode | Created by | What happens after a turn |
7169
7268
  | --- | --- | --- |
@@ -7172,7 +7271,7 @@ observe or manage the stored session.
7172
7271
 
7173
7272
  Task sessions don't accept follow-ups. Trying one returns `409`.
7174
7273
 
7175
- ## What happens when I send a follow-up?
7274
+ ## Follow-ups
7176
7275
 
7177
7276
  A follow-up to an idle chat session starts another turn. Admission when
7178
7277
  the session is already busy depends on the channel:
@@ -7180,23 +7279,21 @@ the session is already busy depends on the channel:
7180
7279
  | Path | Busy-session policy |
7181
7280
  | --- | --- |
7182
7281
  | HTTP playground / `POST /v1/session/:id` / MCP `ask` | **Preempt** (default): interrupt the in-flight turn, wait for it to settle, then run the new message |
7183
- | Slack mentions / DMs / alert-watch | **Coalesce**: leave the active turn running, enqueue the follow-up, and drain queued asks into one follow-up turn when the active turn finishes (no mid-turn tool/hook inject) |
7282
+ | Slack mentions / DMs / alert-watch | **Coalesce**: leave the active turn running and queue the follow-up. The running turn may receive it at a tool boundary; any asks that remain become one follow-up turn when it finishes |
7184
7283
 
7185
7284
  Pass `admission: "coalesce"` on `send()` to opt into the Slack policy from
7186
7285
  other callers. Omit it (or pass `"preempt"`) to keep interrupt semantics.
7187
7286
 
7188
7287
  `POST /v1/session/:id/stop` interrupts a turn without sending a new
7189
- message. Interrupted turns record `turn.failed` with
7190
- `"turn interrupted"`. This means the turn was preempted. A whole-message
7191
- Slack `stop` / `@agent stop` does the same for that thread and clears
7192
- pending coalesced nudges.
7288
+ message. The stream records `turn.failed` with `status: "cancelled"`;
7289
+ the session stays a chat session and can take another follow-up. A
7290
+ whole-message Slack `stop` / `@agent stop` does the same for that thread
7291
+ and clears pending coalesced follow-ups.
7193
7292
 
7194
- Session-bound deterministic tool calls share the lock only for writes: a
7195
- write-effect call returns `409 session_busy` while a model turn is
7196
- running, a read-effect call runs alongside the turn (see
7197
- [Tools](/docs/reference/tools.md#call-a-tool-without-a-model-turn)).
7293
+ See [Tools](/docs/reference/tools.md#call-a-tool-without-a-model-turn) for direct-call
7294
+ behavior while a session is busy.
7198
7295
 
7199
- ## Which events can I stream?
7296
+ ## Stream events
7200
7297
 
7201
7298
  Each NDJSON line uses this envelope:
7202
7299
  `{ type, index, sessionId, turnId?, at, data }`. The `index` increases
@@ -7218,13 +7315,11 @@ within one session. The `at` field is an ISO-8601 timestamp.
7218
7315
 
7219
7316
  Pair `actions.requested` with `action.result` to reconstruct the tool
7220
7317
  trajectory. Read `turn.completed.data.usage` for input, output, and
7221
- cache token counts. When `agent/result.ts` is authored, a thrown
7222
- `commit`, a `ctx.reject` that exhausts the two-repair budget, or empty
7223
- assistant text emits `turn.failed` instead of `turn.completed`.
7224
- `ctx.reject(message)` re-runs the same turn with that message so the
7225
- model can revise while tools and the session filesystem are still up.
7318
+ cache token counts. With `agent/result.ts`, `turn.failed` is emitted when
7319
+ `commit` throws, the turn has no assistant text, or `ctx.reject(message)`
7320
+ doesn't lead to an accepted revision.
7226
7321
 
7227
- ## How do I stream or replay session events?
7322
+ ## Stream or replay events
7228
7323
 
7229
7324
  One endpoint handles both live streaming and replay:
7230
7325
 
@@ -7232,27 +7327,27 @@ One endpoint handles both live streaming and replay:
7232
7327
  curl -N 'http://127.0.0.1:3000/<slug>/v1/session/ses_…/stream?startIndex=0'
7233
7328
  ```
7234
7329
 
7235
- Pass `startIndex` to continue after the last event you received. Omit it
7236
- or pass `0` to replay the full session before following new events.
7330
+ Pass `startIndex` to continue after the last event you received,
7331
+ including after a server restart. Omit it or pass `0` to replay the
7332
+ full session before following new events.
7237
7333
  `GET /v1/session/:id/events` returns a one-time dump without staying
7238
7334
  connected.
7239
7335
 
7240
- Event streams replay from disk after a server restart. Conversation
7241
- state resumes from the Cursor SDK store.
7242
-
7243
- ## What goes into a local session workspace?
7336
+ ## Local session workspace
7244
7337
 
7245
7338
  The Agent SDK creates a workspace before the first local turn:
7246
7339
 
7247
7340
  | Source path | Lands as |
7248
7341
  | ---------------------------------- | ----------------------------------------------------------------------- |
7249
- | `instructions.*` | `AGENTS.md` |
7250
7342
  | `skills/*` | `.cursor/skills/<name>/SKILL.md` |
7251
7343
  | agent tools (`execution: "agent"`) | scripts in the session workspace, with a catalog in `AGENTS.md` |
7252
7344
  | `sandbox/workspace/**` | copied in as seed files |
7253
7345
  | per-send `workspaceFiles` | written before the turn |
7254
7346
 
7255
- The local harness uses this workspace as its working directory. Parent
7347
+ Instructions reach the model as its system prompt; they aren't written
7348
+ into the workspace. See [Instructions](/docs/reference/instructions.md#delivery).
7349
+
7350
+ Local turns use this workspace as their working directory. Parent
7256
7351
  directories can contribute `AGENTS.md` and `.cursor` settings. Set
7257
7352
  `local.cwd` when you need a clean parent directory. A channel can also
7258
7353
  provide a different working directory for one session, such as a PR
@@ -7261,20 +7356,16 @@ worktree.
7261
7356
  See [Agent config: local cwd](/docs/reference/agent-config.md#local-cwd) for the
7262
7357
  inheritance rules.
7263
7358
 
7264
- ## Where does the Agent SDK store session data?
7359
+ ## Session storage
7265
7360
 
7266
- Local state lives under `--state-root`. Slugged mounts store it under
7267
- a subdirectory named for the slug.
7361
+ Local session state lives under `--state-root`. A slugged mount stores
7362
+ it in a subdirectory named for the slug.
7268
7363
 
7269
7364
  Deleting a session directory removes the session from the server: it
7270
7365
  disappears from listings and can no longer be streamed or continued.
7271
7366
  Cloud conversations remain on the Cursor backend.
7272
7367
 
7273
- Nested git checkouts already default `local.cwd` outside the enclosing
7274
- repo. See
7275
- [What goes into a local session workspace?](#what-goes-into-a-local-session-workspace).
7276
-
7277
- ## How do I inspect a saved event stream?
7368
+ ## Inspect a saved event stream
7278
7369
 
7279
7370
  Use `trajectory` with a saved trace:
7280
7371
 
@@ -7283,13 +7374,14 @@ agent-sdk trajectory --events <state-root>/traces/<sessionId>.ndjson
7283
7374
  ```
7284
7375
 
7285
7376
  The command prints tool calls, the reply, and token usage in the same
7286
- JSON shape as `run`. Use **Open trace** in the playground for a visual
7287
- view.
7377
+ JSON shape as `run`. Open the session's **Trace** view in the playground
7378
+ for a visual timeline.
7288
7379
 
7289
7380
  ## Related
7290
7381
 
7291
7382
  - [HTTP API](/docs/reference/http-api.md)
7292
7383
  - [Hooks](/docs/reference/hooks.md)
7384
+ - [Tools](/docs/reference/tools.md)
7293
7385
 
7294
7386
  ---
7295
7387
 
@@ -7305,15 +7397,15 @@ always-on [instructions](/docs/reference/instructions.md).
7305
7397
 
7306
7398
  ## Authoring forms
7307
7399
 
7308
- Three forms cover every case.
7400
+ Skills support three authoring forms.
7309
7401
 
7310
7402
  | Form | Reach for it when |
7311
7403
  | --- | --- |
7312
- | `agent/skills/<name>.md` | Flat markdown. Optional `description` frontmatter; the first body line is the fallback. |
7404
+ | `agent/skills/<name>.md` | Static Markdown. Optional `description` frontmatter; the first body line is the fallback. |
7313
7405
  | `agent/skills/<name>/SKILL.md` plus siblings | A packaged directory with reference files (`references/…`). Requires `description` frontmatter. |
7314
7406
  | `agent/skills/<name>.ts` | Generated content, with `defineSkill` from `@cursor/july/skills`. |
7315
7407
 
7316
- Flat markdown:
7408
+ Use flat Markdown for static content:
7317
7409
 
7318
7410
  ```md
7319
7411
  ---
@@ -7327,8 +7419,8 @@ description: Use when a pull request needs a structured approval checklist.
7327
7419
  3. Call `approve_pr` only after an explicit request; it requires approval.
7328
7420
  ```
7329
7421
 
7330
- TypeScript, when the content must be generated or carry inline sibling
7331
- files:
7422
+ Use TypeScript when the content must be generated or include inline
7423
+ sibling files:
7332
7424
 
7333
7425
  ```ts
7334
7426
  import { defineSkill } from "@cursor/july/skills";
@@ -7340,40 +7432,29 @@ export default defineSkill({
7340
7432
  });
7341
7433
  ```
7342
7434
 
7343
- ## How skills reach the model
7344
-
7345
- On the local runtime, skills land in the session workspace at
7346
- `.cursor/skills/<name>/SKILL.md`, and the harness advertises and loads
7347
- them natively. On the cloud runtime there is no session workspace, so
7348
- the engine copies the same SKILL.md tree onto an Agent Store for native
7349
- discovery:
7350
-
7351
- - Hosted deployments write store-root `skills/<name>/`.
7352
- - `agent-sdk serve` / `run` with a personal `CURSOR_API_KEY` write
7353
- namespaced skills on the USER store so they cannot collide with the
7354
- user's own skills.
7435
+ ## Skill delivery
7355
7436
 
7356
- `validate` still warns about the combination so the store path is
7357
- visible. Cloud turns with neither a hosted store nor an API key see
7358
- only skills already in the cloud repo.
7437
+ Local turns receive authored skills automatically. Hosted deployments
7438
+ and local `serve` or `run` processes with a personal `CURSOR_API_KEY`
7439
+ also make them available to cloud turns. Otherwise, a cloud turn sees
7440
+ only skills already in its checkout. `agent-sdk validate` warns when
7441
+ `runtime: "cloud"` is combined with authored skills.
7359
7442
 
7360
- ## Instructions, skills, or tools?
7443
+ ## Skills, instructions, and tools
7361
7444
 
7362
7445
  Instructions are always in context: identity, tool-choice rules, the
7363
7446
  output contract. Keep them short. Skills load when relevant: procedures,
7364
7447
  checklists, house style. Reach for a skill when the model needs to
7365
- *follow* something but only sometimes needs it loaded. Tools are typed,
7448
+ follow something but only sometimes needs it loaded. Tools are typed,
7366
7449
  executable behavior: anything that must be correct every time belongs in
7367
7450
  tool code, not in prose the model might paraphrase.
7368
7451
 
7369
- A good skill description is a routing rule, not a title. Say *when* to
7452
+ A good skill description is a routing rule, not a title. Say when to
7370
7453
  use it, like "Use when a pull request needs a structured approval
7371
7454
  checklist," because the description is all the model sees before
7372
7455
  deciding to load it.
7373
7456
 
7374
- ## What's next
7375
-
7376
- Continue with these pages:
7457
+ ## Related
7377
7458
 
7378
7459
  - [Instructions](/docs/reference/instructions.md): what stays always-on
7379
7460
  - [Tools](/docs/reference/tools.md): when prose needs to become code
@@ -7388,16 +7469,15 @@ Source: /docs/reference/subagents.md
7388
7469
  # Subagents
7389
7470
 
7390
7471
  A subagent is a specialist child agent the model can delegate to
7391
- mid-turn. Each one is its own directory under `agent/subagents/<id>/`,
7392
- with the same `agent.ts` + `instructions.md` shape as the root. On the
7393
- Cursor harness, subagents run as SDK custom subagents: the parent model
7394
- delegates through the harness `task` tool, and the stream records
7395
- `subagent.called` and `subagent.completed`.
7472
+ mid-turn. Each one lives under `agent/subagents/<id>/` with a required
7473
+ `agent.ts`. Add `instructions.md` or inline `instructions` in `agent.ts`
7474
+ for a custom prompt. The parent model delegates through the `task` tool,
7475
+ and the stream records `subagent.called` and `subagent.completed`.
7396
7476
 
7397
7477
  ```text
7398
7478
  agent/subagents/researcher/
7399
7479
  ├── agent.ts # description (required), model (optional)
7400
- └── instructions.md # the subagent's own system prompt
7480
+ └── instructions.md # optional custom system prompt
7401
7481
  ```
7402
7482
 
7403
7483
  ```ts
@@ -7413,47 +7493,38 @@ export default defineAgent({
7413
7493
 
7414
7494
  ## Subagent rules
7415
7495
 
7416
- `description` is required. It's the only thing the parent model reads
7417
- when deciding whether to delegate, so write it as a routing rule
7418
- ("Background research: …"), the same discipline as a
7419
- [skill](/docs/reference/skills.md) description. `model` is optional; omit it to
7420
- inherit the parent's model, or set it to run the specialist on a
7421
- different one.
7496
+ `description` is required. The parent model uses it to decide whether
7497
+ to delegate, so write it as a routing rule ("Background research: …"),
7498
+ following the same discipline as a [skill](/docs/reference/skills.md) description.
7499
+ `model` is optional; omit it to inherit the parent's model, or set it
7500
+ to run the specialist on a different model.
7422
7501
 
7423
7502
  Subagents inherit the parent's execution surface. Every per-subagent
7424
7503
  capability directory is reported as a warning and ignored: `tools/`,
7425
- `skills/`, `mcp-connections/` (and the legacy `connections/` alias),
7426
- `host-connections/`, `channels/`, `schedules/`, `hooks/`, `sandbox/`,
7427
- and nested `subagents/`.
7504
+ `skills/`, `extensions/`, `mcp-connections/`, `host-connections/`,
7505
+ `channels/`, `schedules/`, `reminders/`, `hooks/`, `sandbox/`, and
7506
+ nested `subagents/`.
7428
7507
 
7429
7508
  Delegation needs both halves: the description makes it possible, and the
7430
7509
  parent's [instructions](/docs/reference/instructions.md) make it happen. "When a
7431
7510
  request needs background research, delegate to the `researcher`
7432
- subagent."
7511
+ subagent." The parent routes; the specialist executes.
7433
7512
 
7434
- ## Subagent or peer?
7513
+ ## Subagents and peers
7435
7514
 
7436
7515
  Subagents split one job into roles inside a single agent. When the
7437
7516
  specialist is independently useful, with its own tools, sessions, and
7438
7517
  playground, make it a full agent. See
7439
- [Agent-to-agent](/docs/guides/agent-to-agent.md#peer-or-subagent).
7518
+ [Peer agents](/docs/guides/agent-to-agent.md#peer-or-subagent).
7440
7519
 
7441
- ## Patterns
7442
-
7443
- Fan-out reviews: a PR-approval agent can delegate to two review
7444
- subagents that read a host-prepared `pr/` evidence tree and report
7445
- prioritized findings, which the parent embeds in its approval comment.
7446
-
7447
- Keep the parent lean: a subagent with focused instructions usually
7448
- works better than a longer parent prompt with conditional sections. The
7449
- parent routes; the specialist executes.
7450
-
7451
- ## What's next
7452
-
7453
- Continue with these pages:
7520
+ ## Related
7454
7521
 
7455
7522
  - [Skills](/docs/reference/skills.md): when a procedure is enough and a child agent
7456
7523
  is overkill
7524
+ - [Peer agents](/docs/guides/agent-to-agent.md): independently useful
7525
+ specialists
7526
+ - [Agent config](/docs/reference/agent-config.md#harness-tools): `"task"` on the
7527
+ parent's tool allowlist
7457
7528
 
7458
7529
  ---
7459
7530
 
@@ -7464,18 +7535,16 @@ Source: /docs/reference/tools.md
7464
7535
  A tool is a typed action the model can call: hit an API, run a query,
7465
7536
  write a file. Each file in `agent/tools/` defines one tool, and the
7466
7537
  filename becomes the tool name the model sees. Tools come in two
7467
- execution flavors: server tools run in-process on the serve host, and
7538
+ execution flavors: server tools run on the serving host, and
7468
7539
  agent tools run as scripts where the agent runs. Every server tool can
7469
7540
  also be called directly, with no model turn.
7470
7541
 
7471
7542
  ## Define a server tool
7472
7543
 
7473
- By default, `execute` runs in-process on the serving host with full
7474
- access to `process.env` and your `agent/lib/` code. Local turns call
7475
- server tools as SDK custom tools. Cloud turns reach them over
7476
- authenticated HTTP MCP back to the serve host when `--public-url` or
7477
- `--cloud-tools-url` is set; without either, the server warns at startup
7478
- and cloud turns omit them.
7544
+ By default, `execute` runs on the serving host with full access to
7545
+ `process.env` and your `agent/lib/` code. Set `--public-url` or
7546
+ `--cloud-tools-url` to make server tools available to cloud turns;
7547
+ without either, startup warns and cloud turns omit them.
7479
7548
 
7480
7549
  ```ts
7481
7550
  // agent/tools/inspect_pr.ts
@@ -7498,6 +7567,10 @@ code that runs it. With a Zod `inputSchema`, the input is validated before
7498
7567
  `execute` runs and the input type is inferred. A plain JSON Schema
7499
7568
  object is forwarded as-is and the input arrives as raw JSON.
7500
7569
 
7570
+ A server tool can call `decide` from
7571
+ [`@cursor/july/extensions/jev`](/docs/guides/jev.md) for a risk tier or a
7572
+ finding gate without a second model turn.
7573
+
7501
7574
  For multi-line descriptions, reminder prompts, and error messages, use
7502
7575
  [`prompt`](/docs/reference/prompt.md) so the string can sit indented with the surrounding
7503
7576
  TypeScript:
@@ -7630,7 +7703,7 @@ check nothing.
7630
7703
  ## Define an agent tool
7631
7704
 
7632
7705
  Set `execution: "agent"` and the tool materializes as a shell script
7633
- that runs where the Cursor agent runs: the local harness workspace or
7706
+ that runs where the Cursor agent runs: the local session workspace or
7634
7707
  the cloud VM. The script receives JSON arguments on stdin and prints its
7635
7708
  result on stdout. Use this flavor when the tool must run next to the
7636
7709
  checkout the agent works in; on cloud, server tools stay available too
@@ -7653,23 +7726,16 @@ printf '%s\\n' "$message"
7653
7726
  ```
7654
7727
 
7655
7728
  On the local runtime, scripts land in the session workspace with a
7656
- catalog in `AGENTS.md`. On cloud, the catalog and script bodies travel
7657
- on the first prompt.
7729
+ catalog in `AGENTS.md`.
7658
7730
 
7659
7731
  ## Tools from an MCP connection, advertised by name
7660
7732
 
7661
- Authored `agent/tools/` files are one catalog for every session. When
7662
- the tools should come from an MCP server, including per-tenant toolsets
7663
- resolved at runtime, declare the connection with
7664
- `advertiseTools: true` (plus per-session `auth` when the credential
7665
- depends on who the session is for) and the engine synthesizes named 1:1
7666
- passthrough server tools from the connection's live `listTools` on every
7667
- local turn. See
7668
- [MCP Connections](/docs/reference/connections.md#advertise-tools).
7669
- Advertised tools ride the same execution path as authored server tools,
7670
- and [direct tool calls](#call-a-tool-without-a-model-turn) address them
7671
- by the same model-facing names: the call's session identity resolves the
7672
- advertised listing when the authored lookup misses.
7733
+ Set `advertiseTools: true` on a connection to expose its MCP tools by
7734
+ name. Add per-session `auth` when credentials depend on the caller.
7735
+ Model calls and [direct tool calls](#call-a-tool-without-a-model-turn)
7736
+ use the same exposed names. See
7737
+ [MCP connections](/docs/reference/connections.md#advertise-tools) for the full
7738
+ contract.
7673
7739
 
7674
7740
  ## Gate a tool on human approval
7675
7741
 
@@ -7700,8 +7766,8 @@ runtime only, and parked calls don't survive a host restart.
7700
7766
  ## Call a tool without a model turn
7701
7767
 
7702
7768
  Server tools can be called deterministically. You pick the tool and the
7703
- input. Validation and execution behave exactly as they would for a
7704
- model-initiated call, and no Cursor API key is needed.
7769
+ input. Validation and execution match a model-initiated call, and no
7770
+ Cursor API key is needed.
7705
7771
 
7706
7772
  Over HTTP, with the same auth chain as the session API:
7707
7773
 
@@ -7727,51 +7793,43 @@ agent-sdk call inspect_pr \
7727
7793
  Programmatically, `callTool(toolName, input, options?)` is available on
7728
7794
  the serve handle, on channel route handlers and `onStart` args, and on
7729
7795
  schedule `run` handlers, so a channel can mix deterministic calls with
7730
- model turns, fetching PR metadata deterministically and then `send()`ing
7731
- the review prompt:
7796
+ model turns:
7732
7797
 
7733
7798
  ```ts
7734
7799
  const outcome = await handle.callTool("inspect_pr", {
7735
7800
  prUrl: "https://github.com/acme/checkout/pull/42",
7736
7801
  });
7802
+ if (outcome.isError) {
7803
+ return outcome;
7804
+ }
7737
7805
  // { toolName, callId, isError, result, durationMs }
7738
7806
  ```
7739
7807
 
7740
- By default the call runs against an ephemeral workspace and is removed
7741
- when the call returns. Pass a `sessionId` (a body field over
7742
- HTTP, `--session` on the CLI, `options.sessionId` programmatically) to
7743
- run inside an existing session instead: the tool sees that session's
7744
- workspace, and the call is recorded on the session's event stream.
7745
- While a model turn is running, a session-bound call is admitted by its
7746
- effect: a read-effect call (a declared `effect: "read"`, or an advertised
7747
- MCP tool whose server annotates it read-only) runs alongside the turn,
7748
- reading the workspace as the turn has left it, and is recorded under its
7749
- own per-call `turnId` so trajectories keep it apart from the turn's own
7750
- calls; a write-effect call including an undeclared tool, which counts
7751
- as a write returns `409 session_busy` until the turn finishes, because
7752
- a running turn owns the workspace. When the
7753
- session's harness cwd cannot be materialized, a read-effect call runs
7754
- in a scratch workspace instead and the outcome carries
7755
- `scratchWorkspace: true`; a write-effect call fails with
7756
- `workspace_unavailable`.
7757
-
7758
- A session can also be addressed by its continuation token: an optional
7759
- `continuationToken` (`<channelId>:<key>`, as `/v1/sessions` lists it;
7760
- mutually exclusive with `sessionId`). A token that maps to a live
7761
- session behaves exactly like passing that session's id — same ownership
7762
- check, same busy semantics, same event recording. A token with no
7763
- session behind it runs the call scratch-bound with the token's channel
7764
- id and continuation key as the call's session identity, so a deployment
7765
- whose tools resolve state from the continuation key can serve it with
7766
- no live session. Malformed tokens are rejected with
7767
- `400 invalid_continuation_token`.
7768
-
7769
- The error semantics match the model path. Unknown tools are rejected
7770
- with the available names, agent-execution tools cannot be called on the
7771
- host (`400`), schema-invalid input is a `400` before the tool body runs
7772
- (Zod validates; plain JSON Schema passes through unvalidated), and a
7773
- tool body that throws reports `isError: true` in the same envelope the
7774
- model would see.
7808
+ Omit both identifiers to run against an ephemeral workspace that is
7809
+ removed when the call returns. Pass `sessionId` (HTTP body, `--session`,
7810
+ or `options.sessionId`) or `continuationToken` (`<channelId>:<key>` as
7811
+ `/v1/sessions` lists it) to bind the call. The two are mutually
7812
+ exclusive.
7813
+
7814
+ | Rule | What happens |
7815
+ | --- | --- |
7816
+ | Existing `sessionId`, or a token that maps to an existing session | Session workspace, owner check, and event recording |
7817
+ | Token with no matching session | Scratch workspace; the token's channel id and key are the call identity |
7818
+ | Busy session, `effect: "read"` (or an advertised MCP tool marked read-only) | Runs alongside the turn, reading the workspace as the turn left it; recorded under its own `turnId` |
7819
+ | Busy session, write or undeclared | `409 session_busy` until the turn finishes |
7820
+ | Session workspace cannot be created, read | Scratch workspace; the outcome has `scratchWorkspace: true` |
7821
+ | Session workspace cannot be created, write | `500 workspace_unavailable` |
7822
+
7823
+ | Error | When |
7824
+ | --- | --- |
7825
+ | `404 unknown_tool` | Name is not a callable server tool; the response lists available names |
7826
+ | `400 tool_not_callable` | Agent-execution tools cannot run on the host |
7827
+ | `400 invalid_tool_input` | Zod rejected the input before `execute`. Plain JSON Schema is forwarded unvalidated |
7828
+ | `400 invalid_continuation_token` | Malformed token, or passed together with `sessionId` |
7829
+ | `404 session_not_found` | `sessionId` does not exist |
7830
+ | `409 session_busy` | Write-effect call while a model turn is running |
7831
+ | `500 workspace_unavailable` | Write-effect call when the session workspace cannot be created |
7832
+ | `isError: true` | The tool body threw; same envelope the model would see |
7775
7833
 
7776
7834
  ## Design habits
7777
7835
 
@@ -7789,9 +7847,7 @@ wrong, no instruction change fixes it.
7789
7847
  Gate side effects with `needsApproval`. Declare each tool's `effect` so
7790
7848
  dry-run sessions can execute reads and stub writes.
7791
7849
 
7792
- ## What's next
7793
-
7794
- Continue with these pages:
7850
+ ## Related
7795
7851
 
7796
7852
  - [MCP connections](/docs/reference/connections.md): tools that come from MCP servers
7797
7853
  instead
@@ -8399,7 +8455,7 @@ Match the visible symptom below. If `agent-sdk` is not on `PATH`, use
8399
8455
  | File reads and greps fail repeatedly with `NGHTTP2_FRAME_SIZE_ERROR` | Run the agent under Node 22.13 or newer, never Bun. |
8400
8456
  | The turn immediately reports a missing API key | Set `CURSOR_API_KEY`, `CURSOR_API_KEY_FILE`, or `CURSOR_SERVICE_ACCOUNT_KEY`, or run `agent-sdk login`. |
8401
8457
  | Login succeeds but the model host rejects the key | Point login and model traffic at the same API host; check `CURSOR_API_BASE_URL` and `CURSOR_BACKEND_URL`. |
8402
- | Cloud turns cannot reach local tools or skills | Give the cloud runtime a reachable `--public-url` and protected tool bridge, or move the required capability into the cloud checkout. See [agent runtime](/docs/reference/agent-config.md#choose-a-runtime). |
8458
+ | Cloud turns cannot reach local tools or skills | Give the cloud runtime a reachable `--public-url` and protected tool bridge, or move the required capability into the cloud checkout. See [agent runtime](/docs/reference/agent-config.md#runtime). |
8403
8459
 
8404
8460
  ## A turn behaves unexpectedly
8405
8461