@cursor/july 0.1.35 → 0.1.37

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (198) hide show
  1. package/AGENTS.md +5 -9
  2. package/README.md +5 -9
  3. package/dist/channels/github/github-channel.d.ts +1 -1
  4. package/dist/channels/github/github-channel.js +1 -1
  5. package/dist/channels/github/index.d.ts +1 -1
  6. package/dist/channels/github/index.js +1 -1
  7. package/dist/channels/slack/api.js +1 -1
  8. package/dist/channels/slack/index.d.ts +1 -1
  9. package/dist/channels/slack/index.js +1 -1
  10. package/dist/channels/slack/init.d.ts.map +1 -1
  11. package/dist/channels/slack/init.js +2 -1
  12. package/dist/docs/404.html +2 -2
  13. package/dist/docs/ab.html +8 -17
  14. package/dist/docs/assets/{ab.md.DAQoJ-up.js → ab.md.6cLOW7--.js} +4 -13
  15. package/dist/docs/assets/{ab.md.DAQoJ-up.lean.js → ab.md.6cLOW7--.lean.js} +1 -1
  16. package/dist/docs/assets/{app.D5Mv1T0U.js → app.Bu9SRvZ9.js} +1 -1
  17. package/dist/docs/assets/{building-with-agents.md.CnHqvYDd.js → building-with-agents.md.txrcGU2B.js} +2 -2
  18. package/dist/docs/assets/chunks/@localSearchIndexroot.pVFW7oJd.js +1 -0
  19. package/dist/docs/assets/chunks/{VPLocalSearchBox.CMq_BQce.js → VPLocalSearchBox.D9u9czHO.js} +1 -1
  20. package/dist/docs/assets/chunks/{theme.C6D9UPLK.js → theme.Bc61GzOf.js} +2 -2
  21. package/dist/docs/assets/{concepts.md.DFaQEFkA.js → concepts.md.CqOsxbMU.js} +1 -1
  22. package/dist/docs/assets/{deployment.md.9MYBuKM1.js → deployment.md.CuK5SNjN.js} +1 -1
  23. package/dist/docs/assets/{evals.md.BIUoVZ6X.js → evals.md.BQXI3rXy.js} +9 -15
  24. package/dist/docs/assets/{evals.md.BIUoVZ6X.lean.js → evals.md.BQXI3rXy.lean.js} +1 -1
  25. package/dist/docs/assets/{example-agents_approval-buddy.md.BhEfleVx.js → example-agents_approval-buddy.md.CIiZ9coo.js} +1 -1
  26. package/dist/docs/assets/{example-agents_benny.md.2Et1qa8f.js → example-agents_benny.md.l7JTmm8X.js} +1 -1
  27. package/dist/docs/assets/{example-agents_bugbot.md.ByUexi5i.js → example-agents_bugbot.md.Dp5JqHSQ.js} +2 -2
  28. package/dist/docs/assets/{example-agents_bugbot.md.ByUexi5i.lean.js → example-agents_bugbot.md.Dp5JqHSQ.lean.js} +1 -1
  29. package/dist/docs/assets/{example-agents_codebase-wiki.md.B4y-7ZVW.js → example-agents_codebase-wiki.md.D-lteFf0.js} +1 -1
  30. package/dist/docs/assets/{example-agents_codebase-wiki.md.B4y-7ZVW.lean.js → example-agents_codebase-wiki.md.D-lteFf0.lean.js} +1 -1
  31. package/dist/docs/assets/{example-agents_codeowners-review.md.D6ay4nvf.js → example-agents_codeowners-review.md.BU2ZXLf-.js} +1 -1
  32. package/dist/docs/assets/{example-agents_codeowners-review.md.D6ay4nvf.lean.js → example-agents_codeowners-review.md.BU2ZXLf-.lean.js} +1 -1
  33. package/dist/docs/assets/{example-agents_concierge.md.lL8rhYlj.js → example-agents_concierge.md.DA2al_NK.js} +2 -2
  34. package/dist/docs/assets/{example-agents_concierge.md.lL8rhYlj.lean.js → example-agents_concierge.md.DA2al_NK.lean.js} +1 -1
  35. package/dist/docs/assets/{example-agents_fsd.md.DfNKQTHz.js → example-agents_fsd.md.DPz9ezO4.js} +1 -1
  36. package/dist/docs/assets/{example-agents_knowledge-base.md.CzyZ2DCr.js → example-agents_knowledge-base.md.IneynQSR.js} +1 -1
  37. package/dist/docs/assets/{example-agents_oncall.md.wFFXXEyW.js → example-agents_oncall.md.ZE0n6ZFN.js} +1 -1
  38. package/dist/docs/assets/{example-agents_slack-agent.md.DvgvT4nn.js → example-agents_slack-agent.md.06jQXTAI.js} +1 -1
  39. package/dist/docs/assets/{example-agents_weather-agent.md.BADkPqxQ.js → example-agents_weather-agent.md.CrGZ0SqR.js} +3 -3
  40. package/dist/docs/assets/{example-agents_weather-agent.md.BADkPqxQ.lean.js → example-agents_weather-agent.md.CrGZ0SqR.lean.js} +1 -1
  41. package/dist/docs/assets/guides_cloud-runtime.md.V5igN4Sq.js +9 -0
  42. package/dist/docs/assets/guides_cloud-runtime.md.V5igN4Sq.lean.js +1 -0
  43. package/dist/docs/assets/{guides_webhooks.md.DiAwSR42.js → guides_webhooks.md.BERuBSJW.js} +1 -1
  44. package/dist/docs/assets/{hillclimbing.md.D9Y1_bYh.js → hillclimbing.md.yXqdlv2R.js} +1 -1
  45. package/dist/docs/assets/index.md.CmhptOmN.js +24 -0
  46. package/dist/docs/assets/{index.md.CZqbBJPB.lean.js → index.md.CmhptOmN.lean.js} +1 -1
  47. package/dist/docs/assets/{quickstart.md.TnEXYgYW.js → quickstart.md.C_b6ESpD.js} +7 -4
  48. package/dist/docs/assets/{quickstart.md.TnEXYgYW.lean.js → quickstart.md.C_b6ESpD.lean.js} +1 -1
  49. package/dist/docs/assets/{reference_agent-config.md.kuN6-OxK.js → reference_agent-config.md.CRmkoxd6.js} +6 -4
  50. package/dist/docs/assets/{reference_agent-config.md.kuN6-OxK.lean.js → reference_agent-config.md.CRmkoxd6.lean.js} +1 -1
  51. package/dist/docs/assets/reference_artifacts.md.BGG4bZo-.js +19 -0
  52. package/dist/docs/assets/reference_artifacts.md.BGG4bZo-.lean.js +1 -0
  53. package/dist/docs/assets/{reference_channels.md.CDhTRfUz.js → reference_channels.md.BIabFUAI.js} +2 -2
  54. package/dist/docs/assets/{reference_channels.md.CDhTRfUz.lean.js → reference_channels.md.BIabFUAI.lean.js} +1 -1
  55. package/dist/docs/assets/{reference_cli.md.sD-IUWjg.js → reference_cli.md.Byvrg8eu.js} +15 -9
  56. package/dist/docs/assets/{reference_cli.md.sD-IUWjg.lean.js → reference_cli.md.Byvrg8eu.lean.js} +1 -1
  57. package/dist/docs/assets/{reference_hooks.md.DyLVfE1O.js → reference_hooks.md.BGDw4VLm.js} +2 -2
  58. package/dist/docs/assets/{reference_hooks.md.DyLVfE1O.lean.js → reference_hooks.md.BGDw4VLm.lean.js} +1 -1
  59. package/dist/docs/assets/reference_http-api.md.DGrw_wOu.js +11 -0
  60. package/dist/docs/assets/reference_http-api.md.DGrw_wOu.lean.js +1 -0
  61. package/dist/docs/assets/{reference_project-layout.md.D8E6ZmHJ.js → reference_project-layout.md._XdeMahr.js} +2 -2
  62. package/dist/docs/assets/{reference_project-layout.md.D8E6ZmHJ.lean.js → reference_project-layout.md._XdeMahr.lean.js} +1 -1
  63. package/dist/docs/assets/{reference_sessions.md.C_ouF_uf.js → reference_sessions.md.DBVFi2Sx.js} +2 -2
  64. package/dist/docs/assets/{reference_subagents.md.zWAMNfi1.js → reference_subagents.md.DSrGLIuB.js} +2 -2
  65. package/dist/docs/assets/{reference_subagents.md.zWAMNfi1.lean.js → reference_subagents.md.DSrGLIuB.lean.js} +1 -1
  66. package/dist/docs/assets/{reference_tools.md.BswAQM41.js → reference_tools.md.lSrsTxYJ.js} +4 -4
  67. package/dist/docs/assets/{reference_tools.md.BswAQM41.lean.js → reference_tools.md.lSrsTxYJ.lean.js} +1 -1
  68. package/dist/docs/assets/scaffolding-agents.md.mkc3B_ZW.js +1 -0
  69. package/dist/docs/assets/{scaffolding-agents.md.Bsr9Pwzu.lean.js → scaffolding-agents.md.mkc3B_ZW.lean.js} +1 -1
  70. package/dist/docs/assets/{storage.md.xZoiGM58.js → storage.md.mQDtIULc.js} +3 -3
  71. package/dist/docs/assets/{storage.md.xZoiGM58.lean.js → storage.md.mQDtIULc.lean.js} +1 -1
  72. package/dist/docs/building-with-agents.html +6 -6
  73. package/dist/docs/concepts.html +5 -5
  74. package/dist/docs/deployment.html +6 -6
  75. package/dist/docs/evals.html +13 -19
  76. package/dist/docs/example-agents/approval-buddy.html +5 -5
  77. package/dist/docs/example-agents/benny.html +5 -5
  78. package/dist/docs/example-agents/bugbot.html +5 -5
  79. package/dist/docs/example-agents/codebase-wiki.html +5 -5
  80. package/dist/docs/example-agents/codeowners-review.html +5 -5
  81. package/dist/docs/example-agents/concierge.html +6 -6
  82. package/dist/docs/example-agents/fsd.html +5 -5
  83. package/dist/docs/example-agents/index.html +4 -4
  84. package/dist/docs/example-agents/knowledge-base.html +5 -5
  85. package/dist/docs/example-agents/oncall.html +5 -5
  86. package/dist/docs/example-agents/security-reviewer.html +4 -4
  87. package/dist/docs/example-agents/slack-agent.html +5 -5
  88. package/dist/docs/example-agents/weather-agent.html +6 -6
  89. package/dist/docs/guides/agent-to-agent.html +5 -5
  90. package/dist/docs/guides/cloud-runtime.html +6 -6
  91. package/dist/docs/guides/github.html +4 -4
  92. package/dist/docs/guides/human-in-the-loop.html +4 -4
  93. package/dist/docs/guides/mcp-oauth.html +5 -5
  94. package/dist/docs/guides/slack.html +4 -4
  95. package/dist/docs/guides/webhooks.html +6 -6
  96. package/dist/docs/hashmap.json +1 -1
  97. package/dist/docs/hillclimbing.html +6 -6
  98. package/dist/docs/index.html +11 -7
  99. package/dist/docs/quickstart.html +10 -7
  100. package/dist/docs/reference/agent-config.html +10 -8
  101. package/dist/docs/reference/artifacts.html +43 -0
  102. package/dist/docs/reference/channels.html +6 -6
  103. package/dist/docs/reference/cli.html +18 -12
  104. package/dist/docs/reference/connections.html +4 -4
  105. package/dist/docs/reference/hooks.html +6 -6
  106. package/dist/docs/reference/http-api.html +7 -7
  107. package/dist/docs/reference/instructions.html +4 -4
  108. package/dist/docs/reference/playground.html +4 -4
  109. package/dist/docs/reference/project-layout.html +6 -6
  110. package/dist/docs/reference/prompt.html +4 -4
  111. package/dist/docs/reference/schedules.html +4 -4
  112. package/dist/docs/reference/sessions.html +7 -7
  113. package/dist/docs/reference/skills.html +4 -4
  114. package/dist/docs/reference/subagents.html +6 -6
  115. package/dist/docs/reference/tools.html +7 -7
  116. package/dist/docs/scaffolding-agents.html +5 -5
  117. package/dist/docs/storage.html +6 -6
  118. package/dist/docs/troubleshooting.html +4 -4
  119. package/dist/files-backends/agent-store-presigned-url.d.ts.map +1 -1
  120. package/dist/files-backends/agent-store-presigned-url.js +15 -22
  121. package/dist/internal/cli-github.d.ts.map +1 -1
  122. package/dist/internal/cli-github.js +8 -7
  123. package/dist/internal/cli-slack.js +9 -9
  124. package/dist/internal/event-mapper.d.ts +3 -3
  125. package/dist/internal/event-mapper.d.ts.map +1 -1
  126. package/dist/internal/event-mapper.js +7 -4
  127. package/dist/internal/session-engine.js +3 -3
  128. package/dist/internal/workspace.d.ts +8 -6
  129. package/dist/internal/workspace.d.ts.map +1 -1
  130. package/dist/internal/workspace.js +15 -11
  131. package/dist/playground/assets/{index-D7OV8B_H.js → index-DRtusTV1.js} +38 -38
  132. package/dist/playground/assets/index-DZp6n4bv.css +1 -0
  133. package/dist/playground/index.html +2 -2
  134. package/docs/README.md +32 -13
  135. package/docs/ab.md +23 -36
  136. package/docs/building-with-agents.md +2 -2
  137. package/docs/concepts.md +3 -2
  138. package/docs/deployment.md +1 -1
  139. package/docs/evals.md +102 -33
  140. package/docs/example-agents/approval-buddy.md +2 -1
  141. package/docs/example-agents/benny.md +2 -0
  142. package/docs/example-agents/bugbot.md +3 -0
  143. package/docs/example-agents/codebase-wiki.md +2 -0
  144. package/docs/example-agents/codeowners-review.md +2 -0
  145. package/docs/example-agents/concierge.md +1 -0
  146. package/docs/example-agents/fsd.md +1 -0
  147. package/docs/example-agents/knowledge-base.md +2 -0
  148. package/docs/example-agents/oncall.md +2 -0
  149. package/docs/example-agents/slack-agent.md +1 -0
  150. package/docs/example-agents/weather-agent.md +9 -4
  151. package/docs/guides/cloud-runtime.md +11 -4
  152. package/docs/guides/webhooks.md +1 -1
  153. package/docs/hillclimbing.md +1 -1
  154. package/docs/quickstart.md +39 -7
  155. package/docs/reference/agent-config.md +74 -14
  156. package/docs/reference/artifacts.md +117 -0
  157. package/docs/reference/channels.md +45 -15
  158. package/docs/reference/cli.md +141 -20
  159. package/docs/reference/hooks.md +11 -4
  160. package/docs/reference/http-api.md +50 -4
  161. package/docs/reference/project-layout.md +6 -0
  162. package/docs/reference/sessions.md +5 -4
  163. package/docs/reference/subagents.md +5 -3
  164. package/docs/reference/tools.md +23 -7
  165. package/docs/scaffolding-agents.md +11 -2
  166. package/docs/storage.md +27 -2
  167. package/package.json +1 -1
  168. package/src/channels/github/github-channel.ts +1 -1
  169. package/src/channels/github/index.ts +1 -1
  170. package/src/channels/slack/api.ts +1 -1
  171. package/src/channels/slack/index.ts +1 -1
  172. package/src/channels/slack/init.ts +2 -1
  173. package/src/files-backends/agent-store-presigned-url.ts +2 -1
  174. package/src/internal/cli-github.ts +8 -7
  175. package/src/internal/cli-slack.ts +9 -9
  176. package/src/internal/event-mapper.ts +9 -4
  177. package/src/internal/session-engine.ts +3 -3
  178. package/src/internal/workspace.ts +15 -11
  179. package/dist/docs/assets/chunks/@localSearchIndexroot.Cu7b6o1D.js +0 -1
  180. package/dist/docs/assets/guides_cloud-runtime.md.CDJGvVC4.js +0 -9
  181. package/dist/docs/assets/guides_cloud-runtime.md.CDJGvVC4.lean.js +0 -1
  182. package/dist/docs/assets/index.md.CZqbBJPB.js +0 -20
  183. package/dist/docs/assets/reference_http-api.md.CfVM_ICa.js +0 -11
  184. package/dist/docs/assets/reference_http-api.md.CfVM_ICa.lean.js +0 -1
  185. package/dist/docs/assets/scaffolding-agents.md.Bsr9Pwzu.js +0 -1
  186. package/dist/playground/assets/index-DOb96C0M.css +0 -1
  187. /package/dist/docs/assets/{building-with-agents.md.CnHqvYDd.lean.js → building-with-agents.md.txrcGU2B.lean.js} +0 -0
  188. /package/dist/docs/assets/{concepts.md.DFaQEFkA.lean.js → concepts.md.CqOsxbMU.lean.js} +0 -0
  189. /package/dist/docs/assets/{deployment.md.9MYBuKM1.lean.js → deployment.md.CuK5SNjN.lean.js} +0 -0
  190. /package/dist/docs/assets/{example-agents_approval-buddy.md.BhEfleVx.lean.js → example-agents_approval-buddy.md.CIiZ9coo.lean.js} +0 -0
  191. /package/dist/docs/assets/{example-agents_benny.md.2Et1qa8f.lean.js → example-agents_benny.md.l7JTmm8X.lean.js} +0 -0
  192. /package/dist/docs/assets/{example-agents_fsd.md.DfNKQTHz.lean.js → example-agents_fsd.md.DPz9ezO4.lean.js} +0 -0
  193. /package/dist/docs/assets/{example-agents_knowledge-base.md.CzyZ2DCr.lean.js → example-agents_knowledge-base.md.IneynQSR.lean.js} +0 -0
  194. /package/dist/docs/assets/{example-agents_oncall.md.wFFXXEyW.lean.js → example-agents_oncall.md.ZE0n6ZFN.lean.js} +0 -0
  195. /package/dist/docs/assets/{example-agents_slack-agent.md.DvgvT4nn.lean.js → example-agents_slack-agent.md.06jQXTAI.lean.js} +0 -0
  196. /package/dist/docs/assets/{guides_webhooks.md.DiAwSR42.lean.js → guides_webhooks.md.BERuBSJW.lean.js} +0 -0
  197. /package/dist/docs/assets/{hillclimbing.md.D9Y1_bYh.lean.js → hillclimbing.md.yXqdlv2R.lean.js} +0 -0
  198. /package/dist/docs/assets/{reference_sessions.md.C_ouF_uf.lean.js → reference_sessions.md.DBVFi2Sx.lean.js} +0 -0
@@ -8,8 +8,8 @@
8
8
  />
9
9
  <meta name="viewport" content="width=device-width, initial-scale=1" />
10
10
  <title>agent-serve playground</title>
11
- <script type="module" crossorigin src="./assets/index-D7OV8B_H.js"></script>
12
- <link rel="stylesheet" crossorigin href="./assets/index-DOb96C0M.css">
11
+ <script type="module" crossorigin src="./assets/index-DRtusTV1.js"></script>
12
+ <link rel="stylesheet" crossorigin href="./assets/index-DZp6n4bv.css">
13
13
  </head>
14
14
  <body>
15
15
  <div id="root"></div>
package/docs/README.md CHANGED
@@ -36,7 +36,19 @@ my-agent/
36
36
  └── evals/ # filesystem evals (regression checks)
37
37
  ```
38
38
 
39
- Browse it locally without serving an agent:
39
+ Bootstrap a project with nothing installed beyond Node:
40
+
41
+ ```bash
42
+ npx @cursor/july init ./my-agent
43
+ cd my-agent
44
+ npx agent-sdk dev
45
+ ```
46
+
47
+ `init` scaffolds the project, runs `npm install`, and offers a Cursor
48
+ sign-in. The install puts the `agent-sdk` bin on the project's path, so
49
+ `npx agent-sdk` resolves locally from then on.
50
+
51
+ Browse the docs locally without serving an agent:
40
52
 
41
53
  ```bash
42
54
  npx @cursor/july docs
@@ -80,6 +92,8 @@ Pick your entry point based on your goal.
80
92
  evals as regression checks.
81
93
  - [Live A/B metrics](./ab.md): assign sticky variants and compare
82
94
  cumulative metrics on live sessions.
95
+ - [Storage](./storage.md): point durable storage at a backend you own
96
+ with `defineStorage`.
83
97
  - [Hillclimbing](./hillclimbing.md): make an agent better one measured
84
98
  round at a time.
85
99
 
@@ -102,7 +116,7 @@ Pick your entry point based on your goal.
102
116
 
103
117
  **Example agents**
104
118
 
105
- - [Choose the right example](./example-agents/index.md): compare all eleven
119
+ - [Choose the right example](./example-agents/index.md): compare all twelve
106
120
  agents by runtime, channels, tools, state, and architecture.
107
121
  - [Weather agent](./example-agents/weather-agent.md): explore tools, MCP,
108
122
  approvals, skills, subagents, schedules, hooks, A/B metrics, and evals.
@@ -112,6 +126,8 @@ Pick your entry point based on your goal.
112
126
  with its own context and sessions.
113
127
  - [Playbook router](./example-agents/benny.md): route Slack intake through inherited
114
128
  repository playbooks.
129
+ - [Alert investigator](./example-agents/oncall.md): watch a Slack alerts
130
+ channel and pin a self-rechecking investigation to every alert thread.
115
131
  - [PR evidence reviewer](./example-agents/bugbot.md): review a host-prepared,
116
132
  diff-first pull-request evidence tree.
117
133
  - [Approval Buddy](./example-agents/approval-buddy.md): keep approval policy
@@ -154,7 +170,14 @@ Pick your entry point based on your goal.
154
170
  ## Run the CLI
155
171
 
156
172
  The docs write commands as `agent-sdk <command>`. Where that command
157
- comes from depends on where you run.
173
+ comes from depends on where you run. Starting fresh? This works with no
174
+ prior install:
175
+
176
+ ```bash
177
+ npx @cursor/july init ./my-agent
178
+ cd my-agent
179
+ npx agent-sdk dev
180
+ ```
158
181
 
159
182
  > [!NOTE]
160
183
  > When running from a source checkout there is no installed bin. Alias it from
@@ -171,22 +194,18 @@ comes from depends on where you run.
171
194
  > runs the `july` bin with that command (no local install required).
172
195
 
173
196
  > [!NOTE]
174
- > The framework is being renamed from agent-serve to the Agent SDK,
175
- > and CLI examples use the new `agent-sdk` name. Paths, package imports,
176
- > and environment variables keep their current names until the code rename
177
- > ships:
197
+ > The framework is being renamed from agent-serve to the Agent SDK. The
198
+ > `agent-sdk` bin already ships (alongside `july`, `agentkit`, and the
199
+ > legacy `agent-serve` alias), and projects already import from
200
+ > `@cursor/july`. A few on-disk names keep their old form until the code
201
+ > rename ships:
178
202
  >
179
- > | Docs say | Current name |
203
+ > | Future name | Current name |
180
204
  > | --- | --- |
181
- > | `@cursor/july` imports and dependency | `@cursor/july` |
182
- > | `agent-sdk` bin | `agent-serve` |
183
- > | `dist/bin/agent-sdk.js` | `dist/bin/agent-serve.js` |
184
205
  > | `.agent-sdk/` state directory | `.agent-serve/` |
185
206
  > | `/var/lib/agent-sdk` (deploy state root) | `/var/lib/agent-serve` |
186
207
  > | `CURSOR_AGENT_SDK_*` env vars | `AGENT_SERVE_*` |
187
- > | `agent-sdk (<hostname>)` API key name | `agent-serve (<hostname>)` |
188
208
  > | Package path `packages/agent-sdk` | `packages/agent-serve` |
189
- > | Package skills `packages/agent-sdk/skills/` | `packages/agent-serve/skills/` |
190
209
 
191
210
  > [!WARNING]
192
211
  > Run the Agent SDK with Node 22.13 or newer, and never with Bun.
package/docs/ab.md CHANGED
@@ -19,7 +19,7 @@ on fixed inputs.
19
19
 
20
20
  > [!NOTE]
21
21
  > Import paths here use `@cursor/july/ab`. On projects still
22
- > using `@cursor/july`, swap the import. See
22
+ > using `@anysphere/agent-serve`, swap the import. See
23
23
  > [Run the CLI](./README.md#run-the-cli) for the full rename table.
24
24
 
25
25
  ## Choose live A/B metrics or evals
@@ -208,6 +208,7 @@ turn and tool events into its own counters.
208
208
  | `toolErrors` | Adds one when `action.result.data.isError` is true |
209
209
  | `inputTokens`, `outputTokens` | Adds usage from completed turns |
210
210
  | `cacheReadTokens`, `cacheWriteTokens` | Adds cache usage from completed turns |
211
+ | `costUsd` | Sums the estimated turn cost recorded on `turn.completed` (turns whose model has no known rates contribute 0) |
211
212
  | `wallTimeMs` | Sums the time from `turn.started` to its completed or failed event |
212
213
  | `custom` | Sums finite numeric deltas returned by `derive` |
213
214
 
@@ -254,44 +255,30 @@ too. This include-all behavior can still apply to `GET /v1/abs` in dev
254
255
  when bearer or custom auth keeps `GET /v1/sessions` owner-scoped.
255
256
 
256
257
  Session `events.ndjson` is the source of truth for assignment + fold.
257
- `GET /v1/abs` recomputes aggregates from those logs. Use
258
- `agent/ab.config.ts` (and/or each experiment's `onSample`) when you want
259
- author-controlled sample/snapshot exports — same idea as eval
260
- `persistRuns`.
258
+ `GET /v1/abs` recomputes aggregates from those logs. For durable
259
+ sample and snapshot export, declare an `abs` table in
260
+ `agent/storage.ts` — same idea as the eval-runs table. See
261
+ [Storage](./storage.md#eval-and-ab-tables).
261
262
 
262
- ## Configure retention and persistence
263
+ ## Configure the playground fold window
263
264
 
264
- Optional project defaults in `agent/ab.config.ts`:
265
+ Assignments and foldable metrics already persist in each session's
266
+ `events.ndjson` under `--state-root`. The optional `agent/ab.config.ts`
267
+ only caps how many sessions the playground and `GET /v1/abs` fold:
265
268
 
266
269
  ```ts
267
- import {
268
- defineABConfig,
269
- persistABSamplesToDir,
270
- persistABSnapshotsToDir,
271
- } from "@cursor/july/ab";
270
+ import { defineABConfig } from "@cursor/july/ab";
272
271
 
273
272
  export default defineABConfig({
274
273
  // Optional — defaults to 200. Only affects GET /v1/abs / A/Bs tab.
275
- // maxPlaygroundSessions: 200,
276
- // retainSnapshots: 20, // prune budget for persistABSnapshotsToDir
277
- // Append onSample payloads (in addition to each experiment's onSample)
278
- persistSamples: persistABSamplesToDir(".agent-serve/ab-samples"),
279
- // Save aggregate snapshots whenever GET /v1/abs runs
280
- persistSnapshots: persistABSnapshotsToDir(".agent-serve/ab-snapshots"),
274
+ maxPlaygroundSessions: 500,
281
275
  });
282
276
  ```
283
277
 
284
- | Option | Default | Meaning |
285
- | --- | --- | --- |
286
- | `maxPlaygroundSessions` | `200` | Max newest sessions folded into the playground / `GET /v1/abs` (not session logs or `persistSamples`) |
287
- | `persistSamples` | unset | Durable sink for metric samples (`save(sample, ctx)`) |
288
- | `persistSnapshots` | unset | Durable store for aggregate snapshots (`load` / `save` / optional `delete`) |
289
- | `retainSnapshots` | `20` when snapshots persist | Snapshot files kept after prune (`persistABSnapshotsToDir` honors this) |
290
-
291
- Implement your own `{ save }` / `{ load, save, delete? }` for S3, a DB, or
292
- your metrics vendor. Without `persistSamples` / `persistSnapshots`,
293
- samples only go where each experiment's `onSample` sends them, and
294
- aggregates exist only as a live fold over session logs.
278
+ `maxPlaygroundSessions` keeps the newest sessions in the fold. It does
279
+ not prune session logs or change assignment. For export to S3, a DB, or
280
+ your metrics vendor, send samples from `onSample` or declare a storage
281
+ `abs` table.
295
282
 
296
283
  ## Keep assignments durable
297
284
 
@@ -301,8 +288,8 @@ metrics come from the turn and tool events that follow it.
301
288
 
302
289
  After a server restart or a parked session resumes, the live collector
303
290
  replays the stream to rebuild cumulative counters. Replay does not call
304
- `onSample` (or `persistSamples`) for historical turns. Only a new
305
- completed or failed turn emits another sample.
291
+ `onSample` (or write to the storage `abs` table) for historical turns.
292
+ Only a new completed or failed turn emits another sample.
306
293
 
307
294
  The snapshot API also replays `derive` across the full stream, so
308
295
  custom totals match the current extractor. Changing a derive function
@@ -332,12 +319,12 @@ does not provide:
332
319
  - Statistical significance calculations
333
320
  - An experiment rollout or lifecycle service
334
321
  - Per-variant model or runtime configuration
335
- - A built-in analytics warehouse (bring your own via `persistSamples` /
336
- `persistSnapshots` / `onSample`)
322
+ - A built-in analytics warehouse (bring your own via `onSample` or the
323
+ storage `abs` table)
337
324
 
338
- Use [evals](./evals.md) to protect known behavior. Configure
339
- `agent/ab.config.ts` (or `onSample`) when you need sample/snapshot
340
- exports beyond the session event log.
325
+ Use [evals](./evals.md) to protect known behavior. Use `onSample` or a
326
+ storage `abs` table when you need sample/snapshot exports beyond the
327
+ session event log.
341
328
 
342
329
  ## What's next
343
330
 
@@ -37,7 +37,7 @@ project, and verifies the result.
37
37
 
38
38
  For example:
39
39
 
40
- > Use the agent-serve create-agent skill to build a PR triage agent
40
+ > Use the Agent SDK create-agent skill to build a PR triage agent
41
41
  > reachable through GitHub. It should summarize failed checks, require
42
42
  > approval before posting a review, and include one smoke eval.
43
43
 
@@ -83,7 +83,7 @@ agent-sdk call inspect_pr --dir . \
83
83
  agent-sdk run --dir . \
84
84
  --message "Is https://github.com/acme/checkout/pull/42 ready to approve?"
85
85
 
86
- agent-sdk trajectory --events .agent-sdk/traces/<sessionId>.ndjson
86
+ agent-sdk trajectory --events .agent-serve/traces/<sessionId>.ndjson
87
87
 
88
88
  agent-sdk eval --dir . --list
89
89
  agent-sdk eval --dir . --json
package/docs/concepts.md CHANGED
@@ -110,7 +110,8 @@ Choose a runtime in `agent/agent.ts`:
110
110
  | | Local (default) | Cloud |
111
111
  | --- | --- | --- |
112
112
  | Turn runs on | The server host | A Cursor cloud agent |
113
- | Server tools and approvals | Supported | Not supported |
113
+ | Server tools | Supported | Supported when the server has `--public-url` or `--cloud-tools-url`; the cloud turn reaches them over authenticated HTTP MCP. Without one of those flags, the server warns and cloud turns omit them. |
114
+ | Approvals (`needsApproval`) | Supported | Not supported (local runtime only) |
114
115
  | Agent tool scripts | Supported | Supported |
115
116
  | Skills and seeded files | Added to the session workspace | Must exist in the cloud repository |
116
117
  | Repository | You provide it | The cloud agent checks it out |
@@ -137,7 +138,7 @@ commands already use temporary state.
137
138
  Durable local state uses this shape:
138
139
 
139
140
  ```text
140
- <project>/.agent-sdk/
141
+ <project>/.agent-serve/
141
142
  sessions/<id>/events.ndjson
142
143
  sessions/<id>/workspace/
143
144
  traces/<sessionId>.ndjson
@@ -305,7 +305,7 @@ A self-hosted server can read these credentials.
305
305
  Use a dedicated Cursor key per host. `agent-sdk whoami` shows the active
306
306
  credential. `logout` removes the stored key from the host; revoke the key
307
307
  in the Cursor dashboard to invalidate it. See
308
- [CLI authentication](./reference/cli.md#login--logout--whoami) for
308
+ [CLI authentication](./reference/cli.md#login-logout-whoami) for
309
309
  credential resolution.
310
310
 
311
311
  ### State
package/docs/evals.md CHANGED
@@ -18,7 +18,7 @@ accepted a message, and did what you asserted.
18
18
 
19
19
  > [!NOTE]
20
20
  > Import paths here use `@cursor/july/evals`. On projects still
21
- > using `@cursor/july`, swap the import and run
21
+ > using `@anysphere/agent-serve`, swap the import and run
22
22
  > `agent-serve eval`. See
23
23
  > [Run the CLI](./README.md#run-the-cli) for the full rename table.
24
24
 
@@ -118,35 +118,39 @@ issues real model-provider requests, so concurrency is capped hard at
118
118
  without this file, but running a case does not.
119
119
 
120
120
  ```ts
121
- import {
122
- defineEvalConfig,
123
- persistEvalRunsToDir,
124
- } from "@cursor/july/evals";
121
+ import { defineEvalConfig } from "@cursor/july/evals";
125
122
 
126
123
  export default defineEvalConfig({
127
124
  maxConcurrency: 20, // required
128
125
  // timeoutMs: 180_000, // optional project-wide default
126
+ // judge: { model: "..." }, // default judge model for t.judge.*
127
+ // reporters: [], // destinations that observe every case
129
128
  // maxPlaygroundRuns: 50, // playground /v1/dev/evals history only (default 20)
130
- //
131
- // Playground batches default to **process memory only** — they disappear
132
- // when `serve` exits. Opt into durable storage:
133
- // persistRuns: persistEvalRunsToDir(".agent-serve/eval-runs"),
134
- // Or implement { load, save, delete } yourself (S3, DB, …).
135
129
  });
136
130
  ```
137
131
 
138
132
  The timeout order is case or file `timeoutMs`, CLI `--timeout-ms`,
139
133
  project config `timeoutMs`, then the 180-second runner default.
140
134
 
141
- Playground batch retention is separate from case concurrency:
135
+ The optional fields:
142
136
 
143
137
  | Option | Default | Meaning |
144
138
  | --- | --- | --- |
139
+ | `timeoutMs` | `180_000` | Project-wide per-case timeout |
140
+ | `judge` | unset | Default judge model for `t.judge.*`; see [Judge free-form output](#judge-free-form-output) |
141
+ | `reporters` | unset | Destinations that observe every case; `--skip-report` suppresses them |
145
142
  | `maxPlaygroundRuns` | `20` | Max batches in the playground / `/v1/dev/evals*` history (not CLI `eval`) |
146
- | `persistRuns` | unset | Optional `{ load, save, delete }` so batches survive process restart (`delete` required for durable prune) |
147
143
 
148
- Without `persistRuns`, navigating away and back still works while the same
149
- `serve` process is up; a restart clears history.
144
+ Reporters come from `@cursor/july/evals/reporters`: `JUnit` writes a
145
+ JUnit XML file for CI, `Artifacts` writes per-case files, and
146
+ `combineReporters` merges several into one (`renderJUnitXml` renders
147
+ the XML for a custom destination). A file or case can add its own
148
+ `reporters` on top of the config list.
149
+
150
+ Playground batches live in process memory and disappear when `serve`
151
+ exits. Navigating away and back still works while the process is up.
152
+ To keep batches across restarts, declare an `evals` table in
153
+ `agent/storage.ts`; see [Storage](./storage.md#eval-and-ab-tables).
150
154
 
151
155
  ## Drive and assert with `t`
152
156
 
@@ -155,33 +159,71 @@ control flow, sending turns and asserting inline.
155
159
 
156
160
  Drive the agent with `t.send(message, options?)`. It runs one turn and
157
161
  waits for the session to park or fail. Multiple sends in one case share
158
- the session, which is how you write multi-turn evals. The return value
159
- contains the turn's `message`, `sessionId`, `events`, `toolCalls`, and
160
- `ok` state.
162
+ the session, which is how you write multi-turn evals.
163
+
164
+ Each `t.send` resolves to a turn result with `message`, `sessionId`,
165
+ `events`, `toolCalls`, `ok`, and `index`. The turn carries the same
166
+ assertion vocabulary as `t`, scoped to that turn, so you can grade an
167
+ intermediate turn before the next send overwrites `t.reply`.
168
+ `turn.expectOk()` throws when the turn failed, for later steps that
169
+ depend on it.
161
170
 
162
171
  Read the full case state with `t.reply` (the last assistant text),
163
- `t.events` (every captured session event across turns), and
164
- `t.sessionId`.
172
+ `t.events` (every captured session event across turns), `t.turns`
173
+ (settled turns, oldest first), and `t.sessionId`. `t.signal` aborts
174
+ when the case hits its timeout; pass it to your own async work.
165
175
 
166
176
  Assert with the gates:
167
177
 
168
178
  | Gate | Checks |
169
179
  | --- | --- |
170
- | `t.succeeded()` | the captured trajectory has at least one turn and did not fail |
171
- | `t.calledTool(name)` | `name` appears in the captured tool calls |
172
- | `t.notCalledTool(name)` | `name` does not appear in the captured tool calls |
173
- | `t.messageIncludes(token)` | the last assistant reply matches a string or `RegExp` |
180
+ | `t.succeeded()` | the run did not fail and is not parked on an unanswered approval |
181
+ | `t.parked()` | the run cleanly parked on an unanswered approval request |
182
+ | `t.messageIncludes(token)` | the joined assistant text matches a string or `RegExp` |
183
+ | `t.calledTool(name, matcher?)` | a matching call to `name` happened |
184
+ | `t.notCalledTool(name)` | no request for `name`, in any lifecycle state |
185
+ | `t.loadedSkill(name)` | the agent opened the skill's `SKILL.md` (read, grep, or shell `cat`) |
186
+ | `t.toolOrder(names)` | tool requests appear in this relative order (extra calls allowed) |
187
+ | `t.usedNoTools()` | no tool calls at all |
188
+ | `t.maxToolCalls(max)` | at most `max` tool calls |
189
+ | `t.noFailedActions()` | no tool call reported an error |
190
+ | `t.calledSubagent(name, matcher?)` | a matching subagent delegation happened |
191
+ | `t.taggedArtifact(kind?, predicate?)` | at least one [artifact](./reference/artifacts.md) was tagged |
192
+ | `t.event(type, matcher?)` | at least one matching event of `type` occurred |
193
+ | `t.notEvent(type, matcher?)` | no matching event of `type` occurred |
194
+ | `t.eventOrder(matchers)` | matching event groups occur in this relative order |
195
+ | `t.eventsSatisfy(label, predicate)` | your predicate over the typed event stream |
174
196
  | `t.check(value, expectation)` | any value, against a builder |
197
+ | `t.score(name, value)` | records a 0–1 score you computed; soft until you add a bar |
198
+ | `t.requireToolCall(name, matcher?)` | gates on a matching call and returns it, so later code can read its input and output |
199
+ | `t.requireInputRequest(filter?)` | gates on exactly one pending approval request and returns it |
200
+
201
+ Every gate returns a handle: `.soft()` demotes it to tracked-only,
202
+ `.atLeast(0.7)` adds a soft score bar, and `.gate(0.8)` promotes a
203
+ scored assertion into a hard gate.
175
204
 
176
- `calledTool` reads the recorded trajectory. A requested call counts even
177
- when its result has not arrived. To require a completed result, inspect
178
- `t.events` for an `action.result` event.
205
+ With no matcher, `calledTool` is request-based: a requested call counts
206
+ even when its result has not arrived. Pass
207
+ `t.calledTool("inspect_pr", { status: "completed" })` to require the
208
+ call to return. `input`, `output`, and `count` matcher fields accept a
209
+ literal, a `RegExp`, or a predicate.
179
210
 
180
211
  The expectation builders are `includes(string | RegExp)`,
181
- `equals(value)`, and `satisfies(predicate, label)`. `includes`
182
- stringifies its input, `equals` compares values deeply, and `satisfies`
183
- runs your predicate. `t.log(message)` records a debug line for the CLI
184
- and playground result.
212
+ `equals(value)`, `matches(schema)`, `similarity(expected)`, and
213
+ `satisfies(predicate, label)`. `includes` stringifies its input,
214
+ `equals` compares values deeply, `matches` validates against a Standard
215
+ Schema (or anything with `safeParse`, like Zod), `similarity` scores
216
+ normalized text similarity, and `satisfies` runs your predicate. The
217
+ plain function `normalizedSimilarity(actual, expected)` returns the
218
+ same 0–1 score for use with `t.score`.
219
+
220
+ A few more context members shape a case: `t.require(value, expectation)`
221
+ records a gate and stops the test body when it fails, without a
222
+ duplicate execution error. `t.skip(reason)` ends the case as skipped
223
+ (reported separately, never changes the exit code; call it before
224
+ sending messages). `t.metric(name, value)` records a structured score
225
+ for the playground case card. `t.log(message)` records a debug line for
226
+ the CLI and playground result.
185
227
 
186
228
  Three `t.send` options apply on session create (first `t.send` only):
187
229
 
@@ -205,6 +247,26 @@ A case with no explicit gates falls back to whether at least one turn
205
247
  completed successfully. Add `t.succeeded()` and behavior-specific gates
206
248
  anyway. They make the contract visible during review.
207
249
 
250
+ ### Judge free-form output
251
+
252
+ When wording matters and no regex captures it, `t.judge` grades the
253
+ reply with an LLM. The built-in graders are `factuality(expected)`,
254
+ `summarizes(expected)`, `closedQA(criteria)`, and `sql(expected)`. Each
255
+ scores `t.reply` by default; pass `{ on }` to grade another value.
256
+
257
+ ```ts
258
+ t.judge.factuality("It is 54°F in NYC right now.").atLeast(0.7);
259
+ ```
260
+
261
+ Judge assertions are soft by default, so a judge never fails a build
262
+ until you give it a bar with `.atLeast(0.7)` or promote it with
263
+ `.gate(0.8)`. The judge model comes from `defineEvalConfig({ judge })`,
264
+ `defineEval({ judge })`, a case-level `judge`, or a per-call
265
+ `{ model }` override; the nearest one wins. For a domain-specific judge
266
+ whose verdict is not a single score, `t.judge.model(prompt)` sends a
267
+ raw prompt to the same model and returns the reply. You then record the
268
+ parsed result with `t.score` or `t.check`.
269
+
208
270
  ## Run evals from the CLI
209
271
 
210
272
  The `eval` command discovers, filters, and runs cases.
@@ -230,7 +292,7 @@ match both groups.
230
292
 
231
293
  `eval` boots an ephemeral server on port 0 with a temp state root
232
294
  outside the project, so cases don't inherit ambient monorepo rules and
233
- don't pollute `.agent-sdk/`. Point `--url` at a running server to eval
295
+ don't pollute `.agent-serve/`. Point `--url` at a running server to eval
234
296
  a live agent instead:
235
297
 
236
298
  ```bash
@@ -299,8 +361,9 @@ agent-sdk serve --dir . --dev
299
361
  Playground runs target the live server instead of an ephemeral one.
300
362
  Their sessions appear in the session list. One eval batch can run at a
301
363
  time. By default those batches are **in-memory only** (capped by
302
- `maxPlaygroundRuns`); set `persistRuns` in `evals.config.ts` if you need them
303
- after a serve restart — see [Configure eval runs](#configure-eval-runs).
364
+ `maxPlaygroundRuns`); declare an `evals` table in `agent/storage.ts` if
365
+ you need them after a serve restart — see
366
+ [Storage](./storage.md#eval-and-ab-tables).
304
367
 
305
368
  The UI uses the playground eval routes (available without `--dev`):
306
369
  `GET /v1/dev/evals` lists datapoints and config (includes `maxPlaygroundRuns` /
@@ -378,6 +441,12 @@ once and commit the rendered fixture before you expand the suite.
378
441
  3. Assert decisions and output shape against the saved evidence.
379
442
  4. Keep a small `smoke` subset for any remaining live pipeline checks.
380
443
 
444
+ Read committed fixtures with `@cursor/july/evals/loaders`: `loadJson`,
445
+ `loadJsonl`, and `loadYaml` resolve relative paths against the project
446
+ root the runner discovered, not the cwd the CLI was invoked from
447
+ (`resolveFixturePath` and `evalFixtureRoot` expose the same
448
+ resolution for other file formats).
449
+
381
450
  `maxConcurrency` limits parallel datapoints. It does not limit model or
382
451
  API fan-out inside one datapoint. Materialized fixtures prevent a large
383
452
  suite from exhausting provider and GitHub rate limits. The
@@ -70,10 +70,11 @@ approving the PR.
70
70
  | Deterministic policy | [`agent/lib/approve.ts`](../../examples/approval-buddy/agent/lib/approve.ts), [`agent/lib/buddies.ts`](../../examples/approval-buddy/agent/lib/buddies.ts) | Own the roster and live eligibility checks. |
71
71
  | Review subagents | [`agent/subagents/`](../../examples/approval-buddy/agent/subagents/) | Run deep audit and code-quality passes over the same evidence. |
72
72
  | Storage | [`agent/storage.ts`](../../examples/approval-buddy/agent/storage.ts) | Persist sessions and events with `cursorHostedStorage` (Bugbot `agent_serve_*`). |
73
+ | Live A/B experiment | [`agent/ab.ts`](../../examples/approval-buddy/agent/ab.ts) | Compare baseline responses with a concise, presentation-only treatment (`concise-results`). |
73
74
  | Evals and unit tests | [`evals/`](../../examples/approval-buddy/evals/), [`agent/lib/`](../../examples/approval-buddy/agent/lib/) | Protect routing, output contracts, policy, and GitHub behavior. |
74
75
 
75
76
  There are no authored skills, MCP connections, schedules, reminders, hooks,
76
- A/B experiments, sandbox seeds, or tool approvals.
77
+ sandbox seeds, or tool approvals.
77
78
 
78
79
  ## Prepare credentials
79
80
 
@@ -52,6 +52,8 @@ need the watched-channel path.
52
52
  | [`agent/instructions.md`](../../examples/benny/agent/instructions.md) | Defines engagement rules, evidence policy, and the playbook routing map. |
53
53
  | [`agent/channels/slack.ts`](../../examples/benny/agent/channels/slack.ts) | Handles account-linked mentions and direct messages. |
54
54
  | [`agent/channels/slack-app.ts`](../../examples/benny/agent/channels/slack-app.ts) | Runs the dedicated app and watches one allowlisted channel. |
55
+ | [`agent/storage.ts`](../../examples/benny/agent/storage.ts) | Persists sessions and events with `cursorHostedStorage`. |
56
+ | [`evals/evals.config.ts`](../../examples/benny/evals/evals.config.ts) | Caps eval run concurrency. |
55
57
  | [`evals/smoke.eval.ts`](../../examples/benny/evals/smoke.eval.ts) | Checks the agent identity and expected triage route. |
56
58
 
57
59
  The playbook router authors no tools, MCP connections, subagents, schedules, hooks, A/B
@@ -58,6 +58,9 @@ mid-turn. That form writes the same files into the active session workspace.
58
58
  | [`agent/channels/review.ts`](../../examples/bugbot/agent/channels/review.ts) | Provides the loopback-only prepare-and-send HTTP route. |
59
59
  | [`agent/channels/slack.ts`](../../examples/bugbot/agent/channels/slack.ts) | Extracts PR references and prepares evidence for mentions and direct messages. |
60
60
  | [`agent/skills/pr-review.md`](../../examples/bugbot/agent/skills/pr-review.md) | Sets finding limits, severities, and the machine-readable review format. |
61
+ | [`agent/lib/log.ts`](../../examples/bugbot/agent/lib/log.ts) | Writes timing logs for the host tools to stderr. |
62
+ | [`agent/storage.ts`](../../examples/bugbot/agent/storage.ts) | Persists sessions and events with `cursorHostedStorage`. |
63
+ | [`evals/evals.config.ts`](../../examples/bugbot/evals/evals.config.ts) | Caps eval run concurrency. |
61
64
  | [`evals/review/smoke.eval.ts`](../../examples/bugbot/evals/review/smoke.eval.ts) | Seeds fake evidence and checks the review path without GitHub. |
62
65
 
63
66
  There is no authored GitHub channel, MCP connection, subagent, schedule,
@@ -67,6 +67,8 @@ tool, which writes the digest into the active session workspace.
67
67
  | [`agent/skills/feature-mapping.md`](../../examples/codebase-wiki/agent/skills/feature-mapping.md) | Maps changes onto features and fixes the page and changelog shape. |
68
68
  | [`agent/schedules/daily-digest.md`](../../examples/codebase-wiki/agent/schedules/daily-digest.md) | Writes `digests/<date>`, rebuilds the index, and flags stale pages. |
69
69
  | [`agent/channels/github.ts`](../../examples/codebase-wiki/agent/channels/github.ts) | Acknowledges closed PRs and starts merged-only ingest turns. |
70
+ | [`agent/storage.ts`](../../examples/codebase-wiki/agent/storage.ts) | Persists sessions and events with `cursorHostedStorage`. |
71
+ | [`evals/evals.config.ts`](../../examples/codebase-wiki/evals/evals.config.ts) | Caps eval run concurrency. |
70
72
  | [`evals/ingest.eval.ts`](../../examples/codebase-wiki/evals/ingest.eval.ts) | Gates ingest decisions against the wiki filesystem. |
71
73
 
72
74
  There is no MCP connection, subagent, hook, A/B experiment, or custom
@@ -74,6 +74,8 @@ APPROVE and commit statuses on top of the same shape.
74
74
  | [`agent/subagents/area-reviewer/`](../../examples/codeowners-review/agent/subagents/area-reviewer/) | Defines the one-area, one-playbook reviewer contract. |
75
75
  | [`agent/channels/github.ts`](../../examples/codeowners-review/agent/channels/github.ts) | Reviews opened, reopened, synchronized, and undrafted PRs. |
76
76
  | [`fixtures/`](../../examples/codeowners-review/fixtures/) | Ships two reviewable PRs with known planted findings. |
77
+ | [`agent/storage.ts`](../../examples/codeowners-review/agent/storage.ts) | Persists sessions and events with `cursorHostedStorage`. |
78
+ | [`evals/evals.config.ts`](../../examples/codeowners-review/evals/evals.config.ts) | Caps eval run concurrency. |
77
79
  | [`evals/review.eval.ts`](../../examples/codeowners-review/evals/review.eval.ts) | Gates routing, fan-out, planted bugs, and verdicts. |
78
80
 
79
81
  There is no MCP connection, schedule, hook, A/B experiment, or custom
@@ -68,6 +68,7 @@ share Concierge's conversation history.
68
68
  | [`agent/agent.ts`](../../examples/concierge/agent/agent.ts) | Describes the root agent and selects the local runtime. |
69
69
  | [`agent/instructions.md`](../../examples/concierge/agent/instructions.md) | Draws a strict weather-only delegation boundary. |
70
70
  | [`agent/mcp-connections/weather.ts`](../../examples/concierge/agent/mcp-connections/weather.ts) | Resolves the peer by its `weather-agent` slug. |
71
+ | [`agent/storage.ts`](../../examples/concierge/agent/storage.ts) | Persists sessions and events with `cursorHostedStorage`. |
71
72
 
72
73
  Concierge doesn't author channels, tools, skills, subagents, schedules,
73
74
  hooks, A/B experiments, or evals. The built-in HTTP and MCP surfaces still
@@ -74,6 +74,7 @@ has `status: "finished"` and no remote session.
74
74
  | Affinity and buffering | [`agent/lib/pr-affinity.ts`](../../examples/fsd/agent/lib/pr-affinity.ts), [`agent/lib/webhook-buffer.ts`](../../examples/fsd/agent/lib/webhook-buffer.ts) | Persist PR identity, sticky mode, and pending wakes. |
75
75
  | Reminders | [`agent/lib/merge-conflict-watch.ts`](../../examples/fsd/agent/lib/merge-conflict-watch.ts) | Recheck merge conflicts every 30 minutes. |
76
76
  | Workflow client | [`agent/lib/fsd-platform.ts`](../../examples/fsd/agent/lib/fsd-platform.ts) | Enroll external runs and read or record findings. |
77
+ | Storage | [`agent/storage.ts`](../../examples/fsd/agent/storage.ts) | Persist sessions and events with `cursorHostedStorage`. |
77
78
 
78
79
  The coordinator has no authored skill, subagent, MCP connection, static
79
80
  schedule, A/B experiment, eval, custom storage definition, or tool approval.
@@ -65,6 +65,8 @@ one-off questions and to ask before saving anything borderline.
65
65
  | [`agent/skills/wiki-conventions.md`](../../examples/knowledge-base/agent/skills/wiki-conventions.md) | Names pages, shapes them, and dates every fact. |
66
66
  | [`agent/schedules/gardener.md`](../../examples/knowledge-base/agent/schedules/gardener.md) | Merges duplicates, rebuilds the index, and flags stale facts daily. |
67
67
  | [`agent/lib/wiki-store.test.ts`](../../examples/knowledge-base/agent/lib/wiki-store.test.ts) | Unit-tests slug safety and store round-trips. |
68
+ | [`agent/storage.ts`](../../examples/knowledge-base/agent/storage.ts) | Persists sessions and events with `cursorHostedStorage`. |
69
+ | [`evals/evals.config.ts`](../../examples/knowledge-base/evals/evals.config.ts) | Caps eval run concurrency. |
68
70
  | [`evals/knowledge.eval.ts`](../../examples/knowledge-base/evals/knowledge.eval.ts) | Seeds a temp knowledge base and gates recall, save, and no-write decisions. |
69
71
 
70
72
  There is no authored channel, MCP connection, subagent, hook, A/B
@@ -51,6 +51,8 @@ Mentions and DMs skip the watch entirely and behave like ordinary chat.
51
51
  | [`agent/lib/slack-api.ts`](../../examples/oncall/agent/lib/slack-api.ts) | Reactions and thread posts on this agent's own token pair. |
52
52
  | [`agent/tools/reminders_create.ts`](../../examples/oncall/agent/tools/reminders_create.ts) | Self-scheduled wakes bound to the thread (plus `reminders_list` and `reminders_cancel`). |
53
53
  | [`agent/tools/post_thread_update.ts`](../../examples/oncall/agent/tools/post_thread_update.ts) | Interim updates to the thread mid-turn. |
54
+ | [`agent/storage.ts`](../../examples/oncall/agent/storage.ts) | Persists sessions and events with `cursorHostedStorage`. |
55
+ | [`evals/evals.config.ts`](../../examples/oncall/evals/evals.config.ts) | Caps eval run concurrency. |
54
56
  | [`evals/smoke.eval.ts`](../../examples/oncall/evals/smoke.eval.ts) | Checks identity and the reminder-tool route. |
55
57
 
56
58
  ## Let bot posts through the watch
@@ -49,6 +49,7 @@ or tool routing.
49
49
  | [`agent/agent.ts`](../../examples/slack-agent/agent/agent.ts) | Names the agent and selects the model. The omitted `runtime` defaults to local. |
50
50
  | [`agent/instructions.md`](../../examples/slack-agent/agent/instructions.md) | Sets the always-on response style. |
51
51
  | [`agent/channels/slack.ts`](../../examples/slack-agent/agent/channels/slack.ts) | Connects the signed-in host account to Slack. |
52
+ | [`agent/storage.ts`](../../examples/slack-agent/agent/storage.ts) | Persists sessions and events with `cursorHostedStorage`. |
52
53
 
53
54
  There are no authored tools, skills, MCP connections, subagents, schedules,
54
55
  hooks, A/B experiments, or evals. This small surface is the lesson.
@@ -141,10 +141,15 @@ first prompt; the cloud model writes and invokes the script in its VM.
141
141
 
142
142
  ## Verify cloud custom-tool execution
143
143
 
144
- Ask the agent to run `probe_cloud_tool`. It writes a deployment-scoped marker
145
- and returns the complete session id, tool-call id, and marker path. Correlate
146
- those ids with `actions.requested` / `action.result` in the session stream and
147
- the host's `cloud HTTP MCP tool ... dispatch/complete` log lines.
144
+ Ask the agent to run `probe_cloud_tool`. On cloud that tool is MCP on
145
+ `agentsdk-tools`, not a local script. A real call writes a deployment-scoped
146
+ marker at `tool-observations/<sessionId>/<toolCallId>.json` and returns those
147
+ ids. Correlate them with `actions.requested` / `action.result` for
148
+ `probe_cloud_tool` (not `shell`) and the host's
149
+ `cloud HTTP MCP tool ... dispatch/complete` log lines.
150
+
151
+ A local `.agent-serve/tools/probe_cloud_tool.sh` or a marker under `probes/`
152
+ means the model invented a substitute and the host never ran.
148
153
 
149
154
  For a body-only smoke test:
150
155
 
@@ -62,17 +62,24 @@ mapping shifts:
62
62
  | Folder or file | Local runtime | Cloud runtime |
63
63
  | --- | --- | --- |
64
64
  | `instructions.*` | `AGENTS.md` in the session workspace | prepended to the first prompt |
65
- | Server tools (`execution: "server"`) | in-process SDK custom tools | authenticated HTTP MCP back to the AgentSDK host |
65
+ | Server tools (`execution: "server"`) | in-process SDK custom tools | authenticated HTTP MCP back to the AgentSDK host, when `--public-url` or `--cloud-tools-url` is set |
66
66
  | Agent tools (`execution: "agent"`) | scripts in the session workspace | catalog + script bodies on the first prompt |
67
67
  | `skills/*` | `.cursor/skills/` in the workspace | only if present in the cloud repo |
68
68
  | `mcp-connections/*.ts` | SDK `mcpServers` | SDK `mcpServers` (peers need `--public-url`) |
69
69
  | `sandbox/workspace/**` | seeded into the session workspace | ignored |
70
- | Tool approvals (`needsApproval`) | supported | supported for server tools |
70
+ | Tool approvals (`needsApproval`) | supported | not supported; keep approval-gated tools on local turns |
71
71
 
72
- Hosted deployments configure the server-tool MCP URL automatically. A
72
+ Hosted deployments configure the server-tool MCP URL automatically
73
+ (`cloudToolsUrl`, authenticated with the resolved Cursor API key). A
73
74
  self-hosted public server needs `--public-url` (and `--bearer-token` when the
74
75
  host is not behind another trusted authentication boundary) so cloud turns
75
- can reach those tools.
76
+ can reach those tools. Without either, the server warns at startup and
77
+ cloud turns omit the server tools.
78
+
79
+ Approvals are a local-runtime contract. On cloud, a `needsApproval` tool
80
+ call rides one HTTP MCP request from the VM, and a parked call would
81
+ hold that request open until it times out; there is no durable approval
82
+ flow for cloud turns.
76
83
 
77
84
  Two more behaviors are cloud-specific. Sessions persist a separate SDK
78
85
  agent id (`bc-…`), emitted on the stream as `agent.bound` with a URL to