@cursor/july 0.1.35 → 0.1.36
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/AGENTS.md +5 -9
- package/README.md +5 -9
- package/dist/channels/github/github-channel.d.ts +1 -1
- package/dist/channels/github/github-channel.js +1 -1
- package/dist/channels/github/index.d.ts +1 -1
- package/dist/channels/github/index.js +1 -1
- package/dist/channels/slack/index.d.ts +1 -1
- package/dist/channels/slack/index.js +1 -1
- package/dist/channels/slack/init.d.ts.map +1 -1
- package/dist/channels/slack/init.js +2 -1
- package/dist/docs/404.html +2 -2
- package/dist/docs/ab.html +8 -17
- package/dist/docs/assets/{ab.md.DAQoJ-up.js → ab.md.6cLOW7--.js} +4 -13
- package/dist/docs/assets/{ab.md.DAQoJ-up.lean.js → ab.md.6cLOW7--.lean.js} +1 -1
- package/dist/docs/assets/{app.D5Mv1T0U.js → app.DEcxy4oz.js} +1 -1
- package/dist/docs/assets/{building-with-agents.md.CnHqvYDd.js → building-with-agents.md.txrcGU2B.js} +2 -2
- package/dist/docs/assets/chunks/@localSearchIndexroot.ByFYcFly.js +1 -0
- package/dist/docs/assets/chunks/{VPLocalSearchBox.CMq_BQce.js → VPLocalSearchBox.n1VOZcy3.js} +1 -1
- package/dist/docs/assets/chunks/{theme.C6D9UPLK.js → theme.BaF1MQ9c.js} +2 -2
- package/dist/docs/assets/{concepts.md.DFaQEFkA.js → concepts.md.CqOsxbMU.js} +1 -1
- package/dist/docs/assets/{deployment.md.9MYBuKM1.js → deployment.md.CuK5SNjN.js} +1 -1
- package/dist/docs/assets/{evals.md.BIUoVZ6X.js → evals.md.BQXI3rXy.js} +9 -15
- package/dist/docs/assets/{evals.md.BIUoVZ6X.lean.js → evals.md.BQXI3rXy.lean.js} +1 -1
- package/dist/docs/assets/{example-agents_approval-buddy.md.BhEfleVx.js → example-agents_approval-buddy.md.CIiZ9coo.js} +1 -1
- package/dist/docs/assets/{example-agents_benny.md.2Et1qa8f.js → example-agents_benny.md.l7JTmm8X.js} +1 -1
- package/dist/docs/assets/{example-agents_bugbot.md.ByUexi5i.js → example-agents_bugbot.md.Dp5JqHSQ.js} +2 -2
- package/dist/docs/assets/{example-agents_bugbot.md.ByUexi5i.lean.js → example-agents_bugbot.md.Dp5JqHSQ.lean.js} +1 -1
- package/dist/docs/assets/{example-agents_codebase-wiki.md.B4y-7ZVW.js → example-agents_codebase-wiki.md.D-lteFf0.js} +1 -1
- package/dist/docs/assets/{example-agents_codebase-wiki.md.B4y-7ZVW.lean.js → example-agents_codebase-wiki.md.D-lteFf0.lean.js} +1 -1
- package/dist/docs/assets/{example-agents_codeowners-review.md.D6ay4nvf.js → example-agents_codeowners-review.md.BU2ZXLf-.js} +1 -1
- package/dist/docs/assets/{example-agents_codeowners-review.md.D6ay4nvf.lean.js → example-agents_codeowners-review.md.BU2ZXLf-.lean.js} +1 -1
- package/dist/docs/assets/{example-agents_concierge.md.lL8rhYlj.js → example-agents_concierge.md.DA2al_NK.js} +2 -2
- package/dist/docs/assets/{example-agents_concierge.md.lL8rhYlj.lean.js → example-agents_concierge.md.DA2al_NK.lean.js} +1 -1
- package/dist/docs/assets/{example-agents_fsd.md.DfNKQTHz.js → example-agents_fsd.md.DPz9ezO4.js} +1 -1
- package/dist/docs/assets/{example-agents_knowledge-base.md.CzyZ2DCr.js → example-agents_knowledge-base.md.IneynQSR.js} +1 -1
- package/dist/docs/assets/{example-agents_oncall.md.wFFXXEyW.js → example-agents_oncall.md.ZE0n6ZFN.js} +1 -1
- package/dist/docs/assets/{example-agents_slack-agent.md.DvgvT4nn.js → example-agents_slack-agent.md.06jQXTAI.js} +1 -1
- package/dist/docs/assets/{example-agents_weather-agent.md.BADkPqxQ.js → example-agents_weather-agent.md.CrGZ0SqR.js} +3 -3
- package/dist/docs/assets/{example-agents_weather-agent.md.BADkPqxQ.lean.js → example-agents_weather-agent.md.CrGZ0SqR.lean.js} +1 -1
- package/dist/docs/assets/guides_cloud-runtime.md.V5igN4Sq.js +9 -0
- package/dist/docs/assets/guides_cloud-runtime.md.V5igN4Sq.lean.js +1 -0
- package/dist/docs/assets/{guides_webhooks.md.DiAwSR42.js → guides_webhooks.md.BERuBSJW.js} +1 -1
- package/dist/docs/assets/{hillclimbing.md.D9Y1_bYh.js → hillclimbing.md.yXqdlv2R.js} +1 -1
- package/dist/docs/assets/index.md.CmhptOmN.js +24 -0
- package/dist/docs/assets/{index.md.CZqbBJPB.lean.js → index.md.CmhptOmN.lean.js} +1 -1
- package/dist/docs/assets/{quickstart.md.TnEXYgYW.js → quickstart.md.C_b6ESpD.js} +7 -4
- package/dist/docs/assets/{quickstart.md.TnEXYgYW.lean.js → quickstart.md.C_b6ESpD.lean.js} +1 -1
- package/dist/docs/assets/{reference_agent-config.md.kuN6-OxK.js → reference_agent-config.md.CRmkoxd6.js} +6 -4
- package/dist/docs/assets/{reference_agent-config.md.kuN6-OxK.lean.js → reference_agent-config.md.CRmkoxd6.lean.js} +1 -1
- package/dist/docs/assets/reference_artifacts.md.BGG4bZo-.js +19 -0
- package/dist/docs/assets/reference_artifacts.md.BGG4bZo-.lean.js +1 -0
- package/dist/docs/assets/{reference_channels.md.CDhTRfUz.js → reference_channels.md.BIabFUAI.js} +2 -2
- package/dist/docs/assets/{reference_channels.md.CDhTRfUz.lean.js → reference_channels.md.BIabFUAI.lean.js} +1 -1
- package/dist/docs/assets/{reference_cli.md.sD-IUWjg.js → reference_cli.md.Byvrg8eu.js} +15 -9
- package/dist/docs/assets/{reference_cli.md.sD-IUWjg.lean.js → reference_cli.md.Byvrg8eu.lean.js} +1 -1
- package/dist/docs/assets/{reference_hooks.md.DyLVfE1O.js → reference_hooks.md.BGDw4VLm.js} +2 -2
- package/dist/docs/assets/{reference_hooks.md.DyLVfE1O.lean.js → reference_hooks.md.BGDw4VLm.lean.js} +1 -1
- package/dist/docs/assets/reference_http-api.md.DGrw_wOu.js +11 -0
- package/dist/docs/assets/reference_http-api.md.DGrw_wOu.lean.js +1 -0
- package/dist/docs/assets/{reference_project-layout.md.D8E6ZmHJ.js → reference_project-layout.md._XdeMahr.js} +2 -2
- package/dist/docs/assets/{reference_project-layout.md.D8E6ZmHJ.lean.js → reference_project-layout.md._XdeMahr.lean.js} +1 -1
- package/dist/docs/assets/{reference_sessions.md.C_ouF_uf.js → reference_sessions.md.DBVFi2Sx.js} +2 -2
- package/dist/docs/assets/{reference_subagents.md.zWAMNfi1.js → reference_subagents.md.DSrGLIuB.js} +2 -2
- package/dist/docs/assets/{reference_subagents.md.zWAMNfi1.lean.js → reference_subagents.md.DSrGLIuB.lean.js} +1 -1
- package/dist/docs/assets/{reference_tools.md.BswAQM41.js → reference_tools.md.lSrsTxYJ.js} +4 -4
- package/dist/docs/assets/{reference_tools.md.BswAQM41.lean.js → reference_tools.md.lSrsTxYJ.lean.js} +1 -1
- package/dist/docs/assets/scaffolding-agents.md.mkc3B_ZW.js +1 -0
- package/dist/docs/assets/{scaffolding-agents.md.Bsr9Pwzu.lean.js → scaffolding-agents.md.mkc3B_ZW.lean.js} +1 -1
- package/dist/docs/assets/{storage.md.xZoiGM58.js → storage.md.mQDtIULc.js} +3 -3
- package/dist/docs/assets/{storage.md.xZoiGM58.lean.js → storage.md.mQDtIULc.lean.js} +1 -1
- package/dist/docs/building-with-agents.html +6 -6
- package/dist/docs/concepts.html +5 -5
- package/dist/docs/deployment.html +6 -6
- package/dist/docs/evals.html +13 -19
- package/dist/docs/example-agents/approval-buddy.html +5 -5
- package/dist/docs/example-agents/benny.html +5 -5
- package/dist/docs/example-agents/bugbot.html +5 -5
- package/dist/docs/example-agents/codebase-wiki.html +5 -5
- package/dist/docs/example-agents/codeowners-review.html +5 -5
- package/dist/docs/example-agents/concierge.html +6 -6
- package/dist/docs/example-agents/fsd.html +5 -5
- package/dist/docs/example-agents/index.html +4 -4
- package/dist/docs/example-agents/knowledge-base.html +5 -5
- package/dist/docs/example-agents/oncall.html +5 -5
- package/dist/docs/example-agents/security-reviewer.html +4 -4
- package/dist/docs/example-agents/slack-agent.html +5 -5
- package/dist/docs/example-agents/weather-agent.html +6 -6
- package/dist/docs/guides/agent-to-agent.html +5 -5
- package/dist/docs/guides/cloud-runtime.html +6 -6
- package/dist/docs/guides/github.html +4 -4
- package/dist/docs/guides/human-in-the-loop.html +4 -4
- package/dist/docs/guides/mcp-oauth.html +5 -5
- package/dist/docs/guides/slack.html +4 -4
- package/dist/docs/guides/webhooks.html +6 -6
- package/dist/docs/hashmap.json +1 -1
- package/dist/docs/hillclimbing.html +6 -6
- package/dist/docs/index.html +11 -7
- package/dist/docs/quickstart.html +10 -7
- package/dist/docs/reference/agent-config.html +10 -8
- package/dist/docs/reference/artifacts.html +43 -0
- package/dist/docs/reference/channels.html +6 -6
- package/dist/docs/reference/cli.html +18 -12
- package/dist/docs/reference/connections.html +4 -4
- package/dist/docs/reference/hooks.html +6 -6
- package/dist/docs/reference/http-api.html +7 -7
- package/dist/docs/reference/instructions.html +4 -4
- package/dist/docs/reference/playground.html +4 -4
- package/dist/docs/reference/project-layout.html +6 -6
- package/dist/docs/reference/prompt.html +4 -4
- package/dist/docs/reference/schedules.html +4 -4
- package/dist/docs/reference/sessions.html +7 -7
- package/dist/docs/reference/skills.html +4 -4
- package/dist/docs/reference/subagents.html +6 -6
- package/dist/docs/reference/tools.html +7 -7
- package/dist/docs/scaffolding-agents.html +5 -5
- package/dist/docs/storage.html +6 -6
- package/dist/docs/troubleshooting.html +4 -4
- package/dist/files-backends/agent-store-presigned-url.d.ts.map +1 -1
- package/dist/files-backends/agent-store-presigned-url.js +15 -22
- package/dist/internal/cli-github.d.ts.map +1 -1
- package/dist/internal/cli-github.js +8 -7
- package/dist/internal/cli-slack.js +9 -9
- package/dist/internal/event-mapper.d.ts +3 -3
- package/dist/internal/event-mapper.d.ts.map +1 -1
- package/dist/internal/event-mapper.js +7 -4
- package/dist/internal/session-engine.js +3 -3
- package/dist/internal/workspace.d.ts +8 -6
- package/dist/internal/workspace.d.ts.map +1 -1
- package/dist/internal/workspace.js +15 -11
- package/dist/playground/assets/{index-D7OV8B_H.js → index-DOnKC85G.js} +18 -18
- package/dist/playground/assets/{index-DOb96C0M.css → index-DoQjqj5w.css} +1 -1
- package/dist/playground/index.html +2 -2
- package/docs/README.md +32 -13
- package/docs/ab.md +23 -36
- package/docs/building-with-agents.md +2 -2
- package/docs/concepts.md +3 -2
- package/docs/deployment.md +1 -1
- package/docs/evals.md +102 -33
- package/docs/example-agents/approval-buddy.md +2 -1
- package/docs/example-agents/benny.md +2 -0
- package/docs/example-agents/bugbot.md +3 -0
- package/docs/example-agents/codebase-wiki.md +2 -0
- package/docs/example-agents/codeowners-review.md +2 -0
- package/docs/example-agents/concierge.md +1 -0
- package/docs/example-agents/fsd.md +1 -0
- package/docs/example-agents/knowledge-base.md +2 -0
- package/docs/example-agents/oncall.md +2 -0
- package/docs/example-agents/slack-agent.md +1 -0
- package/docs/example-agents/weather-agent.md +9 -4
- package/docs/guides/cloud-runtime.md +11 -4
- package/docs/guides/webhooks.md +1 -1
- package/docs/hillclimbing.md +1 -1
- package/docs/quickstart.md +39 -7
- package/docs/reference/agent-config.md +74 -14
- package/docs/reference/artifacts.md +117 -0
- package/docs/reference/channels.md +45 -15
- package/docs/reference/cli.md +141 -20
- package/docs/reference/hooks.md +11 -4
- package/docs/reference/http-api.md +50 -4
- package/docs/reference/project-layout.md +6 -0
- package/docs/reference/sessions.md +5 -4
- package/docs/reference/subagents.md +5 -3
- package/docs/reference/tools.md +23 -7
- package/docs/scaffolding-agents.md +11 -2
- package/docs/storage.md +27 -2
- package/package.json +1 -1
- package/src/channels/github/github-channel.ts +1 -1
- package/src/channels/github/index.ts +1 -1
- package/src/channels/slack/index.ts +1 -1
- package/src/channels/slack/init.ts +2 -1
- package/src/files-backends/agent-store-presigned-url.ts +2 -1
- package/src/internal/cli-github.ts +8 -7
- package/src/internal/cli-slack.ts +9 -9
- package/src/internal/event-mapper.ts +9 -4
- package/src/internal/session-engine.ts +3 -3
- package/src/internal/workspace.ts +15 -11
- package/dist/docs/assets/chunks/@localSearchIndexroot.Cu7b6o1D.js +0 -1
- package/dist/docs/assets/guides_cloud-runtime.md.CDJGvVC4.js +0 -9
- package/dist/docs/assets/guides_cloud-runtime.md.CDJGvVC4.lean.js +0 -1
- package/dist/docs/assets/index.md.CZqbBJPB.js +0 -20
- package/dist/docs/assets/reference_http-api.md.CfVM_ICa.js +0 -11
- package/dist/docs/assets/reference_http-api.md.CfVM_ICa.lean.js +0 -1
- package/dist/docs/assets/scaffolding-agents.md.Bsr9Pwzu.js +0 -1
- /package/dist/docs/assets/{building-with-agents.md.CnHqvYDd.lean.js → building-with-agents.md.txrcGU2B.lean.js} +0 -0
- /package/dist/docs/assets/{concepts.md.DFaQEFkA.lean.js → concepts.md.CqOsxbMU.lean.js} +0 -0
- /package/dist/docs/assets/{deployment.md.9MYBuKM1.lean.js → deployment.md.CuK5SNjN.lean.js} +0 -0
- /package/dist/docs/assets/{example-agents_approval-buddy.md.BhEfleVx.lean.js → example-agents_approval-buddy.md.CIiZ9coo.lean.js} +0 -0
- /package/dist/docs/assets/{example-agents_benny.md.2Et1qa8f.lean.js → example-agents_benny.md.l7JTmm8X.lean.js} +0 -0
- /package/dist/docs/assets/{example-agents_fsd.md.DfNKQTHz.lean.js → example-agents_fsd.md.DPz9ezO4.lean.js} +0 -0
- /package/dist/docs/assets/{example-agents_knowledge-base.md.CzyZ2DCr.lean.js → example-agents_knowledge-base.md.IneynQSR.lean.js} +0 -0
- /package/dist/docs/assets/{example-agents_oncall.md.wFFXXEyW.lean.js → example-agents_oncall.md.ZE0n6ZFN.lean.js} +0 -0
- /package/dist/docs/assets/{example-agents_slack-agent.md.DvgvT4nn.lean.js → example-agents_slack-agent.md.06jQXTAI.lean.js} +0 -0
- /package/dist/docs/assets/{guides_webhooks.md.DiAwSR42.lean.js → guides_webhooks.md.BERuBSJW.lean.js} +0 -0
- /package/dist/docs/assets/{hillclimbing.md.D9Y1_bYh.lean.js → hillclimbing.md.yXqdlv2R.lean.js} +0 -0
- /package/dist/docs/assets/{reference_sessions.md.C_ouF_uf.lean.js → reference_sessions.md.DBVFi2Sx.lean.js} +0 -0
|
@@ -8,8 +8,8 @@
|
|
|
8
8
|
/>
|
|
9
9
|
<meta name="viewport" content="width=device-width, initial-scale=1" />
|
|
10
10
|
<title>agent-serve playground</title>
|
|
11
|
-
<script type="module" crossorigin src="./assets/index-
|
|
12
|
-
<link rel="stylesheet" crossorigin href="./assets/index-
|
|
11
|
+
<script type="module" crossorigin src="./assets/index-DOnKC85G.js"></script>
|
|
12
|
+
<link rel="stylesheet" crossorigin href="./assets/index-DoQjqj5w.css">
|
|
13
13
|
</head>
|
|
14
14
|
<body>
|
|
15
15
|
<div id="root"></div>
|
package/docs/README.md
CHANGED
|
@@ -36,7 +36,19 @@ my-agent/
|
|
|
36
36
|
└── evals/ # filesystem evals (regression checks)
|
|
37
37
|
```
|
|
38
38
|
|
|
39
|
-
|
|
39
|
+
Bootstrap a project with nothing installed beyond Node:
|
|
40
|
+
|
|
41
|
+
```bash
|
|
42
|
+
npx @cursor/july init ./my-agent
|
|
43
|
+
cd my-agent
|
|
44
|
+
npx agent-sdk dev
|
|
45
|
+
```
|
|
46
|
+
|
|
47
|
+
`init` scaffolds the project, runs `npm install`, and offers a Cursor
|
|
48
|
+
sign-in. The install puts the `agent-sdk` bin on the project's path, so
|
|
49
|
+
`npx agent-sdk` resolves locally from then on.
|
|
50
|
+
|
|
51
|
+
Browse the docs locally without serving an agent:
|
|
40
52
|
|
|
41
53
|
```bash
|
|
42
54
|
npx @cursor/july docs
|
|
@@ -80,6 +92,8 @@ Pick your entry point based on your goal.
|
|
|
80
92
|
evals as regression checks.
|
|
81
93
|
- [Live A/B metrics](./ab.md): assign sticky variants and compare
|
|
82
94
|
cumulative metrics on live sessions.
|
|
95
|
+
- [Storage](./storage.md): point durable storage at a backend you own
|
|
96
|
+
with `defineStorage`.
|
|
83
97
|
- [Hillclimbing](./hillclimbing.md): make an agent better one measured
|
|
84
98
|
round at a time.
|
|
85
99
|
|
|
@@ -102,7 +116,7 @@ Pick your entry point based on your goal.
|
|
|
102
116
|
|
|
103
117
|
**Example agents**
|
|
104
118
|
|
|
105
|
-
- [Choose the right example](./example-agents/index.md): compare all
|
|
119
|
+
- [Choose the right example](./example-agents/index.md): compare all twelve
|
|
106
120
|
agents by runtime, channels, tools, state, and architecture.
|
|
107
121
|
- [Weather agent](./example-agents/weather-agent.md): explore tools, MCP,
|
|
108
122
|
approvals, skills, subagents, schedules, hooks, A/B metrics, and evals.
|
|
@@ -112,6 +126,8 @@ Pick your entry point based on your goal.
|
|
|
112
126
|
with its own context and sessions.
|
|
113
127
|
- [Playbook router](./example-agents/benny.md): route Slack intake through inherited
|
|
114
128
|
repository playbooks.
|
|
129
|
+
- [Alert investigator](./example-agents/oncall.md): watch a Slack alerts
|
|
130
|
+
channel and pin a self-rechecking investigation to every alert thread.
|
|
115
131
|
- [PR evidence reviewer](./example-agents/bugbot.md): review a host-prepared,
|
|
116
132
|
diff-first pull-request evidence tree.
|
|
117
133
|
- [Approval Buddy](./example-agents/approval-buddy.md): keep approval policy
|
|
@@ -154,7 +170,14 @@ Pick your entry point based on your goal.
|
|
|
154
170
|
## Run the CLI
|
|
155
171
|
|
|
156
172
|
The docs write commands as `agent-sdk <command>`. Where that command
|
|
157
|
-
comes from depends on where you run.
|
|
173
|
+
comes from depends on where you run. Starting fresh? This works with no
|
|
174
|
+
prior install:
|
|
175
|
+
|
|
176
|
+
```bash
|
|
177
|
+
npx @cursor/july init ./my-agent
|
|
178
|
+
cd my-agent
|
|
179
|
+
npx agent-sdk dev
|
|
180
|
+
```
|
|
158
181
|
|
|
159
182
|
> [!NOTE]
|
|
160
183
|
> When running from a source checkout there is no installed bin. Alias it from
|
|
@@ -171,22 +194,18 @@ comes from depends on where you run.
|
|
|
171
194
|
> runs the `july` bin with that command (no local install required).
|
|
172
195
|
|
|
173
196
|
> [!NOTE]
|
|
174
|
-
> The framework is being renamed from agent-serve to the Agent SDK
|
|
175
|
-
>
|
|
176
|
-
>
|
|
177
|
-
>
|
|
197
|
+
> The framework is being renamed from agent-serve to the Agent SDK. The
|
|
198
|
+
> `agent-sdk` bin already ships (alongside `july`, `agentkit`, and the
|
|
199
|
+
> legacy `agent-serve` alias), and projects already import from
|
|
200
|
+
> `@cursor/july`. A few on-disk names keep their old form until the code
|
|
201
|
+
> rename ships:
|
|
178
202
|
>
|
|
179
|
-
> |
|
|
203
|
+
> | Future name | Current name |
|
|
180
204
|
> | --- | --- |
|
|
181
|
-
> | `@cursor/july` imports and dependency | `@cursor/july` |
|
|
182
|
-
> | `agent-sdk` bin | `agent-serve` |
|
|
183
|
-
> | `dist/bin/agent-sdk.js` | `dist/bin/agent-serve.js` |
|
|
184
205
|
> | `.agent-sdk/` state directory | `.agent-serve/` |
|
|
185
206
|
> | `/var/lib/agent-sdk` (deploy state root) | `/var/lib/agent-serve` |
|
|
186
207
|
> | `CURSOR_AGENT_SDK_*` env vars | `AGENT_SERVE_*` |
|
|
187
|
-
> | `agent-sdk (<hostname>)` API key name | `agent-serve (<hostname>)` |
|
|
188
208
|
> | Package path `packages/agent-sdk` | `packages/agent-serve` |
|
|
189
|
-
> | Package skills `packages/agent-sdk/skills/` | `packages/agent-serve/skills/` |
|
|
190
209
|
|
|
191
210
|
> [!WARNING]
|
|
192
211
|
> Run the Agent SDK with Node 22.13 or newer, and never with Bun.
|
package/docs/ab.md
CHANGED
|
@@ -19,7 +19,7 @@ on fixed inputs.
|
|
|
19
19
|
|
|
20
20
|
> [!NOTE]
|
|
21
21
|
> Import paths here use `@cursor/july/ab`. On projects still
|
|
22
|
-
> using `@
|
|
22
|
+
> using `@anysphere/agent-serve`, swap the import. See
|
|
23
23
|
> [Run the CLI](./README.md#run-the-cli) for the full rename table.
|
|
24
24
|
|
|
25
25
|
## Choose live A/B metrics or evals
|
|
@@ -208,6 +208,7 @@ turn and tool events into its own counters.
|
|
|
208
208
|
| `toolErrors` | Adds one when `action.result.data.isError` is true |
|
|
209
209
|
| `inputTokens`, `outputTokens` | Adds usage from completed turns |
|
|
210
210
|
| `cacheReadTokens`, `cacheWriteTokens` | Adds cache usage from completed turns |
|
|
211
|
+
| `costUsd` | Sums the estimated turn cost recorded on `turn.completed` (turns whose model has no known rates contribute 0) |
|
|
211
212
|
| `wallTimeMs` | Sums the time from `turn.started` to its completed or failed event |
|
|
212
213
|
| `custom` | Sums finite numeric deltas returned by `derive` |
|
|
213
214
|
|
|
@@ -254,44 +255,30 @@ too. This include-all behavior can still apply to `GET /v1/abs` in dev
|
|
|
254
255
|
when bearer or custom auth keeps `GET /v1/sessions` owner-scoped.
|
|
255
256
|
|
|
256
257
|
Session `events.ndjson` is the source of truth for assignment + fold.
|
|
257
|
-
`GET /v1/abs` recomputes aggregates from those logs.
|
|
258
|
-
|
|
259
|
-
|
|
260
|
-
|
|
258
|
+
`GET /v1/abs` recomputes aggregates from those logs. For durable
|
|
259
|
+
sample and snapshot export, declare an `abs` table in
|
|
260
|
+
`agent/storage.ts` — same idea as the eval-runs table. See
|
|
261
|
+
[Storage](./storage.md#eval-and-ab-tables).
|
|
261
262
|
|
|
262
|
-
## Configure
|
|
263
|
+
## Configure the playground fold window
|
|
263
264
|
|
|
264
|
-
|
|
265
|
+
Assignments and foldable metrics already persist in each session's
|
|
266
|
+
`events.ndjson` under `--state-root`. The optional `agent/ab.config.ts`
|
|
267
|
+
only caps how many sessions the playground and `GET /v1/abs` fold:
|
|
265
268
|
|
|
266
269
|
```ts
|
|
267
|
-
import {
|
|
268
|
-
defineABConfig,
|
|
269
|
-
persistABSamplesToDir,
|
|
270
|
-
persistABSnapshotsToDir,
|
|
271
|
-
} from "@cursor/july/ab";
|
|
270
|
+
import { defineABConfig } from "@cursor/july/ab";
|
|
272
271
|
|
|
273
272
|
export default defineABConfig({
|
|
274
273
|
// Optional — defaults to 200. Only affects GET /v1/abs / A/Bs tab.
|
|
275
|
-
|
|
276
|
-
// retainSnapshots: 20, // prune budget for persistABSnapshotsToDir
|
|
277
|
-
// Append onSample payloads (in addition to each experiment's onSample)
|
|
278
|
-
persistSamples: persistABSamplesToDir(".agent-serve/ab-samples"),
|
|
279
|
-
// Save aggregate snapshots whenever GET /v1/abs runs
|
|
280
|
-
persistSnapshots: persistABSnapshotsToDir(".agent-serve/ab-snapshots"),
|
|
274
|
+
maxPlaygroundSessions: 500,
|
|
281
275
|
});
|
|
282
276
|
```
|
|
283
277
|
|
|
284
|
-
|
|
285
|
-
|
|
286
|
-
|
|
287
|
-
|
|
288
|
-
| `persistSnapshots` | unset | Durable store for aggregate snapshots (`load` / `save` / optional `delete`) |
|
|
289
|
-
| `retainSnapshots` | `20` when snapshots persist | Snapshot files kept after prune (`persistABSnapshotsToDir` honors this) |
|
|
290
|
-
|
|
291
|
-
Implement your own `{ save }` / `{ load, save, delete? }` for S3, a DB, or
|
|
292
|
-
your metrics vendor. Without `persistSamples` / `persistSnapshots`,
|
|
293
|
-
samples only go where each experiment's `onSample` sends them, and
|
|
294
|
-
aggregates exist only as a live fold over session logs.
|
|
278
|
+
`maxPlaygroundSessions` keeps the newest sessions in the fold. It does
|
|
279
|
+
not prune session logs or change assignment. For export to S3, a DB, or
|
|
280
|
+
your metrics vendor, send samples from `onSample` or declare a storage
|
|
281
|
+
`abs` table.
|
|
295
282
|
|
|
296
283
|
## Keep assignments durable
|
|
297
284
|
|
|
@@ -301,8 +288,8 @@ metrics come from the turn and tool events that follow it.
|
|
|
301
288
|
|
|
302
289
|
After a server restart or a parked session resumes, the live collector
|
|
303
290
|
replays the stream to rebuild cumulative counters. Replay does not call
|
|
304
|
-
`onSample` (or `
|
|
305
|
-
completed or failed turn emits another sample.
|
|
291
|
+
`onSample` (or write to the storage `abs` table) for historical turns.
|
|
292
|
+
Only a new completed or failed turn emits another sample.
|
|
306
293
|
|
|
307
294
|
The snapshot API also replays `derive` across the full stream, so
|
|
308
295
|
custom totals match the current extractor. Changing a derive function
|
|
@@ -332,12 +319,12 @@ does not provide:
|
|
|
332
319
|
- Statistical significance calculations
|
|
333
320
|
- An experiment rollout or lifecycle service
|
|
334
321
|
- Per-variant model or runtime configuration
|
|
335
|
-
- A built-in analytics warehouse (bring your own via `
|
|
336
|
-
`
|
|
322
|
+
- A built-in analytics warehouse (bring your own via `onSample` or the
|
|
323
|
+
storage `abs` table)
|
|
337
324
|
|
|
338
|
-
Use [evals](./evals.md) to protect known behavior.
|
|
339
|
-
`
|
|
340
|
-
|
|
325
|
+
Use [evals](./evals.md) to protect known behavior. Use `onSample` or a
|
|
326
|
+
storage `abs` table when you need sample/snapshot exports beyond the
|
|
327
|
+
session event log.
|
|
341
328
|
|
|
342
329
|
## What's next
|
|
343
330
|
|
|
@@ -37,7 +37,7 @@ project, and verifies the result.
|
|
|
37
37
|
|
|
38
38
|
For example:
|
|
39
39
|
|
|
40
|
-
> Use the
|
|
40
|
+
> Use the Agent SDK create-agent skill to build a PR triage agent
|
|
41
41
|
> reachable through GitHub. It should summarize failed checks, require
|
|
42
42
|
> approval before posting a review, and include one smoke eval.
|
|
43
43
|
|
|
@@ -83,7 +83,7 @@ agent-sdk call inspect_pr --dir . \
|
|
|
83
83
|
agent-sdk run --dir . \
|
|
84
84
|
--message "Is https://github.com/acme/checkout/pull/42 ready to approve?"
|
|
85
85
|
|
|
86
|
-
agent-sdk trajectory --events .agent-
|
|
86
|
+
agent-sdk trajectory --events .agent-serve/traces/<sessionId>.ndjson
|
|
87
87
|
|
|
88
88
|
agent-sdk eval --dir . --list
|
|
89
89
|
agent-sdk eval --dir . --json
|
package/docs/concepts.md
CHANGED
|
@@ -110,7 +110,8 @@ Choose a runtime in `agent/agent.ts`:
|
|
|
110
110
|
| | Local (default) | Cloud |
|
|
111
111
|
| --- | --- | --- |
|
|
112
112
|
| Turn runs on | The server host | A Cursor cloud agent |
|
|
113
|
-
| Server tools
|
|
113
|
+
| Server tools | Supported | Supported when the server has `--public-url` or `--cloud-tools-url`; the cloud turn reaches them over authenticated HTTP MCP. Without one of those flags, the server warns and cloud turns omit them. |
|
|
114
|
+
| Approvals (`needsApproval`) | Supported | Not supported (local runtime only) |
|
|
114
115
|
| Agent tool scripts | Supported | Supported |
|
|
115
116
|
| Skills and seeded files | Added to the session workspace | Must exist in the cloud repository |
|
|
116
117
|
| Repository | You provide it | The cloud agent checks it out |
|
|
@@ -137,7 +138,7 @@ commands already use temporary state.
|
|
|
137
138
|
Durable local state uses this shape:
|
|
138
139
|
|
|
139
140
|
```text
|
|
140
|
-
<project>/.agent-
|
|
141
|
+
<project>/.agent-serve/
|
|
141
142
|
sessions/<id>/events.ndjson
|
|
142
143
|
sessions/<id>/workspace/
|
|
143
144
|
traces/<sessionId>.ndjson
|
package/docs/deployment.md
CHANGED
|
@@ -305,7 +305,7 @@ A self-hosted server can read these credentials.
|
|
|
305
305
|
Use a dedicated Cursor key per host. `agent-sdk whoami` shows the active
|
|
306
306
|
credential. `logout` removes the stored key from the host; revoke the key
|
|
307
307
|
in the Cursor dashboard to invalidate it. See
|
|
308
|
-
[CLI authentication](./reference/cli.md#login
|
|
308
|
+
[CLI authentication](./reference/cli.md#login-logout-whoami) for
|
|
309
309
|
credential resolution.
|
|
310
310
|
|
|
311
311
|
### State
|
package/docs/evals.md
CHANGED
|
@@ -18,7 +18,7 @@ accepted a message, and did what you asserted.
|
|
|
18
18
|
|
|
19
19
|
> [!NOTE]
|
|
20
20
|
> Import paths here use `@cursor/july/evals`. On projects still
|
|
21
|
-
> using `@
|
|
21
|
+
> using `@anysphere/agent-serve`, swap the import and run
|
|
22
22
|
> `agent-serve eval`. See
|
|
23
23
|
> [Run the CLI](./README.md#run-the-cli) for the full rename table.
|
|
24
24
|
|
|
@@ -118,35 +118,39 @@ issues real model-provider requests, so concurrency is capped hard at
|
|
|
118
118
|
without this file, but running a case does not.
|
|
119
119
|
|
|
120
120
|
```ts
|
|
121
|
-
import {
|
|
122
|
-
defineEvalConfig,
|
|
123
|
-
persistEvalRunsToDir,
|
|
124
|
-
} from "@cursor/july/evals";
|
|
121
|
+
import { defineEvalConfig } from "@cursor/july/evals";
|
|
125
122
|
|
|
126
123
|
export default defineEvalConfig({
|
|
127
124
|
maxConcurrency: 20, // required
|
|
128
125
|
// timeoutMs: 180_000, // optional project-wide default
|
|
126
|
+
// judge: { model: "..." }, // default judge model for t.judge.*
|
|
127
|
+
// reporters: [], // destinations that observe every case
|
|
129
128
|
// maxPlaygroundRuns: 50, // playground /v1/dev/evals history only (default 20)
|
|
130
|
-
//
|
|
131
|
-
// Playground batches default to **process memory only** — they disappear
|
|
132
|
-
// when `serve` exits. Opt into durable storage:
|
|
133
|
-
// persistRuns: persistEvalRunsToDir(".agent-serve/eval-runs"),
|
|
134
|
-
// Or implement { load, save, delete } yourself (S3, DB, …).
|
|
135
129
|
});
|
|
136
130
|
```
|
|
137
131
|
|
|
138
132
|
The timeout order is case or file `timeoutMs`, CLI `--timeout-ms`,
|
|
139
133
|
project config `timeoutMs`, then the 180-second runner default.
|
|
140
134
|
|
|
141
|
-
|
|
135
|
+
The optional fields:
|
|
142
136
|
|
|
143
137
|
| Option | Default | Meaning |
|
|
144
138
|
| --- | --- | --- |
|
|
139
|
+
| `timeoutMs` | `180_000` | Project-wide per-case timeout |
|
|
140
|
+
| `judge` | unset | Default judge model for `t.judge.*`; see [Judge free-form output](#judge-free-form-output) |
|
|
141
|
+
| `reporters` | unset | Destinations that observe every case; `--skip-report` suppresses them |
|
|
145
142
|
| `maxPlaygroundRuns` | `20` | Max batches in the playground / `/v1/dev/evals*` history (not CLI `eval`) |
|
|
146
|
-
| `persistRuns` | unset | Optional `{ load, save, delete }` so batches survive process restart (`delete` required for durable prune) |
|
|
147
143
|
|
|
148
|
-
|
|
149
|
-
|
|
144
|
+
Reporters come from `@cursor/july/evals/reporters`: `JUnit` writes a
|
|
145
|
+
JUnit XML file for CI, `Artifacts` writes per-case files, and
|
|
146
|
+
`combineReporters` merges several into one (`renderJUnitXml` renders
|
|
147
|
+
the XML for a custom destination). A file or case can add its own
|
|
148
|
+
`reporters` on top of the config list.
|
|
149
|
+
|
|
150
|
+
Playground batches live in process memory and disappear when `serve`
|
|
151
|
+
exits. Navigating away and back still works while the process is up.
|
|
152
|
+
To keep batches across restarts, declare an `evals` table in
|
|
153
|
+
`agent/storage.ts`; see [Storage](./storage.md#eval-and-ab-tables).
|
|
150
154
|
|
|
151
155
|
## Drive and assert with `t`
|
|
152
156
|
|
|
@@ -155,33 +159,71 @@ control flow, sending turns and asserting inline.
|
|
|
155
159
|
|
|
156
160
|
Drive the agent with `t.send(message, options?)`. It runs one turn and
|
|
157
161
|
waits for the session to park or fail. Multiple sends in one case share
|
|
158
|
-
the session, which is how you write multi-turn evals.
|
|
159
|
-
|
|
160
|
-
`
|
|
162
|
+
the session, which is how you write multi-turn evals.
|
|
163
|
+
|
|
164
|
+
Each `t.send` resolves to a turn result with `message`, `sessionId`,
|
|
165
|
+
`events`, `toolCalls`, `ok`, and `index`. The turn carries the same
|
|
166
|
+
assertion vocabulary as `t`, scoped to that turn, so you can grade an
|
|
167
|
+
intermediate turn before the next send overwrites `t.reply`.
|
|
168
|
+
`turn.expectOk()` throws when the turn failed, for later steps that
|
|
169
|
+
depend on it.
|
|
161
170
|
|
|
162
171
|
Read the full case state with `t.reply` (the last assistant text),
|
|
163
|
-
`t.events` (every captured session event across turns),
|
|
164
|
-
`t.sessionId`.
|
|
172
|
+
`t.events` (every captured session event across turns), `t.turns`
|
|
173
|
+
(settled turns, oldest first), and `t.sessionId`. `t.signal` aborts
|
|
174
|
+
when the case hits its timeout; pass it to your own async work.
|
|
165
175
|
|
|
166
176
|
Assert with the gates:
|
|
167
177
|
|
|
168
178
|
| Gate | Checks |
|
|
169
179
|
| --- | --- |
|
|
170
|
-
| `t.succeeded()` | the
|
|
171
|
-
| `t.
|
|
172
|
-
| `t.
|
|
173
|
-
| `t.
|
|
180
|
+
| `t.succeeded()` | the run did not fail and is not parked on an unanswered approval |
|
|
181
|
+
| `t.parked()` | the run cleanly parked on an unanswered approval request |
|
|
182
|
+
| `t.messageIncludes(token)` | the joined assistant text matches a string or `RegExp` |
|
|
183
|
+
| `t.calledTool(name, matcher?)` | a matching call to `name` happened |
|
|
184
|
+
| `t.notCalledTool(name)` | no request for `name`, in any lifecycle state |
|
|
185
|
+
| `t.loadedSkill(name)` | the agent opened the skill's `SKILL.md` (read, grep, or shell `cat`) |
|
|
186
|
+
| `t.toolOrder(names)` | tool requests appear in this relative order (extra calls allowed) |
|
|
187
|
+
| `t.usedNoTools()` | no tool calls at all |
|
|
188
|
+
| `t.maxToolCalls(max)` | at most `max` tool calls |
|
|
189
|
+
| `t.noFailedActions()` | no tool call reported an error |
|
|
190
|
+
| `t.calledSubagent(name, matcher?)` | a matching subagent delegation happened |
|
|
191
|
+
| `t.taggedArtifact(kind?, predicate?)` | at least one [artifact](./reference/artifacts.md) was tagged |
|
|
192
|
+
| `t.event(type, matcher?)` | at least one matching event of `type` occurred |
|
|
193
|
+
| `t.notEvent(type, matcher?)` | no matching event of `type` occurred |
|
|
194
|
+
| `t.eventOrder(matchers)` | matching event groups occur in this relative order |
|
|
195
|
+
| `t.eventsSatisfy(label, predicate)` | your predicate over the typed event stream |
|
|
174
196
|
| `t.check(value, expectation)` | any value, against a builder |
|
|
197
|
+
| `t.score(name, value)` | records a 0–1 score you computed; soft until you add a bar |
|
|
198
|
+
| `t.requireToolCall(name, matcher?)` | gates on a matching call and returns it, so later code can read its input and output |
|
|
199
|
+
| `t.requireInputRequest(filter?)` | gates on exactly one pending approval request and returns it |
|
|
200
|
+
|
|
201
|
+
Every gate returns a handle: `.soft()` demotes it to tracked-only,
|
|
202
|
+
`.atLeast(0.7)` adds a soft score bar, and `.gate(0.8)` promotes a
|
|
203
|
+
scored assertion into a hard gate.
|
|
175
204
|
|
|
176
|
-
`calledTool`
|
|
177
|
-
when its result has not arrived.
|
|
178
|
-
`t.
|
|
205
|
+
With no matcher, `calledTool` is request-based: a requested call counts
|
|
206
|
+
even when its result has not arrived. Pass
|
|
207
|
+
`t.calledTool("inspect_pr", { status: "completed" })` to require the
|
|
208
|
+
call to return. `input`, `output`, and `count` matcher fields accept a
|
|
209
|
+
literal, a `RegExp`, or a predicate.
|
|
179
210
|
|
|
180
211
|
The expectation builders are `includes(string | RegExp)`,
|
|
181
|
-
`equals(value)`,
|
|
182
|
-
|
|
183
|
-
|
|
184
|
-
|
|
212
|
+
`equals(value)`, `matches(schema)`, `similarity(expected)`, and
|
|
213
|
+
`satisfies(predicate, label)`. `includes` stringifies its input,
|
|
214
|
+
`equals` compares values deeply, `matches` validates against a Standard
|
|
215
|
+
Schema (or anything with `safeParse`, like Zod), `similarity` scores
|
|
216
|
+
normalized text similarity, and `satisfies` runs your predicate. The
|
|
217
|
+
plain function `normalizedSimilarity(actual, expected)` returns the
|
|
218
|
+
same 0–1 score for use with `t.score`.
|
|
219
|
+
|
|
220
|
+
A few more context members shape a case: `t.require(value, expectation)`
|
|
221
|
+
records a gate and stops the test body when it fails, without a
|
|
222
|
+
duplicate execution error. `t.skip(reason)` ends the case as skipped
|
|
223
|
+
(reported separately, never changes the exit code; call it before
|
|
224
|
+
sending messages). `t.metric(name, value)` records a structured score
|
|
225
|
+
for the playground case card. `t.log(message)` records a debug line for
|
|
226
|
+
the CLI and playground result.
|
|
185
227
|
|
|
186
228
|
Three `t.send` options apply on session create (first `t.send` only):
|
|
187
229
|
|
|
@@ -205,6 +247,26 @@ A case with no explicit gates falls back to whether at least one turn
|
|
|
205
247
|
completed successfully. Add `t.succeeded()` and behavior-specific gates
|
|
206
248
|
anyway. They make the contract visible during review.
|
|
207
249
|
|
|
250
|
+
### Judge free-form output
|
|
251
|
+
|
|
252
|
+
When wording matters and no regex captures it, `t.judge` grades the
|
|
253
|
+
reply with an LLM. The built-in graders are `factuality(expected)`,
|
|
254
|
+
`summarizes(expected)`, `closedQA(criteria)`, and `sql(expected)`. Each
|
|
255
|
+
scores `t.reply` by default; pass `{ on }` to grade another value.
|
|
256
|
+
|
|
257
|
+
```ts
|
|
258
|
+
t.judge.factuality("It is 54°F in NYC right now.").atLeast(0.7);
|
|
259
|
+
```
|
|
260
|
+
|
|
261
|
+
Judge assertions are soft by default, so a judge never fails a build
|
|
262
|
+
until you give it a bar with `.atLeast(0.7)` or promote it with
|
|
263
|
+
`.gate(0.8)`. The judge model comes from `defineEvalConfig({ judge })`,
|
|
264
|
+
`defineEval({ judge })`, a case-level `judge`, or a per-call
|
|
265
|
+
`{ model }` override; the nearest one wins. For a domain-specific judge
|
|
266
|
+
whose verdict is not a single score, `t.judge.model(prompt)` sends a
|
|
267
|
+
raw prompt to the same model and returns the reply. You then record the
|
|
268
|
+
parsed result with `t.score` or `t.check`.
|
|
269
|
+
|
|
208
270
|
## Run evals from the CLI
|
|
209
271
|
|
|
210
272
|
The `eval` command discovers, filters, and runs cases.
|
|
@@ -230,7 +292,7 @@ match both groups.
|
|
|
230
292
|
|
|
231
293
|
`eval` boots an ephemeral server on port 0 with a temp state root
|
|
232
294
|
outside the project, so cases don't inherit ambient monorepo rules and
|
|
233
|
-
don't pollute `.agent-
|
|
295
|
+
don't pollute `.agent-serve/`. Point `--url` at a running server to eval
|
|
234
296
|
a live agent instead:
|
|
235
297
|
|
|
236
298
|
```bash
|
|
@@ -299,8 +361,9 @@ agent-sdk serve --dir . --dev
|
|
|
299
361
|
Playground runs target the live server instead of an ephemeral one.
|
|
300
362
|
Their sessions appear in the session list. One eval batch can run at a
|
|
301
363
|
time. By default those batches are **in-memory only** (capped by
|
|
302
|
-
`maxPlaygroundRuns`);
|
|
303
|
-
after a serve restart — see
|
|
364
|
+
`maxPlaygroundRuns`); declare an `evals` table in `agent/storage.ts` if
|
|
365
|
+
you need them after a serve restart — see
|
|
366
|
+
[Storage](./storage.md#eval-and-ab-tables).
|
|
304
367
|
|
|
305
368
|
The UI uses the playground eval routes (available without `--dev`):
|
|
306
369
|
`GET /v1/dev/evals` lists datapoints and config (includes `maxPlaygroundRuns` /
|
|
@@ -378,6 +441,12 @@ once and commit the rendered fixture before you expand the suite.
|
|
|
378
441
|
3. Assert decisions and output shape against the saved evidence.
|
|
379
442
|
4. Keep a small `smoke` subset for any remaining live pipeline checks.
|
|
380
443
|
|
|
444
|
+
Read committed fixtures with `@cursor/july/evals/loaders`: `loadJson`,
|
|
445
|
+
`loadJsonl`, and `loadYaml` resolve relative paths against the project
|
|
446
|
+
root the runner discovered, not the cwd the CLI was invoked from
|
|
447
|
+
(`resolveFixturePath` and `evalFixtureRoot` expose the same
|
|
448
|
+
resolution for other file formats).
|
|
449
|
+
|
|
381
450
|
`maxConcurrency` limits parallel datapoints. It does not limit model or
|
|
382
451
|
API fan-out inside one datapoint. Materialized fixtures prevent a large
|
|
383
452
|
suite from exhausting provider and GitHub rate limits. The
|
|
@@ -70,10 +70,11 @@ approving the PR.
|
|
|
70
70
|
| Deterministic policy | [`agent/lib/approve.ts`](../../examples/approval-buddy/agent/lib/approve.ts), [`agent/lib/buddies.ts`](../../examples/approval-buddy/agent/lib/buddies.ts) | Own the roster and live eligibility checks. |
|
|
71
71
|
| Review subagents | [`agent/subagents/`](../../examples/approval-buddy/agent/subagents/) | Run deep audit and code-quality passes over the same evidence. |
|
|
72
72
|
| Storage | [`agent/storage.ts`](../../examples/approval-buddy/agent/storage.ts) | Persist sessions and events with `cursorHostedStorage` (Bugbot `agent_serve_*`). |
|
|
73
|
+
| Live A/B experiment | [`agent/ab.ts`](../../examples/approval-buddy/agent/ab.ts) | Compare baseline responses with a concise, presentation-only treatment (`concise-results`). |
|
|
73
74
|
| Evals and unit tests | [`evals/`](../../examples/approval-buddy/evals/), [`agent/lib/`](../../examples/approval-buddy/agent/lib/) | Protect routing, output contracts, policy, and GitHub behavior. |
|
|
74
75
|
|
|
75
76
|
There are no authored skills, MCP connections, schedules, reminders, hooks,
|
|
76
|
-
|
|
77
|
+
sandbox seeds, or tool approvals.
|
|
77
78
|
|
|
78
79
|
## Prepare credentials
|
|
79
80
|
|
|
@@ -52,6 +52,8 @@ need the watched-channel path.
|
|
|
52
52
|
| [`agent/instructions.md`](../../examples/benny/agent/instructions.md) | Defines engagement rules, evidence policy, and the playbook routing map. |
|
|
53
53
|
| [`agent/channels/slack.ts`](../../examples/benny/agent/channels/slack.ts) | Handles account-linked mentions and direct messages. |
|
|
54
54
|
| [`agent/channels/slack-app.ts`](../../examples/benny/agent/channels/slack-app.ts) | Runs the dedicated app and watches one allowlisted channel. |
|
|
55
|
+
| [`agent/storage.ts`](../../examples/benny/agent/storage.ts) | Persists sessions and events with `cursorHostedStorage`. |
|
|
56
|
+
| [`evals/evals.config.ts`](../../examples/benny/evals/evals.config.ts) | Caps eval run concurrency. |
|
|
55
57
|
| [`evals/smoke.eval.ts`](../../examples/benny/evals/smoke.eval.ts) | Checks the agent identity and expected triage route. |
|
|
56
58
|
|
|
57
59
|
The playbook router authors no tools, MCP connections, subagents, schedules, hooks, A/B
|
|
@@ -58,6 +58,9 @@ mid-turn. That form writes the same files into the active session workspace.
|
|
|
58
58
|
| [`agent/channels/review.ts`](../../examples/bugbot/agent/channels/review.ts) | Provides the loopback-only prepare-and-send HTTP route. |
|
|
59
59
|
| [`agent/channels/slack.ts`](../../examples/bugbot/agent/channels/slack.ts) | Extracts PR references and prepares evidence for mentions and direct messages. |
|
|
60
60
|
| [`agent/skills/pr-review.md`](../../examples/bugbot/agent/skills/pr-review.md) | Sets finding limits, severities, and the machine-readable review format. |
|
|
61
|
+
| [`agent/lib/log.ts`](../../examples/bugbot/agent/lib/log.ts) | Writes timing logs for the host tools to stderr. |
|
|
62
|
+
| [`agent/storage.ts`](../../examples/bugbot/agent/storage.ts) | Persists sessions and events with `cursorHostedStorage`. |
|
|
63
|
+
| [`evals/evals.config.ts`](../../examples/bugbot/evals/evals.config.ts) | Caps eval run concurrency. |
|
|
61
64
|
| [`evals/review/smoke.eval.ts`](../../examples/bugbot/evals/review/smoke.eval.ts) | Seeds fake evidence and checks the review path without GitHub. |
|
|
62
65
|
|
|
63
66
|
There is no authored GitHub channel, MCP connection, subagent, schedule,
|
|
@@ -67,6 +67,8 @@ tool, which writes the digest into the active session workspace.
|
|
|
67
67
|
| [`agent/skills/feature-mapping.md`](../../examples/codebase-wiki/agent/skills/feature-mapping.md) | Maps changes onto features and fixes the page and changelog shape. |
|
|
68
68
|
| [`agent/schedules/daily-digest.md`](../../examples/codebase-wiki/agent/schedules/daily-digest.md) | Writes `digests/<date>`, rebuilds the index, and flags stale pages. |
|
|
69
69
|
| [`agent/channels/github.ts`](../../examples/codebase-wiki/agent/channels/github.ts) | Acknowledges closed PRs and starts merged-only ingest turns. |
|
|
70
|
+
| [`agent/storage.ts`](../../examples/codebase-wiki/agent/storage.ts) | Persists sessions and events with `cursorHostedStorage`. |
|
|
71
|
+
| [`evals/evals.config.ts`](../../examples/codebase-wiki/evals/evals.config.ts) | Caps eval run concurrency. |
|
|
70
72
|
| [`evals/ingest.eval.ts`](../../examples/codebase-wiki/evals/ingest.eval.ts) | Gates ingest decisions against the wiki filesystem. |
|
|
71
73
|
|
|
72
74
|
There is no MCP connection, subagent, hook, A/B experiment, or custom
|
|
@@ -74,6 +74,8 @@ APPROVE and commit statuses on top of the same shape.
|
|
|
74
74
|
| [`agent/subagents/area-reviewer/`](../../examples/codeowners-review/agent/subagents/area-reviewer/) | Defines the one-area, one-playbook reviewer contract. |
|
|
75
75
|
| [`agent/channels/github.ts`](../../examples/codeowners-review/agent/channels/github.ts) | Reviews opened, reopened, synchronized, and undrafted PRs. |
|
|
76
76
|
| [`fixtures/`](../../examples/codeowners-review/fixtures/) | Ships two reviewable PRs with known planted findings. |
|
|
77
|
+
| [`agent/storage.ts`](../../examples/codeowners-review/agent/storage.ts) | Persists sessions and events with `cursorHostedStorage`. |
|
|
78
|
+
| [`evals/evals.config.ts`](../../examples/codeowners-review/evals/evals.config.ts) | Caps eval run concurrency. |
|
|
77
79
|
| [`evals/review.eval.ts`](../../examples/codeowners-review/evals/review.eval.ts) | Gates routing, fan-out, planted bugs, and verdicts. |
|
|
78
80
|
|
|
79
81
|
There is no MCP connection, schedule, hook, A/B experiment, or custom
|
|
@@ -68,6 +68,7 @@ share Concierge's conversation history.
|
|
|
68
68
|
| [`agent/agent.ts`](../../examples/concierge/agent/agent.ts) | Describes the root agent and selects the local runtime. |
|
|
69
69
|
| [`agent/instructions.md`](../../examples/concierge/agent/instructions.md) | Draws a strict weather-only delegation boundary. |
|
|
70
70
|
| [`agent/mcp-connections/weather.ts`](../../examples/concierge/agent/mcp-connections/weather.ts) | Resolves the peer by its `weather-agent` slug. |
|
|
71
|
+
| [`agent/storage.ts`](../../examples/concierge/agent/storage.ts) | Persists sessions and events with `cursorHostedStorage`. |
|
|
71
72
|
|
|
72
73
|
Concierge doesn't author channels, tools, skills, subagents, schedules,
|
|
73
74
|
hooks, A/B experiments, or evals. The built-in HTTP and MCP surfaces still
|
|
@@ -74,6 +74,7 @@ has `status: "finished"` and no remote session.
|
|
|
74
74
|
| Affinity and buffering | [`agent/lib/pr-affinity.ts`](../../examples/fsd/agent/lib/pr-affinity.ts), [`agent/lib/webhook-buffer.ts`](../../examples/fsd/agent/lib/webhook-buffer.ts) | Persist PR identity, sticky mode, and pending wakes. |
|
|
75
75
|
| Reminders | [`agent/lib/merge-conflict-watch.ts`](../../examples/fsd/agent/lib/merge-conflict-watch.ts) | Recheck merge conflicts every 30 minutes. |
|
|
76
76
|
| Workflow client | [`agent/lib/fsd-platform.ts`](../../examples/fsd/agent/lib/fsd-platform.ts) | Enroll external runs and read or record findings. |
|
|
77
|
+
| Storage | [`agent/storage.ts`](../../examples/fsd/agent/storage.ts) | Persist sessions and events with `cursorHostedStorage`. |
|
|
77
78
|
|
|
78
79
|
The coordinator has no authored skill, subagent, MCP connection, static
|
|
79
80
|
schedule, A/B experiment, eval, custom storage definition, or tool approval.
|
|
@@ -65,6 +65,8 @@ one-off questions and to ask before saving anything borderline.
|
|
|
65
65
|
| [`agent/skills/wiki-conventions.md`](../../examples/knowledge-base/agent/skills/wiki-conventions.md) | Names pages, shapes them, and dates every fact. |
|
|
66
66
|
| [`agent/schedules/gardener.md`](../../examples/knowledge-base/agent/schedules/gardener.md) | Merges duplicates, rebuilds the index, and flags stale facts daily. |
|
|
67
67
|
| [`agent/lib/wiki-store.test.ts`](../../examples/knowledge-base/agent/lib/wiki-store.test.ts) | Unit-tests slug safety and store round-trips. |
|
|
68
|
+
| [`agent/storage.ts`](../../examples/knowledge-base/agent/storage.ts) | Persists sessions and events with `cursorHostedStorage`. |
|
|
69
|
+
| [`evals/evals.config.ts`](../../examples/knowledge-base/evals/evals.config.ts) | Caps eval run concurrency. |
|
|
68
70
|
| [`evals/knowledge.eval.ts`](../../examples/knowledge-base/evals/knowledge.eval.ts) | Seeds a temp knowledge base and gates recall, save, and no-write decisions. |
|
|
69
71
|
|
|
70
72
|
There is no authored channel, MCP connection, subagent, hook, A/B
|
|
@@ -51,6 +51,8 @@ Mentions and DMs skip the watch entirely and behave like ordinary chat.
|
|
|
51
51
|
| [`agent/lib/slack-api.ts`](../../examples/oncall/agent/lib/slack-api.ts) | Reactions and thread posts on this agent's own token pair. |
|
|
52
52
|
| [`agent/tools/reminders_create.ts`](../../examples/oncall/agent/tools/reminders_create.ts) | Self-scheduled wakes bound to the thread (plus `reminders_list` and `reminders_cancel`). |
|
|
53
53
|
| [`agent/tools/post_thread_update.ts`](../../examples/oncall/agent/tools/post_thread_update.ts) | Interim updates to the thread mid-turn. |
|
|
54
|
+
| [`agent/storage.ts`](../../examples/oncall/agent/storage.ts) | Persists sessions and events with `cursorHostedStorage`. |
|
|
55
|
+
| [`evals/evals.config.ts`](../../examples/oncall/evals/evals.config.ts) | Caps eval run concurrency. |
|
|
54
56
|
| [`evals/smoke.eval.ts`](../../examples/oncall/evals/smoke.eval.ts) | Checks identity and the reminder-tool route. |
|
|
55
57
|
|
|
56
58
|
## Let bot posts through the watch
|
|
@@ -49,6 +49,7 @@ or tool routing.
|
|
|
49
49
|
| [`agent/agent.ts`](../../examples/slack-agent/agent/agent.ts) | Names the agent and selects the model. The omitted `runtime` defaults to local. |
|
|
50
50
|
| [`agent/instructions.md`](../../examples/slack-agent/agent/instructions.md) | Sets the always-on response style. |
|
|
51
51
|
| [`agent/channels/slack.ts`](../../examples/slack-agent/agent/channels/slack.ts) | Connects the signed-in host account to Slack. |
|
|
52
|
+
| [`agent/storage.ts`](../../examples/slack-agent/agent/storage.ts) | Persists sessions and events with `cursorHostedStorage`. |
|
|
52
53
|
|
|
53
54
|
There are no authored tools, skills, MCP connections, subagents, schedules,
|
|
54
55
|
hooks, A/B experiments, or evals. This small surface is the lesson.
|
|
@@ -141,10 +141,15 @@ first prompt; the cloud model writes and invokes the script in its VM.
|
|
|
141
141
|
|
|
142
142
|
## Verify cloud custom-tool execution
|
|
143
143
|
|
|
144
|
-
Ask the agent to run `probe_cloud_tool`.
|
|
145
|
-
|
|
146
|
-
|
|
147
|
-
|
|
144
|
+
Ask the agent to run `probe_cloud_tool`. On cloud that tool is MCP on
|
|
145
|
+
`agentsdk-tools`, not a local script. A real call writes a deployment-scoped
|
|
146
|
+
marker at `tool-observations/<sessionId>/<toolCallId>.json` and returns those
|
|
147
|
+
ids. Correlate them with `actions.requested` / `action.result` for
|
|
148
|
+
`probe_cloud_tool` (not `shell`) and the host's
|
|
149
|
+
`cloud HTTP MCP tool ... dispatch/complete` log lines.
|
|
150
|
+
|
|
151
|
+
A local `.agent-serve/tools/probe_cloud_tool.sh` or a marker under `probes/`
|
|
152
|
+
means the model invented a substitute and the host never ran.
|
|
148
153
|
|
|
149
154
|
For a body-only smoke test:
|
|
150
155
|
|
|
@@ -62,17 +62,24 @@ mapping shifts:
|
|
|
62
62
|
| Folder or file | Local runtime | Cloud runtime |
|
|
63
63
|
| --- | --- | --- |
|
|
64
64
|
| `instructions.*` | `AGENTS.md` in the session workspace | prepended to the first prompt |
|
|
65
|
-
| Server tools (`execution: "server"`) | in-process SDK custom tools | authenticated HTTP MCP back to the AgentSDK host |
|
|
65
|
+
| Server tools (`execution: "server"`) | in-process SDK custom tools | authenticated HTTP MCP back to the AgentSDK host, when `--public-url` or `--cloud-tools-url` is set |
|
|
66
66
|
| Agent tools (`execution: "agent"`) | scripts in the session workspace | catalog + script bodies on the first prompt |
|
|
67
67
|
| `skills/*` | `.cursor/skills/` in the workspace | only if present in the cloud repo |
|
|
68
68
|
| `mcp-connections/*.ts` | SDK `mcpServers` | SDK `mcpServers` (peers need `--public-url`) |
|
|
69
69
|
| `sandbox/workspace/**` | seeded into the session workspace | ignored |
|
|
70
|
-
| Tool approvals (`needsApproval`) | supported | supported
|
|
70
|
+
| Tool approvals (`needsApproval`) | supported | not supported; keep approval-gated tools on local turns |
|
|
71
71
|
|
|
72
|
-
Hosted deployments configure the server-tool MCP URL automatically
|
|
72
|
+
Hosted deployments configure the server-tool MCP URL automatically
|
|
73
|
+
(`cloudToolsUrl`, authenticated with the resolved Cursor API key). A
|
|
73
74
|
self-hosted public server needs `--public-url` (and `--bearer-token` when the
|
|
74
75
|
host is not behind another trusted authentication boundary) so cloud turns
|
|
75
|
-
can reach those tools.
|
|
76
|
+
can reach those tools. Without either, the server warns at startup and
|
|
77
|
+
cloud turns omit the server tools.
|
|
78
|
+
|
|
79
|
+
Approvals are a local-runtime contract. On cloud, a `needsApproval` tool
|
|
80
|
+
call rides one HTTP MCP request from the VM, and a parked call would
|
|
81
|
+
hold that request open until it times out; there is no durable approval
|
|
82
|
+
flow for cloud turns.
|
|
76
83
|
|
|
77
84
|
Two more behaviors are cloud-specific. Sessions persist a separate SDK
|
|
78
85
|
agent id (`bc-…`), emitted on the stream as `agent.bound` with a URL to
|