@andreprado/agentkit 0.1.0-alpha.2 → 0.1.0-alpha.21

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (142) hide show
  1. package/README.md +68 -6
  2. package/docs/guides/add-channel.md +189 -7
  3. package/docs/guides/add-knowledge.md +144 -0
  4. package/docs/guides/add-managed-composio.md +163 -0
  5. package/docs/guides/add-tool.md +1 -1
  6. package/docs/guides/channel-security.md +128 -32
  7. package/docs/guides/connect-discord.md +178 -0
  8. package/docs/guides/connect-slack.md +126 -0
  9. package/docs/guides/connect-telegram.md +78 -1
  10. package/docs/guides/connect-whatsapp-evolution.md +121 -0
  11. package/docs/guides/connect-whatsapp-uazapi.md +126 -0
  12. package/docs/guides/connect-whatsapp-zapster.md +112 -8
  13. package/docs/guides/create-agent.md +45 -4
  14. package/docs/guides/debug-channel.md +147 -0
  15. package/docs/guides/improve-from-production.md +151 -0
  16. package/docs/guides/prepare-deploy.md +47 -17
  17. package/docs/guides/replay-production-traces.md +72 -0
  18. package/docs/guides/run-evals.md +147 -20
  19. package/docs/guides/security-rules.md +7 -6
  20. package/docs/guides/send-feedback.md +135 -0
  21. package/docs/guides/use-provider.md +27 -3
  22. package/docs/llms-full.txt +348 -55
  23. package/docs/llms.txt +62 -7
  24. package/package.json +2 -5
  25. package/src/cli/args.ts +57 -0
  26. package/src/cli/cloud-client.ts +377 -0
  27. package/src/cli/commands/channels.ts +1586 -0
  28. package/src/cli/commands/feedback.ts +438 -0
  29. package/src/cli/commands/knowledge.ts +136 -0
  30. package/src/cli/commands/transcribe.ts +171 -0
  31. package/src/cli/constants.ts +4 -0
  32. package/src/cli/deploy-chat-ui.ts +535 -0
  33. package/src/cli/deploy-readiness.ts +481 -0
  34. package/src/cli/flags.ts +162 -0
  35. package/src/cli/help.ts +236 -0
  36. package/src/cli/index.ts +1167 -1005
  37. package/src/cli/process.ts +31 -0
  38. package/src/cloud/artifact.ts +139 -0
  39. package/src/cloud/client.ts +80 -0
  40. package/src/cloud/contracts.ts +63 -0
  41. package/src/cloud/index.ts +3 -0
  42. package/src/create-project.ts +21 -6
  43. package/src/index.ts +517 -8
  44. package/src/providers/pi.ts +70 -16
  45. package/src/providers/test.ts +88 -1
  46. package/src/providers/types.ts +7 -0
  47. package/src/runtime/channel-buffer.ts +30 -0
  48. package/src/runtime/channel-test-harness.ts +21 -1
  49. package/src/runtime/channels/discord.ts +896 -0
  50. package/src/runtime/channels/generic-webhook.ts +225 -0
  51. package/src/runtime/channels/slack.ts +646 -0
  52. package/src/runtime/channels/telegram.ts +466 -23
  53. package/src/runtime/channels/whatsapp-evolution.ts +1357 -0
  54. package/src/runtime/channels/whatsapp-meta.ts +9 -0
  55. package/src/runtime/channels/whatsapp-uazapi.ts +1327 -0
  56. package/src/runtime/channels/whatsapp-zapster.ts +677 -40
  57. package/src/runtime/channels.ts +87 -4
  58. package/src/runtime/chat.ts +130 -38
  59. package/src/runtime/config.ts +519 -19
  60. package/src/runtime/core/manifest.ts +103 -5
  61. package/src/runtime/core/targets.ts +5 -5
  62. package/src/runtime/database.ts +93 -2
  63. package/src/runtime/db-commands.ts +9 -0
  64. package/src/runtime/deploy-readiness.ts +46 -4
  65. package/src/runtime/deploy.ts +1 -1
  66. package/src/runtime/dev-server.ts +779 -45
  67. package/src/runtime/env.ts +8 -3
  68. package/src/runtime/evals.ts +589 -43
  69. package/src/runtime/improve.ts +868 -0
  70. package/src/runtime/inspect.ts +194 -4
  71. package/src/runtime/integrations/composio.ts +423 -0
  72. package/src/runtime/knowledge/chunk.ts +333 -0
  73. package/src/runtime/knowledge/config.ts +135 -0
  74. package/src/runtime/knowledge/embeddings.ts +133 -0
  75. package/src/runtime/knowledge/ingest.ts +521 -0
  76. package/src/runtime/knowledge/prompt-policy.ts +30 -0
  77. package/src/runtime/knowledge/retrieve.ts +303 -0
  78. package/src/runtime/knowledge/schema.ts +100 -0
  79. package/src/runtime/knowledge/tool.ts +64 -0
  80. package/src/runtime/knowledge/vector.ts +258 -0
  81. package/src/runtime/prompt-context.ts +141 -0
  82. package/src/runtime/runtime-contract.ts +86 -8
  83. package/src/runtime/skills.ts +95 -0
  84. package/src/runtime/spec.ts +152 -0
  85. package/src/runtime/sync.ts +144 -0
  86. package/src/runtime/targets/cloudflare/build.ts +1468 -203
  87. package/src/runtime/targets/container/server.ts +1 -1
  88. package/src/runtime/targets/vps/deploy.ts +26 -9
  89. package/src/runtime/tool-runner.ts +9 -1
  90. package/src/runtime/tools.ts +128 -2
  91. package/src/runtime/traces.ts +41 -0
  92. package/src/runtime/transcription.ts +483 -0
  93. package/src/storage/sqlite.ts +149 -3
  94. package/src/templates/blank.ts +76 -17
  95. package/src/templates/dentista.ts +1011 -0
  96. package/src/templates/index.ts +2 -0
  97. package/src/templates/skills/agentkit-build-agent/SKILL.md +52 -0
  98. package/src/templates/skills/agentkit-build-agent/templates/appointment-intake.instructions.md +21 -0
  99. package/src/templates/skills/agentkit-build-agent/templates/sales-qualifier.instructions.md +17 -0
  100. package/src/templates/skills/agentkit-build-agent/templates/support-agent.instructions.md +16 -0
  101. package/src/templates/skills/agentkit-capsule/SKILL.md +70 -0
  102. package/src/templates/skills/agentkit-capsule/references/docs-router.md +15 -0
  103. package/src/templates/skills/agentkit-channels/SKILL.md +127 -0
  104. package/src/templates/skills/agentkit-channels/references/channel-buffering.md +65 -0
  105. package/src/templates/skills/agentkit-channels/references/channel-debugging.md +66 -0
  106. package/src/templates/skills/agentkit-channels/references/discord.md +93 -0
  107. package/src/templates/skills/agentkit-channels/references/slack.md +56 -0
  108. package/src/templates/skills/agentkit-channels/references/telegram.md +72 -0
  109. package/src/templates/skills/agentkit-channels/references/whatsapp-evolution.md +57 -0
  110. package/src/templates/skills/agentkit-channels/references/whatsapp-uazapi.md +61 -0
  111. package/src/templates/skills/agentkit-channels/references/whatsapp-zapster.md +77 -0
  112. package/src/templates/skills/agentkit-database/SKILL.md +45 -0
  113. package/src/templates/skills/agentkit-database/templates/appointments.schema.sql +15 -0
  114. package/src/templates/skills/agentkit-database/templates/leads.schema.sql +17 -0
  115. package/src/templates/skills/agentkit-deploy/SKILL.md +50 -0
  116. package/src/templates/skills/agentkit-evals/SKILL.md +109 -0
  117. package/src/templates/skills/agentkit-evals/templates/multi-turn.eval.md +29 -0
  118. package/src/templates/skills/agentkit-evals/templates/no-leak.eval.md +18 -0
  119. package/src/templates/skills/agentkit-evals/templates/smoke.eval.md +18 -0
  120. package/src/templates/skills/agentkit-evals/templates/tool-call.eval.md +27 -0
  121. package/src/templates/skills/agentkit-improve/SKILL.md +86 -0
  122. package/src/templates/skills/agentkit-improve/references/replay-side-effects.md +18 -0
  123. package/src/templates/skills/agentkit-improve/references/trace-packets.md +22 -0
  124. package/src/templates/skills/agentkit-improve/templates/regression.eval.md +18 -0
  125. package/src/templates/skills/agentkit-integrations/SKILL.md +76 -0
  126. package/src/templates/skills/agentkit-knowledge/SKILL.md +43 -0
  127. package/src/templates/skills/agentkit-knowledge/templates/faq.md +14 -0
  128. package/src/templates/skills/agentkit-knowledge/templates/policies.md +14 -0
  129. package/src/templates/skills/agentkit-knowledge/templates/prices.csv +3 -0
  130. package/src/templates/skills/agentkit-prompts/SKILL.md +47 -0
  131. package/src/templates/skills/agentkit-prompts/templates/knowledge-grounded-faq.instructions.md +11 -0
  132. package/src/templates/skills/agentkit-provider/SKILL.md +60 -0
  133. package/src/templates/skills/agentkit-security/SKILL.md +56 -0
  134. package/src/templates/skills/agentkit-tools/SKILL.md +37 -0
  135. package/src/templates/skills/agentkit-tools/examples/database-write.tool.md +35 -0
  136. package/src/templates/skills/agentkit-tools/examples/eval-safe-external-action.tool.md +37 -0
  137. package/src/templates/skills/agentkit-tools/examples/lookup-order.tool.md +46 -0
  138. package/src/templates/skills/agentkit-troubleshooting/SKILL.md +76 -0
  139. package/src/templates/support.ts +77 -18
  140. package/docs/guides/channels-production-handoff.md +0 -99
  141. package/docs/portable-deploy-release-checklist.md +0 -41
  142. package/src/runtime/targets/cloudflare/deploy.ts +0 -5475
@@ -0,0 +1,151 @@
1
+ # Improve From Production
2
+
3
+ ## Goal
4
+
5
+ Pull hosted or local conversation evidence into the Agent Capsule, turn it into regression evals, let the local coding agent patch the capsule, and replay before deploying again.
6
+
7
+ ## When To Use This
8
+
9
+ Use this when a deployed agent gave a wrong answer, failed a tool call, mishandled a channel message, or needs production behavior converted into eval coverage.
10
+
11
+ AgentKit Cloud only exports redacted evidence. The local coding agent owns source edits, evals, replay, and deploy.
12
+
13
+ ## Commands
14
+
15
+ Collect evidence from the last hosted deploy:
16
+
17
+ ```sh
18
+ agentkit improve collect --deploy --since 24h
19
+ ```
20
+
21
+ When the CLI is logged in to AgentKit Cloud, this command first asks the control plane for deploy evidence such as failed channel deliveries, deploy errors, and conversation IDs that need review. It then reads replayable hosted conversation traces from the deployed runtime with the deploy access token. If Cloud auth is not available, it still collects hosted conversations directly from the deploy URL.
22
+
23
+ Hosted conversation reads require a deploy access token even when the chat endpoint is public. `agentkit deploy` normally writes `.agentkit/chat-access-token.json`; refresh it with:
24
+
25
+ ```sh
26
+ agentkit access token create agentkit-chat-ui --out .agentkit/chat-access-token.json
27
+ ```
28
+
29
+ Collect one hosted conversation:
30
+
31
+ ```sh
32
+ agentkit improve collect --deploy --conversation-id <conversation-id>
33
+ ```
34
+
35
+ Collect local conversations instead:
36
+
37
+ ```sh
38
+ agentkit improve collect --since 7d
39
+ ```
40
+
41
+ Generate regression evals from the collected bundle:
42
+
43
+ ```sh
44
+ agentkit improve evals .agentkit/improve/<run>
45
+ ```
46
+
47
+ Replay the bundle against the local capsule:
48
+
49
+ ```sh
50
+ agentkit replay .agentkit/improve/<run> --against local
51
+ ```
52
+
53
+ ## Files Created Or Edited
54
+
55
+ Evidence bundle, ignored local state:
56
+
57
+ ```txt
58
+ .agentkit/improve/<run>/
59
+ bundle.json
60
+ report.json
61
+ traces/
62
+ ```
63
+
64
+ Generated regression evals, committed source:
65
+
66
+ ```txt
67
+ evals/regressions/
68
+ improve-<conversation>.eval.ts
69
+ ```
70
+
71
+ The local coding agent may then edit:
72
+
73
+ ```txt
74
+ prompts/instructions.md
75
+ agentkit.config.ts
76
+ tools/
77
+ knowledge/
78
+ evals/
79
+ ```
80
+
81
+ Do not edit `.agentkit/improve/<run>/bundle.json` by hand.
82
+
83
+ ## Workflow
84
+
85
+ ```sh
86
+ agentkit improve collect --deploy --since 24h
87
+ agentkit improve evals .agentkit/improve/<run>
88
+ agentkit replay .agentkit/improve/<run> --against local
89
+ ```
90
+
91
+ Then let the local coding agent inspect `report.json`, the generated eval files, prompts, tools, and Knowledge sources. After edits:
92
+
93
+ ```sh
94
+ npm run typecheck
95
+ npm run agentkit -- inspect
96
+ npm run eval
97
+ agentkit replay .agentkit/improve/<run> --against local
98
+ agentkit deploy --smoke "hello"
99
+ ```
100
+
101
+ ## Safety Rules
102
+
103
+ - Do not paste secrets into evals, prompts, Knowledge files, or reports.
104
+ - Keep `.agentkit/improve/` out of commits.
105
+ - Review generated eval assertions before committing them. AgentKit redacts common email, phone, bearer token, and key patterns in generated eval text, but the local coding agent must still remove or generalize domain-specific client PII.
106
+ - For write, delete, payment, email, or customer-system tools, branch inside the tool on `ctx.runtime.environment === "eval"` and return deterministic non-destructive output.
107
+ - Treat hosted traces as customer evidence.
108
+ - If replay uses a real provider instead of `test/fake`, tell the owner because it may cost money and may be nondeterministic.
109
+
110
+ ## Verification
111
+
112
+ ```sh
113
+ npm run typecheck
114
+ npm run agentkit -- inspect
115
+ npm run eval
116
+ agentkit replay .agentkit/improve/<run> --against local
117
+ ```
118
+
119
+ Expected:
120
+
121
+ - `improve collect` writes a bundle and report under `.agentkit/improve/`.
122
+ - Hosted collection includes a redacted `evidence` summary in `bundle.json` and `report.json` when AgentKit Cloud evidence export is available.
123
+ - `improve evals` writes eval files under `evals/regressions/`.
124
+ - `replay` reports passed, failed, and skipped traces.
125
+ - No production secret values appear in generated files.
126
+
127
+ ## Troubleshooting
128
+
129
+ `No .agentkit/deploy.json found`:
130
+
131
+ Run `agentkit deploy` first, or collect local evidence without `--deploy`.
132
+
133
+ `deploy_conversation_request_failed`:
134
+
135
+ Refresh the deploy chat token with `agentkit access token create agentkit-chat-ui --out .agentkit/chat-access-token.json`, then retry.
136
+
137
+ `improve_evidence_store_not_configured`:
138
+
139
+ The Cloud API does not expose deploy evidence export yet. The CLI falls back to hosted conversation trace collection when possible.
140
+
141
+ `conversation_access_not_configured`:
142
+
143
+ The hosted runtime is not configured with a deploy access-token gate for conversation reads. Redeploy through AgentKit Cloud so the runtime gate is injected, or configure an explicit deploy/private token for local hosted testing.
144
+
145
+ `trace has no user turns`:
146
+
147
+ The trace cannot become a useful conversation eval. Keep the report for diagnosis, but do not commit an empty eval.
148
+
149
+ Generated eval is too strict:
150
+
151
+ Edit the eval to assert the important behavior, such as tool input, safety wording, or knowledge source usage, instead of exact prose.
@@ -19,20 +19,31 @@ Use this before asking a coding agent to make a capsule deployable, or before te
19
19
  From the capsule root:
20
20
 
21
21
  ```sh
22
- npm install
23
22
  npm run typecheck
24
23
  npm run agentkit -- inspect
25
24
  npm run agentkit -- db migrate
26
25
  npm run chat -- --message "hello"
27
26
  ```
28
27
 
29
- All scaffold, local dev, chat, eval, inspect, database, and build commands are token-free. Only hosted production deploy and hosted control-plane mutations require an invited AgentKit Cloud token:
28
+ All scaffold, local dev, chat, eval, inspect, database, and build commands are token-free. Hosted production deploy and hosted AgentKit Cloud changes require an AgentKit Cloud account token with hosted deploy access.
29
+
30
+ If the user does not have a token yet, start checkout from the CLI, finish Stripe Checkout in the browser, then claim the one-time checkout intent:
31
+
32
+ ```sh
33
+ npm run agentkit -- billing checkout --slots 1 --email user@example.com
34
+ npm run agentkit -- billing claim billint_... --secret bsec_...
35
+ ```
36
+
37
+ The claim command stores the returned `agk_user_...` token in the local AgentKit Cloud auth file. Treat the `bsec_...` checkout secret like a password; it exists only to claim the first token after checkout.
38
+
39
+ If the user already has a token:
30
40
 
31
41
  ```sh
32
42
  npm run agentkit -- login --token agk_user_...
33
43
  npm run agentkit -- deploy doctor
34
- npm run agentkit -- secret set OPENAI_API_KEY sk-...
35
- npm run agentkit -- deploy
44
+ npm run agentkit -- secret set OPENAI_API_KEY --from-local-env
45
+ npm run agentkit -- deploy --smoke "hello"
46
+ npm run agentkit -- chat-ui --deploy
36
47
  ```
37
48
 
38
49
  If the capsule uses OpenAI locally, put the user’s local key in `.env`:
@@ -42,7 +53,15 @@ OPENAI_API_KEY=sk-...
42
53
  ```
43
54
 
44
55
  `.env` is local-only. Hosted deploy secrets are handled by AgentKit outside the capsule.
45
- Closed-alpha hosted deploys require an invited account token with `cloudflare_deploy_alpha`.
56
+ Hosted deploys require an account with either `cloudflare_deploy_alpha` or purchased/manual deploy slots. Deploy slots belong to the account, not to a specific token string.
57
+
58
+ To generate or switch account tokens after login:
59
+
60
+ ```sh
61
+ npm run agentkit -- account token create new-laptop --use
62
+ npm run agentkit -- account token list
63
+ npm run agentkit -- account token revoke apitok_...
64
+ ```
46
65
 
47
66
  ## What The Agent Should Edit
48
67
 
@@ -126,12 +145,19 @@ For a final end-to-end test, run:
126
145
 
127
146
  ```sh
128
147
  npm run agentkit -- login --token agk_user_...
148
+ npm run agentkit -- account token list
149
+ npm run agentkit -- secret sync --from-local
129
150
  npm run agentkit -- secret list
130
- npm run agentkit -- deploy
151
+ npm run agentkit -- deploy --smoke "hello"
131
152
  npm run agentkit -- deploy status
153
+ npm run agentkit -- chat-ui --deploy
132
154
  npm run agentkit -- access token list
133
155
  ```
134
156
 
157
+ For hosted deploys, `npm run agentkit -- deploy` writes the local chat/UI access token to `.agentkit/chat-access-token.json`. Use `npm run agentkit -- deploy --smoke "hello"` for the official hosted chat smoke, and use `npm run agentkit -- chat-ui --deploy` for hosted UI testing. The hosted Chat UI shows the conversation id, tool calls, tool errors, and a new-conversation control; use `npm run agentkit -- conversations trace <conversation-id> --deploy` to pull the hosted trace from the last deploy. Use `npm run agentkit -- access token create <name> --out <path>` only for additional clients.
158
+
159
+ When the UI is running, open the printed `Chat:` URL and tell the owner the exact URL. If the capsule is still on `test/fake`, say the UI was tested only with the deterministic fake provider.
160
+
135
161
  ## Readiness Checklist
136
162
 
137
163
  ```txt
@@ -141,6 +167,7 @@ All required secret names are declared
141
167
  Prompt path exists
142
168
  Tools have input schemas
143
169
  Dangerous tools have permissions
170
+ Managed Composio toolkit auth configs are available in AgentKit Cloud when configured
144
171
  Provider model is supported by AgentKit
145
172
  schema.sql is idempotent
146
173
  README/AGENTS/CLAUDE match the capsule
@@ -166,10 +193,16 @@ echo 'OPENAI_API_KEY=sk-...' >> .env
166
193
 
167
194
  Missing hosted secret:
168
195
 
169
- Set the secret in AgentKit Cloud or the configured control plane:
196
+ Set the secret in AgentKit Cloud:
197
+
198
+ ```sh
199
+ npm run agentkit -- secret set OPENAI_API_KEY --from-local-env
200
+ ```
201
+
202
+ To sync all declared user-managed secrets present in local `.env`:
170
203
 
171
204
  ```sh
172
- npm run agentkit -- secret set OPENAI_API_KEY sk-...
205
+ npm run agentkit -- secret sync --from-local
173
206
  ```
174
207
 
175
208
  Do not write hosted secret values into the capsule.
@@ -180,18 +213,15 @@ Before a hosted production deploy, run:
180
213
  npm run agentkit -- deploy doctor
181
214
  ```
182
215
 
183
- The doctor checks AgentKit Cloud login, alpha deploy entitlement, declared hosted secrets, local `.env` names that have not been uploaded with `agentkit secret set`, and private-access runtime token handling. `agentkit deploy` runs the same readiness check automatically before building and uploading the artifact.
216
+ The doctor checks AgentKit Cloud login, Node version, hosted deploy entitlement, online deploy capacity, declared hosted secrets, local `.env` names that have not been uploaded with `npm run agentkit -- secret set`, and private-access runtime token handling. `agentkit deploy` runs the same readiness check automatically before building and uploading the artifact.
217
+
218
+ If the capsule configures `composioManaged({...})`, the doctor also checks the paid `managed_composio` entitlement and AgentKit Cloud toolkit readiness. `COMPOSIO_API_KEY` and toolkit auth config resolution are AgentKit-managed in this path; the user must not set them with `agentkit secret set`.
184
219
 
185
220
  `alpha_access_required`:
186
221
 
187
- Log in with an invited alpha account:
222
+ Log in with an account that has hosted deploy access. If the user needs to buy slots first:
188
223
 
189
224
  ```sh
190
- npm run agentkit -- login --token agk_user_...
225
+ npm run agentkit -- billing checkout --slots 1 --email user@example.com
226
+ npm run agentkit -- billing claim billint_... --secret bsec_...
191
227
  ```
192
-
193
- ## Operator Notes
194
-
195
- This section is for AgentKit maintainers, not for capsule-building agents.
196
-
197
- `agentkit deploy` sends the capsule artifact to the configured AgentKit Cloud API. The control plane owns infrastructure selection, provisioning, managed secrets, and public URL creation. Use `AGENTKIT_CLOUD_API_URL` only when testing a non-default control plane.
@@ -0,0 +1,72 @@
1
+ # Replay Production Traces
2
+
3
+ ## Goal
4
+
5
+ Run collected production or local traces against the local Agent Capsule before redeploying a fix.
6
+
7
+ ## When To Use This
8
+
9
+ Use this after `agentkit improve collect` and before `agentkit deploy` whenever prompts, tools, Knowledge, provider config, or channel behavior changed because of production evidence.
10
+
11
+ ## Commands
12
+
13
+ ```sh
14
+ agentkit replay .agentkit/improve/<run> --against local
15
+ ```
16
+
17
+ Run the full local eval suite too:
18
+
19
+ ```sh
20
+ npm run eval
21
+ ```
22
+
23
+ ## What Replay Checks
24
+
25
+ Replay sends each collected user turn through the local capsule in eval mode:
26
+
27
+ ```txt
28
+ ctx.runtime.environment === "eval"
29
+ ctx.runtime.invocation === "eval"
30
+ ```
31
+
32
+ This proves the current capsule can process the production turns without runtime errors. Generated eval files under `evals/regressions/` add committed behavior assertions.
33
+
34
+ ## Write Tool Safety
35
+
36
+ Replay still runs the registered capsule tools. Any tool that can write externally must guard eval mode:
37
+
38
+ ```ts
39
+ if (ctx.runtime.environment === "eval") {
40
+ return { sent: false, evalFixture: true };
41
+ }
42
+ ```
43
+
44
+ Do not depend on prompt wording alone to prevent side effects.
45
+
46
+ ## Deploy Gate
47
+
48
+ Before redeploying a production fix:
49
+
50
+ ```sh
51
+ npm run typecheck
52
+ npm run agentkit -- inspect
53
+ npm run eval
54
+ agentkit replay .agentkit/improve/<run> --against local
55
+ agentkit deploy --smoke "hello"
56
+ ```
57
+
58
+ If replay fails, inspect the failed trace id in `.agentkit/improve/<run>/traces/`, patch the capsule, and rerun replay.
59
+
60
+ ## Troubleshooting
61
+
62
+ `Replay failed`:
63
+
64
+ Read the printed error and the matching trace file. Common causes are missing local secrets, unsafe tools that do not branch on eval mode, stale Knowledge sources, or provider differences.
65
+
66
+ `Replay skipped`:
67
+
68
+ The trace had no user message. It may still help diagnose delivery or deploy state, but it cannot be replayed as a conversation.
69
+
70
+ `provider_model_unsupported`:
71
+
72
+ The local provider config does not match the replay environment. Keep deterministic regression replay on `test/fake` unless the owner intentionally selected a real provider.
@@ -22,6 +22,30 @@ Run evals:
22
22
  agentkit eval run
23
23
  ```
24
24
 
25
+ If the capsule uses npm scripts and Windows PowerShell blocks `npm.ps1`, use:
26
+
27
+ ```sh
28
+ npm.cmd run eval
29
+ ```
30
+
31
+ Create an eval from a stored conversation:
32
+
33
+ ```sh
34
+ agentkit conversations list
35
+ agentkit conversations trace <conversation-id>
36
+ agentkit eval from-conversation <conversation-id>
37
+ agentkit eval run
38
+ ```
39
+
40
+ Create evals from hosted or local production evidence:
41
+
42
+ ```sh
43
+ agentkit improve collect --deploy --since 24h
44
+ agentkit improve evals .agentkit/improve/<run>
45
+ agentkit replay .agentkit/improve/<run> --against local
46
+ agentkit eval run
47
+ ```
48
+
25
49
  ## Files Created Or Edited
26
50
 
27
51
  Create or edit:
@@ -34,63 +58,166 @@ Use conversations as source material:
34
58
 
35
59
  ```txt
36
60
  .agentkit/agentkit.db
61
+ .agentkit/improve/<run>/
37
62
  ```
38
63
 
39
64
  Do not edit `.agentkit/agentkit.db` by hand.
65
+ Do not edit `.agentkit/improve/<run>/bundle.json` by hand.
40
66
 
41
67
  ## Minimal Working Example
42
68
 
43
69
  Generated smoke eval:
44
70
 
45
71
  ```ts
46
- export default {
72
+ import { defineEval } from "@andreprado/agentkit";
73
+
74
+ export default defineEval({
47
75
  name: "smoke",
48
76
  input: "Say hello in one short sentence.",
49
77
  expect: {
50
- contains: "hello",
78
+ response: {
79
+ caseInsensitiveContains: "hello",
80
+ maxLength: 160,
81
+ },
51
82
  },
52
- };
83
+ });
53
84
  ```
54
85
 
55
86
  Supported assertion types:
56
87
 
57
88
  ```txt
58
- contains
59
- not_contains
60
- regex
61
- matches_regex
62
- persisted_tool_call
89
+ response.contains
90
+ response.containsAll
91
+ response.containsAny
92
+ response.caseInsensitiveContains
93
+ response.notContains
94
+ response.regex
95
+ response.matchesRegex
96
+ response.notRegex
97
+ response.maxLength
98
+ tools.called
99
+ tools.calledOnce
100
+ tools.count
101
+ tools.order
102
+ tools.persisted
63
103
  ```
64
104
 
65
- `persisted_tool_call` validates the tool call saved in local SQLite `tool_calls`, not a provider-specific raw response shape. It can be a tool name string or an object with `name`, `input`, `output`, `status`, and/or `visibility`.
105
+ Older flat aliases still work, including `contains`, `not_contains`, `regex`, `matches_regex`, and `persisted_tool_call`.
106
+
107
+ For date-sensitive evals, set top-level `now` to an ISO timestamp with an explicit timezone designator such as `Z` or `-05:00`. AgentKit uses that fixed clock for every turn and tool call in the eval:
108
+
109
+ ```ts
110
+ export default defineEval({
111
+ name: "appointment relative date",
112
+ now: "2026-02-04T02:30:00.000Z",
113
+ input: "What is today's date?",
114
+ expect: {
115
+ response: {
116
+ containsAll: ["2026-02-03", "Tuesday"],
117
+ },
118
+ },
119
+ });
120
+ ```
121
+
122
+ Multi-turn conversation evals use `turns`:
123
+
124
+ ```ts
125
+ import { defineEval } from "@andreprado/agentkit";
126
+
127
+ export default defineEval({
128
+ name: "buyer under budget",
129
+ turns: [
130
+ {
131
+ input: "I want a house up to 600k near Pinheiros.",
132
+ expect: {
133
+ tools: {
134
+ calledOnce: "buscar_imoveis",
135
+ persisted: {
136
+ name: "buscar_imoveis",
137
+ status: "completed",
138
+ input: { maxPrice: 600000 },
139
+ },
140
+ },
141
+ },
142
+ },
143
+ {
144
+ input: "Show me the best two.",
145
+ expect: {
146
+ response: {
147
+ containsAll: ["Pinheiros", "R$"],
148
+ },
149
+ },
150
+ },
151
+ ],
152
+ });
153
+ ```
154
+
155
+ `tools.persisted` validates the tool call saved in local SQLite `tool_calls`, not a provider-specific raw response shape. It can be a tool name string or an object with `name`, `input`, `output`, `rendered`, `status`, and/or `visibility`.
156
+
157
+ Use `tools.count` for the exact number of persisted calls in that turn, `tools.calledOnce` for exactly one call by name, and `tools.order` for required relative order. `tools.order` allows extra calls before, between, or after the named calls; pair it with `tools.count` when the exact call set matters.
66
158
 
67
159
  Use response assertions and persisted tool assertions together when internal operational output must not leak:
68
160
 
69
161
  ```ts
70
- export default {
162
+ import { defineEval } from "@andreprado/agentkit";
163
+
164
+ export default defineEval({
71
165
  name: "triage lead",
72
166
  input: '{"tool":"triage_real_estate_lead","input":{"email":"ada@example.com"}}',
73
167
  expect: {
74
- not_contains: ["hot", "score"],
75
- persisted_tool_call: {
76
- name: "triage_real_estate_lead",
77
- status: "completed",
78
- visibility: "internal",
79
- output: { status: "hot" },
168
+ response: {
169
+ notContains: ["hot", "score"],
170
+ notRegex: ["API_KEY|secret|token"],
171
+ },
172
+ tools: {
173
+ calledOnce: "triage_real_estate_lead",
174
+ persisted: {
175
+ name: "triage_real_estate_lead",
176
+ status: "completed",
177
+ visibility: "internal",
178
+ output: { status: "hot" },
179
+ },
80
180
  },
81
181
  },
82
- };
182
+ });
83
183
  ```
84
184
 
85
- `tool_call` remains accepted as a backwards-compatible alias, but new evals should use `persisted_tool_call`.
185
+ `tool_call` and `persisted_tool_call` remain accepted as backwards-compatible aliases, but new evals should use `tools.persisted`.
186
+
187
+ Safe external-tool pattern:
188
+
189
+ ```ts
190
+ export const sendFollowupEmail = defineTool({
191
+ name: "send_followup_email",
192
+ description: "Sends a follow-up email.",
193
+ inputSchema: {
194
+ type: "object",
195
+ properties: {
196
+ email: { type: "string" },
197
+ },
198
+ required: ["email"],
199
+ additionalProperties: false,
200
+ },
201
+ async execute(input: { email: string }, ctx) {
202
+ if (ctx.runtime.environment === "eval") {
203
+ return { sent: false, evalFixture: true, email: input.email };
204
+ }
205
+
206
+ // Real provider call here.
207
+ return { sent: true, evalFixture: false, email: input.email };
208
+ },
209
+ });
210
+ ```
86
211
 
87
212
  ## Safety Rules
88
213
 
89
214
  - Do not put provider keys in eval files.
90
215
  - Do not put real client PII in eval files.
91
216
  - Prefer small deterministic assertions.
92
- - Mock tools for evals that would write, delete, charge money, or email people.
217
+ - Keep eval tool calls deterministic and non-destructive.
218
+ - For tools that would write, delete, charge money, send email, or call a real customer system, branch inside the registered tool on `ctx.runtime.environment === "eval"` and return safe fixture output.
93
219
  - Treat conversations as source material, not as automatically safe training data.
220
+ - Review generated regression evals from `agentkit improve evals` before committing them. AgentKit redacts common email, phone, bearer token, and key patterns in generated eval text, but you must still remove or generalize domain-specific client PII and replace brittle exact prose assertions with the important behavior when needed.
94
221
 
95
222
  ## Verification
96
223
 
@@ -111,7 +238,7 @@ Use `test/fake` for deterministic smoke tests, then add provider-specific evals
111
238
 
112
239
  Tool eval hits a real API:
113
240
 
114
- Add tool mocks before running the eval suite.
241
+ Make the registered tool return deterministic fixture output when `ctx.runtime.environment === "eval"`, then rerun the eval suite. AgentKit does not expose a separate eval-only mock registry yet.
115
242
 
116
243
  ## Backend Contracts Used
117
244
 
@@ -29,7 +29,7 @@ Hosted alpha secret commands:
29
29
 
30
30
  ```sh
31
31
  agentkit login --token agk_user_...
32
- agentkit secret set OPENAI_API_KEY
32
+ agentkit secret set OPENAI_API_KEY --from-local-env
33
33
  agentkit secret list
34
34
  agentkit secret unset OPENAI_API_KEY
35
35
  ```
@@ -93,15 +93,16 @@ defineTool({
93
93
  - `.env.schema` is the committed contract for local secret names.
94
94
  - AgentKit local commands load `.env` directly so inspect, chat, tools, and evals share the same secret loader.
95
95
  - Production uses managed secrets.
96
- - Hosted alpha deploys require `cloudflare_deploy_alpha`; local commands do not require login.
96
+ - Hosted deploys require `cloudflare_deploy_alpha` or purchased/manual deploy slots; local commands do not require login.
97
97
  - Secret values must not appear in config, docs, prompts, evals, logs, exports, or SQLite.
98
98
  - A tool receives only secrets listed in that tool.
99
99
  - Avoid direct `process.env` reads inside tools.
100
100
  - Use `permissions` to describe external capabilities.
101
101
  - Add timeouts to network tools.
102
- - Treat public deploy URLs as unauthenticated transport, not access control.
103
- - Use access tokens and limits for clients.
102
+ - Treat hosted deploy URLs as addresses, not access control.
103
+ - Use deploy access tokens for hosted chat, hosted conversation reads, hosted trace reads, and any client app that talks to AgentKit Cloud.
104
104
  - Remove client PII before writing evals.
105
+ - Review `.agentkit/feedback/` drafts before sending AgentKit product feedback. Do not paste `.env` values, provider keys, cookies, client PII, or full private transcripts into feedback messages.
105
106
 
106
107
  ## Verification
107
108
 
@@ -132,9 +133,9 @@ Add the secret name to the tool `secrets` field and set it locally:
132
133
  npm run agentkit -- inspect
133
134
  ```
134
135
 
135
- Public URL is reachable:
136
+ Hosted URL is reachable without a token:
136
137
 
137
- Check hosted access mode. Default should be `private`.
138
+ Treat this as a security bug. Hosted AgentKit Cloud routes should reject missing deploy tokens except authenticated channel webhook ingress.
138
139
 
139
140
  ## Backend Contracts Used
140
141