@andreprado/agentkit 0.1.0-alpha.10

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (122) hide show
  1. package/README.md +69 -0
  2. package/bin/agentkit.mjs +23 -0
  3. package/docs/guides/add-channel.md +114 -0
  4. package/docs/guides/add-knowledge.md +134 -0
  5. package/docs/guides/add-tool.md +342 -0
  6. package/docs/guides/agentkit-skills-architecture.md +471 -0
  7. package/docs/guides/channel-security.md +81 -0
  8. package/docs/guides/channels-implementation-map.md +243 -0
  9. package/docs/guides/channels-production-handoff.md +102 -0
  10. package/docs/guides/connect-telegram.md +110 -0
  11. package/docs/guides/connect-whatsapp-zapster.md +119 -0
  12. package/docs/guides/create-agent.md +220 -0
  13. package/docs/guides/prepare-deploy.md +209 -0
  14. package/docs/guides/run-evals.md +179 -0
  15. package/docs/guides/security-rules.md +156 -0
  16. package/docs/guides/use-provider.md +140 -0
  17. package/docs/llms-full.txt +876 -0
  18. package/docs/llms.txt +83 -0
  19. package/docs/portable-deploy-release-checklist.md +41 -0
  20. package/package.json +47 -0
  21. package/src/cli/args.ts +36 -0
  22. package/src/cli/cloud-client.ts +265 -0
  23. package/src/cli/commands/channels.ts +810 -0
  24. package/src/cli/commands/knowledge.ts +136 -0
  25. package/src/cli/constants.ts +4 -0
  26. package/src/cli/deploy-chat-ui.ts +392 -0
  27. package/src/cli/deploy-readiness.ts +348 -0
  28. package/src/cli/flags.ts +162 -0
  29. package/src/cli/help.ts +184 -0
  30. package/src/cli/index.ts +1276 -0
  31. package/src/cli/process.ts +31 -0
  32. package/src/cloud/artifact.ts +139 -0
  33. package/src/cloud/client.ts +79 -0
  34. package/src/cloud/contracts.ts +63 -0
  35. package/src/cloud/index.ts +3 -0
  36. package/src/create-project.ts +177 -0
  37. package/src/index.ts +408 -0
  38. package/src/providers/index.ts +25 -0
  39. package/src/providers/pi.ts +286 -0
  40. package/src/providers/test.ts +133 -0
  41. package/src/providers/types.ts +34 -0
  42. package/src/runtime/build.ts +43 -0
  43. package/src/runtime/channel-buffer.ts +30 -0
  44. package/src/runtime/channel-test-harness.ts +112 -0
  45. package/src/runtime/channels/telegram.ts +360 -0
  46. package/src/runtime/channels/website.ts +132 -0
  47. package/src/runtime/channels/whatsapp-meta.ts +71 -0
  48. package/src/runtime/channels/whatsapp-zapster.ts +278 -0
  49. package/src/runtime/channels.ts +138 -0
  50. package/src/runtime/chat.ts +218 -0
  51. package/src/runtime/config.ts +684 -0
  52. package/src/runtime/conversations.ts +38 -0
  53. package/src/runtime/core/deploy-state.ts +54 -0
  54. package/src/runtime/core/manifest.ts +213 -0
  55. package/src/runtime/core/targets.ts +133 -0
  56. package/src/runtime/database.ts +256 -0
  57. package/src/runtime/db-commands.ts +167 -0
  58. package/src/runtime/deploy-readiness.ts +105 -0
  59. package/src/runtime/deploy.ts +1 -0
  60. package/src/runtime/dev-server.ts +1247 -0
  61. package/src/runtime/docs.ts +36 -0
  62. package/src/runtime/env.ts +152 -0
  63. package/src/runtime/errors.ts +13 -0
  64. package/src/runtime/evals.ts +509 -0
  65. package/src/runtime/inspect.ts +203 -0
  66. package/src/runtime/knowledge/chunk.ts +333 -0
  67. package/src/runtime/knowledge/config.ts +135 -0
  68. package/src/runtime/knowledge/embeddings.ts +133 -0
  69. package/src/runtime/knowledge/ingest.ts +521 -0
  70. package/src/runtime/knowledge/prompt-policy.ts +30 -0
  71. package/src/runtime/knowledge/retrieve.ts +283 -0
  72. package/src/runtime/knowledge/schema.ts +56 -0
  73. package/src/runtime/knowledge/tool.ts +64 -0
  74. package/src/runtime/knowledge/vector.ts +258 -0
  75. package/src/runtime/runtime-contract.ts +93 -0
  76. package/src/runtime/spec.ts +152 -0
  77. package/src/runtime/sync.ts +144 -0
  78. package/src/runtime/targets/cloudflare/build.ts +2517 -0
  79. package/src/runtime/targets/container/build.ts +146 -0
  80. package/src/runtime/targets/container/server.ts +33 -0
  81. package/src/runtime/targets/vps/deploy.ts +206 -0
  82. package/src/runtime/tool-runner.ts +65 -0
  83. package/src/runtime/tools.ts +470 -0
  84. package/src/runtime/traces.ts +41 -0
  85. package/src/storage/sqlite.ts +1118 -0
  86. package/src/templates/blank.ts +394 -0
  87. package/src/templates/dentista.ts +1003 -0
  88. package/src/templates/index.ts +33 -0
  89. package/src/templates/skills/agentkit-build-agent/SKILL.md +51 -0
  90. package/src/templates/skills/agentkit-build-agent/templates/appointment-intake.instructions.md +20 -0
  91. package/src/templates/skills/agentkit-build-agent/templates/sales-qualifier.instructions.md +17 -0
  92. package/src/templates/skills/agentkit-build-agent/templates/support-agent.instructions.md +16 -0
  93. package/src/templates/skills/agentkit-capsule/SKILL.md +62 -0
  94. package/src/templates/skills/agentkit-capsule/references/docs-router.md +15 -0
  95. package/src/templates/skills/agentkit-channels/SKILL.md +62 -0
  96. package/src/templates/skills/agentkit-channels/references/channel-buffering.md +58 -0
  97. package/src/templates/skills/agentkit-channels/references/channel-debugging.md +41 -0
  98. package/src/templates/skills/agentkit-channels/references/telegram.md +38 -0
  99. package/src/templates/skills/agentkit-channels/references/whatsapp-zapster.md +44 -0
  100. package/src/templates/skills/agentkit-database/SKILL.md +45 -0
  101. package/src/templates/skills/agentkit-database/templates/appointments.schema.sql +15 -0
  102. package/src/templates/skills/agentkit-database/templates/leads.schema.sql +17 -0
  103. package/src/templates/skills/agentkit-deploy/SKILL.md +44 -0
  104. package/src/templates/skills/agentkit-evals/SKILL.md +60 -0
  105. package/src/templates/skills/agentkit-evals/templates/multi-turn.eval.md +22 -0
  106. package/src/templates/skills/agentkit-evals/templates/no-leak.eval.md +14 -0
  107. package/src/templates/skills/agentkit-evals/templates/smoke.eval.md +14 -0
  108. package/src/templates/skills/agentkit-evals/templates/tool-call.eval.md +18 -0
  109. package/src/templates/skills/agentkit-knowledge/SKILL.md +40 -0
  110. package/src/templates/skills/agentkit-knowledge/templates/faq.md +14 -0
  111. package/src/templates/skills/agentkit-knowledge/templates/policies.md +14 -0
  112. package/src/templates/skills/agentkit-knowledge/templates/prices.csv +3 -0
  113. package/src/templates/skills/agentkit-prompts/SKILL.md +45 -0
  114. package/src/templates/skills/agentkit-prompts/templates/knowledge-grounded-faq.instructions.md +11 -0
  115. package/src/templates/skills/agentkit-provider/SKILL.md +57 -0
  116. package/src/templates/skills/agentkit-security/SKILL.md +55 -0
  117. package/src/templates/skills/agentkit-tools/SKILL.md +36 -0
  118. package/src/templates/skills/agentkit-tools/examples/database-write.tool.md +35 -0
  119. package/src/templates/skills/agentkit-tools/examples/eval-safe-external-action.tool.md +37 -0
  120. package/src/templates/skills/agentkit-tools/examples/lookup-order.tool.md +46 -0
  121. package/src/templates/skills/agentkit-troubleshooting/SKILL.md +52 -0
  122. package/src/templates/support.ts +401 -0
@@ -0,0 +1,220 @@
1
+ # Create An Agent Capsule
2
+
3
+ ## Goal
4
+
5
+ Create a local AgentKit Agent Capsule that installs, typechecks, chats, stores conversations, and exposes inspectable runtime state.
6
+
7
+ ## When To Use This
8
+
9
+ Use this when starting a new agent from zero or when verifying that the scaffold still works.
10
+
11
+ ## Commands
12
+
13
+ From a clean working directory:
14
+
15
+ ```sh
16
+ rm -rf /tmp/agentkit-demo
17
+ mkdir -p /tmp/agentkit-demo
18
+ cd /tmp/agentkit-demo
19
+
20
+ npx @andreprado/agentkit@alpha new demo --template blank
21
+ cd demo
22
+ npm run typecheck
23
+ npm run chat -- --message "hello"
24
+ npm run agentkit -- conversations list
25
+ npm run agentkit -- inspect
26
+ ```
27
+
28
+ Then open the folder in Codex, Claude Code, or another coding agent and ask directly:
29
+
30
+ ```txt
31
+ Develop an appointment and intake agent for an ophthalmology office.
32
+ ```
33
+
34
+ For the support template:
35
+
36
+ ```sh
37
+ npx @andreprado/agentkit@alpha new support-demo --template support
38
+ cd support-demo
39
+ npm run agentkit -- tool lookup_order --input '{"orderId":"A100"}'
40
+ ```
41
+
42
+ ## Files Created Or Edited
43
+
44
+ Generated files:
45
+
46
+ ```txt
47
+ agentkit.config.ts
48
+ package.json
49
+ tsconfig.json
50
+ .env.schema
51
+ .gitignore
52
+ AGENTS.md
53
+ AGENTKIT.md
54
+ CLAUDE.md
55
+ README.md
56
+ src/agent.ts
57
+ prompts/instructions.md
58
+ evals/smoke.eval.ts
59
+ ```
60
+
61
+ AgentKit also runs `git init` in the generated capsule when `.git` does not already exist.
62
+
63
+ The `support` template also creates:
64
+
65
+ ```txt
66
+ tools/lookup-order.ts
67
+ ```
68
+
69
+ Runtime files created after chat:
70
+
71
+ ```txt
72
+ .agentkit/agentkit.db
73
+ .agentkit/agentkit.db-shm
74
+ .agentkit/agentkit.db-wal
75
+ ```
76
+
77
+ ## Handoff To A Coding Agent
78
+
79
+ The owner does not need to fill a separate brief file. The natural-language request they type into Codex, Claude Code, or another coding agent is the brief.
80
+
81
+ Primary flow:
82
+
83
+ ```txt
84
+ Develop an appointment and intake agent for an ophthalmology office.
85
+ ```
86
+
87
+ The generated `AGENTS.md`, `AGENTKIT.md`, `CLAUDE.md`, and `skills/` pack tell the coding agent which files to edit, which task skill to load, and which verification commands to run. There is no wizard or recipe layer: the coding agent edits the capsule directly from the scaffold, contract, and owner request. The default router is `skills/agentkit-capsule/SKILL.md`; `llms-full.txt` is reserved for complete-contract checks.
88
+
89
+ After the owner gives the general idea, the coding agent should create or update the implementation contract itself:
90
+
91
+ ```sh
92
+ npm run agentkit -- spec init --brief "Develop an appointment and intake agent for an ophthalmology office."
93
+ npm run agentkit -- spec check
94
+ ```
95
+
96
+ `AGENT_SPEC.md` is an internal working contract for the coding agent. It is not a form the owner must fill before work starts.
97
+
98
+ Optional shortcut when copying a prompt into another coding agent:
99
+
100
+ ```sh
101
+ npm run agentkit -- handoff codex "Develop an appointment and intake agent for an ophthalmology office."
102
+ npm run agentkit -- handoff claude "Develop an appointment and intake agent for an ophthalmology office."
103
+ ```
104
+
105
+ ## Minimal Working Example
106
+
107
+ `agentkit.config.ts` for the offline fake provider:
108
+
109
+ ```ts
110
+ import { defineAgent } from "@andreprado/agentkit";
111
+
112
+ export default defineAgent({
113
+ name: "demo",
114
+ runtime: "edge",
115
+ provider: {
116
+ name: "test",
117
+ model: "fake",
118
+ },
119
+ instructions: "./prompts/instructions.md",
120
+ secrets: [],
121
+ tools: [],
122
+ access: {
123
+ mode: "private",
124
+ },
125
+ storage: {
126
+ driver: "agentkit",
127
+ path: ".agentkit/agentkit.db",
128
+ database: {
129
+ driver: "turso",
130
+ schema: "./schema.sql",
131
+ },
132
+ files: {
133
+ driver: "r2",
134
+ bucket: "agentkit-demo-files",
135
+ prefix: "demo",
136
+ },
137
+ },
138
+ });
139
+ ```
140
+
141
+ Expected chat output:
142
+
143
+ ```txt
144
+ Echo: hello
145
+ ```
146
+
147
+ ## Testing With A UI
148
+
149
+ Local UI:
150
+
151
+ ```sh
152
+ npm run dev
153
+ ```
154
+
155
+ Open the printed `Chat:` URL and tell the owner the exact URL.
156
+
157
+ Hosted deploy UI:
158
+
159
+ ```sh
160
+ npm run agentkit -- deploy
161
+ npm run agentkit -- chat-ui --deploy
162
+ ```
163
+
164
+ Open the printed `Chat:` URL and tell the owner this local UI is connected to the hosted deploy.
165
+
166
+ `test/fake` is deterministic. It validates the scaffold, direct tool checks, and fake-provider evals, but it does not validate natural conversation quality. Before claiming real conversation behavior is tested, ask the owner which provider to use: OpenRouter, OpenAI, Anthropic, or another supported provider.
167
+
168
+ ## Safety Rules
169
+
170
+ - Do not commit `.env`.
171
+ - Do not commit `.agentkit/`.
172
+ - Do not put secret values in `agentkit.config.ts`.
173
+ - Keep `agentkit.config.ts` declarative.
174
+ - Put behavior in prompts and tools.
175
+ - Run commands from the capsule root.
176
+
177
+ ## Verification
178
+
179
+ ```sh
180
+ npm run typecheck
181
+ npm run chat -- --message "hello"
182
+ npm run eval
183
+ npm run agentkit -- conversations list
184
+ npm run agentkit -- inspect
185
+ ```
186
+
187
+ Expected checks:
188
+
189
+ - TypeScript exits with code `0`.
190
+ - Chat prints `Echo: hello`.
191
+ - Evals report `1 passed, 0 failed`.
192
+ - Conversation list shows one row with `2` messages.
193
+ - Inspect prints JSON with `agent`, `runtime`, `provider`, `tools`, `storage`, and `secrets`.
194
+
195
+ ## Troubleshooting
196
+
197
+ `Cannot find module '@andreprado/agentkit'`:
198
+
199
+ ```sh
200
+ npm install
201
+ npm run typecheck
202
+ ```
203
+
204
+ `agentkit new` installs dependencies by default. Run this if the scaffold was created with `--no-install`, the install failed, or `node_modules` was deleted. Make sure `tsconfig.json` has `moduleResolution: "Bundler"`.
205
+
206
+ `No agentkit.config.ts found`:
207
+
208
+ Run the command from the generated capsule root.
209
+
210
+ `Prompt file not found`:
211
+
212
+ Check the `instructions` path in `agentkit.config.ts`.
213
+
214
+ `Unknown template`:
215
+
216
+ Use `blank` or `support`. `research` is planned but not implemented.
217
+
218
+ ## Backend Contracts Used
219
+
220
+ Local commands use local files and SQLite. Hosted deploy sends a build artifact to AgentKit Cloud with `agentkit deploy`; users do not create hosted projects or infrastructure by hand.
@@ -0,0 +1,209 @@
1
+ # Prepare Deploy
2
+
3
+ ## Goal
4
+
5
+ Prepare an Agent Capsule so a coding agent can run the same workflow a user will run:
6
+
7
+ ```sh
8
+ agentkit deploy
9
+ ```
10
+
11
+ The agent should not choose hosting targets or infrastructure providers. AgentKit owns that routing.
12
+
13
+ ## When To Use This
14
+
15
+ Use this before asking a coding agent to make a capsule deployable, or before testing the end-to-end user flow from a clean scaffold.
16
+
17
+ ## User Workflow
18
+
19
+ From the capsule root:
20
+
21
+ ```sh
22
+ npm run typecheck
23
+ npm run agentkit -- inspect
24
+ npm run agentkit -- db migrate
25
+ npm run chat -- --message "hello"
26
+ ```
27
+
28
+ All scaffold, local dev, chat, eval, inspect, database, and build commands are token-free. Only hosted production deploy and hosted control-plane mutations require an invited AgentKit Cloud token:
29
+
30
+ ```sh
31
+ npm run agentkit -- login --token agk_user_...
32
+ npm run agentkit -- deploy doctor
33
+ npm run agentkit -- secret set OPENAI_API_KEY --from-local-env
34
+ npm run agentkit -- deploy --smoke "hello"
35
+ npm run agentkit -- chat-ui --deploy
36
+ ```
37
+
38
+ If the capsule uses OpenAI locally, put the user’s local key in `.env`:
39
+
40
+ ```sh
41
+ OPENAI_API_KEY=sk-...
42
+ ```
43
+
44
+ `.env` is local-only. Hosted deploy secrets are handled by AgentKit outside the capsule.
45
+ Closed-alpha hosted deploys require an invited account token with `cloudflare_deploy_alpha`.
46
+
47
+ ## What The Agent Should Edit
48
+
49
+ Review and edit:
50
+
51
+ ```txt
52
+ agentkit.config.ts
53
+ prompts/instructions.md
54
+ tools/
55
+ evals/
56
+ .env.schema
57
+ AGENTS.md
58
+ CLAUDE.md
59
+ README.md
60
+ schema.sql
61
+ ```
62
+
63
+ Never commit or deploy:
64
+
65
+ ```txt
66
+ .env
67
+ .agentkit/
68
+ node_modules/
69
+ ```
70
+
71
+ ## Deploy-Ready Config Shape
72
+
73
+ Use the AgentKit-managed storage contract. Do not make the user or the coding agent choose a backend target or infrastructure provider.
74
+
75
+ ```ts
76
+ export default defineAgent({
77
+ name: "support-agent",
78
+ runtime: "edge",
79
+ provider: {
80
+ name: "openai",
81
+ model: "gpt-4o-mini",
82
+ },
83
+ instructions: "./prompts/instructions.md",
84
+ secrets: ["OPENAI_API_KEY"],
85
+ tools: [],
86
+ access: {
87
+ mode: "private",
88
+ },
89
+ storage: {
90
+ driver: "agentkit",
91
+ },
92
+ });
93
+ ```
94
+
95
+ This config is user-facing because it describes the agent contract: provider, prompts, tools, access, secrets, and managed storage. The concrete deploy target, database vendor, file bucket, and runtime host are not user-facing.
96
+
97
+ If the generated capsule already includes a managed database block, keep the `schema` path correct and put tables in `schema.sql`. The concrete database vendor is an AgentKit implementation detail.
98
+
99
+ ## Database Rules
100
+
101
+ For agent-owned tables:
102
+
103
+ - Put tables and indexes in `schema.sql`.
104
+ - Keep `schema.sql` idempotent.
105
+ - Prefer `CREATE TABLE IF NOT EXISTS` and `CREATE INDEX IF NOT EXISTS`.
106
+ - Use `agentkit db migrate` before local tool/chat testing.
107
+ - Use `ctx.db` inside tools. `ctx.database` and `ctx.storage.sql` are aliases.
108
+ - Do not import local database drivers from tools.
109
+
110
+ Local commands use `.agentkit/agentkit.db`. Hosted deploy uses AgentKit-managed infrastructure.
111
+
112
+ ## Verification
113
+
114
+ Run this before saying the capsule is deploy-ready:
115
+
116
+ ```sh
117
+ npm run typecheck
118
+ npm run agentkit -- inspect
119
+ npm run agentkit -- db migrate
120
+ npm run chat -- --message "hello"
121
+ npm run agentkit -- deploy --dry-run
122
+ npm run agentkit -- deploy doctor
123
+ ```
124
+
125
+ For a final end-to-end test, run:
126
+
127
+ ```sh
128
+ npm run agentkit -- login --token agk_user_...
129
+ npm run agentkit -- secret sync --from-local
130
+ npm run agentkit -- secret list
131
+ npm run agentkit -- deploy --smoke "hello"
132
+ npm run agentkit -- deploy status
133
+ npm run agentkit -- chat-ui --deploy
134
+ npm run agentkit -- access token list
135
+ ```
136
+
137
+ For private hosted deploys, `npm run agentkit -- deploy` writes the local chat/UI access token to `.agentkit/chat-access-token.json`. Use `npm run agentkit -- deploy --smoke "hello"` for the official hosted chat smoke, and use `npm run agentkit -- chat-ui --deploy` for hosted UI testing. Use `npm run agentkit -- access token create <name> --out <path>` only for additional clients.
138
+
139
+ When the UI is running, open the printed `Chat:` URL and tell the owner the exact URL. If the capsule is still on `test/fake`, say the UI was tested only with the deterministic fake provider.
140
+
141
+ ## Readiness Checklist
142
+
143
+ ```txt
144
+ No .env committed
145
+ No .agentkit committed
146
+ All required secret names are declared
147
+ Prompt path exists
148
+ Tools have input schemas
149
+ Dangerous tools have permissions
150
+ Provider model is supported by AgentKit
151
+ schema.sql is idempotent
152
+ README/AGENTS/CLAUDE match the capsule
153
+ ```
154
+
155
+ ## Troubleshooting
156
+
157
+ `agentkit deploy` fails before publishing:
158
+
159
+ Read the AgentKit error code first. Common fixes are:
160
+
161
+ - run `npm run typecheck`;
162
+ - run `npm run agentkit -- inspect`;
163
+ - make tools edge-safe;
164
+ - declare missing secret names in `agentkit.config.ts`;
165
+ - keep production secret values out of source files.
166
+
167
+ Missing local OpenAI key:
168
+
169
+ ```sh
170
+ echo 'OPENAI_API_KEY=sk-...' >> .env
171
+ ```
172
+
173
+ Missing hosted secret:
174
+
175
+ Set the secret in AgentKit Cloud or the configured control plane:
176
+
177
+ ```sh
178
+ npm run agentkit -- secret set OPENAI_API_KEY --from-local-env
179
+ ```
180
+
181
+ To sync all declared user-managed secrets present in local `.env`:
182
+
183
+ ```sh
184
+ npm run agentkit -- secret sync --from-local
185
+ ```
186
+
187
+ Do not write hosted secret values into the capsule.
188
+
189
+ Before a hosted production deploy, run:
190
+
191
+ ```sh
192
+ npm run agentkit -- deploy doctor
193
+ ```
194
+
195
+ The doctor checks AgentKit Cloud login, alpha deploy entitlement, online deploy capacity, declared hosted secrets, local `.env` names that have not been uploaded with `agentkit secret set`, and private-access runtime token handling. `agentkit deploy` runs the same readiness check automatically before building and uploading the artifact.
196
+
197
+ `alpha_access_required`:
198
+
199
+ Log in with an invited alpha account:
200
+
201
+ ```sh
202
+ npm run agentkit -- login --token agk_user_...
203
+ ```
204
+
205
+ ## Operator Notes
206
+
207
+ This section is for AgentKit maintainers, not for capsule-building agents.
208
+
209
+ `agentkit deploy` sends the capsule artifact to the configured AgentKit Cloud API. The control plane owns infrastructure selection, provisioning, managed secrets, and public URL creation. Use `AGENTKIT_CLOUD_API_URL` only when testing a non-default control plane.
@@ -0,0 +1,179 @@
1
+ # Run Evals
2
+
3
+ ## Goal
4
+
5
+ Prepare and run evals for an Agent Capsule.
6
+
7
+ ## When To Use This
8
+
9
+ Use this after chat works and before changing prompts, tools, or provider config.
10
+
11
+ ## Commands
12
+
13
+ Generated file:
14
+
15
+ ```txt
16
+ evals/smoke.eval.ts
17
+ ```
18
+
19
+ Run evals:
20
+
21
+ ```sh
22
+ agentkit eval run
23
+ ```
24
+
25
+ Create an eval from a stored conversation:
26
+
27
+ ```sh
28
+ agentkit conversations list
29
+ agentkit conversations trace <conversation-id>
30
+ agentkit eval from-conversation <conversation-id>
31
+ agentkit eval run
32
+ ```
33
+
34
+ ## Files Created Or Edited
35
+
36
+ Create or edit:
37
+
38
+ ```txt
39
+ evals/<name>.eval.ts
40
+ ```
41
+
42
+ Use conversations as source material:
43
+
44
+ ```txt
45
+ .agentkit/agentkit.db
46
+ ```
47
+
48
+ Do not edit `.agentkit/agentkit.db` by hand.
49
+
50
+ ## Minimal Working Example
51
+
52
+ Generated smoke eval:
53
+
54
+ ```ts
55
+ export default {
56
+ name: "smoke",
57
+ input: "Say hello in one short sentence.",
58
+ expect: {
59
+ contains: "hello",
60
+ },
61
+ };
62
+ ```
63
+
64
+ Supported assertion types:
65
+
66
+ ```txt
67
+ contains
68
+ not_contains
69
+ regex
70
+ matches_regex
71
+ persisted_tool_call
72
+ ```
73
+
74
+ Multi-turn conversation evals use `turns`:
75
+
76
+ ```ts
77
+ export default {
78
+ name: "buyer under budget",
79
+ turns: [
80
+ {
81
+ input: "I want a house up to 600k near Pinheiros.",
82
+ expect: {
83
+ persisted_tool_call: {
84
+ name: "buscar_imoveis",
85
+ status: "completed",
86
+ input: { maxPrice: 600000 },
87
+ },
88
+ },
89
+ },
90
+ {
91
+ input: "Show me the best two.",
92
+ expect: {
93
+ contains: ["Pinheiros", "R$"],
94
+ },
95
+ },
96
+ ],
97
+ };
98
+ ```
99
+
100
+ `persisted_tool_call` validates the tool call saved in local SQLite `tool_calls`, not a provider-specific raw response shape. It can be a tool name string or an object with `name`, `input`, `output`, `rendered`, `status`, and/or `visibility`.
101
+
102
+ Use response assertions and persisted tool assertions together when internal operational output must not leak:
103
+
104
+ ```ts
105
+ export default {
106
+ name: "triage lead",
107
+ input: '{"tool":"triage_real_estate_lead","input":{"email":"ada@example.com"}}',
108
+ expect: {
109
+ not_contains: ["hot", "score"],
110
+ persisted_tool_call: {
111
+ name: "triage_real_estate_lead",
112
+ status: "completed",
113
+ visibility: "internal",
114
+ output: { status: "hot" },
115
+ },
116
+ },
117
+ };
118
+ ```
119
+
120
+ `tool_call` remains accepted as a backwards-compatible alias, but new evals should use `persisted_tool_call`.
121
+
122
+ Safe external-tool pattern:
123
+
124
+ ```ts
125
+ export const sendFollowupEmail = defineTool({
126
+ name: "send_followup_email",
127
+ description: "Sends a follow-up email.",
128
+ inputSchema: {
129
+ type: "object",
130
+ properties: {
131
+ email: { type: "string" },
132
+ },
133
+ required: ["email"],
134
+ additionalProperties: false,
135
+ },
136
+ async execute(input: { email: string }, ctx) {
137
+ if (ctx.runtime.environment === "eval") {
138
+ return { sent: false, evalFixture: true, email: input.email };
139
+ }
140
+
141
+ // Real provider call here.
142
+ return { sent: true, evalFixture: false, email: input.email };
143
+ },
144
+ });
145
+ ```
146
+
147
+ ## Safety Rules
148
+
149
+ - Do not put provider keys in eval files.
150
+ - Do not put real client PII in eval files.
151
+ - Prefer small deterministic assertions.
152
+ - Keep eval tool calls deterministic and non-destructive.
153
+ - For tools that would write, delete, charge money, send email, or call a real customer system, branch inside the registered tool on `ctx.runtime.environment === "eval"` and return safe fixture output.
154
+ - Treat conversations as source material, not as automatically safe training data.
155
+
156
+ ## Verification
157
+
158
+ ```sh
159
+ npm run typecheck
160
+ npm run eval
161
+ ```
162
+
163
+ ## Troubleshooting
164
+
165
+ `agentkit eval` is unknown:
166
+
167
+ Run `npm install` in the capsule and make sure it depends on the AgentKit version that includes `agentkit eval run`.
168
+
169
+ Eval output changes after provider switch:
170
+
171
+ Use `test/fake` for deterministic smoke tests, then add provider-specific evals separately.
172
+
173
+ Tool eval hits a real API:
174
+
175
+ Make the registered tool return deterministic fixture output when `ctx.runtime.environment === "eval"`, then rerun the eval suite. AgentKit does not expose a separate eval-only mock registry yet.
176
+
177
+ ## Backend Contracts Used
178
+
179
+ Future hosted evals use `EvalCase` and `EvalResult` resources. The current local implementation stores conversations, messages, runs, and tool calls only.