@andreprado/agentkit 0.1.0-alpha.18 → 0.1.0-alpha.19

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (56) hide show
  1. package/README.md +3 -0
  2. package/docs/guides/add-channel.md +12 -6
  3. package/docs/guides/add-knowledge.md +10 -0
  4. package/docs/guides/add-managed-composio.md +4 -2
  5. package/docs/guides/channel-security.md +26 -2
  6. package/docs/guides/connect-discord.md +178 -0
  7. package/docs/guides/create-agent.md +13 -0
  8. package/docs/guides/debug-channel.md +8 -2
  9. package/docs/guides/improve-from-production.md +151 -0
  10. package/docs/guides/prepare-deploy.md +29 -8
  11. package/docs/guides/replay-production-traces.md +72 -0
  12. package/docs/guides/run-evals.md +18 -0
  13. package/docs/guides/security-rules.md +5 -5
  14. package/docs/guides/use-provider.md +11 -1
  15. package/docs/llms-full.txt +100 -11
  16. package/docs/llms.txt +17 -1
  17. package/package.json +1 -1
  18. package/src/cli/args.ts +23 -2
  19. package/src/cli/cloud-client.ts +63 -0
  20. package/src/cli/commands/channels.ts +139 -14
  21. package/src/cli/deploy-readiness.ts +5 -2
  22. package/src/cli/help.ts +26 -2
  23. package/src/cli/index.ts +382 -17
  24. package/src/create-project.ts +13 -2
  25. package/src/index.ts +42 -3
  26. package/src/providers/pi.ts +49 -15
  27. package/src/runtime/channel-test-harness.ts +4 -1
  28. package/src/runtime/channels/discord.ts +887 -0
  29. package/src/runtime/channels.ts +15 -0
  30. package/src/runtime/config.ts +35 -3
  31. package/src/runtime/dev-server.ts +149 -8
  32. package/src/runtime/evals.ts +27 -6
  33. package/src/runtime/improve.ts +868 -0
  34. package/src/runtime/knowledge/retrieve.ts +25 -5
  35. package/src/runtime/knowledge/schema.ts +45 -1
  36. package/src/runtime/runtime-contract.ts +54 -0
  37. package/src/runtime/targets/cloudflare/build.ts +248 -193
  38. package/src/runtime/targets/vps/deploy.ts +1 -1
  39. package/src/storage/sqlite.ts +7 -2
  40. package/src/templates/skills/agentkit-capsule/SKILL.md +8 -1
  41. package/src/templates/skills/agentkit-capsule/references/docs-router.md +1 -2
  42. package/src/templates/skills/agentkit-channels/SKILL.md +6 -1
  43. package/src/templates/skills/agentkit-channels/references/channel-debugging.md +2 -1
  44. package/src/templates/skills/agentkit-channels/references/discord.md +93 -0
  45. package/src/templates/skills/agentkit-deploy/SKILL.md +6 -0
  46. package/src/templates/skills/agentkit-evals/SKILL.md +12 -3
  47. package/src/templates/skills/agentkit-improve/SKILL.md +86 -0
  48. package/src/templates/skills/agentkit-improve/references/replay-side-effects.md +18 -0
  49. package/src/templates/skills/agentkit-improve/references/trace-packets.md +22 -0
  50. package/src/templates/skills/agentkit-improve/templates/regression.eval.md +18 -0
  51. package/src/templates/skills/agentkit-integrations/SKILL.md +1 -0
  52. package/src/templates/skills/agentkit-knowledge/SKILL.md +4 -1
  53. package/src/templates/skills/agentkit-provider/SKILL.md +4 -1
  54. package/src/templates/skills/agentkit-security/SKILL.md +3 -2
  55. package/src/templates/skills/agentkit-troubleshooting/SKILL.md +9 -0
  56. package/src/templates/support.ts +4 -2
@@ -0,0 +1,72 @@
1
+ # Replay Production Traces
2
+
3
+ ## Goal
4
+
5
+ Run collected production or local traces against the local Agent Capsule before redeploying a fix.
6
+
7
+ ## When To Use This
8
+
9
+ Use this after `agentkit improve collect` and before `agentkit deploy` whenever prompts, tools, Knowledge, provider config, or channel behavior changed because of production evidence.
10
+
11
+ ## Commands
12
+
13
+ ```sh
14
+ agentkit replay .agentkit/improve/<run> --against local
15
+ ```
16
+
17
+ Run the full local eval suite too:
18
+
19
+ ```sh
20
+ npm run eval
21
+ ```
22
+
23
+ ## What Replay Checks
24
+
25
+ Replay sends each collected user turn through the local capsule in eval mode:
26
+
27
+ ```txt
28
+ ctx.runtime.environment === "eval"
29
+ ctx.runtime.invocation === "eval"
30
+ ```
31
+
32
+ This proves the current capsule can process the production turns without runtime errors. Generated eval files under `evals/regressions/` add committed behavior assertions.
33
+
34
+ ## Write Tool Safety
35
+
36
+ Replay still runs the registered capsule tools. Any tool that can write externally must guard eval mode:
37
+
38
+ ```ts
39
+ if (ctx.runtime.environment === "eval") {
40
+ return { sent: false, evalFixture: true };
41
+ }
42
+ ```
43
+
44
+ Do not depend on prompt wording alone to prevent side effects.
45
+
46
+ ## Deploy Gate
47
+
48
+ Before redeploying a production fix:
49
+
50
+ ```sh
51
+ npm run typecheck
52
+ npm run agentkit -- inspect
53
+ npm run eval
54
+ agentkit replay .agentkit/improve/<run> --against local
55
+ agentkit deploy --smoke "hello"
56
+ ```
57
+
58
+ If replay fails, inspect the failed trace id in `.agentkit/improve/<run>/traces/`, patch the capsule, and rerun replay.
59
+
60
+ ## Troubleshooting
61
+
62
+ `Replay failed`:
63
+
64
+ Read the printed error and the matching trace file. Common causes are missing local secrets, unsafe tools that do not branch on eval mode, stale Knowledge sources, or provider differences.
65
+
66
+ `Replay skipped`:
67
+
68
+ The trace had no user message. It may still help diagnose delivery or deploy state, but it cannot be replayed as a conversation.
69
+
70
+ `provider_model_unsupported`:
71
+
72
+ The local provider config does not match the replay environment. Keep deterministic regression replay on `test/fake` unless the owner intentionally selected a real provider.
@@ -22,6 +22,12 @@ Run evals:
22
22
  agentkit eval run
23
23
  ```
24
24
 
25
+ If the capsule uses npm scripts and Windows PowerShell blocks `npm.ps1`, use:
26
+
27
+ ```sh
28
+ npm.cmd run eval
29
+ ```
30
+
25
31
  Create an eval from a stored conversation:
26
32
 
27
33
  ```sh
@@ -31,6 +37,15 @@ agentkit eval from-conversation <conversation-id>
31
37
  agentkit eval run
32
38
  ```
33
39
 
40
+ Create evals from hosted or local production evidence:
41
+
42
+ ```sh
43
+ agentkit improve collect --deploy --since 24h
44
+ agentkit improve evals .agentkit/improve/<run>
45
+ agentkit replay .agentkit/improve/<run> --against local
46
+ agentkit eval run
47
+ ```
48
+
34
49
  ## Files Created Or Edited
35
50
 
36
51
  Create or edit:
@@ -43,9 +58,11 @@ Use conversations as source material:
43
58
 
44
59
  ```txt
45
60
  .agentkit/agentkit.db
61
+ .agentkit/improve/<run>/
46
62
  ```
47
63
 
48
64
  Do not edit `.agentkit/agentkit.db` by hand.
65
+ Do not edit `.agentkit/improve/<run>/bundle.json` by hand.
49
66
 
50
67
  ## Minimal Working Example
51
68
 
@@ -200,6 +217,7 @@ export const sendFollowupEmail = defineTool({
200
217
  - Keep eval tool calls deterministic and non-destructive.
201
218
  - For tools that would write, delete, charge money, send email, or call a real customer system, branch inside the registered tool on `ctx.runtime.environment === "eval"` and return safe fixture output.
202
219
  - Treat conversations as source material, not as automatically safe training data.
220
+ - Review generated regression evals from `agentkit improve evals` before committing them. AgentKit redacts common email, phone, bearer token, and key patterns in generated eval text, but you must still remove or generalize domain-specific client PII and replace brittle exact prose assertions with the important behavior when needed.
203
221
 
204
222
  ## Verification
205
223
 
@@ -93,14 +93,14 @@ defineTool({
93
93
  - `.env.schema` is the committed contract for local secret names.
94
94
  - AgentKit local commands load `.env` directly so inspect, chat, tools, and evals share the same secret loader.
95
95
  - Production uses managed secrets.
96
- - Hosted alpha deploys require `cloudflare_deploy_alpha`; local commands do not require login.
96
+ - Hosted deploys require `cloudflare_deploy_alpha` or purchased/manual deploy slots; local commands do not require login.
97
97
  - Secret values must not appear in config, docs, prompts, evals, logs, exports, or SQLite.
98
98
  - A tool receives only secrets listed in that tool.
99
99
  - Avoid direct `process.env` reads inside tools.
100
100
  - Use `permissions` to describe external capabilities.
101
101
  - Add timeouts to network tools.
102
- - Treat public deploy URLs as unauthenticated transport, not access control.
103
- - Use access tokens and limits for clients.
102
+ - Treat hosted deploy URLs as addresses, not access control.
103
+ - Use deploy access tokens for hosted chat, hosted conversation reads, hosted trace reads, and any client app that talks to AgentKit Cloud.
104
104
  - Remove client PII before writing evals.
105
105
 
106
106
  ## Verification
@@ -132,9 +132,9 @@ Add the secret name to the tool `secrets` field and set it locally:
132
132
  npm run agentkit -- inspect
133
133
  ```
134
134
 
135
- Public URL is reachable:
135
+ Hosted URL is reachable without a token:
136
136
 
137
- Check hosted access mode. Default should be `private`.
137
+ Treat this as a security bug. Hosted AgentKit Cloud routes should reject missing deploy tokens except authenticated channel webhook ingress.
138
138
 
139
139
  ## Backend Contracts Used
140
140
 
@@ -21,6 +21,14 @@ npm run chat -- --message "hello"
21
21
  npm run agentkit -- inspect
22
22
  ```
23
23
 
24
+ On Windows PowerShell, if `npm.ps1` is blocked by `PSSecurityException`, use the Windows command shim:
25
+
26
+ ```sh
27
+ npm.cmd run typecheck
28
+ npm.cmd run chat -- --message "hello"
29
+ npm.cmd run agentkit -- inspect
30
+ ```
31
+
24
32
  ## Files Created Or Edited
25
33
 
26
34
  Edit:
@@ -77,6 +85,8 @@ provider: {
77
85
  secrets: ["OPENROUTER_API_KEY"],
78
86
  ```
79
87
 
88
+ Prefer model ids or aliases listed by the installed Pi SDK when available, such as `~google/gemini-flash-latest`. If an OpenRouter model id is newer than the Pi model registry, AgentKit passes the id through to OpenRouter using Pi's OpenAI-compatible transport with conservative unknown-model metadata. The provider may still reject the request if the id is invalid, inaccessible, or does not support the tools/features the agent uses.
89
+
80
90
  ## UI Verification
81
91
 
82
92
  After the provider is configured and the local secret is set, test through chat and UI:
@@ -125,7 +135,7 @@ npm run agentkit -- inspect
125
135
 
126
136
  `provider_model_unsupported`:
127
137
 
128
- Use a model id known to the installed Pi SDK for that provider.
138
+ For OpenAI or Anthropic, use a model id known to the installed Pi SDK for that provider. For OpenRouter, prefer a known Pi alias when possible; otherwise a raw OpenRouter model id is passed through and any remaining model error comes from OpenRouter.
129
139
 
130
140
  Provider returns auth failure:
131
141
 
@@ -49,11 +49,16 @@ agentkit db reset --yes
49
49
  agentkit db shell
50
50
  agentkit db seed [--file <path>]
51
51
  agentkit eval run
52
+ agentkit eval from-conversation <conversation-id> [--out <path>] [--force]
53
+ agentkit improve collect [--deploy] [--since <duration|iso>] [--conversation-id <id>] [--out <directory>]
54
+ agentkit improve evals <bundle-dir-or-json> [--force]
55
+ agentkit replay <bundle-dir-or-json> --against local
52
56
  agentkit conversations list
53
57
  agentkit conversations show <conversation-id>
54
58
  agentkit conversations trace <conversation-id> [--deploy]
55
59
  agentkit channels list
56
- agentkit channels add <website|telegram|whatsapp> <name> [--provider zapster|meta] [--api <url>]
60
+ agentkit channels add <website|telegram|whatsapp|discord> <name> [--provider zapster|meta] [--mode interactions|bot] [--api <url>]
61
+ agentkit channels connect <website|telegram|whatsapp|discord> <name> [--provider zapster|meta] [--mode interactions|bot] [--api <url>]
57
62
  agentkit channels setup <name> [--apply] [--api <url>]
58
63
  agentkit channels status <name> [--api <url>]
59
64
  agentkit channels test <name> [--message <text>] [--fixture <path>] [--api <url>]
@@ -66,8 +71,15 @@ agentkit skills status
66
71
  agentkit skills sync
67
72
  agentkit inspect
68
73
  agentkit build [--target cloudflare|container]
74
+ agentkit billing checkout --slots <count> [--email <email>] [--api <url>]
75
+ agentkit billing status <billing-intent-id> --secret <secret> [--api <url>]
76
+ agentkit billing claim <billing-intent-id> --secret <secret> [--token-name <name>] [--api <url>]
77
+ agentkit billing portal [--api <url>]
69
78
  agentkit login --token <token>
70
79
  agentkit logout
80
+ agentkit account token create <name> [--api <url>] [--use] [--out <path>]
81
+ agentkit account token list [--api <url>]
82
+ agentkit account token revoke <token-id> [--api <url>]
71
83
  agentkit deploy [--target cloudflare|vps] [--host <host>] [--api <url>] [--dry-run] [--anonymous] [--local-wrangler] [--smoke <message>]
72
84
  agentkit deploy doctor [--api <url>] [--anonymous]
73
85
  agentkit deploy smoke [--message <text>] [--api <url>]
@@ -100,13 +112,9 @@ agentkit help commands
100
112
 
101
113
  Prefer `env set --stdin` or `--from-env` for local secret values, and prefer `secret set --stdin`, `--from-env`, or `--from-local-env` for hosted secrets. Inline `<VALUE>` forms exist for simple non-sensitive values, but agents should avoid putting secrets in shell history.
102
114
 
103
- Managed Composio is configured with `composioManaged({...})` in `agentkit.config.ts` and is paid hosted AgentKit infrastructure. It requires a non-anonymous AgentKit Cloud deploy with `managed_composio`, uses one Composio settings profile per agent, injects `COMPOSIO_API_KEY` as an AgentKit-managed secret, validates toolkit auth configs during `agentkit deploy doctor`, and exposes the generated `agentkit_composio_execute` tool only for explicit configured action slugs. Calendar starters should include `GOOGLECALENDAR_EVENTS_LIST`, `GOOGLECALENDAR_CREATE_EVENT`, and `GOOGLECALENDAR_UPDATE_EVENT`; create-event calls must pass UTC `start_datetime` plus explicit duration. Managed external write actions require tool input `confirmed: true` by default unless the integration sets `confirmExternalWrites: false`. See `docs/guides/add-managed-composio.md`.
115
+ On Windows PowerShell, if `npm.ps1` or `npx.ps1` is blocked with `PSSecurityException`, run capsule scripts through the `.cmd` shims instead of changing the workflow. Examples: `npx.cmd @andreprado/agentkit@alpha new demo --template blank`, `npm.cmd run agentkit -- inspect`, `npm.cmd run agentkit -- knowledge sync`, and `npm.cmd run eval`.
104
116
 
105
- Planned commands described by the contract but not implemented yet:
106
-
107
- ```sh
108
- agentkit eval from-conversation <conversation-id>
109
- ```
117
+ Managed Composio is configured with `composioManaged({...})` in `agentkit.config.ts` and is paid hosted AgentKit infrastructure. It requires a non-anonymous AgentKit Cloud deploy with `managed_composio`, uses one Composio settings profile per agent, injects `COMPOSIO_API_KEY` as an AgentKit-managed secret, resolves toolkit auth configs from AgentKit Cloud, validates toolkit readiness during `agentkit deploy doctor`, and exposes the generated `agentkit_composio_execute` tool only for explicit configured action slugs. Calendar starters should include `GOOGLECALENDAR_EVENTS_LIST`, `GOOGLECALENDAR_CREATE_EVENT`, and `GOOGLECALENDAR_UPDATE_EVENT`; create-event calls must pass UTC `start_datetime` plus explicit duration. Managed external write actions require tool input `confirmed: true` by default unless the integration sets `confirmExternalWrites: false`. See `docs/guides/add-managed-composio.md`.
110
118
 
111
119
  ## Create And Test A Capsule
112
120
 
@@ -255,6 +263,8 @@ npm run chat -- --message "hello"
255
263
 
256
264
  If a provider key is missing, the runtime returns `secret_not_found`.
257
265
 
266
+ For OpenRouter, prefer model ids or aliases known to the installed Pi SDK, such as `~google/gemini-flash-latest`. If an OpenRouter id is newer than Pi's registry, AgentKit passes the raw id through to OpenRouter with conservative unknown-model metadata. OpenRouter can still reject invalid, inaccessible, or unsupported models, and unknown-model cost/capability metadata is not authoritative.
267
+
258
268
  ## Knowledge Contract
259
269
 
260
270
  Knowledge is AgentKit's native retrieval layer for facts the agent should ground in source files. Use it for FAQs, prices, policies, service descriptions, procedures, CSV tables, and reference docs. Do not put secrets, credentials, `.env` contents, or live customer/payment records in Knowledge. Use tools for live or authorization-sensitive data.
@@ -304,7 +314,7 @@ agentkit knowledge inspect
304
314
  agentkit knowledge search "refund policy" --top-k 3
305
315
  ```
306
316
 
307
- `knowledge add` indexes one local path. `knowledge sync` indexes all configured `knowledge.sources` and skips unchanged files by content hash. `agentkit dev` and `agentkit chat` also sync configured Knowledge automatically before local runs. `knowledge inspect` lists indexed sources and chunk counts. `knowledge search` validates retrieval before relying on the agent. When embeddings are configured locally, AgentKit stores canonical chunks in `.agentkit/agentkit.db`, rebuilds a local libSQL vector sidecar at `.agentkit/agentkit.vectors.db`, uses native `libsql_vector_idx` semantic search, and falls back to stored JSON embeddings if the native vector path is unavailable.
317
+ `knowledge add` indexes one local path. `knowledge sync` indexes all configured `knowledge.sources` and skips unchanged files by content hash. `agentkit dev` and `agentkit chat` also sync configured Knowledge automatically before local runs. `knowledge inspect` lists indexed sources and chunk counts. `knowledge search` validates retrieval before relying on the agent. Local lexical search uses SQLite FTS5 when the local SQLite build provides it; when it does not, AgentKit automatically keeps indexing and searching with a normal SQLite table and simpler text matching. When embeddings are configured locally, AgentKit stores canonical chunks in `.agentkit/agentkit.db`, rebuilds a local libSQL vector sidecar at `.agentkit/agentkit.vectors.db`, uses native `libsql_vector_idx` semantic search, and falls back to stored JSON embeddings if the native vector path is unavailable.
308
318
 
309
319
  When `knowledge` is configured, AgentKit automatically registers the internal chat tool `agentkit_search_knowledge` and appends a prompt policy. The policy tells the agent to search before answering business-specific factual questions and not to expose raw retrieval JSON, scores, chunk IDs, or tool output objects. With `test/fake`, verify the internal tool directly:
310
320
 
@@ -329,7 +339,7 @@ Channels are hosted inbound/outbound conversation transports. They are separate
329
339
  Use these helpers in `agentkit.config.ts`:
330
340
 
331
341
  ```ts
332
- import { defineAgent, telegramChannel, whatsappChannel, websiteChannel } from "@andreprado/agentkit";
342
+ import { defineAgent, discordChannel, telegramChannel, whatsappChannel, websiteChannel } from "@andreprado/agentkit";
333
343
 
334
344
  export default defineAgent({
335
345
  name: "support-agent",
@@ -342,6 +352,8 @@ export default defineAgent({
342
352
  websiteChannel({ name: "website-chat" }),
343
353
  telegramChannel({ name: "support-telegram" }),
344
354
  whatsappChannel({ name: "support-whatsapp", provider: "zapster" }),
355
+ discordChannel({ name: "support-discord" }),
356
+ discordChannel({ name: "server-discord", mode: "bot" }),
345
357
  ],
346
358
  access: { mode: "public" },
347
359
  storage: { driver: "agentkit" },
@@ -377,6 +389,7 @@ whatsappChannel({
377
389
  Useful guides:
378
390
 
379
391
  - Add a channel: `docs/guides/add-channel.md`
392
+ - Connect Discord: `docs/guides/connect-discord.md`
380
393
  - Connect Telegram: `docs/guides/connect-telegram.md`
381
394
  - Connect WhatsApp through Zapster: `docs/guides/connect-whatsapp-zapster.md`
382
395
  - Debug a channel: `docs/guides/debug-channel.md`
@@ -397,6 +410,18 @@ ZAPSTER_INSTANCE_ID
397
410
  ZAPSTER_WEBHOOK_ID
398
411
  ```
399
412
 
413
+ Discord slash-command required secret:
414
+
415
+ ```txt
416
+ DISCORD_PUBLIC_KEY
417
+ ```
418
+
419
+ Discord bot-mode required secret:
420
+
421
+ ```txt
422
+ DISCORD_BOT_TOKEN
423
+ ```
424
+
400
425
  Common channel verification:
401
426
 
402
427
  ```sh
@@ -404,6 +429,8 @@ agentkit inspect
404
429
  agentkit deploy
405
430
  agentkit channels list
406
431
  agentkit channels add telegram support-telegram
432
+ agentkit channels connect discord support-discord
433
+ agentkit channels connect discord server-discord --mode bot
407
434
  agentkit channels setup support-telegram
408
435
  agentkit channels test support-telegram --message "hello"
409
436
  agentkit channels test-audio support-telegram --fixture voice-note
@@ -414,6 +441,8 @@ agentkit channels deliveries show <delivery-id>
414
441
 
415
442
  `channels connect` creates or refreshes the channel resource, validates secrets, runs provider setup when supported, then runs the official synthetic smoke. `channels setup` is read-only by default. `channels setup <telegram-name> --apply` calls Telegram `setWebhook` and requires `TELEGRAM_BOT_TOKEN` plus `TELEGRAM_WEBHOOK_SECRET`.
416
443
 
444
+ Discord slash-command mode validates `X-Signature-Ed25519` and `X-Signature-Timestamp` against `DISCORD_PUBLIC_KEY`, answers signed `PING` requests with `type: 1`, acknowledges slash commands with a deferred response, then sends the final answer as an interaction follow-up. Discord bot mode uses `DISCORD_BOT_TOKEN`, Discord Gateway `MESSAGE_CREATE`, Message Content Intent, and `/channels/<channel_id>/messages` bot replies. Discord channels support buffering but do not support `audio` in V1.
445
+
417
446
  Default tests are offline. Real provider smoke tests are opt-in:
418
447
 
419
448
  ```sh
@@ -755,6 +784,62 @@ For date-sensitive evals, set top-level `now` to an ISO timestamp with an explic
755
784
 
756
785
  Evals run the normal capsule tools. If a tool would write externally, delete, charge money, send email, or call a real customer system, make its `execute` implementation branch on `ctx.runtime.environment === "eval"` and return deterministic non-destructive output for eval runs. Do not invent an eval-only mock API; keep the behavior inside the registered tool contract unless AgentKit adds a first-class mock facility later.
757
786
 
787
+ ## Improve From Production
788
+
789
+ Use AgentKit Improve when a hosted or local conversation should become a reproducible local fix loop. Hosted AgentKit Cloud exports evidence; the local coding agent edits the Agent Capsule, writes evals, replays, and deploys.
790
+
791
+ Collect hosted evidence from the last deploy:
792
+
793
+ ```sh
794
+ agentkit improve collect --deploy --since 24h
795
+ ```
796
+
797
+ When the CLI is logged in to AgentKit Cloud, hosted collection first exports deploy evidence such as failed channel deliveries, deploy errors, and conversation IDs from the control plane, then reads replayable conversation traces from the deployed runtime. Without Cloud auth, it falls back to deploy-token conversation trace reads.
798
+
799
+ Hosted conversation reads require a deploy access token even when the chat endpoint is public. `agentkit deploy` normally writes `.agentkit/chat-access-token.json`; use `agentkit access token create agentkit-chat-ui --out .agentkit/chat-access-token.json` to refresh it.
800
+
801
+ Collect one hosted conversation:
802
+
803
+ ```sh
804
+ agentkit improve collect --deploy --conversation-id <conversation-id>
805
+ ```
806
+
807
+ Collect local evidence:
808
+
809
+ ```sh
810
+ agentkit improve collect --since 7d
811
+ ```
812
+
813
+ The command writes ignored local state:
814
+
815
+ ```txt
816
+ .agentkit/improve/<run>/
817
+ bundle.json
818
+ report.json
819
+ traces/
820
+ ```
821
+
822
+ Generate committed regression evals:
823
+
824
+ ```sh
825
+ agentkit improve evals .agentkit/improve/<run>
826
+ ```
827
+
828
+ Then replay before deploying:
829
+
830
+ ```sh
831
+ agentkit replay .agentkit/improve/<run> --against local
832
+ npm run eval
833
+ agentkit deploy --smoke "hello"
834
+ ```
835
+
836
+ Replay runs collected user turns through the local capsule with `ctx.runtime.environment === "eval"` and `ctx.runtime.invocation === "eval"`. Generated evals live under `evals/regressions/`; review them before committing, especially when traces contain real client details or overly strict prose assertions.
837
+
838
+ Full guides:
839
+
840
+ - `docs/guides/improve-from-production.md`
841
+ - `docs/guides/replay-production-traces.md`
842
+
758
843
  ## Security Rules
759
844
 
760
845
  Never commit:
@@ -778,7 +863,7 @@ prompts/
778
863
  docs/
779
864
  ```
780
865
 
781
- Local `.env` is development only. Use `.env.schema` as the committed secret-name contract; local AgentKit commands load `.env` directly. Hosted alpha deploys use `agentkit login --token ...` and managed secrets through `agentkit secret set/list/unset` or `agentkit secret sync --from-local`. Prefer `--stdin`, `--from-env`, `--from-local-env`, or sync from local `.env` so secret values do not appear in shell history. Inline `<VALUE>` forms exist only for compatibility and simple non-sensitive values.
866
+ Local `.env` is development only. Use `.env.schema` as the committed secret-name contract; local AgentKit commands load `.env` directly. Hosted deploys require an AgentKit Cloud account with `cloudflare_deploy_alpha` or purchased/manual deploy slots. First-time paid access uses `agentkit billing checkout --slots <count>` and `agentkit billing claim billint_... --secret bsec_...`; existing accounts use `agentkit login --token ...`. Hosted secrets use `agentkit secret set/list/unset` or `agentkit secret sync --from-local`. Prefer `--stdin`, `--from-env`, `--from-local-env`, or sync from local `.env` so secret values do not appear in shell history. Inline `<VALUE>` forms exist only for compatibility and simple non-sensitive values.
782
867
 
783
868
  Tools are a security boundary. A tool must declare every secret it needs. The runtime injects only tool-declared secrets.
784
869
 
@@ -790,7 +875,10 @@ Current flow:
790
875
 
791
876
  ```sh
792
877
  agentkit deploy --dry-run
878
+ agentkit billing checkout --slots 1 --email user@example.com
879
+ agentkit billing claim billint_... --secret bsec_...
793
880
  agentkit login --token agk_user_...
881
+ agentkit account token create new-laptop --use
794
882
  agentkit deploy doctor
795
883
  agentkit deploy --smoke "hello"
796
884
  agentkit deploy status
@@ -798,7 +886,7 @@ agentkit deploy smoke --message "hello"
798
886
  agentkit chat-ui --deploy
799
887
  ```
800
888
 
801
- `agentkit deploy doctor` checks AgentKit Cloud login, `cloudflare_deploy_alpha`, online deploy capacity, hosted secrets, local `.env` names that still need `agentkit secret set`, managed Composio API/auth-config readiness by toolkit, and private-access runtime token handling. `agentkit deploy` sends the capsule to AgentKit Cloud, runs the same readiness check automatically before building and uploading, updates the current project deploy slot by default, and writes the local chat/UI deploy access token to `.agentkit/chat-access-token.json` for private hosted deploys. `agentkit deploy --smoke "hello"` deploys and then tests `/v1/chat` with the deploy access token. `agentkit deploy smoke --message "hello"` repeats that smoke against the last local deploy. `agentkit chat-ui --deploy` serves a local UI pointed at the hosted deploy using that token without exposing it to browser code, shows the conversation id and tool calls, and supports starting a new conversation. Use `agentkit conversations trace <conversation-id> --deploy` to pull hosted conversation messages and tool calls from the last deploy. Production alpha deploys require an account with `cloudflare_deploy_alpha`; local commands and dry-run builds do not require login. AgentKit owns infrastructure selection, backend migration, managed secrets, and public URL creation.
889
+ `agentkit deploy doctor` checks AgentKit Cloud login, hosted deploy entitlement, online deploy capacity, hosted secrets, local `.env` names that still need `agentkit secret set`, managed Composio API/auth-config readiness by toolkit, and private-access runtime token handling. `agentkit deploy` sends the capsule to AgentKit Cloud, runs the same readiness check automatically before building and uploading, updates the current project deploy slot by default, and writes the local chat/UI deploy access token to `.agentkit/chat-access-token.json` for hosted deploys. `agentkit deploy --smoke "hello"` deploys and then tests `/v1/chat` with the deploy access token. `agentkit deploy smoke --message "hello"` repeats that smoke against the last local deploy. `agentkit chat-ui --deploy` serves a local UI pointed at the hosted deploy using that token without exposing it to browser code, shows the conversation id and tool calls, and supports starting a new conversation. Use `agentkit conversations trace <conversation-id> --deploy` to pull hosted conversation messages and tool calls from the last deploy. Production deploys require an account with `cloudflare_deploy_alpha` or purchased/manual deploy slots; local commands and dry-run builds do not require login. AgentKit owns infrastructure selection, backend migration, managed secrets, and public URL creation.
802
890
 
803
891
  The CLI defaults to the hosted AgentKit Cloud API at `https://agentkit-cloud.aibuilders.com.br`. Use `AGENTKIT_CLOUD_API_URL` or `agentkit deploy --api <url>` only when the owner gives you a non-default AgentKit Cloud API URL.
804
892
 
@@ -808,6 +896,7 @@ Account/access flow:
808
896
  agentkit secret set OPENAI_API_KEY --from-local-env
809
897
  agentkit secret sync --from-local
810
898
  agentkit secret list
899
+ agentkit account token list
811
900
  agentkit skills status
812
901
  agentkit skills sync
813
902
  agentkit access token create website-chat --out .agentkit/website-chat-access-token.json
package/docs/llms.txt CHANGED
@@ -10,6 +10,8 @@ Task guides:
10
10
  - Add a TypeScript tool: `docs/guides/add-tool.md`
11
11
  - Add Knowledge from local docs/CSVs: `docs/guides/add-knowledge.md`
12
12
  - Run or prepare evals: `docs/guides/run-evals.md`
13
+ - Improve from production traces: `docs/guides/improve-from-production.md`
14
+ - Replay collected traces locally: `docs/guides/replay-production-traces.md`
13
15
  - Switch from `test/fake` to a real provider: `docs/guides/use-provider.md`
14
16
  - Prepare for hosted deploy: `docs/guides/prepare-deploy.md`
15
17
  - Build Cloudflare artifact: `agentkit build --target cloudflare`
@@ -18,6 +20,7 @@ Task guides:
18
20
  - Add hosted channels: `docs/guides/add-channel.md`
19
21
  - Add AgentKit-managed Composio: `docs/guides/add-managed-composio.md`
20
22
  - Buffer rapid channel messages: `docs/guides/add-channel.md#buffer-bursty-messages`
23
+ - Connect Discord: `docs/guides/connect-discord.md`
21
24
  - Connect Telegram: `docs/guides/connect-telegram.md`
22
25
  - Connect WhatsApp through Zapster: `docs/guides/connect-whatsapp-zapster.md`
23
26
  - Follow channel webhook and delivery-log safety rules: `docs/guides/channel-security.md`
@@ -40,11 +43,16 @@ agentkit knowledge sync
40
43
  agentkit knowledge inspect
41
44
  agentkit knowledge search <query> [--top-k <number>]
42
45
  agentkit eval run
46
+ agentkit improve collect --deploy --since 24h
47
+ agentkit improve evals .agentkit/improve/<run>
48
+ agentkit replay .agentkit/improve/<run> --against local
43
49
  agentkit conversations list
44
50
  agentkit conversations show <conversation-id>
45
51
  agentkit conversations trace <conversation-id> [--deploy]
46
52
  agentkit channels list
47
53
  agentkit channels add telegram support-telegram
54
+ agentkit channels connect discord support-discord
55
+ agentkit channels connect discord server-discord --mode bot
48
56
  agentkit channels setup support-telegram
49
57
  agentkit channels status support-telegram
50
58
  agentkit channels test support-telegram --message "hello"
@@ -55,7 +63,11 @@ agentkit integrations status [--toolkit googlecalendar]
55
63
  agentkit integrations connect composio --toolkit gmail
56
64
  agentkit skills status
57
65
  agentkit skills sync
66
+ agentkit billing checkout --slots 1 --email user@example.com
67
+ agentkit billing claim billint_... --secret bsec_...
58
68
  agentkit login --token agk_user_...
69
+ agentkit account token create new-laptop --use
70
+ agentkit account token list
59
71
  agentkit deploy doctor
60
72
  agentkit deploy --smoke "hello"
61
73
  agentkit deploy status
@@ -71,6 +83,8 @@ agentkit dev
71
83
  agentkit open
72
84
  ```
73
85
 
86
+ On Windows PowerShell, if `npm.ps1` or `npx.ps1` is blocked with `PSSecurityException`, run capsule scripts through the `.cmd` shims, for example `npm.cmd run agentkit -- inspect`, `npm.cmd run agentkit -- knowledge sync`, or `npm.cmd run eval`.
87
+
74
88
  Generated capsules include `AGENTKIT.md`, `AGENTS.md`, and a repo-local `skills/` pack so Codex, Claude Code, or another coding agent can treat the owner's natural-language request as the brief and start building immediately without loading the full contract by default. Start with `skills/agentkit-capsule/SKILL.md`, then load the task skill for the current work. `agentkit handoff codex "Develop an ophthalmology office intake agent"` is an optional prompt-printing shortcut for users who are not already inside a coding-agent workspace. There is no wizard or recipe layer: the coding agent edits the capsule directly from the scaffold and contract.
75
89
 
76
90
  UI testing is part of the handoff. For local UI testing, run `agentkit dev`, open the printed `Chat:` URL, and tell the owner the exact URL. After hosted deploy, run `agentkit chat-ui --deploy`, open the printed `Chat:` URL, and tell the owner it is connected to the deploy.
@@ -79,6 +93,8 @@ Hosted Chat UI shows the current conversation id, tool calls, tool errors, and a
79
93
 
80
94
  `test/fake` is deterministic and validates scaffold, direct tool calls, and fake-provider evals. It does not validate natural conversation quality. Before claiming real conversation behavior has been tested, ask the owner which provider to use: OpenRouter, OpenAI, Anthropic, or another supported provider. Do not choose for them.
81
95
 
96
+ For OpenRouter, prefer model ids or aliases known to the installed Pi SDK, such as `~google/gemini-flash-latest`. If an OpenRouter id is newer than Pi's registry, AgentKit passes the raw id through to OpenRouter with conservative unknown-model metadata; OpenRouter can still reject invalid, inaccessible, or unsupported models.
97
+
82
98
  AgentKit injects the current ISO timestamp, local date, weekday, local date/time, and timezone dynamically into every chat run. Set `timeZone` in `agentkit.config.ts` for scheduling agents so "today", "tomorrow", and weekdays resolve in the business/user timezone; otherwise AgentKit falls back to `AGENTKIT_TIME_ZONE`, valid `TZ`, then the runtime default. Do not hardcode today's date in prompts.
83
99
 
84
100
  Current local endpoints from `agentkit dev`:
@@ -91,4 +107,4 @@ GET /v1/conversations/:id
91
107
  GET /v1/conversations/:id/trace
92
108
  ```
93
109
 
94
- Local `.env` is for development only. Use `.env.schema` as the committed secret-name contract; local AgentKit commands load `.env` directly. Hosted alpha deploys require `cloudflare_deploy_alpha` through `agentkit login --token ...`; production uses managed secrets in the hosted contract.
110
+ Local `.env` is for development only. Use `.env.schema` as the committed secret-name contract; local AgentKit commands load `.env` directly. Hosted deploys require an AgentKit Cloud account with `cloudflare_deploy_alpha` or purchased/manual deploy slots. Use `agentkit billing checkout` + `agentkit billing claim` for first-time paid access, or `agentkit login --token ...` when the user already has an `agk_user_...` token. Production uses managed secrets in the hosted contract.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@andreprado/agentkit",
3
- "version": "0.1.0-alpha.18",
3
+ "version": "0.1.0-alpha.19",
4
4
  "private": false,
5
5
  "type": "module",
6
6
  "repository": {
package/src/cli/args.ts CHANGED
@@ -4,6 +4,15 @@ export type ParsedArgs = {
4
4
  flags: Record<string, string | boolean>;
5
5
  };
6
6
 
7
+ const dashPrefixedValueFlags = new Set([
8
+ "brief",
9
+ "input",
10
+ "input-json",
11
+ "message",
12
+ "prompt",
13
+ "smoke",
14
+ ]);
15
+
7
16
  export function parseArgs(argv: string[]): ParsedArgs {
8
17
  const [command, ...rest] = argv;
9
18
  const positional: string[] = [];
@@ -12,15 +21,27 @@ export function parseArgs(argv: string[]): ParsedArgs {
12
21
  for (let index = 0; index < rest.length; index += 1) {
13
22
  const value = rest[index];
14
23
 
24
+ if (value === "--") {
25
+ positional.push(...rest.slice(index + 1));
26
+ break;
27
+ }
28
+
15
29
  if (!value.startsWith("--")) {
16
30
  positional.push(value);
17
31
  continue;
18
32
  }
19
33
 
20
- const name = value.slice(2);
34
+ const equalsIndex = value.indexOf("=");
35
+ const name = equalsIndex === -1 ? value.slice(2) : value.slice(2, equalsIndex);
36
+
37
+ if (equalsIndex !== -1) {
38
+ flags[name] = value.slice(equalsIndex + 1);
39
+ continue;
40
+ }
41
+
21
42
  const next = rest[index + 1];
22
43
 
23
- if (next && !next.startsWith("--")) {
44
+ if (next && (dashPrefixedValueFlags.has(name) || !next.startsWith("--"))) {
24
45
  flags[name] = next;
25
46
  index += 1;
26
47
  } else {
@@ -47,6 +47,7 @@ export type CloudManagedComposioResponse = {
47
47
  toolkit?: string;
48
48
  auth_config_env?: string;
49
49
  auth_config?: "set" | "missing" | string;
50
+ auth_config_source?: "project" | "account" | "global" | "env" | string;
50
51
  }>;
51
52
  };
52
53
  };
@@ -84,6 +85,68 @@ export type DeployAccessTokenListResponse = {
84
85
  }>;
85
86
  };
86
87
 
88
+ export type AccountApiTokenCreateResponse = {
89
+ token?: {
90
+ id?: string;
91
+ account_id?: string;
92
+ name?: string;
93
+ token?: string;
94
+ created_at?: string;
95
+ expires_at?: string;
96
+ last_used_at?: string;
97
+ };
98
+ };
99
+
100
+ export type AccountApiTokenListResponse = {
101
+ tokens?: Array<{
102
+ id?: string;
103
+ account_id?: string;
104
+ name?: string;
105
+ created_at?: string;
106
+ expires_at?: string;
107
+ last_used_at?: string;
108
+ revoked_at?: string;
109
+ }>;
110
+ };
111
+
112
+ export type BillingCheckoutResponse = {
113
+ checkout?: {
114
+ id?: string;
115
+ url?: string;
116
+ };
117
+ billing_intent?: {
118
+ id?: string;
119
+ account_id?: string;
120
+ email?: string;
121
+ requested_slots?: number;
122
+ status?: string;
123
+ secret?: string;
124
+ };
125
+ };
126
+
127
+ export type BillingCheckoutIntentResponse = {
128
+ billing_intent?: {
129
+ id?: string;
130
+ account_id?: string;
131
+ email?: string;
132
+ requested_slots?: number;
133
+ status?: string;
134
+ checkout_session_id?: string;
135
+ stripe_customer_id?: string;
136
+ stripe_subscription_id?: string;
137
+ completed_at?: string;
138
+ token_claimed_at?: string;
139
+ };
140
+ };
141
+
142
+ export type BillingClaimResponse = BillingCheckoutIntentResponse & AccountApiTokenCreateResponse;
143
+
144
+ export type BillingPortalResponse = {
145
+ portal?: {
146
+ url?: string;
147
+ };
148
+ };
149
+
87
150
  export type CloudSecretsResponse = {
88
151
  secrets?: Array<{
89
152
  name?: string;