@andreprado/agentkit 0.1.0-alpha.18 → 0.1.0-alpha.19
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +3 -0
- package/docs/guides/add-channel.md +12 -6
- package/docs/guides/add-knowledge.md +10 -0
- package/docs/guides/add-managed-composio.md +4 -2
- package/docs/guides/channel-security.md +26 -2
- package/docs/guides/connect-discord.md +178 -0
- package/docs/guides/create-agent.md +13 -0
- package/docs/guides/debug-channel.md +8 -2
- package/docs/guides/improve-from-production.md +151 -0
- package/docs/guides/prepare-deploy.md +29 -8
- package/docs/guides/replay-production-traces.md +72 -0
- package/docs/guides/run-evals.md +18 -0
- package/docs/guides/security-rules.md +5 -5
- package/docs/guides/use-provider.md +11 -1
- package/docs/llms-full.txt +100 -11
- package/docs/llms.txt +17 -1
- package/package.json +1 -1
- package/src/cli/args.ts +23 -2
- package/src/cli/cloud-client.ts +63 -0
- package/src/cli/commands/channels.ts +139 -14
- package/src/cli/deploy-readiness.ts +5 -2
- package/src/cli/help.ts +26 -2
- package/src/cli/index.ts +382 -17
- package/src/create-project.ts +13 -2
- package/src/index.ts +42 -3
- package/src/providers/pi.ts +49 -15
- package/src/runtime/channel-test-harness.ts +4 -1
- package/src/runtime/channels/discord.ts +887 -0
- package/src/runtime/channels.ts +15 -0
- package/src/runtime/config.ts +35 -3
- package/src/runtime/dev-server.ts +149 -8
- package/src/runtime/evals.ts +27 -6
- package/src/runtime/improve.ts +868 -0
- package/src/runtime/knowledge/retrieve.ts +25 -5
- package/src/runtime/knowledge/schema.ts +45 -1
- package/src/runtime/runtime-contract.ts +54 -0
- package/src/runtime/targets/cloudflare/build.ts +248 -193
- package/src/runtime/targets/vps/deploy.ts +1 -1
- package/src/storage/sqlite.ts +7 -2
- package/src/templates/skills/agentkit-capsule/SKILL.md +8 -1
- package/src/templates/skills/agentkit-capsule/references/docs-router.md +1 -2
- package/src/templates/skills/agentkit-channels/SKILL.md +6 -1
- package/src/templates/skills/agentkit-channels/references/channel-debugging.md +2 -1
- package/src/templates/skills/agentkit-channels/references/discord.md +93 -0
- package/src/templates/skills/agentkit-deploy/SKILL.md +6 -0
- package/src/templates/skills/agentkit-evals/SKILL.md +12 -3
- package/src/templates/skills/agentkit-improve/SKILL.md +86 -0
- package/src/templates/skills/agentkit-improve/references/replay-side-effects.md +18 -0
- package/src/templates/skills/agentkit-improve/references/trace-packets.md +22 -0
- package/src/templates/skills/agentkit-improve/templates/regression.eval.md +18 -0
- package/src/templates/skills/agentkit-integrations/SKILL.md +1 -0
- package/src/templates/skills/agentkit-knowledge/SKILL.md +4 -1
- package/src/templates/skills/agentkit-provider/SKILL.md +4 -1
- package/src/templates/skills/agentkit-security/SKILL.md +3 -2
- package/src/templates/skills/agentkit-troubleshooting/SKILL.md +9 -0
- package/src/templates/support.ts +4 -2
|
@@ -0,0 +1,72 @@
|
|
|
1
|
+
# Replay Production Traces
|
|
2
|
+
|
|
3
|
+
## Goal
|
|
4
|
+
|
|
5
|
+
Run collected production or local traces against the local Agent Capsule before redeploying a fix.
|
|
6
|
+
|
|
7
|
+
## When To Use This
|
|
8
|
+
|
|
9
|
+
Use this after `agentkit improve collect` and before `agentkit deploy` whenever prompts, tools, Knowledge, provider config, or channel behavior changed because of production evidence.
|
|
10
|
+
|
|
11
|
+
## Commands
|
|
12
|
+
|
|
13
|
+
```sh
|
|
14
|
+
agentkit replay .agentkit/improve/<run> --against local
|
|
15
|
+
```
|
|
16
|
+
|
|
17
|
+
Run the full local eval suite too:
|
|
18
|
+
|
|
19
|
+
```sh
|
|
20
|
+
npm run eval
|
|
21
|
+
```
|
|
22
|
+
|
|
23
|
+
## What Replay Checks
|
|
24
|
+
|
|
25
|
+
Replay sends each collected user turn through the local capsule in eval mode:
|
|
26
|
+
|
|
27
|
+
```txt
|
|
28
|
+
ctx.runtime.environment === "eval"
|
|
29
|
+
ctx.runtime.invocation === "eval"
|
|
30
|
+
```
|
|
31
|
+
|
|
32
|
+
This proves the current capsule can process the production turns without runtime errors. Generated eval files under `evals/regressions/` add committed behavior assertions.
|
|
33
|
+
|
|
34
|
+
## Write Tool Safety
|
|
35
|
+
|
|
36
|
+
Replay still runs the registered capsule tools. Any tool that can write externally must guard eval mode:
|
|
37
|
+
|
|
38
|
+
```ts
|
|
39
|
+
if (ctx.runtime.environment === "eval") {
|
|
40
|
+
return { sent: false, evalFixture: true };
|
|
41
|
+
}
|
|
42
|
+
```
|
|
43
|
+
|
|
44
|
+
Do not depend on prompt wording alone to prevent side effects.
|
|
45
|
+
|
|
46
|
+
## Deploy Gate
|
|
47
|
+
|
|
48
|
+
Before redeploying a production fix:
|
|
49
|
+
|
|
50
|
+
```sh
|
|
51
|
+
npm run typecheck
|
|
52
|
+
npm run agentkit -- inspect
|
|
53
|
+
npm run eval
|
|
54
|
+
agentkit replay .agentkit/improve/<run> --against local
|
|
55
|
+
agentkit deploy --smoke "hello"
|
|
56
|
+
```
|
|
57
|
+
|
|
58
|
+
If replay fails, inspect the failed trace id in `.agentkit/improve/<run>/traces/`, patch the capsule, and rerun replay.
|
|
59
|
+
|
|
60
|
+
## Troubleshooting
|
|
61
|
+
|
|
62
|
+
`Replay failed`:
|
|
63
|
+
|
|
64
|
+
Read the printed error and the matching trace file. Common causes are missing local secrets, unsafe tools that do not branch on eval mode, stale Knowledge sources, or provider differences.
|
|
65
|
+
|
|
66
|
+
`Replay skipped`:
|
|
67
|
+
|
|
68
|
+
The trace had no user message. It may still help diagnose delivery or deploy state, but it cannot be replayed as a conversation.
|
|
69
|
+
|
|
70
|
+
`provider_model_unsupported`:
|
|
71
|
+
|
|
72
|
+
The local provider config does not match the replay environment. Keep deterministic regression replay on `test/fake` unless the owner intentionally selected a real provider.
|
package/docs/guides/run-evals.md
CHANGED
|
@@ -22,6 +22,12 @@ Run evals:
|
|
|
22
22
|
agentkit eval run
|
|
23
23
|
```
|
|
24
24
|
|
|
25
|
+
If the capsule uses npm scripts and Windows PowerShell blocks `npm.ps1`, use:
|
|
26
|
+
|
|
27
|
+
```sh
|
|
28
|
+
npm.cmd run eval
|
|
29
|
+
```
|
|
30
|
+
|
|
25
31
|
Create an eval from a stored conversation:
|
|
26
32
|
|
|
27
33
|
```sh
|
|
@@ -31,6 +37,15 @@ agentkit eval from-conversation <conversation-id>
|
|
|
31
37
|
agentkit eval run
|
|
32
38
|
```
|
|
33
39
|
|
|
40
|
+
Create evals from hosted or local production evidence:
|
|
41
|
+
|
|
42
|
+
```sh
|
|
43
|
+
agentkit improve collect --deploy --since 24h
|
|
44
|
+
agentkit improve evals .agentkit/improve/<run>
|
|
45
|
+
agentkit replay .agentkit/improve/<run> --against local
|
|
46
|
+
agentkit eval run
|
|
47
|
+
```
|
|
48
|
+
|
|
34
49
|
## Files Created Or Edited
|
|
35
50
|
|
|
36
51
|
Create or edit:
|
|
@@ -43,9 +58,11 @@ Use conversations as source material:
|
|
|
43
58
|
|
|
44
59
|
```txt
|
|
45
60
|
.agentkit/agentkit.db
|
|
61
|
+
.agentkit/improve/<run>/
|
|
46
62
|
```
|
|
47
63
|
|
|
48
64
|
Do not edit `.agentkit/agentkit.db` by hand.
|
|
65
|
+
Do not edit `.agentkit/improve/<run>/bundle.json` by hand.
|
|
49
66
|
|
|
50
67
|
## Minimal Working Example
|
|
51
68
|
|
|
@@ -200,6 +217,7 @@ export const sendFollowupEmail = defineTool({
|
|
|
200
217
|
- Keep eval tool calls deterministic and non-destructive.
|
|
201
218
|
- For tools that would write, delete, charge money, send email, or call a real customer system, branch inside the registered tool on `ctx.runtime.environment === "eval"` and return safe fixture output.
|
|
202
219
|
- Treat conversations as source material, not as automatically safe training data.
|
|
220
|
+
- Review generated regression evals from `agentkit improve evals` before committing them. AgentKit redacts common email, phone, bearer token, and key patterns in generated eval text, but you must still remove or generalize domain-specific client PII and replace brittle exact prose assertions with the important behavior when needed.
|
|
203
221
|
|
|
204
222
|
## Verification
|
|
205
223
|
|
|
@@ -93,14 +93,14 @@ defineTool({
|
|
|
93
93
|
- `.env.schema` is the committed contract for local secret names.
|
|
94
94
|
- AgentKit local commands load `.env` directly so inspect, chat, tools, and evals share the same secret loader.
|
|
95
95
|
- Production uses managed secrets.
|
|
96
|
-
- Hosted
|
|
96
|
+
- Hosted deploys require `cloudflare_deploy_alpha` or purchased/manual deploy slots; local commands do not require login.
|
|
97
97
|
- Secret values must not appear in config, docs, prompts, evals, logs, exports, or SQLite.
|
|
98
98
|
- A tool receives only secrets listed in that tool.
|
|
99
99
|
- Avoid direct `process.env` reads inside tools.
|
|
100
100
|
- Use `permissions` to describe external capabilities.
|
|
101
101
|
- Add timeouts to network tools.
|
|
102
|
-
- Treat
|
|
103
|
-
- Use access tokens and
|
|
102
|
+
- Treat hosted deploy URLs as addresses, not access control.
|
|
103
|
+
- Use deploy access tokens for hosted chat, hosted conversation reads, hosted trace reads, and any client app that talks to AgentKit Cloud.
|
|
104
104
|
- Remove client PII before writing evals.
|
|
105
105
|
|
|
106
106
|
## Verification
|
|
@@ -132,9 +132,9 @@ Add the secret name to the tool `secrets` field and set it locally:
|
|
|
132
132
|
npm run agentkit -- inspect
|
|
133
133
|
```
|
|
134
134
|
|
|
135
|
-
|
|
135
|
+
Hosted URL is reachable without a token:
|
|
136
136
|
|
|
137
|
-
|
|
137
|
+
Treat this as a security bug. Hosted AgentKit Cloud routes should reject missing deploy tokens except authenticated channel webhook ingress.
|
|
138
138
|
|
|
139
139
|
## Backend Contracts Used
|
|
140
140
|
|
|
@@ -21,6 +21,14 @@ npm run chat -- --message "hello"
|
|
|
21
21
|
npm run agentkit -- inspect
|
|
22
22
|
```
|
|
23
23
|
|
|
24
|
+
On Windows PowerShell, if `npm.ps1` is blocked by `PSSecurityException`, use the Windows command shim:
|
|
25
|
+
|
|
26
|
+
```sh
|
|
27
|
+
npm.cmd run typecheck
|
|
28
|
+
npm.cmd run chat -- --message "hello"
|
|
29
|
+
npm.cmd run agentkit -- inspect
|
|
30
|
+
```
|
|
31
|
+
|
|
24
32
|
## Files Created Or Edited
|
|
25
33
|
|
|
26
34
|
Edit:
|
|
@@ -77,6 +85,8 @@ provider: {
|
|
|
77
85
|
secrets: ["OPENROUTER_API_KEY"],
|
|
78
86
|
```
|
|
79
87
|
|
|
88
|
+
Prefer model ids or aliases listed by the installed Pi SDK when available, such as `~google/gemini-flash-latest`. If an OpenRouter model id is newer than the Pi model registry, AgentKit passes the id through to OpenRouter using Pi's OpenAI-compatible transport with conservative unknown-model metadata. The provider may still reject the request if the id is invalid, inaccessible, or does not support the tools/features the agent uses.
|
|
89
|
+
|
|
80
90
|
## UI Verification
|
|
81
91
|
|
|
82
92
|
After the provider is configured and the local secret is set, test through chat and UI:
|
|
@@ -125,7 +135,7 @@ npm run agentkit -- inspect
|
|
|
125
135
|
|
|
126
136
|
`provider_model_unsupported`:
|
|
127
137
|
|
|
128
|
-
|
|
138
|
+
For OpenAI or Anthropic, use a model id known to the installed Pi SDK for that provider. For OpenRouter, prefer a known Pi alias when possible; otherwise a raw OpenRouter model id is passed through and any remaining model error comes from OpenRouter.
|
|
129
139
|
|
|
130
140
|
Provider returns auth failure:
|
|
131
141
|
|
package/docs/llms-full.txt
CHANGED
|
@@ -49,11 +49,16 @@ agentkit db reset --yes
|
|
|
49
49
|
agentkit db shell
|
|
50
50
|
agentkit db seed [--file <path>]
|
|
51
51
|
agentkit eval run
|
|
52
|
+
agentkit eval from-conversation <conversation-id> [--out <path>] [--force]
|
|
53
|
+
agentkit improve collect [--deploy] [--since <duration|iso>] [--conversation-id <id>] [--out <directory>]
|
|
54
|
+
agentkit improve evals <bundle-dir-or-json> [--force]
|
|
55
|
+
agentkit replay <bundle-dir-or-json> --against local
|
|
52
56
|
agentkit conversations list
|
|
53
57
|
agentkit conversations show <conversation-id>
|
|
54
58
|
agentkit conversations trace <conversation-id> [--deploy]
|
|
55
59
|
agentkit channels list
|
|
56
|
-
agentkit channels add <website|telegram|whatsapp> <name> [--provider zapster|meta] [--api <url>]
|
|
60
|
+
agentkit channels add <website|telegram|whatsapp|discord> <name> [--provider zapster|meta] [--mode interactions|bot] [--api <url>]
|
|
61
|
+
agentkit channels connect <website|telegram|whatsapp|discord> <name> [--provider zapster|meta] [--mode interactions|bot] [--api <url>]
|
|
57
62
|
agentkit channels setup <name> [--apply] [--api <url>]
|
|
58
63
|
agentkit channels status <name> [--api <url>]
|
|
59
64
|
agentkit channels test <name> [--message <text>] [--fixture <path>] [--api <url>]
|
|
@@ -66,8 +71,15 @@ agentkit skills status
|
|
|
66
71
|
agentkit skills sync
|
|
67
72
|
agentkit inspect
|
|
68
73
|
agentkit build [--target cloudflare|container]
|
|
74
|
+
agentkit billing checkout --slots <count> [--email <email>] [--api <url>]
|
|
75
|
+
agentkit billing status <billing-intent-id> --secret <secret> [--api <url>]
|
|
76
|
+
agentkit billing claim <billing-intent-id> --secret <secret> [--token-name <name>] [--api <url>]
|
|
77
|
+
agentkit billing portal [--api <url>]
|
|
69
78
|
agentkit login --token <token>
|
|
70
79
|
agentkit logout
|
|
80
|
+
agentkit account token create <name> [--api <url>] [--use] [--out <path>]
|
|
81
|
+
agentkit account token list [--api <url>]
|
|
82
|
+
agentkit account token revoke <token-id> [--api <url>]
|
|
71
83
|
agentkit deploy [--target cloudflare|vps] [--host <host>] [--api <url>] [--dry-run] [--anonymous] [--local-wrangler] [--smoke <message>]
|
|
72
84
|
agentkit deploy doctor [--api <url>] [--anonymous]
|
|
73
85
|
agentkit deploy smoke [--message <text>] [--api <url>]
|
|
@@ -100,13 +112,9 @@ agentkit help commands
|
|
|
100
112
|
|
|
101
113
|
Prefer `env set --stdin` or `--from-env` for local secret values, and prefer `secret set --stdin`, `--from-env`, or `--from-local-env` for hosted secrets. Inline `<VALUE>` forms exist for simple non-sensitive values, but agents should avoid putting secrets in shell history.
|
|
102
114
|
|
|
103
|
-
|
|
115
|
+
On Windows PowerShell, if `npm.ps1` or `npx.ps1` is blocked with `PSSecurityException`, run capsule scripts through the `.cmd` shims instead of changing the workflow. Examples: `npx.cmd @andreprado/agentkit@alpha new demo --template blank`, `npm.cmd run agentkit -- inspect`, `npm.cmd run agentkit -- knowledge sync`, and `npm.cmd run eval`.
|
|
104
116
|
|
|
105
|
-
|
|
106
|
-
|
|
107
|
-
```sh
|
|
108
|
-
agentkit eval from-conversation <conversation-id>
|
|
109
|
-
```
|
|
117
|
+
Managed Composio is configured with `composioManaged({...})` in `agentkit.config.ts` and is paid hosted AgentKit infrastructure. It requires a non-anonymous AgentKit Cloud deploy with `managed_composio`, uses one Composio settings profile per agent, injects `COMPOSIO_API_KEY` as an AgentKit-managed secret, resolves toolkit auth configs from AgentKit Cloud, validates toolkit readiness during `agentkit deploy doctor`, and exposes the generated `agentkit_composio_execute` tool only for explicit configured action slugs. Calendar starters should include `GOOGLECALENDAR_EVENTS_LIST`, `GOOGLECALENDAR_CREATE_EVENT`, and `GOOGLECALENDAR_UPDATE_EVENT`; create-event calls must pass UTC `start_datetime` plus explicit duration. Managed external write actions require tool input `confirmed: true` by default unless the integration sets `confirmExternalWrites: false`. See `docs/guides/add-managed-composio.md`.
|
|
110
118
|
|
|
111
119
|
## Create And Test A Capsule
|
|
112
120
|
|
|
@@ -255,6 +263,8 @@ npm run chat -- --message "hello"
|
|
|
255
263
|
|
|
256
264
|
If a provider key is missing, the runtime returns `secret_not_found`.
|
|
257
265
|
|
|
266
|
+
For OpenRouter, prefer model ids or aliases known to the installed Pi SDK, such as `~google/gemini-flash-latest`. If an OpenRouter id is newer than Pi's registry, AgentKit passes the raw id through to OpenRouter with conservative unknown-model metadata. OpenRouter can still reject invalid, inaccessible, or unsupported models, and unknown-model cost/capability metadata is not authoritative.
|
|
267
|
+
|
|
258
268
|
## Knowledge Contract
|
|
259
269
|
|
|
260
270
|
Knowledge is AgentKit's native retrieval layer for facts the agent should ground in source files. Use it for FAQs, prices, policies, service descriptions, procedures, CSV tables, and reference docs. Do not put secrets, credentials, `.env` contents, or live customer/payment records in Knowledge. Use tools for live or authorization-sensitive data.
|
|
@@ -304,7 +314,7 @@ agentkit knowledge inspect
|
|
|
304
314
|
agentkit knowledge search "refund policy" --top-k 3
|
|
305
315
|
```
|
|
306
316
|
|
|
307
|
-
`knowledge add` indexes one local path. `knowledge sync` indexes all configured `knowledge.sources` and skips unchanged files by content hash. `agentkit dev` and `agentkit chat` also sync configured Knowledge automatically before local runs. `knowledge inspect` lists indexed sources and chunk counts. `knowledge search` validates retrieval before relying on the agent. When embeddings are configured locally, AgentKit stores canonical chunks in `.agentkit/agentkit.db`, rebuilds a local libSQL vector sidecar at `.agentkit/agentkit.vectors.db`, uses native `libsql_vector_idx` semantic search, and falls back to stored JSON embeddings if the native vector path is unavailable.
|
|
317
|
+
`knowledge add` indexes one local path. `knowledge sync` indexes all configured `knowledge.sources` and skips unchanged files by content hash. `agentkit dev` and `agentkit chat` also sync configured Knowledge automatically before local runs. `knowledge inspect` lists indexed sources and chunk counts. `knowledge search` validates retrieval before relying on the agent. Local lexical search uses SQLite FTS5 when the local SQLite build provides it; when it does not, AgentKit automatically keeps indexing and searching with a normal SQLite table and simpler text matching. When embeddings are configured locally, AgentKit stores canonical chunks in `.agentkit/agentkit.db`, rebuilds a local libSQL vector sidecar at `.agentkit/agentkit.vectors.db`, uses native `libsql_vector_idx` semantic search, and falls back to stored JSON embeddings if the native vector path is unavailable.
|
|
308
318
|
|
|
309
319
|
When `knowledge` is configured, AgentKit automatically registers the internal chat tool `agentkit_search_knowledge` and appends a prompt policy. The policy tells the agent to search before answering business-specific factual questions and not to expose raw retrieval JSON, scores, chunk IDs, or tool output objects. With `test/fake`, verify the internal tool directly:
|
|
310
320
|
|
|
@@ -329,7 +339,7 @@ Channels are hosted inbound/outbound conversation transports. They are separate
|
|
|
329
339
|
Use these helpers in `agentkit.config.ts`:
|
|
330
340
|
|
|
331
341
|
```ts
|
|
332
|
-
import { defineAgent, telegramChannel, whatsappChannel, websiteChannel } from "@andreprado/agentkit";
|
|
342
|
+
import { defineAgent, discordChannel, telegramChannel, whatsappChannel, websiteChannel } from "@andreprado/agentkit";
|
|
333
343
|
|
|
334
344
|
export default defineAgent({
|
|
335
345
|
name: "support-agent",
|
|
@@ -342,6 +352,8 @@ export default defineAgent({
|
|
|
342
352
|
websiteChannel({ name: "website-chat" }),
|
|
343
353
|
telegramChannel({ name: "support-telegram" }),
|
|
344
354
|
whatsappChannel({ name: "support-whatsapp", provider: "zapster" }),
|
|
355
|
+
discordChannel({ name: "support-discord" }),
|
|
356
|
+
discordChannel({ name: "server-discord", mode: "bot" }),
|
|
345
357
|
],
|
|
346
358
|
access: { mode: "public" },
|
|
347
359
|
storage: { driver: "agentkit" },
|
|
@@ -377,6 +389,7 @@ whatsappChannel({
|
|
|
377
389
|
Useful guides:
|
|
378
390
|
|
|
379
391
|
- Add a channel: `docs/guides/add-channel.md`
|
|
392
|
+
- Connect Discord: `docs/guides/connect-discord.md`
|
|
380
393
|
- Connect Telegram: `docs/guides/connect-telegram.md`
|
|
381
394
|
- Connect WhatsApp through Zapster: `docs/guides/connect-whatsapp-zapster.md`
|
|
382
395
|
- Debug a channel: `docs/guides/debug-channel.md`
|
|
@@ -397,6 +410,18 @@ ZAPSTER_INSTANCE_ID
|
|
|
397
410
|
ZAPSTER_WEBHOOK_ID
|
|
398
411
|
```
|
|
399
412
|
|
|
413
|
+
Discord slash-command required secret:
|
|
414
|
+
|
|
415
|
+
```txt
|
|
416
|
+
DISCORD_PUBLIC_KEY
|
|
417
|
+
```
|
|
418
|
+
|
|
419
|
+
Discord bot-mode required secret:
|
|
420
|
+
|
|
421
|
+
```txt
|
|
422
|
+
DISCORD_BOT_TOKEN
|
|
423
|
+
```
|
|
424
|
+
|
|
400
425
|
Common channel verification:
|
|
401
426
|
|
|
402
427
|
```sh
|
|
@@ -404,6 +429,8 @@ agentkit inspect
|
|
|
404
429
|
agentkit deploy
|
|
405
430
|
agentkit channels list
|
|
406
431
|
agentkit channels add telegram support-telegram
|
|
432
|
+
agentkit channels connect discord support-discord
|
|
433
|
+
agentkit channels connect discord server-discord --mode bot
|
|
407
434
|
agentkit channels setup support-telegram
|
|
408
435
|
agentkit channels test support-telegram --message "hello"
|
|
409
436
|
agentkit channels test-audio support-telegram --fixture voice-note
|
|
@@ -414,6 +441,8 @@ agentkit channels deliveries show <delivery-id>
|
|
|
414
441
|
|
|
415
442
|
`channels connect` creates or refreshes the channel resource, validates secrets, runs provider setup when supported, then runs the official synthetic smoke. `channels setup` is read-only by default. `channels setup <telegram-name> --apply` calls Telegram `setWebhook` and requires `TELEGRAM_BOT_TOKEN` plus `TELEGRAM_WEBHOOK_SECRET`.
|
|
416
443
|
|
|
444
|
+
Discord slash-command mode validates `X-Signature-Ed25519` and `X-Signature-Timestamp` against `DISCORD_PUBLIC_KEY`, answers signed `PING` requests with `type: 1`, acknowledges slash commands with a deferred response, then sends the final answer as an interaction follow-up. Discord bot mode uses `DISCORD_BOT_TOKEN`, Discord Gateway `MESSAGE_CREATE`, Message Content Intent, and `/channels/<channel_id>/messages` bot replies. Discord channels support buffering but do not support `audio` in V1.
|
|
445
|
+
|
|
417
446
|
Default tests are offline. Real provider smoke tests are opt-in:
|
|
418
447
|
|
|
419
448
|
```sh
|
|
@@ -755,6 +784,62 @@ For date-sensitive evals, set top-level `now` to an ISO timestamp with an explic
|
|
|
755
784
|
|
|
756
785
|
Evals run the normal capsule tools. If a tool would write externally, delete, charge money, send email, or call a real customer system, make its `execute` implementation branch on `ctx.runtime.environment === "eval"` and return deterministic non-destructive output for eval runs. Do not invent an eval-only mock API; keep the behavior inside the registered tool contract unless AgentKit adds a first-class mock facility later.
|
|
757
786
|
|
|
787
|
+
## Improve From Production
|
|
788
|
+
|
|
789
|
+
Use AgentKit Improve when a hosted or local conversation should become a reproducible local fix loop. Hosted AgentKit Cloud exports evidence; the local coding agent edits the Agent Capsule, writes evals, replays, and deploys.
|
|
790
|
+
|
|
791
|
+
Collect hosted evidence from the last deploy:
|
|
792
|
+
|
|
793
|
+
```sh
|
|
794
|
+
agentkit improve collect --deploy --since 24h
|
|
795
|
+
```
|
|
796
|
+
|
|
797
|
+
When the CLI is logged in to AgentKit Cloud, hosted collection first exports deploy evidence such as failed channel deliveries, deploy errors, and conversation IDs from the control plane, then reads replayable conversation traces from the deployed runtime. Without Cloud auth, it falls back to deploy-token conversation trace reads.
|
|
798
|
+
|
|
799
|
+
Hosted conversation reads require a deploy access token even when the chat endpoint is public. `agentkit deploy` normally writes `.agentkit/chat-access-token.json`; use `agentkit access token create agentkit-chat-ui --out .agentkit/chat-access-token.json` to refresh it.
|
|
800
|
+
|
|
801
|
+
Collect one hosted conversation:
|
|
802
|
+
|
|
803
|
+
```sh
|
|
804
|
+
agentkit improve collect --deploy --conversation-id <conversation-id>
|
|
805
|
+
```
|
|
806
|
+
|
|
807
|
+
Collect local evidence:
|
|
808
|
+
|
|
809
|
+
```sh
|
|
810
|
+
agentkit improve collect --since 7d
|
|
811
|
+
```
|
|
812
|
+
|
|
813
|
+
The command writes ignored local state:
|
|
814
|
+
|
|
815
|
+
```txt
|
|
816
|
+
.agentkit/improve/<run>/
|
|
817
|
+
bundle.json
|
|
818
|
+
report.json
|
|
819
|
+
traces/
|
|
820
|
+
```
|
|
821
|
+
|
|
822
|
+
Generate committed regression evals:
|
|
823
|
+
|
|
824
|
+
```sh
|
|
825
|
+
agentkit improve evals .agentkit/improve/<run>
|
|
826
|
+
```
|
|
827
|
+
|
|
828
|
+
Then replay before deploying:
|
|
829
|
+
|
|
830
|
+
```sh
|
|
831
|
+
agentkit replay .agentkit/improve/<run> --against local
|
|
832
|
+
npm run eval
|
|
833
|
+
agentkit deploy --smoke "hello"
|
|
834
|
+
```
|
|
835
|
+
|
|
836
|
+
Replay runs collected user turns through the local capsule with `ctx.runtime.environment === "eval"` and `ctx.runtime.invocation === "eval"`. Generated evals live under `evals/regressions/`; review them before committing, especially when traces contain real client details or overly strict prose assertions.
|
|
837
|
+
|
|
838
|
+
Full guides:
|
|
839
|
+
|
|
840
|
+
- `docs/guides/improve-from-production.md`
|
|
841
|
+
- `docs/guides/replay-production-traces.md`
|
|
842
|
+
|
|
758
843
|
## Security Rules
|
|
759
844
|
|
|
760
845
|
Never commit:
|
|
@@ -778,7 +863,7 @@ prompts/
|
|
|
778
863
|
docs/
|
|
779
864
|
```
|
|
780
865
|
|
|
781
|
-
Local `.env` is development only. Use `.env.schema` as the committed secret-name contract; local AgentKit commands load `.env` directly. Hosted
|
|
866
|
+
Local `.env` is development only. Use `.env.schema` as the committed secret-name contract; local AgentKit commands load `.env` directly. Hosted deploys require an AgentKit Cloud account with `cloudflare_deploy_alpha` or purchased/manual deploy slots. First-time paid access uses `agentkit billing checkout --slots <count>` and `agentkit billing claim billint_... --secret bsec_...`; existing accounts use `agentkit login --token ...`. Hosted secrets use `agentkit secret set/list/unset` or `agentkit secret sync --from-local`. Prefer `--stdin`, `--from-env`, `--from-local-env`, or sync from local `.env` so secret values do not appear in shell history. Inline `<VALUE>` forms exist only for compatibility and simple non-sensitive values.
|
|
782
867
|
|
|
783
868
|
Tools are a security boundary. A tool must declare every secret it needs. The runtime injects only tool-declared secrets.
|
|
784
869
|
|
|
@@ -790,7 +875,10 @@ Current flow:
|
|
|
790
875
|
|
|
791
876
|
```sh
|
|
792
877
|
agentkit deploy --dry-run
|
|
878
|
+
agentkit billing checkout --slots 1 --email user@example.com
|
|
879
|
+
agentkit billing claim billint_... --secret bsec_...
|
|
793
880
|
agentkit login --token agk_user_...
|
|
881
|
+
agentkit account token create new-laptop --use
|
|
794
882
|
agentkit deploy doctor
|
|
795
883
|
agentkit deploy --smoke "hello"
|
|
796
884
|
agentkit deploy status
|
|
@@ -798,7 +886,7 @@ agentkit deploy smoke --message "hello"
|
|
|
798
886
|
agentkit chat-ui --deploy
|
|
799
887
|
```
|
|
800
888
|
|
|
801
|
-
`agentkit deploy doctor` checks AgentKit Cloud login,
|
|
889
|
+
`agentkit deploy doctor` checks AgentKit Cloud login, hosted deploy entitlement, online deploy capacity, hosted secrets, local `.env` names that still need `agentkit secret set`, managed Composio API/auth-config readiness by toolkit, and private-access runtime token handling. `agentkit deploy` sends the capsule to AgentKit Cloud, runs the same readiness check automatically before building and uploading, updates the current project deploy slot by default, and writes the local chat/UI deploy access token to `.agentkit/chat-access-token.json` for hosted deploys. `agentkit deploy --smoke "hello"` deploys and then tests `/v1/chat` with the deploy access token. `agentkit deploy smoke --message "hello"` repeats that smoke against the last local deploy. `agentkit chat-ui --deploy` serves a local UI pointed at the hosted deploy using that token without exposing it to browser code, shows the conversation id and tool calls, and supports starting a new conversation. Use `agentkit conversations trace <conversation-id> --deploy` to pull hosted conversation messages and tool calls from the last deploy. Production deploys require an account with `cloudflare_deploy_alpha` or purchased/manual deploy slots; local commands and dry-run builds do not require login. AgentKit owns infrastructure selection, backend migration, managed secrets, and public URL creation.
|
|
802
890
|
|
|
803
891
|
The CLI defaults to the hosted AgentKit Cloud API at `https://agentkit-cloud.aibuilders.com.br`. Use `AGENTKIT_CLOUD_API_URL` or `agentkit deploy --api <url>` only when the owner gives you a non-default AgentKit Cloud API URL.
|
|
804
892
|
|
|
@@ -808,6 +896,7 @@ Account/access flow:
|
|
|
808
896
|
agentkit secret set OPENAI_API_KEY --from-local-env
|
|
809
897
|
agentkit secret sync --from-local
|
|
810
898
|
agentkit secret list
|
|
899
|
+
agentkit account token list
|
|
811
900
|
agentkit skills status
|
|
812
901
|
agentkit skills sync
|
|
813
902
|
agentkit access token create website-chat --out .agentkit/website-chat-access-token.json
|
package/docs/llms.txt
CHANGED
|
@@ -10,6 +10,8 @@ Task guides:
|
|
|
10
10
|
- Add a TypeScript tool: `docs/guides/add-tool.md`
|
|
11
11
|
- Add Knowledge from local docs/CSVs: `docs/guides/add-knowledge.md`
|
|
12
12
|
- Run or prepare evals: `docs/guides/run-evals.md`
|
|
13
|
+
- Improve from production traces: `docs/guides/improve-from-production.md`
|
|
14
|
+
- Replay collected traces locally: `docs/guides/replay-production-traces.md`
|
|
13
15
|
- Switch from `test/fake` to a real provider: `docs/guides/use-provider.md`
|
|
14
16
|
- Prepare for hosted deploy: `docs/guides/prepare-deploy.md`
|
|
15
17
|
- Build Cloudflare artifact: `agentkit build --target cloudflare`
|
|
@@ -18,6 +20,7 @@ Task guides:
|
|
|
18
20
|
- Add hosted channels: `docs/guides/add-channel.md`
|
|
19
21
|
- Add AgentKit-managed Composio: `docs/guides/add-managed-composio.md`
|
|
20
22
|
- Buffer rapid channel messages: `docs/guides/add-channel.md#buffer-bursty-messages`
|
|
23
|
+
- Connect Discord: `docs/guides/connect-discord.md`
|
|
21
24
|
- Connect Telegram: `docs/guides/connect-telegram.md`
|
|
22
25
|
- Connect WhatsApp through Zapster: `docs/guides/connect-whatsapp-zapster.md`
|
|
23
26
|
- Follow channel webhook and delivery-log safety rules: `docs/guides/channel-security.md`
|
|
@@ -40,11 +43,16 @@ agentkit knowledge sync
|
|
|
40
43
|
agentkit knowledge inspect
|
|
41
44
|
agentkit knowledge search <query> [--top-k <number>]
|
|
42
45
|
agentkit eval run
|
|
46
|
+
agentkit improve collect --deploy --since 24h
|
|
47
|
+
agentkit improve evals .agentkit/improve/<run>
|
|
48
|
+
agentkit replay .agentkit/improve/<run> --against local
|
|
43
49
|
agentkit conversations list
|
|
44
50
|
agentkit conversations show <conversation-id>
|
|
45
51
|
agentkit conversations trace <conversation-id> [--deploy]
|
|
46
52
|
agentkit channels list
|
|
47
53
|
agentkit channels add telegram support-telegram
|
|
54
|
+
agentkit channels connect discord support-discord
|
|
55
|
+
agentkit channels connect discord server-discord --mode bot
|
|
48
56
|
agentkit channels setup support-telegram
|
|
49
57
|
agentkit channels status support-telegram
|
|
50
58
|
agentkit channels test support-telegram --message "hello"
|
|
@@ -55,7 +63,11 @@ agentkit integrations status [--toolkit googlecalendar]
|
|
|
55
63
|
agentkit integrations connect composio --toolkit gmail
|
|
56
64
|
agentkit skills status
|
|
57
65
|
agentkit skills sync
|
|
66
|
+
agentkit billing checkout --slots 1 --email user@example.com
|
|
67
|
+
agentkit billing claim billint_... --secret bsec_...
|
|
58
68
|
agentkit login --token agk_user_...
|
|
69
|
+
agentkit account token create new-laptop --use
|
|
70
|
+
agentkit account token list
|
|
59
71
|
agentkit deploy doctor
|
|
60
72
|
agentkit deploy --smoke "hello"
|
|
61
73
|
agentkit deploy status
|
|
@@ -71,6 +83,8 @@ agentkit dev
|
|
|
71
83
|
agentkit open
|
|
72
84
|
```
|
|
73
85
|
|
|
86
|
+
On Windows PowerShell, if `npm.ps1` or `npx.ps1` is blocked with `PSSecurityException`, run capsule scripts through the `.cmd` shims, for example `npm.cmd run agentkit -- inspect`, `npm.cmd run agentkit -- knowledge sync`, or `npm.cmd run eval`.
|
|
87
|
+
|
|
74
88
|
Generated capsules include `AGENTKIT.md`, `AGENTS.md`, and a repo-local `skills/` pack so Codex, Claude Code, or another coding agent can treat the owner's natural-language request as the brief and start building immediately without loading the full contract by default. Start with `skills/agentkit-capsule/SKILL.md`, then load the task skill for the current work. `agentkit handoff codex "Develop an ophthalmology office intake agent"` is an optional prompt-printing shortcut for users who are not already inside a coding-agent workspace. There is no wizard or recipe layer: the coding agent edits the capsule directly from the scaffold and contract.
|
|
75
89
|
|
|
76
90
|
UI testing is part of the handoff. For local UI testing, run `agentkit dev`, open the printed `Chat:` URL, and tell the owner the exact URL. After hosted deploy, run `agentkit chat-ui --deploy`, open the printed `Chat:` URL, and tell the owner it is connected to the deploy.
|
|
@@ -79,6 +93,8 @@ Hosted Chat UI shows the current conversation id, tool calls, tool errors, and a
|
|
|
79
93
|
|
|
80
94
|
`test/fake` is deterministic and validates scaffold, direct tool calls, and fake-provider evals. It does not validate natural conversation quality. Before claiming real conversation behavior has been tested, ask the owner which provider to use: OpenRouter, OpenAI, Anthropic, or another supported provider. Do not choose for them.
|
|
81
95
|
|
|
96
|
+
For OpenRouter, prefer model ids or aliases known to the installed Pi SDK, such as `~google/gemini-flash-latest`. If an OpenRouter id is newer than Pi's registry, AgentKit passes the raw id through to OpenRouter with conservative unknown-model metadata; OpenRouter can still reject invalid, inaccessible, or unsupported models.
|
|
97
|
+
|
|
82
98
|
AgentKit injects the current ISO timestamp, local date, weekday, local date/time, and timezone dynamically into every chat run. Set `timeZone` in `agentkit.config.ts` for scheduling agents so "today", "tomorrow", and weekdays resolve in the business/user timezone; otherwise AgentKit falls back to `AGENTKIT_TIME_ZONE`, valid `TZ`, then the runtime default. Do not hardcode today's date in prompts.
|
|
83
99
|
|
|
84
100
|
Current local endpoints from `agentkit dev`:
|
|
@@ -91,4 +107,4 @@ GET /v1/conversations/:id
|
|
|
91
107
|
GET /v1/conversations/:id/trace
|
|
92
108
|
```
|
|
93
109
|
|
|
94
|
-
Local `.env` is for development only. Use `.env.schema` as the committed secret-name contract; local AgentKit commands load `.env` directly. Hosted
|
|
110
|
+
Local `.env` is for development only. Use `.env.schema` as the committed secret-name contract; local AgentKit commands load `.env` directly. Hosted deploys require an AgentKit Cloud account with `cloudflare_deploy_alpha` or purchased/manual deploy slots. Use `agentkit billing checkout` + `agentkit billing claim` for first-time paid access, or `agentkit login --token ...` when the user already has an `agk_user_...` token. Production uses managed secrets in the hosted contract.
|
package/package.json
CHANGED
package/src/cli/args.ts
CHANGED
|
@@ -4,6 +4,15 @@ export type ParsedArgs = {
|
|
|
4
4
|
flags: Record<string, string | boolean>;
|
|
5
5
|
};
|
|
6
6
|
|
|
7
|
+
const dashPrefixedValueFlags = new Set([
|
|
8
|
+
"brief",
|
|
9
|
+
"input",
|
|
10
|
+
"input-json",
|
|
11
|
+
"message",
|
|
12
|
+
"prompt",
|
|
13
|
+
"smoke",
|
|
14
|
+
]);
|
|
15
|
+
|
|
7
16
|
export function parseArgs(argv: string[]): ParsedArgs {
|
|
8
17
|
const [command, ...rest] = argv;
|
|
9
18
|
const positional: string[] = [];
|
|
@@ -12,15 +21,27 @@ export function parseArgs(argv: string[]): ParsedArgs {
|
|
|
12
21
|
for (let index = 0; index < rest.length; index += 1) {
|
|
13
22
|
const value = rest[index];
|
|
14
23
|
|
|
24
|
+
if (value === "--") {
|
|
25
|
+
positional.push(...rest.slice(index + 1));
|
|
26
|
+
break;
|
|
27
|
+
}
|
|
28
|
+
|
|
15
29
|
if (!value.startsWith("--")) {
|
|
16
30
|
positional.push(value);
|
|
17
31
|
continue;
|
|
18
32
|
}
|
|
19
33
|
|
|
20
|
-
const
|
|
34
|
+
const equalsIndex = value.indexOf("=");
|
|
35
|
+
const name = equalsIndex === -1 ? value.slice(2) : value.slice(2, equalsIndex);
|
|
36
|
+
|
|
37
|
+
if (equalsIndex !== -1) {
|
|
38
|
+
flags[name] = value.slice(equalsIndex + 1);
|
|
39
|
+
continue;
|
|
40
|
+
}
|
|
41
|
+
|
|
21
42
|
const next = rest[index + 1];
|
|
22
43
|
|
|
23
|
-
if (next && !next.startsWith("--")) {
|
|
44
|
+
if (next && (dashPrefixedValueFlags.has(name) || !next.startsWith("--"))) {
|
|
24
45
|
flags[name] = next;
|
|
25
46
|
index += 1;
|
|
26
47
|
} else {
|
package/src/cli/cloud-client.ts
CHANGED
|
@@ -47,6 +47,7 @@ export type CloudManagedComposioResponse = {
|
|
|
47
47
|
toolkit?: string;
|
|
48
48
|
auth_config_env?: string;
|
|
49
49
|
auth_config?: "set" | "missing" | string;
|
|
50
|
+
auth_config_source?: "project" | "account" | "global" | "env" | string;
|
|
50
51
|
}>;
|
|
51
52
|
};
|
|
52
53
|
};
|
|
@@ -84,6 +85,68 @@ export type DeployAccessTokenListResponse = {
|
|
|
84
85
|
}>;
|
|
85
86
|
};
|
|
86
87
|
|
|
88
|
+
export type AccountApiTokenCreateResponse = {
|
|
89
|
+
token?: {
|
|
90
|
+
id?: string;
|
|
91
|
+
account_id?: string;
|
|
92
|
+
name?: string;
|
|
93
|
+
token?: string;
|
|
94
|
+
created_at?: string;
|
|
95
|
+
expires_at?: string;
|
|
96
|
+
last_used_at?: string;
|
|
97
|
+
};
|
|
98
|
+
};
|
|
99
|
+
|
|
100
|
+
export type AccountApiTokenListResponse = {
|
|
101
|
+
tokens?: Array<{
|
|
102
|
+
id?: string;
|
|
103
|
+
account_id?: string;
|
|
104
|
+
name?: string;
|
|
105
|
+
created_at?: string;
|
|
106
|
+
expires_at?: string;
|
|
107
|
+
last_used_at?: string;
|
|
108
|
+
revoked_at?: string;
|
|
109
|
+
}>;
|
|
110
|
+
};
|
|
111
|
+
|
|
112
|
+
export type BillingCheckoutResponse = {
|
|
113
|
+
checkout?: {
|
|
114
|
+
id?: string;
|
|
115
|
+
url?: string;
|
|
116
|
+
};
|
|
117
|
+
billing_intent?: {
|
|
118
|
+
id?: string;
|
|
119
|
+
account_id?: string;
|
|
120
|
+
email?: string;
|
|
121
|
+
requested_slots?: number;
|
|
122
|
+
status?: string;
|
|
123
|
+
secret?: string;
|
|
124
|
+
};
|
|
125
|
+
};
|
|
126
|
+
|
|
127
|
+
export type BillingCheckoutIntentResponse = {
|
|
128
|
+
billing_intent?: {
|
|
129
|
+
id?: string;
|
|
130
|
+
account_id?: string;
|
|
131
|
+
email?: string;
|
|
132
|
+
requested_slots?: number;
|
|
133
|
+
status?: string;
|
|
134
|
+
checkout_session_id?: string;
|
|
135
|
+
stripe_customer_id?: string;
|
|
136
|
+
stripe_subscription_id?: string;
|
|
137
|
+
completed_at?: string;
|
|
138
|
+
token_claimed_at?: string;
|
|
139
|
+
};
|
|
140
|
+
};
|
|
141
|
+
|
|
142
|
+
export type BillingClaimResponse = BillingCheckoutIntentResponse & AccountApiTokenCreateResponse;
|
|
143
|
+
|
|
144
|
+
export type BillingPortalResponse = {
|
|
145
|
+
portal?: {
|
|
146
|
+
url?: string;
|
|
147
|
+
};
|
|
148
|
+
};
|
|
149
|
+
|
|
87
150
|
export type CloudSecretsResponse = {
|
|
88
151
|
secrets?: Array<{
|
|
89
152
|
name?: string;
|