@andreprado/agentkit 0.1.0-alpha.17 → 0.1.0-alpha.19
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +3 -0
- package/docs/guides/add-channel.md +14 -8
- package/docs/guides/add-knowledge.md +10 -0
- package/docs/guides/add-managed-composio.md +43 -17
- package/docs/guides/channel-security.md +60 -39
- package/docs/guides/connect-discord.md +178 -0
- package/docs/guides/create-agent.md +13 -0
- package/docs/guides/debug-channel.md +147 -0
- package/docs/guides/improve-from-production.md +151 -0
- package/docs/guides/prepare-deploy.md +30 -14
- package/docs/guides/replay-production-traces.md +72 -0
- package/docs/guides/run-evals.md +18 -0
- package/docs/guides/security-rules.md +5 -5
- package/docs/guides/use-provider.md +11 -1
- package/docs/llms-full.txt +106 -15
- package/docs/llms.txt +22 -4
- package/package.json +1 -3
- package/src/cli/args.ts +23 -2
- package/src/cli/cloud-client.ts +75 -0
- package/src/cli/commands/channels.ts +139 -14
- package/src/cli/deploy-chat-ui.ts +146 -3
- package/src/cli/deploy-readiness.ts +57 -1
- package/src/cli/help.ts +32 -8
- package/src/cli/index.ts +447 -17
- package/src/create-project.ts +13 -2
- package/src/index.ts +46 -3
- package/src/providers/pi.ts +49 -15
- package/src/runtime/channel-test-harness.ts +4 -1
- package/src/runtime/channels/discord.ts +887 -0
- package/src/runtime/channels.ts +15 -0
- package/src/runtime/config.ts +39 -3
- package/src/runtime/core/manifest.ts +2 -0
- package/src/runtime/dev-server.ts +149 -8
- package/src/runtime/evals.ts +27 -6
- package/src/runtime/improve.ts +868 -0
- package/src/runtime/inspect.ts +1 -0
- package/src/runtime/integrations/composio.ts +168 -2
- package/src/runtime/knowledge/retrieve.ts +25 -5
- package/src/runtime/knowledge/schema.ts +45 -1
- package/src/runtime/runtime-contract.ts +54 -0
- package/src/runtime/targets/cloudflare/build.ts +479 -194
- package/src/runtime/targets/vps/deploy.ts +1 -1
- package/src/storage/sqlite.ts +7 -2
- package/src/templates/skills/agentkit-capsule/SKILL.md +8 -1
- package/src/templates/skills/agentkit-capsule/references/docs-router.md +1 -2
- package/src/templates/skills/agentkit-channels/SKILL.md +6 -1
- package/src/templates/skills/agentkit-channels/references/channel-debugging.md +2 -1
- package/src/templates/skills/agentkit-channels/references/discord.md +93 -0
- package/src/templates/skills/agentkit-deploy/SKILL.md +6 -0
- package/src/templates/skills/agentkit-evals/SKILL.md +12 -3
- package/src/templates/skills/agentkit-improve/SKILL.md +86 -0
- package/src/templates/skills/agentkit-improve/references/replay-side-effects.md +18 -0
- package/src/templates/skills/agentkit-improve/references/trace-packets.md +22 -0
- package/src/templates/skills/agentkit-improve/templates/regression.eval.md +18 -0
- package/src/templates/skills/agentkit-integrations/SKILL.md +12 -2
- package/src/templates/skills/agentkit-knowledge/SKILL.md +4 -1
- package/src/templates/skills/agentkit-provider/SKILL.md +4 -1
- package/src/templates/skills/agentkit-security/SKILL.md +3 -2
- package/src/templates/skills/agentkit-troubleshooting/SKILL.md +9 -0
- package/src/templates/support.ts +4 -2
- package/docs/guides/agentkit-skills-architecture.md +0 -472
- package/docs/guides/channels-implementation-map.md +0 -243
- package/docs/guides/channels-production-handoff.md +0 -118
- package/docs/portable-deploy-release-checklist.md +0 -41
|
@@ -118,7 +118,7 @@ function renderCompose(agentName: string): string {
|
|
|
118
118
|
- agentkit_data:/data
|
|
119
119
|
- agentkit_files:/capsule/.agentkit/files
|
|
120
120
|
healthcheck:
|
|
121
|
-
test: ["CMD", "
|
|
121
|
+
test: ["CMD", "node", "-e", "fetch('http://127.0.0.1:4123/health').then((r)=>process.exit(r.ok?0:1)).catch(()=>process.exit(1))"]
|
|
122
122
|
interval: 30s
|
|
123
123
|
timeout: 5s
|
|
124
124
|
retries: 3
|
package/src/storage/sqlite.ts
CHANGED
|
@@ -6,7 +6,7 @@ import type { DatabaseArgs, DatabaseResult, DatabaseRow, DatabaseStatement } fro
|
|
|
6
6
|
import type { AgentMessageRole, ProviderRunResult } from "../providers";
|
|
7
7
|
import type { LoadedAgentCapsule } from "../runtime/config";
|
|
8
8
|
import { AgentKitError } from "../runtime/errors";
|
|
9
|
-
import { KNOWLEDGE_MIGRATION_ID, KNOWLEDGE_SCHEMA_SQL } from "../runtime/knowledge/schema";
|
|
9
|
+
import { applyKnowledgeSchema, KNOWLEDGE_MIGRATION_ID, KNOWLEDGE_SCHEMA_SQL } from "../runtime/knowledge/schema";
|
|
10
10
|
|
|
11
11
|
export type ConversationSummary = {
|
|
12
12
|
id: string;
|
|
@@ -806,7 +806,12 @@ export class SqliteAgentKitStore {
|
|
|
806
806
|
continue;
|
|
807
807
|
}
|
|
808
808
|
|
|
809
|
-
|
|
809
|
+
if (migration.id === KNOWLEDGE_MIGRATION_ID) {
|
|
810
|
+
applyKnowledgeSchema(this.db);
|
|
811
|
+
} else {
|
|
812
|
+
this.db.exec(migration.sql);
|
|
813
|
+
}
|
|
814
|
+
|
|
810
815
|
this.db
|
|
811
816
|
.prepare("INSERT INTO agentkit_migrations (id, applied_at) VALUES ($id, $appliedAt)")
|
|
812
817
|
.run({ $id: migration.id, $appliedAt: new Date().toISOString() });
|
|
@@ -24,9 +24,10 @@ Use this first inside an AgentKit Agent Capsule.
|
|
|
24
24
|
- Add docs, FAQs, prices, policies, or CSV facts: `skills/agentkit-knowledge/SKILL.md`
|
|
25
25
|
- Switch from `test/fake` to a real model provider: `skills/agentkit-provider/SKILL.md`
|
|
26
26
|
- Add or run evals: `skills/agentkit-evals/SKILL.md`
|
|
27
|
+
- Improve from hosted or local production evidence: `skills/agentkit-improve/SKILL.md`
|
|
27
28
|
- Prepare hosted deploy: `skills/agentkit-deploy/SKILL.md`
|
|
28
29
|
- Work with secrets, external APIs, public access, channels, or real data: `skills/agentkit-security/SKILL.md`
|
|
29
|
-
- Add or debug website, Telegram, or
|
|
30
|
+
- Add or debug website, Telegram, WhatsApp, or Discord channels: `skills/agentkit-channels/SKILL.md`
|
|
30
31
|
- Investigate command failures: `skills/agentkit-troubleshooting/SKILL.md`
|
|
31
32
|
|
|
32
33
|
For a compact docs map, read `references/docs-router.md`.
|
|
@@ -53,6 +54,12 @@ If you change behavior, add or update an eval and run:
|
|
|
53
54
|
npm run eval
|
|
54
55
|
```
|
|
55
56
|
|
|
57
|
+
If the change fixes production behavior, also collect or use an improve bundle and run:
|
|
58
|
+
|
|
59
|
+
```sh
|
|
60
|
+
npm run agentkit -- replay .agentkit/improve/<run> --against local
|
|
61
|
+
```
|
|
62
|
+
|
|
56
63
|
## Rules
|
|
57
64
|
|
|
58
65
|
- Keep `.env`, `.agentkit/`, and `node_modules/` out of commits.
|
|
@@ -8,8 +8,7 @@ Prefer the narrowest source that covers the task.
|
|
|
8
8
|
- Evals and deterministic side-effect guards: `docs/guides/run-evals.md`
|
|
9
9
|
- Real provider setup: `docs/guides/use-provider.md`
|
|
10
10
|
- Deploy readiness, managed secrets, smoke checks, hosted UI: `docs/guides/prepare-deploy.md`
|
|
11
|
-
- Channels: `docs/guides/add-channel.md`, `connect-telegram.md`, `connect-whatsapp-zapster.md`, `debug-channel.md`
|
|
11
|
+
- Channels: `docs/guides/add-channel.md`, `connect-discord.md`, `connect-telegram.md`, `connect-whatsapp-zapster.md`, `debug-channel.md`
|
|
12
12
|
- Security: `docs/guides/security-rules.md`
|
|
13
13
|
|
|
14
14
|
Use `npm run agentkit -- docs full` only for a complete-contract audit, framework internals, or a behavior not covered by the task guide.
|
|
15
|
-
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
---
|
|
2
2
|
name: agentkit-channels
|
|
3
|
-
description: Use when adding, connecting, testing, buffering, transcribing audio, or debugging AgentKit website, Telegram, or
|
|
3
|
+
description: Use when adding, connecting, testing, buffering, transcribing audio, or debugging AgentKit website, Telegram, WhatsApp, or Discord channels, including channel config helpers, provider secrets, webhook setup, channel tests, delivery logs, burst-message buffers, and transcription provider secrets.
|
|
4
4
|
---
|
|
5
5
|
|
|
6
6
|
# AgentKit Channels
|
|
@@ -49,6 +49,8 @@ V1 providers:
|
|
|
49
49
|
|
|
50
50
|
Telegram voice notes are usually OGG/Opus, so use Groq for the default Telegram voice-note path in V1. Zapster audio needs a usable HTTPS Zapster media download URL in the webhook payload; arbitrary hosts are rejected before bearer auth is sent. Hosted channel creation requires the transcription secret automatically when the channel enables transcription. Webhooks only enqueue audio jobs; download and transcription run in the retryable channel worker before the agent run.
|
|
51
51
|
|
|
52
|
+
Discord channels do not support audio in V1.
|
|
53
|
+
|
|
52
54
|
## Buffering
|
|
53
55
|
|
|
54
56
|
Enable `buffer.mode: "debounce"` when clients send several short messages in a row and the agent should answer once.
|
|
@@ -78,6 +80,8 @@ npm run agentkit -- channels list
|
|
|
78
80
|
npm run agentkit -- channels add website website-chat
|
|
79
81
|
npm run agentkit -- channels connect telegram support-telegram
|
|
80
82
|
npm run agentkit -- channels add whatsapp support-whatsapp --provider zapster
|
|
83
|
+
npm run agentkit -- channels connect discord support-discord
|
|
84
|
+
npm run agentkit -- channels connect discord server-discord --mode bot
|
|
81
85
|
npm run agentkit -- channels doctor support-telegram
|
|
82
86
|
npm run agentkit -- channels test support-telegram --message "hello"
|
|
83
87
|
npm run agentkit -- channels test-audio support-telegram --fixture voice-note
|
|
@@ -87,6 +91,7 @@ npm run agentkit -- channels deliveries list support-telegram
|
|
|
87
91
|
|
|
88
92
|
## References
|
|
89
93
|
|
|
94
|
+
- `references/discord.md`
|
|
90
95
|
- `references/telegram.md`
|
|
91
96
|
- `references/whatsapp-zapster.md`
|
|
92
97
|
- `references/channel-buffering.md`
|
|
@@ -46,7 +46,8 @@ Common errors:
|
|
|
46
46
|
|
|
47
47
|
- `channel_not_found`: webhook URL points to an unknown channel. `channels test` is the official synthetic smoke and should resolve the current `.agentkit/deploy.json` channel.
|
|
48
48
|
- `channel_secret_missing`: required hosted secret is not set.
|
|
49
|
-
- `channel_signature_invalid`: webhook secret, token,
|
|
49
|
+
- `channel_signature_invalid`: webhook secret, token, origin header, or Discord public key mismatch.
|
|
50
|
+
- Discord bot mode with no deliveries: confirm `--mode bot`, `DISCORD_BOT_TOKEN`, Message Content Intent, server install, and channel permissions for View Channel, Read Message History, and Send Messages.
|
|
50
51
|
- `channel_payload_invalid`: malformed or unsupported provider payload.
|
|
51
52
|
- `channel_event_duplicate`: provider retry; do not create a second run.
|
|
52
53
|
- `audio_received`: audio message was accepted and normalized.
|
|
@@ -0,0 +1,93 @@
|
|
|
1
|
+
# Discord Channel
|
|
2
|
+
|
|
3
|
+
Discord supports two modes:
|
|
4
|
+
|
|
5
|
+
- `interactions`: slash commands through the Discord Interactions Endpoint URL.
|
|
6
|
+
- `bot`: normal server messages through a Discord Bot user and Gateway connection.
|
|
7
|
+
|
|
8
|
+
Use bot mode when the owner asks for the agent to answer without slash commands.
|
|
9
|
+
|
|
10
|
+
## Slash Commands
|
|
11
|
+
|
|
12
|
+
Required secret:
|
|
13
|
+
|
|
14
|
+
```txt
|
|
15
|
+
DISCORD_PUBLIC_KEY
|
|
16
|
+
```
|
|
17
|
+
|
|
18
|
+
Config:
|
|
19
|
+
|
|
20
|
+
```ts
|
|
21
|
+
discordChannel({ name: "support-discord" })
|
|
22
|
+
```
|
|
23
|
+
|
|
24
|
+
Commands:
|
|
25
|
+
|
|
26
|
+
```sh
|
|
27
|
+
agentkit deploy
|
|
28
|
+
agentkit secret set DISCORD_PUBLIC_KEY --stdin
|
|
29
|
+
agentkit channels connect discord support-discord
|
|
30
|
+
agentkit channels test support-discord --message "hello"
|
|
31
|
+
agentkit channels deliveries list support-discord
|
|
32
|
+
```
|
|
33
|
+
|
|
34
|
+
Paste the printed webhook URL into the application's Interactions Endpoint URL. Register a slash command with a string option named `message`, `text`, `prompt`, `question`, `query`, or `input`.
|
|
35
|
+
|
|
36
|
+
## Bot Server Messages
|
|
37
|
+
|
|
38
|
+
Required secret:
|
|
39
|
+
|
|
40
|
+
```txt
|
|
41
|
+
DISCORD_BOT_TOKEN
|
|
42
|
+
```
|
|
43
|
+
|
|
44
|
+
Config:
|
|
45
|
+
|
|
46
|
+
```ts
|
|
47
|
+
discordChannel({
|
|
48
|
+
name: "server-discord",
|
|
49
|
+
mode: "bot",
|
|
50
|
+
})
|
|
51
|
+
```
|
|
52
|
+
|
|
53
|
+
Commands:
|
|
54
|
+
|
|
55
|
+
```sh
|
|
56
|
+
agentkit deploy
|
|
57
|
+
agentkit secret set DISCORD_BOT_TOKEN --stdin
|
|
58
|
+
agentkit channels connect discord server-discord --mode bot
|
|
59
|
+
agentkit channels test server-discord --message "hello"
|
|
60
|
+
agentkit channels deliveries list server-discord
|
|
61
|
+
```
|
|
62
|
+
|
|
63
|
+
In the Discord Developer Portal, open the Bot page, enable Message Content Intent, install the app into the server, and grant the bot `View Channel`, `Read Message History`, and `Send Messages` in the channels it should answer.
|
|
64
|
+
|
|
65
|
+
Bot mode processes every readable non-bot text message Discord sends over the Gateway. Keep server permissions narrow if the agent should answer only in specific channels.
|
|
66
|
+
|
|
67
|
+
## Buffering
|
|
68
|
+
|
|
69
|
+
Buffer rapid Discord messages:
|
|
70
|
+
|
|
71
|
+
```ts
|
|
72
|
+
discordChannel({
|
|
73
|
+
name: "server-discord",
|
|
74
|
+
mode: "bot",
|
|
75
|
+
buffer: {
|
|
76
|
+
mode: "debounce",
|
|
77
|
+
quietWindowMs: 1500,
|
|
78
|
+
maxWaitMs: 8000,
|
|
79
|
+
maxMessages: 20,
|
|
80
|
+
maxChars: 8000,
|
|
81
|
+
},
|
|
82
|
+
})
|
|
83
|
+
```
|
|
84
|
+
|
|
85
|
+
## Runtime Behavior
|
|
86
|
+
|
|
87
|
+
Slash-command mode validates `X-Signature-Ed25519` and `X-Signature-Timestamp` against the raw body and `DISCORD_PUBLIC_KEY`, returns `type: 1` for Discord `PING`, acknowledges slash commands with a deferred response, then sends the final answer as an interaction follow-up.
|
|
88
|
+
|
|
89
|
+
Bot mode connects to Discord Gateway with `DISCORD_BOT_TOKEN`, requests guild message and message-content intents, ignores bot-authored messages, normalizes `MESSAGE_CREATE`, and sends the final answer through `/channels/<channel_id>/messages`.
|
|
90
|
+
|
|
91
|
+
All outbound Discord sends use `allowed_mentions: { parse: [] }`. Discord interaction tokens and bot tokens must not appear in delivery, queue, buffer, or doctor API responses.
|
|
92
|
+
|
|
93
|
+
Discord audio is not supported in V1; do not configure `audio` on `discordChannel`.
|
|
@@ -26,6 +26,12 @@ npm run agentkit -- deploy --dry-run
|
|
|
26
26
|
npm run agentkit -- deploy doctor
|
|
27
27
|
```
|
|
28
28
|
|
|
29
|
+
If this deploy fixes production behavior, replay the collected evidence first:
|
|
30
|
+
|
|
31
|
+
```sh
|
|
32
|
+
npm run agentkit -- replay .agentkit/improve/<run> --against local
|
|
33
|
+
```
|
|
34
|
+
|
|
29
35
|
## Hosted Flow
|
|
30
36
|
|
|
31
37
|
```sh
|
|
@@ -16,9 +16,10 @@ Use evals after chat works and before claiming behavior is stable.
|
|
|
16
16
|
5. Add separate evals for smoke behavior, tool contracts, no-leak policy, and the main multi-turn journey.
|
|
17
17
|
6. For date-sensitive flows, set top-level `now` to an ISO timestamp with `Z` or a numeric offset so today, tomorrow, weekdays, and tool date validation stay deterministic.
|
|
18
18
|
7. Use `turns` for full conversation flows, such as user asks, agent calls a tool, then the answer follows the required format.
|
|
19
|
-
8. Convert
|
|
20
|
-
9.
|
|
21
|
-
10.
|
|
19
|
+
8. Convert local failures into regression tests with `npm run agentkit -- eval from-conversation <conversation-id>`.
|
|
20
|
+
9. Convert hosted or local production evidence into regression tests with `npm run agentkit -- improve collect --deploy --since 24h`, then `npm run agentkit -- improve evals .agentkit/improve/<run>`.
|
|
21
|
+
10. Do not put secrets or real client PII in evals.
|
|
22
|
+
11. For tools that write externally, delete, charge money, send email, or call real customer systems, branch on `ctx.runtime.environment === "eval"` inside the registered tool.
|
|
22
23
|
|
|
23
24
|
## Assertion Shape
|
|
24
25
|
|
|
@@ -97,4 +98,12 @@ npm run typecheck
|
|
|
97
98
|
npm run eval
|
|
98
99
|
```
|
|
99
100
|
|
|
101
|
+
On Windows PowerShell, if `npm.ps1` is blocked with `PSSecurityException`, use `npm.cmd run typecheck` and `npm.cmd run eval`.
|
|
102
|
+
|
|
103
|
+
When evals came from an improve bundle, also run:
|
|
104
|
+
|
|
105
|
+
```sh
|
|
106
|
+
npm run agentkit -- replay .agentkit/improve/<run> --against local
|
|
107
|
+
```
|
|
108
|
+
|
|
100
109
|
If eval output changes after switching providers, keep deterministic smoke evals on `test/fake` and add provider-specific evals separately.
|
|
@@ -0,0 +1,86 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: agentkit-improve
|
|
3
|
+
description: Use when improving an AgentKit Agent Capsule from hosted or local production evidence, including collected traces, generated regression evals, local replay, channel failures, or post-deploy behavior fixes.
|
|
4
|
+
---
|
|
5
|
+
|
|
6
|
+
# AgentKit Improve
|
|
7
|
+
|
|
8
|
+
Use this when production or local conversation evidence should drive a fix.
|
|
9
|
+
|
|
10
|
+
## Boundary
|
|
11
|
+
|
|
12
|
+
AgentKit Cloud exports evidence. The local coding agent edits the capsule, writes evals, runs replay, and deploys. Do not expect hosted AgentKit Cloud to change source files.
|
|
13
|
+
|
|
14
|
+
## Workflow
|
|
15
|
+
|
|
16
|
+
1. Collect evidence:
|
|
17
|
+
|
|
18
|
+
```sh
|
|
19
|
+
npm run agentkit -- improve collect --deploy --since 24h
|
|
20
|
+
```
|
|
21
|
+
|
|
22
|
+
Hosted conversation reads require a deploy access token even when a deploy manifest says `access.mode: "public"`. If collection fails with auth, refresh the local token:
|
|
23
|
+
|
|
24
|
+
```sh
|
|
25
|
+
npm run agentkit -- access token create agentkit-chat-ui --out .agentkit/chat-access-token.json
|
|
26
|
+
```
|
|
27
|
+
|
|
28
|
+
For one known conversation:
|
|
29
|
+
|
|
30
|
+
```sh
|
|
31
|
+
npm run agentkit -- improve collect --deploy --conversation-id <conversation-id>
|
|
32
|
+
```
|
|
33
|
+
|
|
34
|
+
2. Read the generated report:
|
|
35
|
+
|
|
36
|
+
```txt
|
|
37
|
+
.agentkit/improve/<run>/report.json
|
|
38
|
+
.agentkit/improve/<run>/traces/
|
|
39
|
+
```
|
|
40
|
+
|
|
41
|
+
3. Generate regression evals:
|
|
42
|
+
|
|
43
|
+
```sh
|
|
44
|
+
npm run agentkit -- improve evals .agentkit/improve/<run>
|
|
45
|
+
```
|
|
46
|
+
|
|
47
|
+
4. Patch the capsule. Likely files:
|
|
48
|
+
|
|
49
|
+
```txt
|
|
50
|
+
prompts/instructions.md
|
|
51
|
+
agentkit.config.ts
|
|
52
|
+
tools/
|
|
53
|
+
knowledge/
|
|
54
|
+
evals/
|
|
55
|
+
```
|
|
56
|
+
|
|
57
|
+
5. Verify:
|
|
58
|
+
|
|
59
|
+
```sh
|
|
60
|
+
npm run typecheck
|
|
61
|
+
npm run agentkit -- inspect
|
|
62
|
+
npm run eval
|
|
63
|
+
npm run agentkit -- replay .agentkit/improve/<run> --against local
|
|
64
|
+
```
|
|
65
|
+
|
|
66
|
+
6. Deploy only after local evals and replay pass:
|
|
67
|
+
|
|
68
|
+
```sh
|
|
69
|
+
npm run agentkit -- deploy --smoke "hello"
|
|
70
|
+
```
|
|
71
|
+
|
|
72
|
+
## Rules
|
|
73
|
+
|
|
74
|
+
- Keep `.agentkit/improve/` out of commits.
|
|
75
|
+
- Review generated evals before committing them.
|
|
76
|
+
- AgentKit redacts common email, phone, bearer token, and key patterns in generated eval text, but you must still remove or generalize domain-specific client PII.
|
|
77
|
+
- If a tool writes externally, deletes, charges money, sends email, or touches real customer systems, make the tool branch on `ctx.runtime.environment === "eval"`.
|
|
78
|
+
- Do not paste secret values into reports, prompts, evals, or Knowledge files.
|
|
79
|
+
- Do not try to read the hosted database directly. Use authenticated AgentKit CLI/API routes only.
|
|
80
|
+
- If replay uses a real provider instead of `test/fake`, say that in the final response.
|
|
81
|
+
|
|
82
|
+
## References
|
|
83
|
+
|
|
84
|
+
- `references/trace-packets.md`
|
|
85
|
+
- `references/replay-side-effects.md`
|
|
86
|
+
- `templates/regression.eval.md`
|
|
@@ -0,0 +1,18 @@
|
|
|
1
|
+
# Replay Side Effects
|
|
2
|
+
|
|
3
|
+
Replay runs collected user turns through the local capsule with:
|
|
4
|
+
|
|
5
|
+
```txt
|
|
6
|
+
ctx.runtime.environment === "eval"
|
|
7
|
+
ctx.runtime.invocation === "eval"
|
|
8
|
+
```
|
|
9
|
+
|
|
10
|
+
Tools still execute. Any tool that writes externally, deletes data, charges money, sends email, sends messages, or calls a real customer system must guard eval mode:
|
|
11
|
+
|
|
12
|
+
```ts
|
|
13
|
+
if (ctx.runtime.environment === "eval") {
|
|
14
|
+
return { ok: true, evalFixture: true };
|
|
15
|
+
}
|
|
16
|
+
```
|
|
17
|
+
|
|
18
|
+
Do not rely on prompt text alone to prevent side effects.
|
|
@@ -0,0 +1,22 @@
|
|
|
1
|
+
# Trace Packets
|
|
2
|
+
|
|
3
|
+
`agentkit improve collect` writes an ignored evidence bundle:
|
|
4
|
+
|
|
5
|
+
```txt
|
|
6
|
+
.agentkit/improve/<run>/
|
|
7
|
+
bundle.json
|
|
8
|
+
report.json
|
|
9
|
+
traces/
|
|
10
|
+
```
|
|
11
|
+
|
|
12
|
+
Use `report.json` for a quick index and `traces/<trace_id>.json` for the full conversation trace.
|
|
13
|
+
|
|
14
|
+
The bundle can contain hosted or local traces. Treat both as sensitive source material. Do not commit `.agentkit/improve/`.
|
|
15
|
+
|
|
16
|
+
Generated evals belong in:
|
|
17
|
+
|
|
18
|
+
```txt
|
|
19
|
+
evals/regressions/
|
|
20
|
+
```
|
|
21
|
+
|
|
22
|
+
Before committing generated evals, remove real client PII and replace brittle exact response assertions with the important behavior when needed. AgentKit redacts common email, phone, bearer token, and key patterns in generated eval text, but it cannot know every domain-specific identifier.
|
|
@@ -0,0 +1,18 @@
|
|
|
1
|
+
```ts
|
|
2
|
+
import { defineEval } from "@andreprado/agentkit";
|
|
3
|
+
|
|
4
|
+
export default defineEval({
|
|
5
|
+
name: "production regression",
|
|
6
|
+
turns: [
|
|
7
|
+
{
|
|
8
|
+
input: "User message from the production trace.",
|
|
9
|
+
expect: {
|
|
10
|
+
response: {
|
|
11
|
+
containsAny: ["required phrase", "acceptable alternative"],
|
|
12
|
+
notRegex: ["API_KEY|secret|token"],
|
|
13
|
+
},
|
|
14
|
+
},
|
|
15
|
+
},
|
|
16
|
+
],
|
|
17
|
+
});
|
|
18
|
+
```
|
|
@@ -26,8 +26,13 @@ export default defineAgent({
|
|
|
26
26
|
toolkits: ["gmail", "googlecalendar"],
|
|
27
27
|
tools: {
|
|
28
28
|
gmail: ["GMAIL_FETCH_EMAILS", "GMAIL_SEND_EMAIL"],
|
|
29
|
-
googlecalendar: [
|
|
29
|
+
googlecalendar: [
|
|
30
|
+
"GOOGLECALENDAR_EVENTS_LIST",
|
|
31
|
+
"GOOGLECALENDAR_CREATE_EVENT",
|
|
32
|
+
"GOOGLECALENDAR_UPDATE_EVENT",
|
|
33
|
+
],
|
|
30
34
|
},
|
|
35
|
+
confirmExternalWrites: true,
|
|
31
36
|
}),
|
|
32
37
|
],
|
|
33
38
|
});
|
|
@@ -35,13 +40,17 @@ export default defineAgent({
|
|
|
35
40
|
|
|
36
41
|
Keep the action list explicit. Do not expose the whole Composio catalog by default.
|
|
37
42
|
|
|
43
|
+
For Google Calendar, do not configure create-only access. Include `GOOGLECALENDAR_EVENTS_LIST` so the agent can inspect availability before writing. For `GOOGLECALENDAR_CREATE_EVENT`, pass UTC `start_datetime` and explicit `event_duration_minutes` or `event_duration_hour`; AgentKit blocks Composio's implicit 30-minute duration default.
|
|
44
|
+
|
|
45
|
+
Managed Composio write actions require tool input `confirmed: true` by default. Set it only after the owner/user confirms the exact external change. Use `confirmExternalWrites: false` only when the capsule implements an equivalent confirmation guard elsewhere.
|
|
46
|
+
|
|
38
47
|
## Commands
|
|
39
48
|
|
|
40
49
|
```sh
|
|
41
50
|
npm run agentkit -- inspect
|
|
42
51
|
npm run agentkit -- deploy doctor
|
|
43
52
|
npm run agentkit -- deploy
|
|
44
|
-
npm run agentkit -- integrations status
|
|
53
|
+
npm run agentkit -- integrations status --toolkit googlecalendar
|
|
45
54
|
npm run agentkit -- integrations connect composio --toolkit gmail
|
|
46
55
|
```
|
|
47
56
|
|
|
@@ -49,6 +58,7 @@ npm run agentkit -- integrations connect composio --toolkit gmail
|
|
|
49
58
|
|
|
50
59
|
- Do not add `COMPOSIO_API_KEY` to `.env.schema` for managed Composio.
|
|
51
60
|
- Do not ask the owner for a Composio key when using managed Composio.
|
|
61
|
+
- Do not ask the owner for Composio auth config ids; AgentKit Cloud resolves toolkit auth configs.
|
|
52
62
|
- Managed Composio requires a non-anonymous hosted deploy and `managed_composio` entitlement.
|
|
53
63
|
- The generated tool is `agentkit_composio_execute`.
|
|
54
64
|
- Use one Composio settings profile per agent.
|
|
@@ -31,10 +31,13 @@ npm run agentkit -- knowledge inspect
|
|
|
31
31
|
npm run agentkit -- knowledge search "refund policy" --top-k 3
|
|
32
32
|
```
|
|
33
33
|
|
|
34
|
+
On Windows PowerShell, if `npm.ps1` is blocked with `PSSecurityException`, use `npm.cmd run agentkit -- knowledge sync` and `npm.cmd run agentkit -- knowledge search "refund policy" --top-k 3`.
|
|
35
|
+
|
|
36
|
+
Local lexical search uses SQLite FTS5 when available. If the local SQLite build does not provide FTS5, AgentKit automatically uses a plain SQLite fallback table and simpler text matching.
|
|
37
|
+
|
|
34
38
|
## Safety
|
|
35
39
|
|
|
36
40
|
- Do not put secrets, credentials, `.env` contents, or private tokens in Knowledge files.
|
|
37
41
|
- Treat committed Knowledge files as repo content.
|
|
38
42
|
- Use a private repo for private business docs.
|
|
39
43
|
- Do not expose raw retrieval JSON, scores, chunk IDs, or tool output objects to users.
|
|
40
|
-
|
|
@@ -42,6 +42,8 @@ provider: { name: "openrouter", model: "gpt-4o-mini" },
|
|
|
42
42
|
secrets: ["OPENROUTER_API_KEY"],
|
|
43
43
|
```
|
|
44
44
|
|
|
45
|
+
Prefer OpenRouter model ids or aliases known to the installed Pi SDK, such as `~google/gemini-flash-latest`. If a raw OpenRouter id is newer than Pi's registry, AgentKit passes it through to OpenRouter with conservative unknown-model metadata; OpenRouter can still reject invalid, inaccessible, or unsupported models.
|
|
46
|
+
|
|
45
47
|
## Verification
|
|
46
48
|
|
|
47
49
|
```sh
|
|
@@ -51,7 +53,8 @@ npm run chat -- --message "hello"
|
|
|
51
53
|
npm run dev
|
|
52
54
|
```
|
|
53
55
|
|
|
56
|
+
On Windows PowerShell, if `npm.ps1` is blocked with `PSSecurityException`, use `npm.cmd run typecheck`, `npm.cmd run agentkit -- inspect`, and `npm.cmd run chat -- --message "hello"`.
|
|
57
|
+
|
|
54
58
|
Open the printed `Chat:` URL and report it to the owner.
|
|
55
59
|
|
|
56
60
|
Do not import provider SDKs in the capsule. AgentKit resolves providers internally through Pi-backed adapters.
|
|
57
|
-
|
|
@@ -40,7 +40,9 @@ skills/
|
|
|
40
40
|
- Add `permissions` for external capabilities.
|
|
41
41
|
- Add timeouts to network tools.
|
|
42
42
|
- Remove client PII before writing evals.
|
|
43
|
-
-
|
|
43
|
+
- Keep `.agentkit/improve/` bundles out of commits and review generated regression evals before committing.
|
|
44
|
+
- Guard replay/eval mode inside external write tools with `ctx.runtime.environment === "eval"`.
|
|
45
|
+
- Treat hosted deploy URLs as addresses, not access control. Hosted chat, conversation reads, and trace reads require a deploy access token even if a config says `access.mode: "public"`.
|
|
44
46
|
|
|
45
47
|
## Checks
|
|
46
48
|
|
|
@@ -52,4 +54,3 @@ git diff --check
|
|
|
52
54
|
```
|
|
53
55
|
|
|
54
56
|
Expected: secret names may appear, secret values do not.
|
|
55
|
-
|
|
@@ -42,6 +42,14 @@ For deploy issues:
|
|
|
42
42
|
npm run agentkit -- deploy doctor
|
|
43
43
|
```
|
|
44
44
|
|
|
45
|
+
For production behavior issues:
|
|
46
|
+
|
|
47
|
+
```sh
|
|
48
|
+
npm run agentkit -- improve collect --deploy --since 24h
|
|
49
|
+
npm run agentkit -- improve evals .agentkit/improve/<run>
|
|
50
|
+
npm run agentkit -- replay .agentkit/improve/<run> --against local
|
|
51
|
+
```
|
|
52
|
+
|
|
45
53
|
## Common Causes
|
|
46
54
|
|
|
47
55
|
- missing dependencies: run `npm install`;
|
|
@@ -50,4 +58,5 @@ npm run agentkit -- deploy doctor
|
|
|
50
58
|
- schema missing: run `db migrate` and check `schema.sql`;
|
|
51
59
|
- tool validation failed: check `inputSchema` and `outputSchema`;
|
|
52
60
|
- channel secret missing: set hosted managed secret, not source files;
|
|
61
|
+
- production behavior drift: collect an improve bundle and convert it to regression evals before patching;
|
|
53
62
|
- local AgentKit skills are stale: run `npm run agentkit -- skills sync`.
|
package/src/templates/support.ts
CHANGED
|
@@ -111,7 +111,9 @@ export default defineAgent({
|
|
|
111
111
|
},
|
|
112
112
|
{
|
|
113
113
|
path: "tools/lookup-order.ts",
|
|
114
|
-
contents: `
|
|
114
|
+
contents: `import type { AgentTool } from "@andreprado/agentkit";
|
|
115
|
+
|
|
116
|
+
const orders: Record<string, { status: string; eta: string }> = {
|
|
115
117
|
A100: { status: "preparing", eta: "today" },
|
|
116
118
|
B200: { status: "shipped", eta: "tomorrow" },
|
|
117
119
|
};
|
|
@@ -157,7 +159,7 @@ export const lookupOrder = {
|
|
|
157
159
|
eta: order.eta,
|
|
158
160
|
};
|
|
159
161
|
},
|
|
160
|
-
};
|
|
162
|
+
} satisfies AgentTool;
|
|
161
163
|
`,
|
|
162
164
|
},
|
|
163
165
|
{
|