@andreprado/agentkit 0.1.0-alpha.9 → 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +18 -1
- package/docs/guides/add-channel.md +251 -7
- package/docs/guides/add-knowledge.md +10 -0
- package/docs/guides/add-managed-composio.md +165 -0
- package/docs/guides/add-tool.md +10 -3
- package/docs/guides/channel-security.md +162 -32
- package/docs/guides/connect-discord.md +178 -0
- package/docs/guides/connect-slack.md +126 -0
- package/docs/guides/connect-telegram.md +61 -1
- package/docs/guides/connect-whatsapp-evolution.md +121 -0
- package/docs/guides/connect-whatsapp-uazapi.md +139 -0
- package/docs/guides/connect-whatsapp-zapster.md +119 -16
- package/docs/guides/create-agent.md +31 -4
- package/docs/guides/debug-channel.md +159 -0
- package/docs/guides/improve-from-production.md +151 -0
- package/docs/guides/prepare-deploy.md +32 -14
- package/docs/guides/replay-production-traces.md +72 -0
- package/docs/guides/run-evals.md +95 -25
- package/docs/guides/security-rules.md +9 -5
- package/docs/guides/send-feedback.md +135 -0
- package/docs/guides/use-jev.md +67 -0
- package/docs/guides/use-provider.md +70 -3
- package/docs/llms-full.txt +295 -25
- package/docs/llms.txt +54 -7
- package/package.json +3 -7
- package/src/cli/args.ts +23 -2
- package/src/cli/cloud-client.ts +121 -9
- package/src/cli/commands/channels.ts +856 -36
- package/src/cli/commands/feedback.ts +438 -0
- package/src/cli/commands/provider.ts +47 -0
- package/src/cli/commands/transcribe.ts +171 -0
- package/src/cli/deploy-chat-ui.ts +232 -18
- package/src/cli/deploy-readiness.ts +227 -14
- package/src/cli/help.ts +67 -9
- package/src/cli/index.ts +740 -35
- package/src/cli/new-command.ts +41 -0
- package/src/cloud/client.ts +4 -3
- package/src/cloud/contracts.ts +1 -1
- package/src/create-project.ts +18 -35
- package/src/index.ts +565 -11
- package/src/providers/codex-auth.ts +111 -0
- package/src/providers/pi.ts +88 -19
- package/src/providers/test.ts +36 -0
- package/src/providers/types.ts +8 -0
- package/src/runtime/channel-test-harness.ts +21 -1
- package/src/runtime/channels/discord.ts +896 -0
- package/src/runtime/channels/generic-webhook.ts +974 -0
- package/src/runtime/channels/slack.ts +646 -0
- package/src/runtime/channels/telegram.ts +343 -12
- package/src/runtime/channels/whatsapp-evolution.ts +1357 -0
- package/src/runtime/channels/whatsapp-meta.ts +9 -0
- package/src/runtime/channels/whatsapp-uazapi.ts +1323 -0
- package/src/runtime/channels/whatsapp-zapster.ts +674 -40
- package/src/runtime/channels.ts +83 -3
- package/src/runtime/chat.ts +70 -44
- package/src/runtime/config.ts +489 -20
- package/src/runtime/core/manifest.ts +75 -5
- package/src/runtime/core/targets.ts +5 -5
- package/src/runtime/deploy-readiness.ts +34 -4
- package/src/runtime/dev-server.ts +639 -39
- package/src/runtime/env.ts +8 -3
- package/src/runtime/evals.ts +445 -74
- package/src/runtime/improve.ts +868 -0
- package/src/runtime/inspect.ts +173 -4
- package/src/runtime/integrations/composio.ts +425 -0
- package/src/runtime/knowledge/retrieve.ts +25 -5
- package/src/runtime/knowledge/schema.ts +45 -1
- package/src/runtime/prompt-context.ts +141 -0
- package/src/runtime/runtime-contract.ts +71 -7
- package/src/runtime/skills.ts +95 -0
- package/src/runtime/targets/cloudflare/build.ts +1010 -208
- package/src/runtime/targets/container/server.ts +1 -1
- package/src/runtime/targets/vps/deploy.ts +26 -9
- package/src/runtime/tool-runner.ts +9 -1
- package/src/runtime/tools.ts +26 -2
- package/src/runtime/transcription.ts +483 -0
- package/src/storage/sqlite.ts +7 -2
- package/src/templates/blank.ts +37 -9
- package/src/templates/dentista.ts +40 -14
- package/src/templates/skills/agentkit-build-agent/SKILL.md +34 -5
- package/src/templates/skills/agentkit-build-agent/templates/appointment-intake.instructions.md +2 -1
- package/src/templates/skills/agentkit-capsule/SKILL.md +32 -3
- package/src/templates/skills/agentkit-capsule/references/docs-router.md +2 -2
- package/src/templates/skills/agentkit-channels/SKILL.md +66 -1
- package/src/templates/skills/agentkit-channels/references/channel-buffering.md +8 -1
- package/src/templates/skills/agentkit-channels/references/channel-debugging.md +28 -3
- package/src/templates/skills/agentkit-channels/references/discord.md +93 -0
- package/src/templates/skills/agentkit-channels/references/slack.md +56 -0
- package/src/templates/skills/agentkit-channels/references/telegram.md +34 -0
- package/src/templates/skills/agentkit-channels/references/whatsapp-evolution.md +57 -0
- package/src/templates/skills/agentkit-channels/references/whatsapp-uazapi.md +54 -0
- package/src/templates/skills/agentkit-channels/references/whatsapp-zapster.md +42 -8
- package/src/templates/skills/agentkit-database/SKILL.md +11 -0
- package/src/templates/skills/agentkit-deploy/SKILL.md +9 -1
- package/src/templates/skills/agentkit-evals/SKILL.md +77 -13
- package/src/templates/skills/agentkit-evals/templates/multi-turn.eval.md +13 -6
- package/src/templates/skills/agentkit-evals/templates/no-leak.eval.md +8 -4
- package/src/templates/skills/agentkit-evals/templates/smoke.eval.md +8 -4
- package/src/templates/skills/agentkit-evals/templates/tool-call.eval.md +16 -7
- package/src/templates/skills/agentkit-improve/SKILL.md +96 -0
- package/src/templates/skills/agentkit-improve/references/replay-side-effects.md +18 -0
- package/src/templates/skills/agentkit-improve/references/trace-packets.md +22 -0
- package/src/templates/skills/agentkit-improve/templates/regression.eval.md +18 -0
- package/src/templates/skills/agentkit-integrations/SKILL.md +98 -0
- package/src/templates/skills/agentkit-knowledge/SKILL.md +4 -1
- package/src/templates/skills/agentkit-prompts/SKILL.md +3 -1
- package/src/templates/skills/agentkit-provider/SKILL.md +29 -4
- package/src/templates/skills/agentkit-security/SKILL.md +5 -2
- package/src/templates/skills/agentkit-tools/SKILL.md +8 -1
- package/src/templates/skills/agentkit-tools/examples/eval-safe-external-action.tool.md +8 -8
- package/src/templates/skills/agentkit-tools/examples/jev-service-fit.tool.md +110 -0
- package/src/templates/skills/agentkit-troubleshooting/SKILL.md +25 -1
- package/src/templates/support.ts +42 -12
- package/docs/guides/agentkit-skills-architecture.md +0 -471
- package/docs/guides/channels-implementation-map.md +0 -243
- package/docs/guides/channels-production-handoff.md +0 -101
- package/docs/portable-deploy-release-checklist.md +0 -41
|
@@ -0,0 +1,159 @@
|
|
|
1
|
+
# Debug A Channel
|
|
2
|
+
|
|
3
|
+
## Goal
|
|
4
|
+
|
|
5
|
+
Diagnose website, Telegram, WhatsApp, Discord, Slack, generic inbound webhook, generic outbound webhook, cross-channel reply routing, webhook validation, dedupe, buffering, outbound sends, and delivery failures from the Agent Capsule CLI.
|
|
6
|
+
|
|
7
|
+
## When To Use It
|
|
8
|
+
|
|
9
|
+
Use this when a hosted channel is not responding, a provider is retrying messages, audio transcription is failing, cross-channel replies are going to the wrong target, or delivery status does not match the user's expectation.
|
|
10
|
+
|
|
11
|
+
## Commands
|
|
12
|
+
|
|
13
|
+
```sh
|
|
14
|
+
agentkit channels list
|
|
15
|
+
agentkit channels status <name>
|
|
16
|
+
agentkit channels doctor <name>
|
|
17
|
+
agentkit channels test <name> --message "hello"
|
|
18
|
+
agentkit channels test-audio <name> --fixture voice-note
|
|
19
|
+
agentkit transcribe smoke --provider groq
|
|
20
|
+
agentkit channels test <name> --fixture ./fixtures/provider-event.json
|
|
21
|
+
agentkit channels deliveries list <name> --since 24h
|
|
22
|
+
agentkit channels deliveries show <delivery-id>
|
|
23
|
+
agentkit channels deliveries list <reply-target-name> --since 24h
|
|
24
|
+
agentkit channels buffers list <name>
|
|
25
|
+
agentkit channels buffers show <conversation-id>
|
|
26
|
+
agentkit channels buffers flush <conversation-id>
|
|
27
|
+
agentkit channels buffers clear <conversation-id>
|
|
28
|
+
agentkit channels buffers retry <conversation-id>
|
|
29
|
+
```
|
|
30
|
+
|
|
31
|
+
## Files Created Or Edited
|
|
32
|
+
|
|
33
|
+
Debugging should not require source edits. Use fixtures only when reproducing provider payload shape, and never commit provider secrets, full phone numbers, raw audio, or private client data.
|
|
34
|
+
|
|
35
|
+
## Delivery States
|
|
36
|
+
|
|
37
|
+
```txt
|
|
38
|
+
received
|
|
39
|
+
validated
|
|
40
|
+
duplicate
|
|
41
|
+
audio_received
|
|
42
|
+
audio_downloaded
|
|
43
|
+
transcribing
|
|
44
|
+
transcribed
|
|
45
|
+
buffered
|
|
46
|
+
queued
|
|
47
|
+
running
|
|
48
|
+
agent_completed
|
|
49
|
+
provider_request_built
|
|
50
|
+
provider_sent
|
|
51
|
+
adapter_stubbed
|
|
52
|
+
delivered
|
|
53
|
+
provider_failed
|
|
54
|
+
failed
|
|
55
|
+
dead_lettered
|
|
56
|
+
skipped
|
|
57
|
+
```
|
|
58
|
+
|
|
59
|
+
## Minimal Working Example
|
|
60
|
+
|
|
61
|
+
```sh
|
|
62
|
+
agentkit channels test support-telegram --message "hello"
|
|
63
|
+
agentkit channels deliveries list support-telegram --since 1h
|
|
64
|
+
agentkit channels deliveries show del_123
|
|
65
|
+
```
|
|
66
|
+
|
|
67
|
+
Expected `show` output includes signature status, dedupe key, queue/run state, outbound provider request ID when available, and a normalized error envelope.
|
|
68
|
+
|
|
69
|
+
## Safety Rules
|
|
70
|
+
|
|
71
|
+
- Inspect delivery IDs, hashes, statuses, and redacted metadata rather than raw provider payloads.
|
|
72
|
+
- Never paste provider secrets into fixtures or prompts.
|
|
73
|
+
- Treat `adapter_stubbed` as "no outbound provider call happened", not proof that a client received a message. It can mean explicit dry-run mode or an inbound-only channel such as a generic webhook.
|
|
74
|
+
- For `replyTo` routes, inspect the source channel delivery for validation, dedupe, buffer, queue, and agent-run state, then inspect the target channel delivery for the outbound provider result.
|
|
75
|
+
- Generic output webhook channels do not have inbound endpoints. Use `agentkit channels doctor <name>` to inspect secret readiness and inspect target-channel deliveries after a source event triggers the route.
|
|
76
|
+
- Before replaying or flushing buffered conversations, confirm the channel name and conversation ID.
|
|
77
|
+
- Keep local `.env` values ignored and upload hosted production values through AgentKit secret commands.
|
|
78
|
+
|
|
79
|
+
## Troubleshooting By Error Code
|
|
80
|
+
|
|
81
|
+
`channel_not_found`:
|
|
82
|
+
The webhook URL points to an unknown or deleted channel. Stable hosted URLs use the channel name, for example `/channels/support-whatsapp/whatsapp/zapster/webhook`; old `chn_*` URLs are accepted only for compatibility.
|
|
83
|
+
|
|
84
|
+
`agentkit channels test <name>` is the official synthetic smoke. If it fails while `channels list`, `status`, or `doctor` find the channel, rerun against the current deploy state in `.agentkit/deploy.json`.
|
|
85
|
+
|
|
86
|
+
`channel_disabled`:
|
|
87
|
+
The channel was disabled. Re-add or recreate it.
|
|
88
|
+
|
|
89
|
+
`channel_secret_missing`:
|
|
90
|
+
The hosted secret metadata says a required channel secret is missing. Set it with `agentkit secret set <NAME> --from-local-env` or `agentkit secret sync --from-local`.
|
|
91
|
+
|
|
92
|
+
`channel_signature_invalid`:
|
|
93
|
+
Provider authenticity validation failed. Check the provider webhook secret, token, origin settings, or Discord public key.
|
|
94
|
+
|
|
95
|
+
`channel_payload_invalid`:
|
|
96
|
+
The provider payload is malformed or does not match the route provider.
|
|
97
|
+
|
|
98
|
+
Discord endpoint validation fails in the Developer Portal:
|
|
99
|
+
Confirm `DISCORD_PUBLIC_KEY` is set from the application's public key and the webhook URL is `/channels/<name>/discord/discord/webhook`. AgentKit must validate Discord's signature headers before returning the `PING` PONG.
|
|
100
|
+
|
|
101
|
+
Discord bot does not answer normal server messages:
|
|
102
|
+
Confirm the channel was created with `--mode bot`, `DISCORD_BOT_TOKEN` is set, Message Content Intent is enabled in the Discord Developer Portal, the app is installed into the server, and the bot has `View Channel`, `Read Message History`, and `Send Messages` permissions for the channel.
|
|
103
|
+
|
|
104
|
+
`audio_received`:
|
|
105
|
+
The webhook contained a supported audio message and the channel is entering the audio handling path.
|
|
106
|
+
|
|
107
|
+
`audio_downloaded`:
|
|
108
|
+
The retryable channel worker downloaded the provider media file into memory. The delivery metadata should include only redacted size, MIME, and provider IDs.
|
|
109
|
+
|
|
110
|
+
`transcribing`:
|
|
111
|
+
AgentKit is calling the configured transcription provider with the user's managed transcription secret.
|
|
112
|
+
|
|
113
|
+
`transcribed`:
|
|
114
|
+
Transcription succeeded and the queued agent message contains transcript text instead of raw audio.
|
|
115
|
+
|
|
116
|
+
`channel_event_duplicate` or `duplicate`:
|
|
117
|
+
The provider retried an already-processed event. No second agent run should be created.
|
|
118
|
+
|
|
119
|
+
`buffered`:
|
|
120
|
+
The message is accepted and waiting inside a per-conversation channel buffer. It should move to `queued` after the quiet window, max wait, max message count, or max character count.
|
|
121
|
+
|
|
122
|
+
`adapter_stubbed`:
|
|
123
|
+
The adapter completed without calling an outbound provider. This can happen in explicit dry-run mode or for an inbound-only channel such as a generic webhook without `replyTo`.
|
|
124
|
+
|
|
125
|
+
`provider_sent`:
|
|
126
|
+
The provider API accepted the outbound request and returned a provider message ID. For cross-channel replies, this status appears on the reply target channel, such as WhatsApp or a generic output webhook.
|
|
127
|
+
|
|
128
|
+
`channel_reply_target_missing`:
|
|
129
|
+
The source channel has `replyTo.channel`, but the hosted channel resource for that target does not exist in the same deploy. Create or reconnect the target channel, then retry the source event.
|
|
130
|
+
|
|
131
|
+
`channel_reply_recipient_missing`:
|
|
132
|
+
The route needs a destination identity but `recipientFrom` was missing, resolved to an empty value, or the source identity cannot be reused by the target channel. Check the inbound payload and the configured dotted JSON path.
|
|
133
|
+
|
|
134
|
+
`channel_reply_target_invalid`:
|
|
135
|
+
The route points to an invalid target, such as an inbound-only generic webhook, or tries to use `recipientFrom` for a provider that requires native source identities such as Discord or Slack.
|
|
136
|
+
|
|
137
|
+
`channel_limit_exceeded`:
|
|
138
|
+
Backpressure skipped the message before queueing.
|
|
139
|
+
|
|
140
|
+
`channel_audio_download_unavailable`:
|
|
141
|
+
The channel provider reported audio but did not include enough metadata for AgentKit to download it.
|
|
142
|
+
|
|
143
|
+
`transcription_secret_missing`:
|
|
144
|
+
The transcription provider secret declared in `agentkit inspect` is not set as a managed hosted secret.
|
|
145
|
+
|
|
146
|
+
`transcription_audio_too_large` or `transcription_audio_too_long`:
|
|
147
|
+
The audio exceeded `transcription.limits` or the channel-level `audio.limits`.
|
|
148
|
+
|
|
149
|
+
`transcription_audio_format_unsupported`:
|
|
150
|
+
The configured transcription provider does not accept this audio MIME type or file extension.
|
|
151
|
+
|
|
152
|
+
`transcription_provider_unavailable`:
|
|
153
|
+
The transcription provider returned a retryable error, usually HTTP 429 or a server-side failure.
|
|
154
|
+
|
|
155
|
+
`channel_provider_unavailable`:
|
|
156
|
+
The agent run or provider send path failed with a retryable provider condition.
|
|
157
|
+
|
|
158
|
+
`channel_delivery_dead_lettered`:
|
|
159
|
+
The queue exhausted retries. Inspect the delivery and replay manually only after fixing the cause.
|
|
@@ -0,0 +1,151 @@
|
|
|
1
|
+
# Improve From Production
|
|
2
|
+
|
|
3
|
+
## Goal
|
|
4
|
+
|
|
5
|
+
Pull hosted or local conversation evidence into the Agent Capsule, turn it into regression evals, let the local coding agent patch the capsule, and replay before deploying again.
|
|
6
|
+
|
|
7
|
+
## When To Use This
|
|
8
|
+
|
|
9
|
+
Use this when a deployed agent gave a wrong answer, failed a tool call, mishandled a channel message, or needs production behavior converted into eval coverage.
|
|
10
|
+
|
|
11
|
+
AgentKit Cloud only exports redacted evidence. The local coding agent owns source edits, evals, replay, and deploy.
|
|
12
|
+
|
|
13
|
+
## Commands
|
|
14
|
+
|
|
15
|
+
Collect evidence from the last hosted deploy:
|
|
16
|
+
|
|
17
|
+
```sh
|
|
18
|
+
agentkit improve collect --deploy --since 24h
|
|
19
|
+
```
|
|
20
|
+
|
|
21
|
+
When the CLI is logged in to AgentKit Cloud, this command first asks the control plane for deploy evidence such as failed channel deliveries, deploy errors, and conversation IDs that need review. It then reads replayable hosted conversation traces from the deployed runtime with the deploy access token. If Cloud auth is not available, it still collects hosted conversations directly from the deploy URL.
|
|
22
|
+
|
|
23
|
+
Hosted conversation reads require a deploy access token even when the chat endpoint is public. `agentkit deploy` normally writes `.agentkit/chat-access-token.json`; refresh it with:
|
|
24
|
+
|
|
25
|
+
```sh
|
|
26
|
+
agentkit access token create agentkit-chat-ui --out .agentkit/chat-access-token.json
|
|
27
|
+
```
|
|
28
|
+
|
|
29
|
+
Collect one hosted conversation:
|
|
30
|
+
|
|
31
|
+
```sh
|
|
32
|
+
agentkit improve collect --deploy --conversation-id <conversation-id>
|
|
33
|
+
```
|
|
34
|
+
|
|
35
|
+
Collect local conversations instead:
|
|
36
|
+
|
|
37
|
+
```sh
|
|
38
|
+
agentkit improve collect --since 7d
|
|
39
|
+
```
|
|
40
|
+
|
|
41
|
+
Generate regression evals from the collected bundle:
|
|
42
|
+
|
|
43
|
+
```sh
|
|
44
|
+
agentkit improve evals .agentkit/improve/<run>
|
|
45
|
+
```
|
|
46
|
+
|
|
47
|
+
Replay the bundle against the local capsule:
|
|
48
|
+
|
|
49
|
+
```sh
|
|
50
|
+
agentkit replay .agentkit/improve/<run> --against local
|
|
51
|
+
```
|
|
52
|
+
|
|
53
|
+
## Files Created Or Edited
|
|
54
|
+
|
|
55
|
+
Evidence bundle, ignored local state:
|
|
56
|
+
|
|
57
|
+
```txt
|
|
58
|
+
.agentkit/improve/<run>/
|
|
59
|
+
bundle.json
|
|
60
|
+
report.json
|
|
61
|
+
traces/
|
|
62
|
+
```
|
|
63
|
+
|
|
64
|
+
Generated regression evals, committed source:
|
|
65
|
+
|
|
66
|
+
```txt
|
|
67
|
+
evals/regressions/
|
|
68
|
+
improve-<conversation>.eval.ts
|
|
69
|
+
```
|
|
70
|
+
|
|
71
|
+
The local coding agent may then edit:
|
|
72
|
+
|
|
73
|
+
```txt
|
|
74
|
+
prompts/instructions.md
|
|
75
|
+
agentkit.config.ts
|
|
76
|
+
tools/
|
|
77
|
+
knowledge/
|
|
78
|
+
evals/
|
|
79
|
+
```
|
|
80
|
+
|
|
81
|
+
Do not edit `.agentkit/improve/<run>/bundle.json` by hand.
|
|
82
|
+
|
|
83
|
+
## Workflow
|
|
84
|
+
|
|
85
|
+
```sh
|
|
86
|
+
agentkit improve collect --deploy --since 24h
|
|
87
|
+
agentkit improve evals .agentkit/improve/<run>
|
|
88
|
+
agentkit replay .agentkit/improve/<run> --against local
|
|
89
|
+
```
|
|
90
|
+
|
|
91
|
+
Then let the local coding agent inspect `report.json`, the generated eval files, prompts, tools, and Knowledge sources. After edits:
|
|
92
|
+
|
|
93
|
+
```sh
|
|
94
|
+
npm run typecheck
|
|
95
|
+
npm run agentkit -- inspect
|
|
96
|
+
npm run eval
|
|
97
|
+
agentkit replay .agentkit/improve/<run> --against local
|
|
98
|
+
agentkit deploy --smoke "hello"
|
|
99
|
+
```
|
|
100
|
+
|
|
101
|
+
## Safety Rules
|
|
102
|
+
|
|
103
|
+
- Do not paste secrets into evals, prompts, Knowledge files, or reports.
|
|
104
|
+
- Keep `.agentkit/improve/` out of commits.
|
|
105
|
+
- Review generated eval assertions before committing them. AgentKit redacts common email, phone, bearer token, and key patterns in generated eval text, but the local coding agent must still remove or generalize domain-specific client PII.
|
|
106
|
+
- For write, delete, payment, email, or customer-system tools, branch inside the tool on `ctx.runtime.environment === "eval"` and return deterministic non-destructive output.
|
|
107
|
+
- Treat hosted traces as customer evidence.
|
|
108
|
+
- If replay uses a real provider instead of `test/fake`, tell the owner because it may cost money and may be nondeterministic.
|
|
109
|
+
|
|
110
|
+
## Verification
|
|
111
|
+
|
|
112
|
+
```sh
|
|
113
|
+
npm run typecheck
|
|
114
|
+
npm run agentkit -- inspect
|
|
115
|
+
npm run eval
|
|
116
|
+
agentkit replay .agentkit/improve/<run> --against local
|
|
117
|
+
```
|
|
118
|
+
|
|
119
|
+
Expected:
|
|
120
|
+
|
|
121
|
+
- `improve collect` writes a bundle and report under `.agentkit/improve/`.
|
|
122
|
+
- Hosted collection includes a redacted `evidence` summary in `bundle.json` and `report.json` when AgentKit Cloud evidence export is available.
|
|
123
|
+
- `improve evals` writes eval files under `evals/regressions/`.
|
|
124
|
+
- `replay` reports passed, failed, and skipped traces.
|
|
125
|
+
- No production secret values appear in generated files.
|
|
126
|
+
|
|
127
|
+
## Troubleshooting
|
|
128
|
+
|
|
129
|
+
`No .agentkit/deploy.json found`:
|
|
130
|
+
|
|
131
|
+
Run `agentkit deploy` first, or collect local evidence without `--deploy`.
|
|
132
|
+
|
|
133
|
+
`deploy_conversation_request_failed`:
|
|
134
|
+
|
|
135
|
+
Refresh the deploy chat token with `agentkit access token create agentkit-chat-ui --out .agentkit/chat-access-token.json`, then retry.
|
|
136
|
+
|
|
137
|
+
`improve_evidence_store_not_configured`:
|
|
138
|
+
|
|
139
|
+
The Cloud API does not expose deploy evidence export yet. The CLI falls back to hosted conversation trace collection when possible.
|
|
140
|
+
|
|
141
|
+
`conversation_access_not_configured`:
|
|
142
|
+
|
|
143
|
+
The hosted runtime is not configured with a deploy access-token gate for conversation reads. Redeploy through AgentKit Cloud so the runtime gate is injected, or configure an explicit deploy/private token for local hosted testing.
|
|
144
|
+
|
|
145
|
+
`trace has no user turns`:
|
|
146
|
+
|
|
147
|
+
The trace cannot become a useful conversation eval. Keep the report for diagnosis, but do not commit an empty eval.
|
|
148
|
+
|
|
149
|
+
Generated eval is too strict:
|
|
150
|
+
|
|
151
|
+
Edit the eval to assert the important behavior, such as tool input, safety wording, or knowledge source usage, instead of exact prose.
|
|
@@ -25,7 +25,18 @@ npm run agentkit -- db migrate
|
|
|
25
25
|
npm run chat -- --message "hello"
|
|
26
26
|
```
|
|
27
27
|
|
|
28
|
-
All scaffold, local dev, chat, eval, inspect, database, and build commands are token-free.
|
|
28
|
+
All scaffold, local dev, chat, eval, inspect, database, and build commands are token-free. Hosted production deploy and hosted AgentKit Cloud changes require an AgentKit Cloud account token with hosted deploy access.
|
|
29
|
+
|
|
30
|
+
If the user does not have a token yet, start checkout from the CLI, finish Stripe Checkout in the browser, then claim the one-time checkout intent:
|
|
31
|
+
|
|
32
|
+
```sh
|
|
33
|
+
npm run agentkit -- billing checkout --slots 1 --email user@example.com
|
|
34
|
+
npm run agentkit -- billing claim billint_... --secret bsec_...
|
|
35
|
+
```
|
|
36
|
+
|
|
37
|
+
The claim command stores the returned `agk_user_...` token in the local AgentKit Cloud auth file. Treat the `bsec_...` checkout secret like a password; it exists only to claim the first token after checkout.
|
|
38
|
+
|
|
39
|
+
If the user already has a token:
|
|
29
40
|
|
|
30
41
|
```sh
|
|
31
42
|
npm run agentkit -- login --token agk_user_...
|
|
@@ -42,7 +53,15 @@ OPENAI_API_KEY=sk-...
|
|
|
42
53
|
```
|
|
43
54
|
|
|
44
55
|
`.env` is local-only. Hosted deploy secrets are handled by AgentKit outside the capsule.
|
|
45
|
-
|
|
56
|
+
Hosted deploys require an account with either `cloudflare_deploy_alpha` or purchased/manual deploy slots. Deploy slots belong to the account, not to a specific token string.
|
|
57
|
+
|
|
58
|
+
To generate or switch account tokens after login:
|
|
59
|
+
|
|
60
|
+
```sh
|
|
61
|
+
npm run agentkit -- account token create new-laptop --use
|
|
62
|
+
npm run agentkit -- account token list
|
|
63
|
+
npm run agentkit -- account token revoke apitok_...
|
|
64
|
+
```
|
|
46
65
|
|
|
47
66
|
## What The Agent Should Edit
|
|
48
67
|
|
|
@@ -126,6 +145,7 @@ For a final end-to-end test, run:
|
|
|
126
145
|
|
|
127
146
|
```sh
|
|
128
147
|
npm run agentkit -- login --token agk_user_...
|
|
148
|
+
npm run agentkit -- account token list
|
|
129
149
|
npm run agentkit -- secret sync --from-local
|
|
130
150
|
npm run agentkit -- secret list
|
|
131
151
|
npm run agentkit -- deploy --smoke "hello"
|
|
@@ -134,7 +154,7 @@ npm run agentkit -- chat-ui --deploy
|
|
|
134
154
|
npm run agentkit -- access token list
|
|
135
155
|
```
|
|
136
156
|
|
|
137
|
-
For
|
|
157
|
+
For hosted deploys, `npm run agentkit -- deploy` writes the local chat/UI access token to `.agentkit/chat-access-token.json` and prints a production handoff with the deploy URL, UI command, secret status, database/schema artifact, integration connect commands, smoke status, and next recommended command. Use `npm run agentkit -- deploy --smoke "hello"` for the official hosted chat smoke, and use `npm run agentkit -- chat-ui --deploy` for hosted UI testing. The hosted Chat UI shows the conversation id, tool calls, tool errors, and a new-conversation control; use `npm run agentkit -- conversations trace <conversation-id> --deploy` to pull the hosted trace from the last deploy. Use `npm run agentkit -- access token create <name> --out <path>` only for additional clients.
|
|
138
158
|
|
|
139
159
|
When the UI is running, open the printed `Chat:` URL and tell the owner the exact URL. If the capsule is still on `test/fake`, say the UI was tested only with the deterministic fake provider.
|
|
140
160
|
|
|
@@ -146,7 +166,8 @@ No .agentkit committed
|
|
|
146
166
|
All required secret names are declared
|
|
147
167
|
Prompt path exists
|
|
148
168
|
Tools have input schemas
|
|
149
|
-
|
|
169
|
+
Read-only tools use `:read` permissions; dangerous tools use operator-only non-read permissions
|
|
170
|
+
Managed Composio toolkit auth configs are available in AgentKit Cloud when configured
|
|
150
171
|
Provider model is supported by AgentKit
|
|
151
172
|
schema.sql is idempotent
|
|
152
173
|
README/AGENTS/CLAUDE match the capsule
|
|
@@ -172,7 +193,7 @@ echo 'OPENAI_API_KEY=sk-...' >> .env
|
|
|
172
193
|
|
|
173
194
|
Missing hosted secret:
|
|
174
195
|
|
|
175
|
-
Set the secret in AgentKit Cloud
|
|
196
|
+
Set the secret in AgentKit Cloud:
|
|
176
197
|
|
|
177
198
|
```sh
|
|
178
199
|
npm run agentkit -- secret set OPENAI_API_KEY --from-local-env
|
|
@@ -192,18 +213,15 @@ Before a hosted production deploy, run:
|
|
|
192
213
|
npm run agentkit -- deploy doctor
|
|
193
214
|
```
|
|
194
215
|
|
|
195
|
-
The doctor checks AgentKit Cloud login,
|
|
216
|
+
The doctor checks AgentKit Cloud login, Node version, hosted deploy entitlement, online deploy capacity, declared hosted secrets, local `.env` names that have not been uploaded with `npm run agentkit -- secret set`, and private-access runtime token handling. `agentkit deploy` runs the same readiness check automatically before building and uploading the artifact.
|
|
217
|
+
|
|
218
|
+
If the capsule configures `composioManaged({...})`, the doctor also checks the paid `managed_composio` entitlement and AgentKit Cloud toolkit readiness. `COMPOSIO_API_KEY` and toolkit auth config resolution are AgentKit-managed in this path; the user must not set them with `agentkit secret set`.
|
|
196
219
|
|
|
197
220
|
`alpha_access_required`:
|
|
198
221
|
|
|
199
|
-
Log in with an
|
|
222
|
+
Log in with an account that has hosted deploy access. If the user needs to buy slots first:
|
|
200
223
|
|
|
201
224
|
```sh
|
|
202
|
-
npm run agentkit --
|
|
225
|
+
npm run agentkit -- billing checkout --slots 1 --email user@example.com
|
|
226
|
+
npm run agentkit -- billing claim billint_... --secret bsec_...
|
|
203
227
|
```
|
|
204
|
-
|
|
205
|
-
## Operator Notes
|
|
206
|
-
|
|
207
|
-
This section is for AgentKit maintainers, not for capsule-building agents.
|
|
208
|
-
|
|
209
|
-
`agentkit deploy` sends the capsule artifact to the configured AgentKit Cloud API. The control plane owns infrastructure selection, provisioning, managed secrets, and public URL creation. Use `AGENTKIT_CLOUD_API_URL` only when testing a non-default control plane.
|
|
@@ -0,0 +1,72 @@
|
|
|
1
|
+
# Replay Production Traces
|
|
2
|
+
|
|
3
|
+
## Goal
|
|
4
|
+
|
|
5
|
+
Run collected production or local traces against the local Agent Capsule before redeploying a fix.
|
|
6
|
+
|
|
7
|
+
## When To Use This
|
|
8
|
+
|
|
9
|
+
Use this after `agentkit improve collect` and before `agentkit deploy` whenever prompts, tools, Knowledge, provider config, or channel behavior changed because of production evidence.
|
|
10
|
+
|
|
11
|
+
## Commands
|
|
12
|
+
|
|
13
|
+
```sh
|
|
14
|
+
agentkit replay .agentkit/improve/<run> --against local
|
|
15
|
+
```
|
|
16
|
+
|
|
17
|
+
Run the full local eval suite too:
|
|
18
|
+
|
|
19
|
+
```sh
|
|
20
|
+
npm run eval
|
|
21
|
+
```
|
|
22
|
+
|
|
23
|
+
## What Replay Checks
|
|
24
|
+
|
|
25
|
+
Replay sends each collected user turn through the local capsule in eval mode:
|
|
26
|
+
|
|
27
|
+
```txt
|
|
28
|
+
ctx.runtime.environment === "eval"
|
|
29
|
+
ctx.runtime.invocation === "eval"
|
|
30
|
+
```
|
|
31
|
+
|
|
32
|
+
This proves the current capsule can process the production turns without runtime errors. Generated eval files under `evals/regressions/` add committed behavior assertions.
|
|
33
|
+
|
|
34
|
+
## Write Tool Safety
|
|
35
|
+
|
|
36
|
+
Replay still runs the registered capsule tools. Any tool that can write externally must guard eval mode:
|
|
37
|
+
|
|
38
|
+
```ts
|
|
39
|
+
if (ctx.runtime.environment === "eval") {
|
|
40
|
+
return { sent: false, evalFixture: true };
|
|
41
|
+
}
|
|
42
|
+
```
|
|
43
|
+
|
|
44
|
+
Do not depend on prompt wording alone to prevent side effects.
|
|
45
|
+
|
|
46
|
+
## Deploy Gate
|
|
47
|
+
|
|
48
|
+
Before redeploying a production fix:
|
|
49
|
+
|
|
50
|
+
```sh
|
|
51
|
+
npm run typecheck
|
|
52
|
+
npm run agentkit -- inspect
|
|
53
|
+
npm run eval
|
|
54
|
+
agentkit replay .agentkit/improve/<run> --against local
|
|
55
|
+
agentkit deploy --smoke "hello"
|
|
56
|
+
```
|
|
57
|
+
|
|
58
|
+
If replay fails, inspect the failed trace id in `.agentkit/improve/<run>/traces/`, patch the capsule, and rerun replay.
|
|
59
|
+
|
|
60
|
+
## Troubleshooting
|
|
61
|
+
|
|
62
|
+
`Replay failed`:
|
|
63
|
+
|
|
64
|
+
Read the printed error and the matching trace file. Common causes are missing local secrets, unsafe tools that do not branch on eval mode, stale Knowledge sources, or provider differences.
|
|
65
|
+
|
|
66
|
+
`Replay skipped`:
|
|
67
|
+
|
|
68
|
+
The trace had no user message. It may still help diagnose delivery or deploy state, but it cannot be replayed as a conversation.
|
|
69
|
+
|
|
70
|
+
`provider_model_unsupported`:
|
|
71
|
+
|
|
72
|
+
The local provider config does not match the replay environment. Keep deterministic regression replay on `test/fake` unless the owner intentionally selected a real provider.
|
package/docs/guides/run-evals.md
CHANGED
|
@@ -22,6 +22,14 @@ Run evals:
|
|
|
22
22
|
agentkit eval run
|
|
23
23
|
```
|
|
24
24
|
|
|
25
|
+
`agentkit eval run` uses temporary local SQLite storage for eval execution. This keeps eval conversations and tool calls isolated from `.agentkit/agentkit.db`, so evals can run while a local chat or dev server is using the normal development database.
|
|
26
|
+
|
|
27
|
+
If the capsule uses npm scripts and Windows PowerShell blocks `npm.ps1`, use:
|
|
28
|
+
|
|
29
|
+
```sh
|
|
30
|
+
npm.cmd run eval
|
|
31
|
+
```
|
|
32
|
+
|
|
25
33
|
Create an eval from a stored conversation:
|
|
26
34
|
|
|
27
35
|
```sh
|
|
@@ -31,6 +39,15 @@ agentkit eval from-conversation <conversation-id>
|
|
|
31
39
|
agentkit eval run
|
|
32
40
|
```
|
|
33
41
|
|
|
42
|
+
Create evals from hosted or local production evidence:
|
|
43
|
+
|
|
44
|
+
```sh
|
|
45
|
+
agentkit improve collect --deploy --since 24h
|
|
46
|
+
agentkit improve evals .agentkit/improve/<run>
|
|
47
|
+
agentkit replay .agentkit/improve/<run> --against local
|
|
48
|
+
agentkit eval run
|
|
49
|
+
```
|
|
50
|
+
|
|
34
51
|
## Files Created Or Edited
|
|
35
52
|
|
|
36
53
|
Create or edit:
|
|
@@ -39,85 +56,137 @@ Create or edit:
|
|
|
39
56
|
evals/<name>.eval.ts
|
|
40
57
|
```
|
|
41
58
|
|
|
59
|
+
Create evals proactively from the agent brief and `AGENT_SPEC.md`. High-value evals cover identity and scope, required intake fields, confirmation before writes, no-leak/privacy rules, timezone and business-hour behavior, default durations or limits, tool-call payloads, unavailable slots, empty results, missing auth, rate limits, and timeouts. If a real conversation reveals a bug, add the smallest regression eval that would have failed before the fix.
|
|
60
|
+
|
|
42
61
|
Use conversations as source material:
|
|
43
62
|
|
|
44
63
|
```txt
|
|
45
64
|
.agentkit/agentkit.db
|
|
65
|
+
.agentkit/improve/<run>/
|
|
46
66
|
```
|
|
47
67
|
|
|
48
68
|
Do not edit `.agentkit/agentkit.db` by hand.
|
|
69
|
+
Do not edit `.agentkit/improve/<run>/bundle.json` by hand.
|
|
49
70
|
|
|
50
71
|
## Minimal Working Example
|
|
51
72
|
|
|
52
73
|
Generated smoke eval:
|
|
53
74
|
|
|
54
75
|
```ts
|
|
55
|
-
|
|
76
|
+
import { defineEval } from "@andreprado/agentkit";
|
|
77
|
+
|
|
78
|
+
export default defineEval({
|
|
56
79
|
name: "smoke",
|
|
57
80
|
input: "Say hello in one short sentence.",
|
|
58
81
|
expect: {
|
|
59
|
-
|
|
82
|
+
response: {
|
|
83
|
+
caseInsensitiveContains: "hello",
|
|
84
|
+
maxLength: 160,
|
|
85
|
+
},
|
|
60
86
|
},
|
|
61
|
-
};
|
|
87
|
+
});
|
|
62
88
|
```
|
|
63
89
|
|
|
64
90
|
Supported assertion types:
|
|
65
91
|
|
|
66
92
|
```txt
|
|
67
|
-
contains
|
|
68
|
-
|
|
69
|
-
|
|
70
|
-
|
|
71
|
-
|
|
93
|
+
response.contains
|
|
94
|
+
response.containsAll
|
|
95
|
+
response.containsAny
|
|
96
|
+
response.caseInsensitiveContains
|
|
97
|
+
response.notContains
|
|
98
|
+
response.regex
|
|
99
|
+
response.matchesRegex
|
|
100
|
+
response.notRegex
|
|
101
|
+
response.maxLength
|
|
102
|
+
tools.called
|
|
103
|
+
tools.calledOnce
|
|
104
|
+
tools.count
|
|
105
|
+
tools.order
|
|
106
|
+
tools.persisted
|
|
107
|
+
```
|
|
108
|
+
|
|
109
|
+
Older flat aliases still work, including `contains`, `not_contains`, `regex`, `matches_regex`, and `persisted_tool_call`.
|
|
110
|
+
|
|
111
|
+
For date-sensitive evals, set top-level `now` to an ISO timestamp with an explicit timezone designator such as `Z` or `-05:00`. AgentKit uses that fixed clock for every turn and tool call in the eval:
|
|
112
|
+
|
|
113
|
+
```ts
|
|
114
|
+
export default defineEval({
|
|
115
|
+
name: "appointment relative date",
|
|
116
|
+
now: "2026-02-04T02:30:00.000Z",
|
|
117
|
+
input: "What is today's date?",
|
|
118
|
+
expect: {
|
|
119
|
+
response: {
|
|
120
|
+
containsAll: ["2026-02-03", "Tuesday"],
|
|
121
|
+
},
|
|
122
|
+
},
|
|
123
|
+
});
|
|
72
124
|
```
|
|
73
125
|
|
|
74
126
|
Multi-turn conversation evals use `turns`:
|
|
75
127
|
|
|
76
128
|
```ts
|
|
77
|
-
|
|
129
|
+
import { defineEval } from "@andreprado/agentkit";
|
|
130
|
+
|
|
131
|
+
export default defineEval({
|
|
78
132
|
name: "buyer under budget",
|
|
79
133
|
turns: [
|
|
80
134
|
{
|
|
81
135
|
input: "I want a house up to 600k near Pinheiros.",
|
|
82
136
|
expect: {
|
|
83
|
-
|
|
84
|
-
|
|
85
|
-
|
|
86
|
-
|
|
137
|
+
tools: {
|
|
138
|
+
calledOnce: "buscar_imoveis",
|
|
139
|
+
persisted: {
|
|
140
|
+
name: "buscar_imoveis",
|
|
141
|
+
status: "completed",
|
|
142
|
+
input: { maxPrice: 600000 },
|
|
143
|
+
},
|
|
87
144
|
},
|
|
88
145
|
},
|
|
89
146
|
},
|
|
90
147
|
{
|
|
91
148
|
input: "Show me the best two.",
|
|
92
149
|
expect: {
|
|
93
|
-
|
|
150
|
+
response: {
|
|
151
|
+
containsAll: ["Pinheiros", "R$"],
|
|
152
|
+
},
|
|
94
153
|
},
|
|
95
154
|
},
|
|
96
155
|
],
|
|
97
|
-
};
|
|
156
|
+
});
|
|
98
157
|
```
|
|
99
158
|
|
|
100
|
-
`
|
|
159
|
+
`tools.persisted` validates the tool call saved in the eval run's local SQLite `tool_calls`, not a provider-specific raw response shape. It can be a tool name string or an object with `name`, `input`, `output`, `rendered`, `status`, and/or `visibility`.
|
|
160
|
+
|
|
161
|
+
Use `tools.count` for the exact number of persisted calls in that turn, `tools.calledOnce` for exactly one call by name, and `tools.order` for required relative order. `tools.order` allows extra calls before, between, or after the named calls; pair it with `tools.count` when the exact call set matters.
|
|
101
162
|
|
|
102
163
|
Use response assertions and persisted tool assertions together when internal operational output must not leak:
|
|
103
164
|
|
|
104
165
|
```ts
|
|
105
|
-
|
|
166
|
+
import { defineEval } from "@andreprado/agentkit";
|
|
167
|
+
|
|
168
|
+
export default defineEval({
|
|
106
169
|
name: "triage lead",
|
|
107
170
|
input: '{"tool":"triage_real_estate_lead","input":{"email":"ada@example.com"}}',
|
|
108
171
|
expect: {
|
|
109
|
-
|
|
110
|
-
|
|
111
|
-
|
|
112
|
-
|
|
113
|
-
|
|
114
|
-
|
|
172
|
+
response: {
|
|
173
|
+
notContains: ["hot", "score"],
|
|
174
|
+
notRegex: ["API_KEY|secret|token"],
|
|
175
|
+
},
|
|
176
|
+
tools: {
|
|
177
|
+
calledOnce: "triage_real_estate_lead",
|
|
178
|
+
persisted: {
|
|
179
|
+
name: "triage_real_estate_lead",
|
|
180
|
+
status: "completed",
|
|
181
|
+
visibility: "internal",
|
|
182
|
+
output: { status: "hot" },
|
|
183
|
+
},
|
|
115
184
|
},
|
|
116
185
|
},
|
|
117
|
-
};
|
|
186
|
+
});
|
|
118
187
|
```
|
|
119
188
|
|
|
120
|
-
`tool_call`
|
|
189
|
+
`tool_call` and `persisted_tool_call` remain accepted as backwards-compatible aliases, but new evals should use `tools.persisted`.
|
|
121
190
|
|
|
122
191
|
Safe external-tool pattern:
|
|
123
192
|
|
|
@@ -152,6 +221,7 @@ export const sendFollowupEmail = defineTool({
|
|
|
152
221
|
- Keep eval tool calls deterministic and non-destructive.
|
|
153
222
|
- For tools that would write, delete, charge money, send email, or call a real customer system, branch inside the registered tool on `ctx.runtime.environment === "eval"` and return safe fixture output.
|
|
154
223
|
- Treat conversations as source material, not as automatically safe training data.
|
|
224
|
+
- Review generated regression evals from `agentkit improve evals` before committing them. AgentKit redacts common email, phone, bearer token, and key patterns in generated eval text, but you must still remove or generalize domain-specific client PII and replace brittle exact prose assertions with the important behavior when needed.
|
|
155
225
|
|
|
156
226
|
## Verification
|
|
157
227
|
|