@andreprado/agentkit 0.1.0-alpha.14 → 0.1.0-alpha.16
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +2 -0
- package/docs/guides/add-channel.md +63 -0
- package/docs/guides/channel-security.md +32 -0
- package/docs/guides/connect-telegram.md +58 -0
- package/docs/guides/connect-whatsapp-zapster.md +65 -0
- package/docs/guides/run-evals.md +73 -25
- package/docs/llms-full.txt +24 -9
- package/docs/llms.txt +2 -0
- package/package.json +1 -1
- package/src/cli/cloud-client.ts +30 -10
- package/src/cli/commands/channels.ts +2 -0
- package/src/cli/deploy-readiness.ts +32 -11
- package/src/cli/index.ts +20 -6
- package/src/cloud/client.ts +4 -3
- package/src/cloud/contracts.ts +1 -1
- package/src/create-project.ts +1 -1
- package/src/index.ts +110 -1
- package/src/providers/pi.ts +14 -1
- package/src/providers/test.ts +36 -0
- package/src/runtime/channel-test-harness.ts +2 -0
- package/src/runtime/channels/telegram.ts +326 -10
- package/src/runtime/channels/whatsapp-zapster.ts +319 -0
- package/src/runtime/channels.ts +47 -1
- package/src/runtime/chat.ts +59 -42
- package/src/runtime/config.ts +96 -4
- package/src/runtime/core/manifest.ts +35 -3
- package/src/runtime/deploy-readiness.ts +3 -3
- package/src/runtime/dev-server.ts +243 -17
- package/src/runtime/env.ts +8 -3
- package/src/runtime/evals.ts +404 -69
- package/src/runtime/inspect.ts +46 -0
- package/src/runtime/prompt-context.ts +141 -0
- package/src/runtime/runtime-contract.ts +17 -7
- package/src/runtime/targets/cloudflare/build.ts +25 -3
- package/src/runtime/targets/container/server.ts +1 -1
- package/src/runtime/targets/vps/deploy.ts +25 -8
- package/src/runtime/tool-runner.ts +7 -0
- package/src/runtime/tools.ts +8 -2
- package/src/runtime/transcription.ts +483 -0
- package/src/templates/blank.ts +8 -3
- package/src/templates/dentista.ts +18 -10
- package/src/templates/skills/agentkit-build-agent/SKILL.md +6 -5
- package/src/templates/skills/agentkit-build-agent/templates/appointment-intake.instructions.md +2 -1
- package/src/templates/skills/agentkit-capsule/SKILL.md +1 -1
- package/src/templates/skills/agentkit-channels/SKILL.md +34 -1
- package/src/templates/skills/agentkit-channels/references/channel-debugging.md +13 -0
- package/src/templates/skills/agentkit-channels/references/telegram.md +32 -0
- package/src/templates/skills/agentkit-channels/references/whatsapp-zapster.md +29 -0
- package/src/templates/skills/agentkit-evals/SKILL.md +53 -13
- package/src/templates/skills/agentkit-evals/templates/multi-turn.eval.md +13 -6
- package/src/templates/skills/agentkit-evals/templates/no-leak.eval.md +8 -4
- package/src/templates/skills/agentkit-evals/templates/smoke.eval.md +8 -4
- package/src/templates/skills/agentkit-evals/templates/tool-call.eval.md +16 -7
- package/src/templates/skills/agentkit-prompts/SKILL.md +3 -1
- package/src/templates/skills/agentkit-tools/SKILL.md +2 -1
- package/src/templates/support.ts +8 -3
package/README.md
CHANGED
|
@@ -48,6 +48,8 @@ agentkit conversations trace <conversation-id>
|
|
|
48
48
|
|
|
49
49
|
Knowledge indexes local `.md`, `.txt`, and `.csv` sources into the capsule database and registers the internal `agentkit_search_knowledge` chat tool when `knowledge` is configured in `agentkit.config.ts`. `agentkit dev` and `agentkit chat` sync configured sources automatically; when embeddings are configured, local semantic search uses a libSQL vector sidecar. Hosted Cloudflare deploys package configured local Knowledge sources and sync them into the project Turso database automatically during `agentkit deploy`; when embeddings are configured, the deploy also creates and populates the hosted Turso vector index.
|
|
50
50
|
|
|
51
|
+
Every chat run receives dynamic runtime date context: current ISO timestamp, local date, weekday, local date/time, and timezone. Set `timeZone` in `agentkit.config.ts` for scheduling agents so relative dates like "today" and "next Friday" resolve in the right business/user timezone.
|
|
52
|
+
|
|
51
53
|
## Deploy Later
|
|
52
54
|
|
|
53
55
|
```sh
|
|
@@ -87,6 +87,62 @@ agentkit channels buffers clear support-whatsapp <conversation-id>
|
|
|
87
87
|
agentkit channels buffers retry support-whatsapp <conversation-id>
|
|
88
88
|
```
|
|
89
89
|
|
|
90
|
+
## Auto Transcribe Audio
|
|
91
|
+
|
|
92
|
+
Use `transcription` at the agent level and `audio.mode: "transcribe"` on each channel that should accept voice notes or audio files.
|
|
93
|
+
|
|
94
|
+
```ts
|
|
95
|
+
export default defineAgent({
|
|
96
|
+
name: "support-agent",
|
|
97
|
+
runtime: "edge",
|
|
98
|
+
provider: { name: "openai", model: "gpt-5.4-mini" },
|
|
99
|
+
instructions: "./prompts/instructions.md",
|
|
100
|
+
transcription: {
|
|
101
|
+
provider: "groq",
|
|
102
|
+
model: "whisper-large-v3-turbo",
|
|
103
|
+
secret: "GROQ_API_KEY",
|
|
104
|
+
language: "pt",
|
|
105
|
+
limits: {
|
|
106
|
+
maxDurationSeconds: 180,
|
|
107
|
+
maxBytes: 20_000_000,
|
|
108
|
+
},
|
|
109
|
+
},
|
|
110
|
+
channels: [
|
|
111
|
+
telegramChannel({
|
|
112
|
+
name: "support-telegram",
|
|
113
|
+
audio: { mode: "transcribe" },
|
|
114
|
+
}),
|
|
115
|
+
whatsappChannel({
|
|
116
|
+
name: "support-whatsapp",
|
|
117
|
+
provider: "zapster",
|
|
118
|
+
audio: { mode: "transcribe" },
|
|
119
|
+
}),
|
|
120
|
+
],
|
|
121
|
+
access: { mode: "public" },
|
|
122
|
+
storage: { driver: "agentkit" },
|
|
123
|
+
});
|
|
124
|
+
```
|
|
125
|
+
|
|
126
|
+
Supported transcription providers in V1:
|
|
127
|
+
|
|
128
|
+
| Provider | Default secret | Supported models |
|
|
129
|
+
| --- | --- | --- |
|
|
130
|
+
| `openai` | `OPENAI_API_KEY` | `gpt-4o-mini-transcribe`, `gpt-4o-transcribe`, `whisper-1` |
|
|
131
|
+
| `groq` | `GROQ_API_KEY` | `whisper-large-v3-turbo`, `whisper-large-v3`, `distil-whisper-large-v3-en` |
|
|
132
|
+
|
|
133
|
+
Audio transcription is paid by the capsule owner because AgentKit only passes through the configured provider secret. Hosted channel creation automatically requires the transcription secret when a channel enables `audio.mode: "transcribe"`.
|
|
134
|
+
|
|
135
|
+
Processing order:
|
|
136
|
+
|
|
137
|
+
1. Provider webhook is validated and deduped.
|
|
138
|
+
2. Channel adapter normalizes the audio metadata.
|
|
139
|
+
3. AgentKit records `audio_received` and enqueues a channel job before acknowledging the webhook.
|
|
140
|
+
4. The retryable channel worker downloads the audio using the channel provider secret.
|
|
141
|
+
5. The transcription adapter sends the file to the configured transcription provider.
|
|
142
|
+
6. The agent receives a text message containing the transcript.
|
|
143
|
+
|
|
144
|
+
V1 keeps the raw audio in memory for the request path and delivery metadata only records redacted status/error fields. `rawAudioTtlSeconds` is part of the manifest contract for future object-storage retention, but V1 does not persist raw audio by default.
|
|
145
|
+
|
|
90
146
|
## Safety Rules
|
|
91
147
|
|
|
92
148
|
- Never put provider token values in `agentkit.config.ts`.
|
|
@@ -94,6 +150,7 @@ agentkit channels buffers retry support-whatsapp <conversation-id>
|
|
|
94
150
|
- Treat channel webhook URLs as public transport endpoints. Provider validation or the AgentKit website channel token controls authenticity.
|
|
95
151
|
- Keep channels separate from tools. Channels deliver user messages; tools let the agent call external systems.
|
|
96
152
|
- Keep `maxMessages` and `maxChars` bounded so one burst cannot create an oversized prompt or unexpected model spend.
|
|
153
|
+
- Keep `audio.limits` bounded so one voice note cannot create unexpected transcription spend.
|
|
97
154
|
|
|
98
155
|
## Verification
|
|
99
156
|
|
|
@@ -124,4 +181,10 @@ The provider webhook secret, token, or origin header does not match the managed
|
|
|
124
181
|
`channel_limit_exceeded`:
|
|
125
182
|
The channel daily message limit was reached. Website requests return `429`; Telegram and WhatsApp are acknowledged and skipped to avoid provider retry storms.
|
|
126
183
|
|
|
184
|
+
`transcription_secret_missing`:
|
|
185
|
+
Set the managed transcription secret declared by `agentkit inspect`, for example `OPENAI_API_KEY` or `GROQ_API_KEY`.
|
|
186
|
+
|
|
187
|
+
`transcription_audio_format_unsupported`:
|
|
188
|
+
The channel delivered an audio format the configured transcription provider does not accept. Telegram voice notes are OGG/Opus and work with Groq in V1; OpenAI accepts MP3, MP4, MPEG, MPGA, M4A, WAV, and WEBM in the AgentKit adapter.
|
|
189
|
+
|
|
127
190
|
Buffered messages stay in `buffered` delivery state until the quiet window or max wait flushes them into one queued run.
|
|
@@ -15,6 +15,7 @@ agentkit inspect
|
|
|
15
15
|
agentkit channels status <name>
|
|
16
16
|
agentkit channels deliveries show <delivery-id>
|
|
17
17
|
bun test packages/agentkit/src/runtime/channels/adapters.test.ts
|
|
18
|
+
bun test packages/agentkit/src/runtime/transcription.test.ts
|
|
18
19
|
bun test packages/agentkit/src/runtime/deploy.test.ts
|
|
19
20
|
```
|
|
20
21
|
|
|
@@ -41,9 +42,12 @@ Zapster smoke also requires `ZAPSTER_API_KEY`, `ZAPSTER_INSTANCE_ID`, and `AGENT
|
|
|
41
42
|
- Validate provider authenticity when the provider supports it.
|
|
42
43
|
- Do not store channel plumbing in the user's Turso database.
|
|
43
44
|
- Do not store raw webhook bodies in delivery records; store a SHA-256 hash.
|
|
45
|
+
- Do not store raw audio in delivery records. V1 audio transcription downloads provider media into memory and stores only redacted metadata and state transitions.
|
|
44
46
|
- Redact bearer tokens, bot tokens, signing secrets, provider API tokens, and phone numbers.
|
|
45
47
|
- Inject only channel-declared secrets into adapter code.
|
|
48
|
+
- Inject transcription secrets only when the deploy manifest declares transcription and the channel uses `audio.mode: "transcribe"`.
|
|
46
49
|
- Prefer acknowledging Telegram/WhatsApp over retry storms when a channel-level limit is exceeded.
|
|
50
|
+
- Keep audio duration and byte limits bounded before provider calls to avoid uncontrolled transcription spend.
|
|
47
51
|
|
|
48
52
|
## Minimal Working Example
|
|
49
53
|
|
|
@@ -54,17 +58,42 @@ telegramChannel({
|
|
|
54
58
|
});
|
|
55
59
|
```
|
|
56
60
|
|
|
61
|
+
With audio transcription:
|
|
62
|
+
|
|
63
|
+
```ts
|
|
64
|
+
export default defineAgent({
|
|
65
|
+
// ...
|
|
66
|
+
transcription: {
|
|
67
|
+
provider: "groq",
|
|
68
|
+
model: "whisper-large-v3-turbo",
|
|
69
|
+
secret: "GROQ_API_KEY",
|
|
70
|
+
limits: {
|
|
71
|
+
maxDurationSeconds: 180,
|
|
72
|
+
maxBytes: 20_000_000,
|
|
73
|
+
},
|
|
74
|
+
},
|
|
75
|
+
channels: [
|
|
76
|
+
telegramChannel({
|
|
77
|
+
name: "support-telegram",
|
|
78
|
+
audio: { mode: "transcribe" },
|
|
79
|
+
}),
|
|
80
|
+
],
|
|
81
|
+
});
|
|
82
|
+
```
|
|
83
|
+
|
|
57
84
|
The config contains secret names only. Hosted responses report:
|
|
58
85
|
|
|
59
86
|
```txt
|
|
60
87
|
TELEGRAM_BOT_TOKEN: set
|
|
61
88
|
TELEGRAM_WEBHOOK_SECRET: missing
|
|
89
|
+
GROQ_API_KEY: set
|
|
62
90
|
```
|
|
63
91
|
|
|
64
92
|
## Verification
|
|
65
93
|
|
|
66
94
|
```sh
|
|
67
95
|
bun test packages/agentkit/src/runtime/channels/adapters.test.ts
|
|
96
|
+
bun test packages/agentkit/src/runtime/transcription.test.ts
|
|
68
97
|
bun test packages/agentkit/src/runtime/deploy.test.ts
|
|
69
98
|
npm run typecheck
|
|
70
99
|
```
|
|
@@ -79,3 +108,6 @@ Redact it to a stable partial form such as `5511******9999`.
|
|
|
79
108
|
|
|
80
109
|
Webhook accepts invalid signatures or origin headers:
|
|
81
110
|
Fix `verifyWebhook` for the adapter before enabling provider setup docs.
|
|
111
|
+
|
|
112
|
+
Raw audio or transcript provider secret appears in output:
|
|
113
|
+
Stop and add a regression test before changing behavior. Delivery APIs may include transcript text in normalized agent messages after successful transcription, but must never include raw bytes or provider secret values.
|
|
@@ -27,6 +27,12 @@ TELEGRAM_BOT_TOKEN
|
|
|
27
27
|
TELEGRAM_WEBHOOK_SECRET
|
|
28
28
|
```
|
|
29
29
|
|
|
30
|
+
If Telegram audio transcription is enabled, also set the transcription provider secret declared by `agentkit inspect`, usually:
|
|
31
|
+
|
|
32
|
+
```txt
|
|
33
|
+
GROQ_API_KEY
|
|
34
|
+
```
|
|
35
|
+
|
|
30
36
|
## Files Created Or Edited
|
|
31
37
|
|
|
32
38
|
- `agentkit.config.ts`: `telegramChannel({ name: "support-telegram" })`.
|
|
@@ -66,6 +72,52 @@ telegramChannel({
|
|
|
66
72
|
})
|
|
67
73
|
```
|
|
68
74
|
|
|
75
|
+
## Auto Transcribe Telegram Audio
|
|
76
|
+
|
|
77
|
+
Telegram voice notes arrive as OGG/Opus. In V1, use Groq for the most obvious Telegram voice-note path because the AgentKit Groq adapter accepts `audio/ogg`.
|
|
78
|
+
|
|
79
|
+
```ts
|
|
80
|
+
import { defineAgent, telegramChannel } from "@andreprado/agentkit";
|
|
81
|
+
|
|
82
|
+
export default defineAgent({
|
|
83
|
+
name: "support-agent",
|
|
84
|
+
runtime: "edge",
|
|
85
|
+
provider: { name: "openai", model: "gpt-5.4-mini" },
|
|
86
|
+
instructions: "./prompts/instructions.md",
|
|
87
|
+
transcription: {
|
|
88
|
+
provider: "groq",
|
|
89
|
+
model: "whisper-large-v3-turbo",
|
|
90
|
+
secret: "GROQ_API_KEY",
|
|
91
|
+
language: "pt",
|
|
92
|
+
limits: {
|
|
93
|
+
maxDurationSeconds: 180,
|
|
94
|
+
maxBytes: 20_000_000,
|
|
95
|
+
},
|
|
96
|
+
},
|
|
97
|
+
channels: [
|
|
98
|
+
telegramChannel({
|
|
99
|
+
name: "support-telegram",
|
|
100
|
+
audio: {
|
|
101
|
+
mode: "transcribe",
|
|
102
|
+
},
|
|
103
|
+
}),
|
|
104
|
+
],
|
|
105
|
+
access: { mode: "public" },
|
|
106
|
+
storage: { driver: "agentkit" },
|
|
107
|
+
});
|
|
108
|
+
```
|
|
109
|
+
|
|
110
|
+
Processing order:
|
|
111
|
+
|
|
112
|
+
1. AgentKit validates `X-Telegram-Bot-Api-Secret-Token`.
|
|
113
|
+
2. Telegram `voice` or `audio` payloads become normalized audio messages.
|
|
114
|
+
3. AgentKit records `audio_received` and enqueues a channel job before acknowledging Telegram.
|
|
115
|
+
4. The retryable channel worker calls Telegram `getFile`, downloads the file with `TELEGRAM_BOT_TOKEN`, and does not log the token.
|
|
116
|
+
5. AgentKit sends the audio bytes to the configured transcription provider using the user's managed secret.
|
|
117
|
+
6. The agent run receives a text message with the transcript.
|
|
118
|
+
|
|
119
|
+
OpenAI transcription can be used for Telegram files that arrive as MP3, MP4, MPEG, MPGA, M4A, WAV, or WEBM. Telegram voice notes are usually OGG/Opus, so they should use Groq in V1 unless the provider payload is converted before it reaches AgentKit.
|
|
120
|
+
|
|
69
121
|
## Setup Behavior
|
|
70
122
|
|
|
71
123
|
`agentkit channels setup support-telegram` is read-only and prints the webhook URL.
|
|
@@ -108,3 +160,9 @@ The incoming Telegram secret token does not match `TELEGRAM_WEBHOOK_SECRET`.
|
|
|
108
160
|
|
|
109
161
|
`channel_payload_invalid`:
|
|
110
162
|
The update is malformed or is not a supported private text message.
|
|
163
|
+
|
|
164
|
+
`transcription_secret_missing`:
|
|
165
|
+
Set `GROQ_API_KEY` or the custom secret named in `transcription.secret` as a managed hosted secret.
|
|
166
|
+
|
|
167
|
+
`transcription_audio_format_unsupported`:
|
|
168
|
+
The configured transcription provider does not accept the Telegram file format. Use Groq for OGG/Opus voice notes in V1.
|
|
@@ -33,6 +33,18 @@ Optional hardening secret:
|
|
|
33
33
|
ZAPSTER_WEBHOOK_TOKEN
|
|
34
34
|
```
|
|
35
35
|
|
|
36
|
+
If WhatsApp audio transcription is enabled, also set the transcription provider secret declared by `agentkit inspect`, usually:
|
|
37
|
+
|
|
38
|
+
```txt
|
|
39
|
+
OPENAI_API_KEY
|
|
40
|
+
```
|
|
41
|
+
|
|
42
|
+
or:
|
|
43
|
+
|
|
44
|
+
```txt
|
|
45
|
+
GROQ_API_KEY
|
|
46
|
+
```
|
|
47
|
+
|
|
36
48
|
## Files Created Or Edited
|
|
37
49
|
|
|
38
50
|
- `agentkit.config.ts`: `whatsappChannel({ name: "support-whatsapp", provider: "zapster" })`.
|
|
@@ -73,6 +85,53 @@ whatsappChannel({
|
|
|
73
85
|
})
|
|
74
86
|
```
|
|
75
87
|
|
|
88
|
+
## Auto Transcribe WhatsApp Audio
|
|
89
|
+
|
|
90
|
+
Use `transcription` at the agent level and `audio.mode: "transcribe"` on the Zapster channel.
|
|
91
|
+
|
|
92
|
+
```ts
|
|
93
|
+
import { defineAgent, whatsappChannel } from "@andreprado/agentkit";
|
|
94
|
+
|
|
95
|
+
export default defineAgent({
|
|
96
|
+
name: "support-agent",
|
|
97
|
+
runtime: "edge",
|
|
98
|
+
provider: { name: "openai", model: "gpt-5.4-mini" },
|
|
99
|
+
instructions: "./prompts/instructions.md",
|
|
100
|
+
transcription: {
|
|
101
|
+
provider: "openai",
|
|
102
|
+
model: "gpt-4o-mini-transcribe",
|
|
103
|
+
secret: "OPENAI_API_KEY",
|
|
104
|
+
language: "pt",
|
|
105
|
+
limits: {
|
|
106
|
+
maxDurationSeconds: 180,
|
|
107
|
+
maxBytes: 20_000_000,
|
|
108
|
+
},
|
|
109
|
+
},
|
|
110
|
+
channels: [
|
|
111
|
+
whatsappChannel({
|
|
112
|
+
name: "support-whatsapp",
|
|
113
|
+
provider: "zapster",
|
|
114
|
+
audio: {
|
|
115
|
+
mode: "transcribe",
|
|
116
|
+
},
|
|
117
|
+
}),
|
|
118
|
+
],
|
|
119
|
+
access: { mode: "public" },
|
|
120
|
+
storage: { driver: "agentkit" },
|
|
121
|
+
});
|
|
122
|
+
```
|
|
123
|
+
|
|
124
|
+
Processing order:
|
|
125
|
+
|
|
126
|
+
1. AgentKit validates Zapster origin headers and the optional webhook token.
|
|
127
|
+
2. Zapster audio payloads become normalized audio messages.
|
|
128
|
+
3. AgentKit records `audio_received` and enqueues a channel job before acknowledging Zapster.
|
|
129
|
+
4. The retryable channel worker downloads the media URL from a trusted Zapster HTTPS host using `ZAPSTER_API_KEY`.
|
|
130
|
+
5. AgentKit sends the audio bytes to the configured transcription provider using the user's managed secret.
|
|
131
|
+
6. The agent run receives a text message with the transcript.
|
|
132
|
+
|
|
133
|
+
Zapster payloads must include a media download URL such as `audio.downloadUrl`, `audio.url`, `audio.mediaUrl`, or the snake_case equivalents. The URL must be HTTPS and hosted by Zapster; AgentKit rejects arbitrary webhook-provided hosts before sending `ZAPSTER_API_KEY`. If Zapster sends only a media ID without a download URL, AgentKit records `channel_audio_download_unavailable` and does not create an agent run for that audio in V1.
|
|
134
|
+
|
|
76
135
|
## Setup Behavior
|
|
77
136
|
|
|
78
137
|
`agentkit channels setup support-whatsapp` prints the stable AgentKit webhook URL. Paste it into Zapster webhook settings.
|
|
@@ -125,3 +184,9 @@ Zapster is not sending the expected instance/webhook IDs, or the optional query
|
|
|
125
184
|
|
|
126
185
|
`channel_unsupported_message_type` or skipped delivery:
|
|
127
186
|
The inbound WhatsApp event was not supported text. Inspect the delivery record for provider metadata.
|
|
187
|
+
|
|
188
|
+
`channel_audio_download_unavailable`:
|
|
189
|
+
Zapster sent an audio event without a usable media download URL, or the URL was not an HTTPS Zapster media host. Configure Zapster to include a trusted Zapster media URL in webhook payloads, or add a Zapster media lookup adapter before enabling `audio.mode: "transcribe"`.
|
|
190
|
+
|
|
191
|
+
`transcription_secret_missing`:
|
|
192
|
+
Set `OPENAI_API_KEY`, `GROQ_API_KEY`, or the custom secret named in `transcription.secret` as a managed hosted secret.
|
package/docs/guides/run-evals.md
CHANGED
|
@@ -52,72 +52,120 @@ Do not edit `.agentkit/agentkit.db` by hand.
|
|
|
52
52
|
Generated smoke eval:
|
|
53
53
|
|
|
54
54
|
```ts
|
|
55
|
-
|
|
55
|
+
import { defineEval } from "@andreprado/agentkit";
|
|
56
|
+
|
|
57
|
+
export default defineEval({
|
|
56
58
|
name: "smoke",
|
|
57
59
|
input: "Say hello in one short sentence.",
|
|
58
60
|
expect: {
|
|
59
|
-
|
|
61
|
+
response: {
|
|
62
|
+
caseInsensitiveContains: "hello",
|
|
63
|
+
maxLength: 160,
|
|
64
|
+
},
|
|
60
65
|
},
|
|
61
|
-
};
|
|
66
|
+
});
|
|
62
67
|
```
|
|
63
68
|
|
|
64
69
|
Supported assertion types:
|
|
65
70
|
|
|
66
71
|
```txt
|
|
67
|
-
contains
|
|
68
|
-
|
|
69
|
-
|
|
70
|
-
|
|
71
|
-
|
|
72
|
+
response.contains
|
|
73
|
+
response.containsAll
|
|
74
|
+
response.containsAny
|
|
75
|
+
response.caseInsensitiveContains
|
|
76
|
+
response.notContains
|
|
77
|
+
response.regex
|
|
78
|
+
response.matchesRegex
|
|
79
|
+
response.notRegex
|
|
80
|
+
response.maxLength
|
|
81
|
+
tools.called
|
|
82
|
+
tools.calledOnce
|
|
83
|
+
tools.count
|
|
84
|
+
tools.order
|
|
85
|
+
tools.persisted
|
|
86
|
+
```
|
|
87
|
+
|
|
88
|
+
Older flat aliases still work, including `contains`, `not_contains`, `regex`, `matches_regex`, and `persisted_tool_call`.
|
|
89
|
+
|
|
90
|
+
For date-sensitive evals, set top-level `now` to an ISO timestamp with an explicit timezone designator such as `Z` or `-05:00`. AgentKit uses that fixed clock for every turn and tool call in the eval:
|
|
91
|
+
|
|
92
|
+
```ts
|
|
93
|
+
export default defineEval({
|
|
94
|
+
name: "appointment relative date",
|
|
95
|
+
now: "2026-02-04T02:30:00.000Z",
|
|
96
|
+
input: "What is today's date?",
|
|
97
|
+
expect: {
|
|
98
|
+
response: {
|
|
99
|
+
containsAll: ["2026-02-03", "Tuesday"],
|
|
100
|
+
},
|
|
101
|
+
},
|
|
102
|
+
});
|
|
72
103
|
```
|
|
73
104
|
|
|
74
105
|
Multi-turn conversation evals use `turns`:
|
|
75
106
|
|
|
76
107
|
```ts
|
|
77
|
-
|
|
108
|
+
import { defineEval } from "@andreprado/agentkit";
|
|
109
|
+
|
|
110
|
+
export default defineEval({
|
|
78
111
|
name: "buyer under budget",
|
|
79
112
|
turns: [
|
|
80
113
|
{
|
|
81
114
|
input: "I want a house up to 600k near Pinheiros.",
|
|
82
115
|
expect: {
|
|
83
|
-
|
|
84
|
-
|
|
85
|
-
|
|
86
|
-
|
|
116
|
+
tools: {
|
|
117
|
+
calledOnce: "buscar_imoveis",
|
|
118
|
+
persisted: {
|
|
119
|
+
name: "buscar_imoveis",
|
|
120
|
+
status: "completed",
|
|
121
|
+
input: { maxPrice: 600000 },
|
|
122
|
+
},
|
|
87
123
|
},
|
|
88
124
|
},
|
|
89
125
|
},
|
|
90
126
|
{
|
|
91
127
|
input: "Show me the best two.",
|
|
92
128
|
expect: {
|
|
93
|
-
|
|
129
|
+
response: {
|
|
130
|
+
containsAll: ["Pinheiros", "R$"],
|
|
131
|
+
},
|
|
94
132
|
},
|
|
95
133
|
},
|
|
96
134
|
],
|
|
97
|
-
};
|
|
135
|
+
});
|
|
98
136
|
```
|
|
99
137
|
|
|
100
|
-
`
|
|
138
|
+
`tools.persisted` validates the tool call saved in local SQLite `tool_calls`, not a provider-specific raw response shape. It can be a tool name string or an object with `name`, `input`, `output`, `rendered`, `status`, and/or `visibility`.
|
|
139
|
+
|
|
140
|
+
Use `tools.count` for the exact number of persisted calls in that turn, `tools.calledOnce` for exactly one call by name, and `tools.order` for required relative order. `tools.order` allows extra calls before, between, or after the named calls; pair it with `tools.count` when the exact call set matters.
|
|
101
141
|
|
|
102
142
|
Use response assertions and persisted tool assertions together when internal operational output must not leak:
|
|
103
143
|
|
|
104
144
|
```ts
|
|
105
|
-
|
|
145
|
+
import { defineEval } from "@andreprado/agentkit";
|
|
146
|
+
|
|
147
|
+
export default defineEval({
|
|
106
148
|
name: "triage lead",
|
|
107
149
|
input: '{"tool":"triage_real_estate_lead","input":{"email":"ada@example.com"}}',
|
|
108
150
|
expect: {
|
|
109
|
-
|
|
110
|
-
|
|
111
|
-
|
|
112
|
-
|
|
113
|
-
|
|
114
|
-
|
|
151
|
+
response: {
|
|
152
|
+
notContains: ["hot", "score"],
|
|
153
|
+
notRegex: ["API_KEY|secret|token"],
|
|
154
|
+
},
|
|
155
|
+
tools: {
|
|
156
|
+
calledOnce: "triage_real_estate_lead",
|
|
157
|
+
persisted: {
|
|
158
|
+
name: "triage_real_estate_lead",
|
|
159
|
+
status: "completed",
|
|
160
|
+
visibility: "internal",
|
|
161
|
+
output: { status: "hot" },
|
|
162
|
+
},
|
|
115
163
|
},
|
|
116
164
|
},
|
|
117
|
-
};
|
|
165
|
+
});
|
|
118
166
|
```
|
|
119
167
|
|
|
120
|
-
`tool_call`
|
|
168
|
+
`tool_call` and `persisted_tool_call` remain accepted as backwards-compatible aliases, but new evals should use `tools.persisted`.
|
|
121
169
|
|
|
122
170
|
Safe external-tool pattern:
|
|
123
171
|
|
package/docs/llms-full.txt
CHANGED
|
@@ -96,7 +96,7 @@ Prefer `env set --stdin` or `--from-env` for local secret values, and prefer `se
|
|
|
96
96
|
Planned commands described by the contract but not implemented yet:
|
|
97
97
|
|
|
98
98
|
```sh
|
|
99
|
-
agentkit eval
|
|
99
|
+
agentkit eval from-conversation <conversation-id>
|
|
100
100
|
```
|
|
101
101
|
|
|
102
102
|
## Create And Test A Capsule
|
|
@@ -166,6 +166,7 @@ export default defineAgent({
|
|
|
166
166
|
name: "test",
|
|
167
167
|
model: "fake",
|
|
168
168
|
},
|
|
169
|
+
timeZone: "America/New_York",
|
|
169
170
|
instructions: "./prompts/instructions.md",
|
|
170
171
|
secrets: [],
|
|
171
172
|
tools: [],
|
|
@@ -178,6 +179,8 @@ export default defineAgent({
|
|
|
178
179
|
});
|
|
179
180
|
```
|
|
180
181
|
|
|
182
|
+
`timeZone` is optional and must be an IANA time zone when set. AgentKit injects dynamic runtime context into every chat run: current ISO timestamp, local date, local weekday, local date/time, and timezone. Use `timeZone` for scheduling, appointments, reminders, deadlines, and any prompt behavior that interprets "today", "tomorrow", weekdays, or relative dates. Do not hardcode today's date in `prompts/instructions.md`. If `timeZone` is omitted, AgentKit falls back to `AGENTKIT_TIME_ZONE`, then valid `TZ`, then the runtime default timezone.
|
|
183
|
+
|
|
181
184
|
Valid runtime values:
|
|
182
185
|
|
|
183
186
|
```txt
|
|
@@ -473,6 +476,7 @@ Tool runtime rules:
|
|
|
473
476
|
- tools that need SQL use canonical `ctx.db`; `ctx.database` and `ctx.storage.sql` are supported aliases;
|
|
474
477
|
- tools can use `ctx.db.batch([...])` for atomic writes; local tools can also use `ctx.db.transaction(async (tx) => ...)`;
|
|
475
478
|
- tools can inspect `ctx.runtime` with `{ environment, invocation, target, database }`;
|
|
479
|
+
- tools can use `ctx.clock` for the same runtime clock injected into the agent prompt, including `now`, `isoTimestamp`, `timeZone`, `localDate`, `localWeekday`, and `localDateTime`;
|
|
476
480
|
- tools must not import local database drivers or Node-only APIs. Use AgentKit runtime services instead.
|
|
477
481
|
|
|
478
482
|
## Database Tools And Dual Storage
|
|
@@ -717,14 +721,25 @@ npm run eval
|
|
|
717
721
|
Supported assertion types:
|
|
718
722
|
|
|
719
723
|
```txt
|
|
720
|
-
contains
|
|
721
|
-
|
|
722
|
-
|
|
723
|
-
|
|
724
|
-
|
|
725
|
-
|
|
726
|
-
|
|
727
|
-
|
|
724
|
+
response.contains
|
|
725
|
+
response.containsAll
|
|
726
|
+
response.containsAny
|
|
727
|
+
response.caseInsensitiveContains
|
|
728
|
+
response.notContains
|
|
729
|
+
response.regex
|
|
730
|
+
response.matchesRegex
|
|
731
|
+
response.notRegex
|
|
732
|
+
response.maxLength
|
|
733
|
+
tools.called
|
|
734
|
+
tools.calledOnce
|
|
735
|
+
tools.count
|
|
736
|
+
tools.order
|
|
737
|
+
tools.persisted
|
|
738
|
+
```
|
|
739
|
+
|
|
740
|
+
Import `defineEval` from `@andreprado/agentkit` when writing new evals. `tools.persisted` validates the tool call saved in local SQLite `tool_calls`, not a provider-specific raw response shape. It can be a tool name string or an object with `name`, `input`, `output`, `rendered`, `status`, and/or `visibility`. `tool_call` and `persisted_tool_call` remain accepted as backwards-compatible aliases, but new evals should use `tools.persisted`.
|
|
741
|
+
|
|
742
|
+
For date-sensitive evals, set top-level `now` to an ISO timestamp with an explicit timezone designator such as `Z` or `-05:00`. AgentKit uses that fixed clock for every turn and tool call in the eval so "today", "tomorrow", and weekdays remain deterministic while normal chat continues to use the real current date.
|
|
728
743
|
|
|
729
744
|
Evals run the normal capsule tools. If a tool would write externally, delete, charge money, send email, or call a real customer system, make its `execute` implementation branch on `ctx.runtime.environment === "eval"` and return deterministic non-destructive output for eval runs. Do not invent an eval-only mock API; keep the behavior inside the registered tool contract unless AgentKit adds a first-class mock facility later.
|
|
730
745
|
|
package/docs/llms.txt
CHANGED
|
@@ -71,6 +71,8 @@ UI testing is part of the handoff. For local UI testing, run `agentkit dev`, ope
|
|
|
71
71
|
|
|
72
72
|
`test/fake` is deterministic and validates scaffold, direct tool calls, and fake-provider evals. It does not validate natural conversation quality. Before claiming real conversation behavior has been tested, ask the owner which provider to use: OpenRouter, OpenAI, Anthropic, or another supported provider. Do not choose for them.
|
|
73
73
|
|
|
74
|
+
AgentKit injects the current ISO timestamp, local date, weekday, local date/time, and timezone dynamically into every chat run. Set `timeZone` in `agentkit.config.ts` for scheduling agents so "today", "tomorrow", and weekdays resolve in the business/user timezone; otherwise AgentKit falls back to `AGENTKIT_TIME_ZONE`, valid `TZ`, then the runtime default. Do not hardcode today's date in prompts.
|
|
75
|
+
|
|
74
76
|
Current local endpoints from `agentkit dev`:
|
|
75
77
|
|
|
76
78
|
```txt
|
package/package.json
CHANGED
package/src/cli/cloud-client.ts
CHANGED
|
@@ -39,12 +39,24 @@ export type CloudLimitsResponse = {
|
|
|
39
39
|
};
|
|
40
40
|
};
|
|
41
41
|
|
|
42
|
+
export type CloudProjectResolveResponse = {
|
|
43
|
+
project?: {
|
|
44
|
+
id?: string;
|
|
45
|
+
name?: string;
|
|
46
|
+
owner_state?: "anonymous" | "claimed";
|
|
47
|
+
};
|
|
48
|
+
};
|
|
49
|
+
|
|
42
50
|
export type DeployAccessTokenCreateResponse = {
|
|
43
51
|
access_token?: {
|
|
44
52
|
id?: string;
|
|
45
53
|
deploy_id?: string;
|
|
54
|
+
account_id?: string;
|
|
46
55
|
name?: string;
|
|
47
56
|
token?: string;
|
|
57
|
+
created_at?: string;
|
|
58
|
+
expires_at?: string;
|
|
59
|
+
last_used_at?: string;
|
|
48
60
|
};
|
|
49
61
|
};
|
|
50
62
|
|
|
@@ -52,7 +64,11 @@ export type DeployAccessTokenListResponse = {
|
|
|
52
64
|
access_tokens?: Array<{
|
|
53
65
|
id?: string;
|
|
54
66
|
deploy_id?: string;
|
|
67
|
+
account_id?: string;
|
|
55
68
|
name?: string;
|
|
69
|
+
created_at?: string;
|
|
70
|
+
expires_at?: string;
|
|
71
|
+
last_used_at?: string;
|
|
56
72
|
}>;
|
|
57
73
|
};
|
|
58
74
|
|
|
@@ -94,6 +110,16 @@ export async function readCloudAuth(): Promise<CloudAuthConfig | null> {
|
|
|
94
110
|
return null;
|
|
95
111
|
}
|
|
96
112
|
|
|
113
|
+
export async function readCloudAuthForApiUrl(apiUrl: string): Promise<CloudAuthConfig | null> {
|
|
114
|
+
const auth = await readCloudAuth();
|
|
115
|
+
|
|
116
|
+
if (auth?.source === "file" && !cloudAuthMatchesApiUrl(auth, apiUrl)) {
|
|
117
|
+
return null;
|
|
118
|
+
}
|
|
119
|
+
|
|
120
|
+
return auth;
|
|
121
|
+
}
|
|
122
|
+
|
|
97
123
|
export async function writeCloudAuth(config: CloudAuthConfig): Promise<void> {
|
|
98
124
|
await mkdir(join(homedir(), ".agentkit"), { recursive: true });
|
|
99
125
|
await writeFile(cloudAuthPath(), `${JSON.stringify(config, null, 2)}\n`, { mode: 0o600 });
|
|
@@ -156,11 +182,8 @@ export async function cloudPost<T>(apiUrl: string, path: string, body: unknown):
|
|
|
156
182
|
|
|
157
183
|
export async function cloudFetch(apiUrl: string, path: string, init: RequestInit = {}): Promise<Response> {
|
|
158
184
|
const headers = new Headers(init.headers);
|
|
159
|
-
const auth = await
|
|
160
|
-
const apiToken =
|
|
161
|
-
auth?.source === "file" && !cloudAuthMatchesApiUrl(auth, apiUrl)
|
|
162
|
-
? undefined
|
|
163
|
-
: auth?.token;
|
|
185
|
+
const auth = await readCloudAuthForApiUrl(apiUrl);
|
|
186
|
+
const apiToken = auth?.token;
|
|
164
187
|
|
|
165
188
|
if (apiToken && !headers.has("Authorization")) {
|
|
166
189
|
headers.set("Authorization", `Bearer ${apiToken}`);
|
|
@@ -177,12 +200,9 @@ function cloudAuthMatchesApiUrl(auth: CloudAuthConfig, apiUrl: string): boolean
|
|
|
177
200
|
}
|
|
178
201
|
|
|
179
202
|
export async function cloudApiRequest(apiUrl: string, path: string, init: RequestInit): Promise<unknown> {
|
|
180
|
-
const auth = await
|
|
203
|
+
const auth = await readCloudAuthForApiUrl(apiUrl);
|
|
181
204
|
const headers = new Headers(init.headers);
|
|
182
|
-
const apiToken =
|
|
183
|
-
auth?.source === "file" && !cloudAuthMatchesApiUrl(auth, apiUrl)
|
|
184
|
-
? undefined
|
|
185
|
-
: auth?.token;
|
|
205
|
+
const apiToken = auth?.token;
|
|
186
206
|
|
|
187
207
|
if (!headers.has("Content-Type") && init.body !== undefined) {
|
|
188
208
|
headers.set("Content-Type", "application/json");
|
|
@@ -20,6 +20,7 @@ type CloudChannel = {
|
|
|
20
20
|
webhook_url: string;
|
|
21
21
|
required_secrets: Array<{ name: string; status: string }>;
|
|
22
22
|
buffer?: AgentChannel["buffer"];
|
|
23
|
+
audio?: AgentChannel["audio"];
|
|
23
24
|
};
|
|
24
25
|
|
|
25
26
|
type CloudDelivery = {
|
|
@@ -386,6 +387,7 @@ async function createHostedChannel(
|
|
|
386
387
|
secrets: localChannel?.secrets ?? defaultChannelSecretsForCli(type, provider),
|
|
387
388
|
...(localChannel?.limits ? { limits: channelLimitsForApi(localChannel.limits) } : {}),
|
|
388
389
|
...(localChannel?.buffer ? { buffer: localChannel.buffer } : {}),
|
|
390
|
+
...(localChannel?.audio ? { audio: localChannel.audio } : {}),
|
|
389
391
|
};
|
|
390
392
|
|
|
391
393
|
return cloudPost<{ channel: CloudChannel }>(
|