@andreprado/agentkit 0.1.0-alpha.21 → 0.1.0-alpha.22
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/docs/guides/add-managed-composio.md +3 -1
- package/docs/guides/add-tool.md +6 -3
- package/docs/guides/create-agent.md +6 -4
- package/docs/guides/prepare-deploy.md +1 -1
- package/docs/guides/run-evals.md +5 -1
- package/docs/guides/use-provider.md +26 -2
- package/docs/llms-full.txt +33 -5
- package/docs/llms.txt +4 -2
- package/package.json +1 -1
- package/src/cli/deploy-chat-ui.ts +86 -15
- package/src/cli/index.ts +107 -8
- package/src/index.ts +1 -1
- package/src/providers/pi.ts +12 -2
- package/src/runtime/chat.ts +7 -1
- package/src/runtime/config.ts +1 -1
- package/src/runtime/evals.ts +39 -24
- package/src/templates/blank.ts +29 -6
- package/src/templates/dentista.ts +22 -4
- package/src/templates/skills/agentkit-build-agent/SKILL.md +30 -2
- package/src/templates/skills/agentkit-capsule/SKILL.md +21 -0
- package/src/templates/skills/agentkit-database/SKILL.md +11 -0
- package/src/templates/skills/agentkit-deploy/SKILL.md +2 -0
- package/src/templates/skills/agentkit-evals/SKILL.md +15 -0
- package/src/templates/skills/agentkit-improve/SKILL.md +14 -4
- package/src/templates/skills/agentkit-integrations/SKILL.md +22 -0
- package/src/templates/skills/agentkit-provider/SKILL.md +20 -2
- package/src/templates/skills/agentkit-tools/SKILL.md +2 -0
- package/src/templates/support.ts +30 -7
package/src/providers/pi.ts
CHANGED
|
@@ -17,7 +17,7 @@ import type { AgentProviderName, AgentTool } from "../index";
|
|
|
17
17
|
import { AgentKitError } from "../runtime/errors";
|
|
18
18
|
import type { ProviderAdapter, ProviderRunInput, ProviderRunResult } from "./types";
|
|
19
19
|
|
|
20
|
-
const PI_PROVIDER_NAMES = ["openai", "anthropic", "openrouter"] as const satisfies AgentProviderName[];
|
|
20
|
+
const PI_PROVIDER_NAMES = ["openai", "anthropic", "openrouter", "opencode", "opencode-go"] as const satisfies AgentProviderName[];
|
|
21
21
|
const MAX_TOOL_TURNS = 8;
|
|
22
22
|
const OPENROUTER_BASE_URL = "https://openrouter.ai/api/v1";
|
|
23
23
|
const UNKNOWN_OPENROUTER_CONTEXT_WINDOW = 128000;
|
|
@@ -159,7 +159,13 @@ async function runPiTurn(
|
|
|
159
159
|
}
|
|
160
160
|
|
|
161
161
|
function toPiProvider(provider: string): Provider {
|
|
162
|
-
if (
|
|
162
|
+
if (
|
|
163
|
+
provider === "openai" ||
|
|
164
|
+
provider === "anthropic" ||
|
|
165
|
+
provider === "openrouter" ||
|
|
166
|
+
provider === "opencode" ||
|
|
167
|
+
provider === "opencode-go"
|
|
168
|
+
) {
|
|
163
169
|
return provider;
|
|
164
170
|
}
|
|
165
171
|
|
|
@@ -181,6 +187,10 @@ function requireProviderApiKey(provider: Provider, env: Record<string, string |
|
|
|
181
187
|
}
|
|
182
188
|
|
|
183
189
|
function defaultProviderSecretNames(provider: Provider): string[] {
|
|
190
|
+
if (provider === "opencode" || provider === "opencode-go") {
|
|
191
|
+
return ["OPENCODE_API_KEY"];
|
|
192
|
+
}
|
|
193
|
+
|
|
184
194
|
return [`${provider.toUpperCase().replace(/[^A-Z0-9]+/g, "_")}_API_KEY`];
|
|
185
195
|
}
|
|
186
196
|
|
package/src/runtime/chat.ts
CHANGED
|
@@ -20,6 +20,7 @@ export type RunAgentMessageOptions = {
|
|
|
20
20
|
signal?: AbortSignal;
|
|
21
21
|
runtime?: Partial<ToolRuntimeContext>;
|
|
22
22
|
now?: Date;
|
|
23
|
+
localStoragePath?: string;
|
|
23
24
|
};
|
|
24
25
|
|
|
25
26
|
export type AgentRunResult = ProviderRunResult & PersistedChatRun;
|
|
@@ -28,7 +29,8 @@ export async function runAgentMessageFromCwd(
|
|
|
28
29
|
cwd: string,
|
|
29
30
|
options: RunAgentMessageOptions,
|
|
30
31
|
): Promise<AgentRunResult> {
|
|
31
|
-
|
|
32
|
+
const capsule = await loadAgentCapsule(cwd);
|
|
33
|
+
return runAgentMessage(withLocalStoragePath(capsule, options.localStoragePath), options);
|
|
32
34
|
}
|
|
33
35
|
|
|
34
36
|
export async function runAgentMessage(
|
|
@@ -208,6 +210,10 @@ function normalizeConversationId(value: string | undefined): string | undefined
|
|
|
208
210
|
return conversationId;
|
|
209
211
|
}
|
|
210
212
|
|
|
213
|
+
function withLocalStoragePath(capsule: LoadedAgentCapsule, storagePath?: string): LoadedAgentCapsule {
|
|
214
|
+
return storagePath ? { ...capsule, storagePath } : capsule;
|
|
215
|
+
}
|
|
216
|
+
|
|
211
217
|
function withToolVisibilityInstructions(instructions: string, tools: LoadedAgentCapsule["config"]["tools"] = []): string {
|
|
212
218
|
const internalTools = tools.filter((tool) => tool.visibility === "internal").map((tool) => tool.name).sort();
|
|
213
219
|
|
package/src/runtime/config.ts
CHANGED
|
@@ -27,7 +27,7 @@ import { isValidTimeZone } from "./prompt-context";
|
|
|
27
27
|
|
|
28
28
|
const CONFIG_FILE = "agentkit.config.ts";
|
|
29
29
|
const AGENT_RUNTIMES = new Set<AgentRuntime>(["local", "edge"]);
|
|
30
|
-
const PROVIDERS = new Set<AgentProviderName>(["test", "openai", "anthropic", "openrouter", "custom"]);
|
|
30
|
+
const PROVIDERS = new Set<AgentProviderName>(["test", "openai", "anthropic", "openrouter", "opencode", "opencode-go", "custom"]);
|
|
31
31
|
const ACCESS_MODES = new Set<AccessMode>(["private", "token", "public"]);
|
|
32
32
|
const TOOL_VISIBILITIES = new Set<ToolVisibility>(["user", "internal"]);
|
|
33
33
|
const CHANNEL_TYPES = new Set<ChannelType>(["website", "telegram", "whatsapp", "discord", "slack", "webhook"]);
|
package/src/runtime/evals.ts
CHANGED
|
@@ -1,5 +1,6 @@
|
|
|
1
|
-
import { mkdir, readdir, writeFile } from "node:fs/promises";
|
|
1
|
+
import { mkdir, mkdtemp, readdir, rm, writeFile } from "node:fs/promises";
|
|
2
2
|
import type { Dirent } from "node:fs";
|
|
3
|
+
import { tmpdir } from "node:os";
|
|
3
4
|
import { dirname, isAbsolute, join, relative, resolve } from "node:path";
|
|
4
5
|
import { pathToFileURL } from "node:url";
|
|
5
6
|
|
|
@@ -243,35 +244,49 @@ async function runEvalCase(
|
|
|
243
244
|
const failures: string[] = [];
|
|
244
245
|
let output = "";
|
|
245
246
|
const now = resolveEvalNow(file, evalCase.now);
|
|
247
|
+
const evalStore = await createEvalStore();
|
|
246
248
|
|
|
247
|
-
|
|
248
|
-
|
|
249
|
-
|
|
250
|
-
|
|
251
|
-
|
|
252
|
-
|
|
253
|
-
|
|
254
|
-
|
|
255
|
-
|
|
256
|
-
|
|
257
|
-
|
|
258
|
-
|
|
259
|
-
|
|
260
|
-
|
|
261
|
-
|
|
262
|
-
|
|
249
|
+
try {
|
|
250
|
+
for (const [index, turn] of turns.entries()) {
|
|
251
|
+
try {
|
|
252
|
+
const run = await runAgentMessageFromCwd(root, {
|
|
253
|
+
message: turn.input,
|
|
254
|
+
conversationId,
|
|
255
|
+
now,
|
|
256
|
+
localStoragePath: evalStore.databasePath,
|
|
257
|
+
runtime: {
|
|
258
|
+
environment: "eval",
|
|
259
|
+
invocation: "eval",
|
|
260
|
+
},
|
|
261
|
+
});
|
|
262
|
+
const persistedToolCalls = await loadPersistedToolCalls(root, run.runId, evalStore.databasePath);
|
|
263
|
+
const turnFailures = evaluateExpectations(turn.expect ?? {}, run, persistedToolCalls);
|
|
264
|
+
|
|
265
|
+
for (const failure of turnFailures) {
|
|
266
|
+
failures.push(turns.length === 1 ? failure : `turn ${index + 1}: ${failure}`);
|
|
267
|
+
}
|
|
268
|
+
|
|
269
|
+
output = run.message.content;
|
|
270
|
+
} catch (error) {
|
|
271
|
+
failures.push(turns.length === 1 ? formatEvalFailure(error) : `turn ${index + 1}: ${formatEvalFailure(error)}`);
|
|
272
|
+
break;
|
|
263
273
|
}
|
|
264
|
-
|
|
265
|
-
output = run.message.content;
|
|
266
|
-
} catch (error) {
|
|
267
|
-
failures.push(turns.length === 1 ? formatEvalFailure(error) : `turn ${index + 1}: ${formatEvalFailure(error)}`);
|
|
268
|
-
break;
|
|
269
274
|
}
|
|
275
|
+
} finally {
|
|
276
|
+
await rm(evalStore.directory, { recursive: true, force: true });
|
|
270
277
|
}
|
|
271
278
|
|
|
272
279
|
return { failures, output };
|
|
273
280
|
}
|
|
274
281
|
|
|
282
|
+
async function createEvalStore(): Promise<{ directory: string; databasePath: string }> {
|
|
283
|
+
const directory = await mkdtemp(join(tmpdir(), "agentkit-eval-"));
|
|
284
|
+
return {
|
|
285
|
+
directory,
|
|
286
|
+
databasePath: join(directory, "agentkit.db"),
|
|
287
|
+
};
|
|
288
|
+
}
|
|
289
|
+
|
|
275
290
|
function resolveEvalNow(file: string, value: EvalCase["now"]): Date | undefined {
|
|
276
291
|
if (value === undefined) {
|
|
277
292
|
return undefined;
|
|
@@ -382,9 +397,9 @@ async function loadEvalCase(file: string): Promise<EvalCase> {
|
|
|
382
397
|
return module.default;
|
|
383
398
|
}
|
|
384
399
|
|
|
385
|
-
async function loadPersistedToolCalls(root: string, runId: string): Promise<ToolCallSnapshot[]> {
|
|
400
|
+
async function loadPersistedToolCalls(root: string, runId: string, localStoragePath?: string): Promise<ToolCallSnapshot[]> {
|
|
386
401
|
const capsule = await loadAgentCapsule(root);
|
|
387
|
-
const store = await openCapsuleStore(capsule);
|
|
402
|
+
const store = await openCapsuleStore(localStoragePath ? { ...capsule, storagePath: localStoragePath } : capsule);
|
|
388
403
|
|
|
389
404
|
try {
|
|
390
405
|
return store.listToolCallsForRun(runId).map(toolCallSnapshotFromStored);
|
package/src/templates/blank.ts
CHANGED
|
@@ -66,6 +66,7 @@ node_modules/
|
|
|
66
66
|
OPENAI_API_KEY=
|
|
67
67
|
ANTHROPIC_API_KEY=
|
|
68
68
|
OPENROUTER_API_KEY=
|
|
69
|
+
OPENCODE_API_KEY=
|
|
69
70
|
`,
|
|
70
71
|
},
|
|
71
72
|
{
|
|
@@ -148,7 +149,7 @@ The capsule root is the runtime boundary. Run, inspect, evaluate, and deploy fro
|
|
|
148
149
|
|
|
149
150
|
## Coding Agent Workflow
|
|
150
151
|
|
|
151
|
-
When the owner opens this folder in Codex, Claude Code, or another coding agent and asks for a specific agent, treat that request as the product brief.
|
|
152
|
+
When the owner opens this folder in Codex, Claude Code, or another coding agent and asks for a specific agent, treat that request as the product brief. The owner should not need to run a separate AgentKit wizard or prepare a brief file.
|
|
152
153
|
|
|
153
154
|
Start building immediately:
|
|
154
155
|
|
|
@@ -158,11 +159,26 @@ Start building immediately:
|
|
|
158
159
|
- Edit \`prompts/instructions.md\` for the agent behavior.
|
|
159
160
|
- Edit \`agentkit.config.ts\` for provider, tools, secrets, access, and storage.
|
|
160
161
|
- Add TypeScript tools under \`tools/\` when the requested agent needs actions or external data.
|
|
162
|
+
- When the agent needs to save durable records, complete the full slice: schema/migration, tool, config registration, prompt instructions, direct tool check, and eval.
|
|
161
163
|
- Add \`sync.ts\`, \`seed.sql\`, and ordered \`migrations/*.sql\` when the requested agent depends on external catalogs or production-shaped data changes.
|
|
162
164
|
- Do not wait for a wizard or recipe. AgentKit provides the scaffold and contract; you decide the implementation from the owner's brief.
|
|
163
165
|
- Ask follow-up questions only when missing information blocks a safe local implementation.
|
|
164
166
|
- State assumptions in the final response.
|
|
165
167
|
|
|
168
|
+
## Proactive Agent Builder Contract
|
|
169
|
+
|
|
170
|
+
Do not only edit prompts. For every meaningful requirement in the owner's request or \`AGENT_SPEC.md\`, decide what should enforce it:
|
|
171
|
+
|
|
172
|
+
- spec entry for the product contract;
|
|
173
|
+
- prompt instruction for behavior, tone, boundaries, intake, and escalation;
|
|
174
|
+
- tool plus config registration for actions, live data, external writes, or authorization-sensitive data;
|
|
175
|
+
- schema/migration plus tool for durable records;
|
|
176
|
+
- eval for privacy, confirmation, required fields, date/time behavior, business rules, and regressions;
|
|
177
|
+
- fixture, seed data, fake branch, or direct tool check for integrations and failure paths;
|
|
178
|
+
- deploy/readiness check for hosted secrets, channels, integrations, or production access.
|
|
179
|
+
|
|
180
|
+
If a rule protects privacy, money, bookings, external writes, customer data, business hours, or safety, it must have an eval or deterministic check before you call the capsule done. If a real conversation exposes a bug, convert it into the smallest regression eval before or alongside the fix.
|
|
181
|
+
|
|
166
182
|
## Local Commands
|
|
167
183
|
|
|
168
184
|
- \`npm install\`: restore capsule dependencies if this capsule used \`--no-install\`, install failed, or \`node_modules\` was deleted.
|
|
@@ -188,7 +204,7 @@ Start building immediately:
|
|
|
188
204
|
- Local UI: run \`npm run dev\`, open the printed \`Chat:\` URL, and tell the owner the exact URL.
|
|
189
205
|
- Hosted UI: after \`npm run agentkit -- deploy\`, run \`npm run agentkit -- chat-ui --deploy\`, open the printed \`Chat:\` URL, and tell the owner it is connected to the hosted deploy.
|
|
190
206
|
- \`test/fake\` is deterministic. It is useful for scaffold checks, direct tool checks, and fake-provider evals, but it does not validate natural conversation quality.
|
|
191
|
-
- Before claiming real conversation behavior is tested, ask the owner which provider to use: OpenRouter, OpenAI, Anthropic, or another supported provider. Do not choose for them.
|
|
207
|
+
- Before claiming real conversation behavior is tested, ask the owner which provider to use: OpenRouter, OpenAI, Anthropic, OpenCode Zen, OpenCode Go, or another supported provider. Do not choose for them.
|
|
192
208
|
|
|
193
209
|
This blank capsule starts with the built-in \`test/fake\` provider, so local chat works without secrets or internet access. It is also deploy-ready by default: the user can edit the prompt, add tools, and run \`npm run agentkit -- deploy\`.
|
|
194
210
|
|
|
@@ -253,7 +269,9 @@ This folder is an AgentKit Agent Capsule.
|
|
|
253
269
|
|
|
254
270
|
## Start Here
|
|
255
271
|
|
|
256
|
-
If the owner asks you to build an agent in natural language, that request is the brief. Do not ask them to fill another file first.
|
|
272
|
+
If the owner asks you to build an agent in natural language, that request is the brief. Do not ask them to run a wizard or fill another file first.
|
|
273
|
+
|
|
274
|
+
Build a testable capsule, not only a prompt.
|
|
257
275
|
|
|
258
276
|
Example owner request:
|
|
259
277
|
|
|
@@ -265,7 +283,10 @@ Turn the request into a working local capsule:
|
|
|
265
283
|
- Create or update \`AGENT_SPEC.md\` with \`npm run agentkit -- spec init --brief "<owner request>"\`. The owner gives the general idea; the coding agent turns it into the structured contract.
|
|
266
284
|
- Update \`agentkit.config.ts\` when tools, secrets, provider, or access rules change.
|
|
267
285
|
- Add TypeScript tools under \`tools/\` for real actions or external data.
|
|
286
|
+
- For durable records, implement the full schema/tool/prompt/eval slice instead of only adding a table or only adding a tool.
|
|
268
287
|
- Use \`npm run agentkit -- sync init\` when the agent needs catalog sync, fixture seed data, or ordered migrations.
|
|
288
|
+
- For every privacy, confirmation, required-intake, timezone, business-hour, integration-error, or no-leak rule, add an eval, fixture, fake branch, or direct tool check.
|
|
289
|
+
- Convert failed or surprising real conversations into regression evals with \`npm run agentkit -- eval from-conversation <conversation-id>\`.
|
|
269
290
|
- Keep the first version runnable with \`test/fake\` unless the owner explicitly asks for a real provider.
|
|
270
291
|
- Do not use a wizard or recipe. Build the capsule directly from the scaffold, the AgentKit contract, and the owner's brief.
|
|
271
292
|
- Make practical assumptions and list them in your final response.
|
|
@@ -281,6 +302,8 @@ npm run chat -- --message "hello"
|
|
|
281
302
|
|
|
282
303
|
\`test/fake\` proves the scaffold and deterministic tool paths. It does not prove natural conversation quality.
|
|
283
304
|
|
|
305
|
+
Before saying the agent is done, make sure important requirements have matching checks. Prompt-only changes are not enough for privacy, external writes, bookings, customer data, business hours, or integration failures.
|
|
306
|
+
|
|
284
307
|
\`agentkit new\` installs dependencies by default. Run \`npm install\` only if the capsule was created with \`--no-install\`, install failed, or \`node_modules\` was deleted.
|
|
285
308
|
|
|
286
309
|
Set local development secrets without opening code:
|
|
@@ -310,7 +333,7 @@ npm run agentkit -- chat-ui --deploy
|
|
|
310
333
|
|
|
311
334
|
Open the printed \`Chat:\` URL and tell the owner this local UI is connected to the hosted deploy.
|
|
312
335
|
|
|
313
|
-
Before claiming real conversation behavior has been tested, ask the owner which provider to use: OpenRouter, OpenAI, Anthropic, or another supported provider. Do not choose for them. After they choose, update \`agentkit.config.ts\`, \`.env.schema\`, local secrets, hosted secrets if deploying, then rerun chat/UI checks.
|
|
336
|
+
Before claiming real conversation behavior has been tested, ask the owner which provider to use: OpenRouter, OpenAI, Anthropic, OpenCode Zen, OpenCode Go, or another supported provider. Do not choose for them. After they choose, update \`agentkit.config.ts\`, \`.env.schema\`, local secrets, hosted secrets if deploying, then rerun chat/UI checks.
|
|
314
337
|
|
|
315
338
|
If you add a tool, also run a fake-provider tool smoke test:
|
|
316
339
|
|
|
@@ -354,7 +377,7 @@ Use AgentKit conventions when editing this project.
|
|
|
354
377
|
- Local runtime state lives in \`.agentkit/\` and should not be committed.
|
|
355
378
|
- The default provider is \`test/fake\`, which needs no secrets.
|
|
356
379
|
- Real providers run through AgentKit's internal Pi SDK backend.
|
|
357
|
-
- Ask the owner which real provider to use before switching from \`test/fake\`; do not choose OpenRouter, OpenAI, or
|
|
380
|
+
- Ask the owner which real provider to use before switching from \`test/fake\`; do not choose OpenRouter, OpenAI, Anthropic, OpenCode Zen, or OpenCode Go automatically.
|
|
358
381
|
- Production secrets must be managed secrets, not committed files.
|
|
359
382
|
- Keep required local secret names in \`.env.schema\` and values in ignored \`.env\`. AgentKit local commands load \`.env\` directly.
|
|
360
383
|
- Treat the owner's natural-language request as the brief and start implementing inside this capsule.
|
|
@@ -383,7 +406,7 @@ When a real provider or tool needs a local development secret, keep the required
|
|
|
383
406
|
|
|
384
407
|
For UI testing, run \`npm run dev\` and open the printed \`Chat:\` URL. After hosted deploy, run \`npm run agentkit -- chat-ui --deploy\` and open its printed \`Chat:\` URL.
|
|
385
408
|
|
|
386
|
-
\`test/fake\` does not validate real conversation quality. The owner must choose OpenRouter, OpenAI, Anthropic, or another supported provider before real model behavior is tested.
|
|
409
|
+
\`test/fake\` does not validate real conversation quality. The owner must choose OpenRouter, OpenAI, Anthropic, OpenCode Zen, OpenCode Go, or another supported provider before real model behavior is tested.
|
|
387
410
|
|
|
388
411
|
## Files
|
|
389
412
|
|
|
@@ -64,6 +64,7 @@ node_modules/
|
|
|
64
64
|
OPENAI_API_KEY=
|
|
65
65
|
ANTHROPIC_API_KEY=
|
|
66
66
|
OPENROUTER_API_KEY=
|
|
67
|
+
OPENCODE_API_KEY=
|
|
67
68
|
`,
|
|
68
69
|
},
|
|
69
70
|
{
|
|
@@ -865,7 +866,21 @@ Esta é uma AgentKit Agent Capsule para a Clara, atendente em português de um c
|
|
|
865
866
|
|
|
866
867
|
A Clara conversa com clientes, coleta nome, email e telefone, consulta disponibilidade e agenda consultas em slots de 30 minutos. Ela também consulta e altera o próprio horário do cliente usando email e telefone como verificação mínima.
|
|
867
868
|
|
|
868
|
-
Quando o dono pedir mudanças em linguagem natural,
|
|
869
|
+
Quando o dono pedir mudanças em linguagem natural, trate a mensagem como o brief. O dono não deve precisar rodar wizard ou preparar arquivo de brief. Crie ou atualize \`AGENT_SPEC.md\` com \`npm run agentkit -- spec init --brief "<pedido do dono>"\`; o agente de código transforma a ideia geral no contrato estruturado.
|
|
870
|
+
|
|
871
|
+
## Contrato Proativo De Produto Testável
|
|
872
|
+
|
|
873
|
+
Não edite apenas o prompt. Para cada requisito relevante do pedido ou do \`AGENT_SPEC.md\`, decida qual artefato deve garantir o comportamento:
|
|
874
|
+
|
|
875
|
+
- entrada no spec para o contrato de produto;
|
|
876
|
+
- instrução no prompt para comportamento, tom, limites, coleta de dados e escalonamento;
|
|
877
|
+
- ferramenta mais registro no config para ações, dados vivos, escritas externas ou dados sensíveis;
|
|
878
|
+
- schema/migration mais ferramenta para dados duráveis;
|
|
879
|
+
- eval para privacidade, confirmação, campos obrigatórios, horário comercial, datas, duração e regressões;
|
|
880
|
+
- fixture, seed, branch fake ou teste direto de ferramenta para integrações e caminhos de erro;
|
|
881
|
+
- checagem de deploy/readiness para secrets hosted, canais, integrações ou acesso de produção.
|
|
882
|
+
|
|
883
|
+
Regras que protegem privacidade, agendamento, confirmação, dados de cliente, horário comercial ou segurança precisam de eval ou checagem determinística antes de dizer que a cápsula está pronta. Se uma conversa real revelar bug, transforme em uma eval de regressão pequena antes ou junto da correção.
|
|
869
884
|
|
|
870
885
|
## Regras de Produto
|
|
871
886
|
|
|
@@ -908,16 +923,19 @@ Quando o dono pedir mudanças em linguagem natural, crie ou atualize \`AGENT_SPE
|
|
|
908
923
|
- UI local: rode \`npm run dev\`, abra a URL impressa em \`Chat:\` e informe essa URL exata ao dono.
|
|
909
924
|
- UI pós-deploy: depois de \`npm run agentkit -- deploy\`, rode \`npm run agentkit -- chat-ui --deploy\`, abra a URL impressa em \`Chat:\` e informe que ela está conectada ao deploy hospedado.
|
|
910
925
|
- \`test/fake\` é determinístico. Ele valida scaffold, ferramentas e evals previsíveis, mas não valida qualidade de conversa natural.
|
|
911
|
-
- Antes de dizer que a conversa real foi testada, pergunte ao dono qual provider usar: OpenRouter, OpenAI, Anthropic ou outro provider suportado. Não escolha pelo dono.
|
|
926
|
+
- Antes de dizer que a conversa real foi testada, pergunte ao dono qual provider usar: OpenRouter, OpenAI, Anthropic, OpenCode Zen, OpenCode Go ou outro provider suportado. Não escolha pelo dono.
|
|
912
927
|
|
|
913
928
|
## Regras de Implementação
|
|
914
929
|
|
|
915
930
|
- Use \`ctx.db\` nas ferramentas. Não importe drivers SQLite, Turso ou Node-only APIs.
|
|
916
931
|
- Mantenha \`schema.sql\` idempotente com \`CREATE TABLE IF NOT EXISTS\` e \`CREATE INDEX IF NOT EXISTS\`. Use \`migrations/*.sql\` ordenadas para evolução de schema com cara de produção.
|
|
932
|
+
- Quando a mudança exigir persistência nova, implemente a fatia completa: schema/migration, ferramenta, registro no config, instruções no prompt, teste direto da ferramenta e eval.
|
|
917
933
|
- Use \`npm run agentkit -- sync init\` quando a agente depender de catálogo externo, fixture local ou seed padronizado.
|
|
934
|
+
- Para regras de privacidade, confirmação, campos obrigatórios, timezone, horário comercial, duração, erro de integração ou no-leak, adicione uma eval, fixture, branch fake ou teste direto de ferramenta.
|
|
935
|
+
- Transforme conversas reais com falhas em regressão com \`npm run agentkit -- eval from-conversation <conversation-id>\`.
|
|
918
936
|
- Mantenha segredos em \`.env\`, nunca em arquivos versionados.
|
|
919
937
|
- Não espere wizard ou recipe. AgentKit fornece o scaffold e o contrato; implemente diretamente conforme o brief do dono.
|
|
920
|
-
- Para trocar para um provider real, o dono deve escolher OpenRouter, OpenAI, Anthropic ou outro provider suportado. Depois edite \`agentkit.config.ts\`, atualize \`.env.schema\` e configure secrets locais/hosted.
|
|
938
|
+
- Para trocar para um provider real, o dono deve escolher OpenRouter, OpenAI, Anthropic, OpenCode Zen, OpenCode Go ou outro provider suportado. Depois edite \`agentkit.config.ts\`, atualize \`.env.schema\` e configure secrets locais/hosted.
|
|
921
939
|
- Quando a próxima ação não for óbvia, comece por \`skills/agentkit-capsule/SKILL.md\`.
|
|
922
940
|
`,
|
|
923
941
|
},
|
|
@@ -995,7 +1013,7 @@ npm run agentkit -- chat-ui --deploy
|
|
|
995
1013
|
|
|
996
1014
|
Abra a URL impressa em \`Chat:\` e informe que essa UI local está conectada ao deploy hospedado.
|
|
997
1015
|
|
|
998
|
-
Antes de dizer que a conversa real foi testada, pergunte ao dono qual provider usar: OpenRouter, OpenAI, Anthropic ou outro provider suportado. Não escolha pelo dono. Depois configure \`agentkit.config.ts\`, \`.env.schema\`, secrets locais e secrets hosted se for deployar.
|
|
1016
|
+
Antes de dizer que a conversa real foi testada, pergunte ao dono qual provider usar: OpenRouter, OpenAI, Anthropic, OpenCode Zen, OpenCode Go ou outro provider suportado. Não escolha pelo dono. Depois configure \`agentkit.config.ts\`, \`.env.schema\`, secrets locais e secrets hosted se for deployar.
|
|
999
1017
|
|
|
1000
1018
|
## Deploy
|
|
1001
1019
|
|
|
@@ -7,6 +7,8 @@ description: Use when the owner gives a natural-language brief for a new or chan
|
|
|
7
7
|
|
|
8
8
|
Use this when the owner asks for an agent in plain language.
|
|
9
9
|
|
|
10
|
+
The owner should only need to run `agentkit new <name>`, open the capsule in Codex or another coding agent, and say what agent they want. Do not send them back to the CLI for a brief wizard. Treat their chat message as the brief and build the first useful local capsule.
|
|
11
|
+
|
|
10
12
|
## Workflow
|
|
11
13
|
|
|
12
14
|
1. Read `agentkit.config.ts`, `prompts/instructions.md`, `schema.sql`, `evals/`, and existing `tools/`.
|
|
@@ -17,8 +19,34 @@ Use this when the owner asks for an agent in plain language.
|
|
|
17
19
|
6. Add tools only when the agent needs action, live data, authorization-sensitive data, or durable writes.
|
|
18
20
|
7. Add database tables to `schema.sql` or ordered `migrations/*.sql` when the agent owns records.
|
|
19
21
|
8. Add `sync.ts` and `seed.sql` with `npm run agentkit -- sync init` when the agent depends on external catalogs or recurring imports.
|
|
20
|
-
9.
|
|
21
|
-
10.
|
|
22
|
+
9. Turn requirements into checks as you build. Every privacy rule, external write, confirmation step, business-hour rule, timezone rule, required intake field, and customer-data boundary needs an eval, direct tool check, fixture, or deterministic fake path.
|
|
23
|
+
10. Add or update evals for the main flow. Prefer multi-turn `turns` evals for real conversations.
|
|
24
|
+
11. Keep the capsule runnable on `test/fake` unless the owner has chosen a real provider.
|
|
25
|
+
|
|
26
|
+
## Requirement-To-Verification Loop
|
|
27
|
+
|
|
28
|
+
For each item in the brief or `AGENT_SPEC.md`, classify it before finishing:
|
|
29
|
+
|
|
30
|
+
- prompt-only behavior: add a prompt instruction and a response eval when the wording or boundary matters;
|
|
31
|
+
- required intake such as name, phone, email, budget, or account id: add a multi-turn eval that proves the agent asks before acting;
|
|
32
|
+
- tool call or external write: add a persisted tool-call eval and a direct `agentkit tool` check;
|
|
33
|
+
- database record: implement the complete slice in the next section;
|
|
34
|
+
- scheduling, deadlines, or timezone: set `timeZone`, freeze eval `now`, and assert the date/time sent to tools;
|
|
35
|
+
- privacy/no-leak rule: add a no-leak eval with `notContains` or `notRegex`;
|
|
36
|
+
- integration failure such as rate limit, timeout, unavailable slot, or missing auth: add a fixture, fake branch, or eval-safe tool behavior.
|
|
37
|
+
|
|
38
|
+
If a requirement is important enough to mention in the spec, it is usually important enough to test. State any untested requirement in the final response.
|
|
39
|
+
|
|
40
|
+
## Complete Slice
|
|
41
|
+
|
|
42
|
+
When the brief implies durable records such as leads, clients, bookings, tickets, or notes, do the whole local slice:
|
|
43
|
+
|
|
44
|
+
- add or update `schema.sql` and ordered `migrations/*.sql` when appropriate;
|
|
45
|
+
- add the `defineTool` implementation under `tools/`;
|
|
46
|
+
- register the tool in `agentkit.config.ts`;
|
|
47
|
+
- teach `prompts/instructions.md` when and how to use the tool;
|
|
48
|
+
- add a direct tool check fixture or command;
|
|
49
|
+
- add an eval that proves the agent calls the tool in the expected flow.
|
|
22
50
|
|
|
23
51
|
## Templates
|
|
24
52
|
|
|
@@ -7,6 +7,27 @@ description: Use when working inside an AgentKit Agent Capsule, especially after
|
|
|
7
7
|
|
|
8
8
|
Use this first inside an AgentKit Agent Capsule.
|
|
9
9
|
|
|
10
|
+
## Owner-To-Codex Contract
|
|
11
|
+
|
|
12
|
+
The owner has already done the setup work by running `agentkit new <name>` and opening this folder in a coding agent. When they ask for an agent in natural language, that message is the brief.
|
|
13
|
+
|
|
14
|
+
- Do not ask the owner to run a wizard, fill a form, or prepare `AGENT_SPEC.md`.
|
|
15
|
+
- Route immediately to `skills/agentkit-build-agent/SKILL.md`.
|
|
16
|
+
- You create or update `AGENT_SPEC.md`, prompts, tools, schema, evals, and docs as needed.
|
|
17
|
+
- Use the CLI for validation and deployment, not for business inference.
|
|
18
|
+
|
|
19
|
+
## Proactive Builder Contract
|
|
20
|
+
|
|
21
|
+
Do not stop at "the prompt was edited." For every meaningful requirement in the owner's brief or `AGENT_SPEC.md`, decide which artifact must enforce it:
|
|
22
|
+
|
|
23
|
+
- behavior or tone: `prompts/instructions.md`;
|
|
24
|
+
- action, live data, or external write: `tools/` plus `agentkit.config.ts`;
|
|
25
|
+
- durable records: schema/migration, tool, config registration, prompt guidance, direct tool check, and eval;
|
|
26
|
+
- privacy, confirmation, business hours, dates, money, safety, or client data: at least one eval or deterministic check;
|
|
27
|
+
- production or integration dependency: deploy/readiness check and safe fixture or fake path.
|
|
28
|
+
|
|
29
|
+
If a real conversation reveals a bug or risky behavior, convert it into a regression eval before or alongside the fix.
|
|
30
|
+
|
|
10
31
|
## Start
|
|
11
32
|
|
|
12
33
|
1. Treat the directory containing `agentkit.config.ts` as the capsule root.
|
|
@@ -7,6 +7,17 @@ description: Use when adding AgentKit-managed database tables, editing schema.sq
|
|
|
7
7
|
|
|
8
8
|
Use this when a capsule owns durable application records.
|
|
9
9
|
|
|
10
|
+
## Complete Database Tool Slice
|
|
11
|
+
|
|
12
|
+
When the owner asks the agent to save or update records, do not stop after adding a table. Ship the whole slice:
|
|
13
|
+
|
|
14
|
+
1. Add or update `schema.sql`; for production-shaped evolution, add an ordered `migrations/*.sql` file too.
|
|
15
|
+
2. Add a `defineTool` under `tools/` that uses `ctx.db`.
|
|
16
|
+
3. Register the tool in `agentkit.config.ts`.
|
|
17
|
+
4. Update `prompts/instructions.md` so the agent knows when to collect fields, confirm writes, and call the tool.
|
|
18
|
+
5. Add or update an eval for the user flow that should persist the record.
|
|
19
|
+
6. Run `db migrate`, a direct tool check, `typecheck`, and `eval`.
|
|
20
|
+
|
|
10
21
|
## Rules
|
|
11
22
|
|
|
12
23
|
- Put the first idempotent bootstrap schema in `schema.sql`.
|
|
@@ -47,4 +47,6 @@ Open the printed `Chat:` URL and report it to the owner.
|
|
|
47
47
|
|
|
48
48
|
## Production Handoff
|
|
49
49
|
|
|
50
|
+
`npm run agentkit -- deploy` prints a production handoff after a successful deploy. Use it as the source of truth for the deploy URL, UI command, secret status, database/schema artifact, integration connect commands, smoke status, and next recommended command.
|
|
51
|
+
|
|
50
52
|
Report changed files, required env/secret names, database schema changes, deploy order, smoke checks, rollback concerns, and whether the provider was still `test/fake`.
|
|
@@ -7,6 +7,8 @@ description: Use when adding, editing, or running AgentKit eval files, including
|
|
|
7
7
|
|
|
8
8
|
Use evals after chat works and before claiming behavior is stable.
|
|
9
9
|
|
|
10
|
+
`npm run eval` uses temporary local SQLite storage for eval execution. Eval conversations and tool calls do not write to the normal `.agentkit/agentkit.db`, so evals can run while local chat or `npm run dev` is using the development database.
|
|
11
|
+
|
|
10
12
|
## Workflow
|
|
11
13
|
|
|
12
14
|
1. Create or edit `evals/<name>.eval.ts`.
|
|
@@ -21,6 +23,19 @@ Use evals after chat works and before claiming behavior is stable.
|
|
|
21
23
|
10. Do not put secrets or real client PII in evals.
|
|
22
24
|
11. For tools that write externally, delete, charge money, send email, or call real customer systems, branch on `ctx.runtime.environment === "eval"` inside the registered tool.
|
|
23
25
|
|
|
26
|
+
## What To Test
|
|
27
|
+
|
|
28
|
+
Create evals proactively from the brief and spec. Common high-value evals:
|
|
29
|
+
|
|
30
|
+
- identity and scope: the agent says who it is and refuses out-of-scope work;
|
|
31
|
+
- intake: required fields such as name, phone, email, account id, date, or budget are collected before action;
|
|
32
|
+
- confirmation: external writes, deletes, messages, charges, and bookings do not happen before explicit confirmation;
|
|
33
|
+
- privacy: raw tool output, full calendars, internal IDs, retrieval metadata, secrets, and stack traces are not shown to the client;
|
|
34
|
+
- timezone and schedule: eval `now` is frozen, `timeZone` is honored, and tool payloads use the intended local date/time;
|
|
35
|
+
- defaults and constraints: durations, allowed hours, allowed regions, max/min values, and business rules are asserted;
|
|
36
|
+
- unhappy paths: rate limits, timeouts, missing auth, unavailable slots, empty results, and validation errors produce safe user-facing responses;
|
|
37
|
+
- regression: every real conversation bug gets the smallest eval that would have failed before the fix.
|
|
38
|
+
|
|
24
39
|
## Assertion Shape
|
|
25
40
|
|
|
26
41
|
Use this shape first:
|
|
@@ -38,13 +38,23 @@ npm run agentkit -- improve collect --deploy --conversation-id <conversation-id>
|
|
|
38
38
|
.agentkit/improve/<run>/traces/
|
|
39
39
|
```
|
|
40
40
|
|
|
41
|
-
3.
|
|
41
|
+
3. Before patching, identify the smallest testable lesson from each relevant trace:
|
|
42
|
+
|
|
43
|
+
- Did the agent miss a required field?
|
|
44
|
+
- Did it expose raw tool output, a full schedule, an internal id, or a technical error?
|
|
45
|
+
- Did it write externally without confirmation?
|
|
46
|
+
- Did it use the wrong timezone, duration, business hour, or availability assumption?
|
|
47
|
+
- Did a tool error or provider limit produce a bad client response?
|
|
48
|
+
|
|
49
|
+
4. Generate regression evals:
|
|
42
50
|
|
|
43
51
|
```sh
|
|
44
52
|
npm run agentkit -- improve evals .agentkit/improve/<run>
|
|
45
53
|
```
|
|
46
54
|
|
|
47
|
-
|
|
55
|
+
5. Review or rewrite generated evals so they assert the behavior, not brittle transcript wording. If the bug involved a tool call, assert the persisted tool call input or absence of the unsafe call.
|
|
56
|
+
|
|
57
|
+
6. Patch the capsule. Likely files:
|
|
48
58
|
|
|
49
59
|
```txt
|
|
50
60
|
prompts/instructions.md
|
|
@@ -54,7 +64,7 @@ knowledge/
|
|
|
54
64
|
evals/
|
|
55
65
|
```
|
|
56
66
|
|
|
57
|
-
|
|
67
|
+
7. Verify:
|
|
58
68
|
|
|
59
69
|
```sh
|
|
60
70
|
npm run typecheck
|
|
@@ -63,7 +73,7 @@ npm run eval
|
|
|
63
73
|
npm run agentkit -- replay .agentkit/improve/<run> --against local
|
|
64
74
|
```
|
|
65
75
|
|
|
66
|
-
|
|
76
|
+
8. Deploy only after local evals and replay pass:
|
|
67
77
|
|
|
68
78
|
```sh
|
|
69
79
|
npm run agentkit -- deploy --smoke "hello"
|
|
@@ -44,6 +44,18 @@ For Google Calendar, do not configure create-only access. Include `GOOGLECALENDA
|
|
|
44
44
|
|
|
45
45
|
Managed Composio write actions require tool input `confirmed: true` by default. Set it only after the owner/user confirms the exact external change. Use `confirmExternalWrites: false` only when the capsule implements an equivalent confirmation guard elsewhere.
|
|
46
46
|
|
|
47
|
+
## Testability
|
|
48
|
+
|
|
49
|
+
Do not rely on the real connected app for ordinary evals. When adding an integration, also add deterministic coverage for:
|
|
50
|
+
|
|
51
|
+
- the safe path, such as free/busy before calendar create;
|
|
52
|
+
- missing confirmation before an external write;
|
|
53
|
+
- provider errors such as 429, timeout, missing auth, empty result, or unavailable slot;
|
|
54
|
+
- privacy rules, such as not showing a full calendar or raw provider payload to the client;
|
|
55
|
+
- payload invariants, such as timezone conversion, duration, recipients, or record ids.
|
|
56
|
+
|
|
57
|
+
Use eval-safe branches inside capsule tools, local fixtures, or `test/fake` behavior when the hosted integration cannot run locally without touching the real provider.
|
|
58
|
+
|
|
47
59
|
## Commands
|
|
48
60
|
|
|
49
61
|
```sh
|
|
@@ -54,6 +66,16 @@ npm run agentkit -- integrations status --toolkit googlecalendar
|
|
|
54
66
|
npm run agentkit -- integrations connect composio --toolkit gmail
|
|
55
67
|
```
|
|
56
68
|
|
|
69
|
+
`integrations connect composio` requires a hosted deploy because the connect link is deploy-scoped. After declaring `composioManaged(...)`, tell the owner the sequence is:
|
|
70
|
+
|
|
71
|
+
```sh
|
|
72
|
+
npm run agentkit -- deploy doctor
|
|
73
|
+
npm run agentkit -- deploy
|
|
74
|
+
npm run agentkit -- integrations connect composio --toolkit googlecalendar
|
|
75
|
+
```
|
|
76
|
+
|
|
77
|
+
The deploy handoff prints the connect command for each configured toolkit.
|
|
78
|
+
|
|
57
79
|
## Rules
|
|
58
80
|
|
|
59
81
|
- Do not add `COMPOSIO_API_KEY` to `.env.schema` for managed Composio.
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
---
|
|
2
2
|
name: agentkit-provider
|
|
3
|
-
description: Use when switching an AgentKit capsule from the deterministic test/fake provider to a real Pi-backed provider such as OpenAI, Anthropic, or
|
|
3
|
+
description: Use when switching an AgentKit capsule from the deterministic test/fake provider to a real Pi-backed provider such as OpenAI, Anthropic, OpenRouter, OpenCode Zen, or OpenCode Go, or when verifying provider secrets and model behavior.
|
|
4
4
|
---
|
|
5
5
|
|
|
6
6
|
# AgentKit Provider
|
|
@@ -9,7 +9,7 @@ Use this when `test/fake` is no longer enough.
|
|
|
9
9
|
|
|
10
10
|
## Rule
|
|
11
11
|
|
|
12
|
-
Do not choose a real provider automatically. Ask the owner which provider to use: OpenRouter, OpenAI, Anthropic, or another supported provider.
|
|
12
|
+
Do not choose a real provider automatically. Ask the owner which provider to use: OpenRouter, OpenAI, Anthropic, OpenCode Zen, OpenCode Go, or another supported provider.
|
|
13
13
|
|
|
14
14
|
## Workflow
|
|
15
15
|
|
|
@@ -44,6 +44,24 @@ secrets: ["OPENROUTER_API_KEY"],
|
|
|
44
44
|
|
|
45
45
|
Prefer OpenRouter model ids or aliases known to the installed Pi SDK, such as `~google/gemini-flash-latest`. If a raw OpenRouter id is newer than Pi's registry, AgentKit passes it through to OpenRouter with conservative unknown-model metadata; OpenRouter can still reject invalid, inaccessible, or unsupported models.
|
|
46
46
|
|
|
47
|
+
OpenCode Zen:
|
|
48
|
+
|
|
49
|
+
```ts
|
|
50
|
+
provider: { name: "opencode", model: "big-pickle" },
|
|
51
|
+
secrets: ["OPENCODE_API_KEY"],
|
|
52
|
+
```
|
|
53
|
+
|
|
54
|
+
Use an OpenCode Zen model id listed by the installed Pi SDK, such as `big-pickle`, `deepseek-v4-flash-free`, `claude-sonnet-4-5`, or `gpt-5.4-mini`.
|
|
55
|
+
|
|
56
|
+
OpenCode Go:
|
|
57
|
+
|
|
58
|
+
```ts
|
|
59
|
+
provider: { name: "opencode-go", model: "deepseek-v4-flash" },
|
|
60
|
+
secrets: ["OPENCODE_API_KEY"],
|
|
61
|
+
```
|
|
62
|
+
|
|
63
|
+
Use an OpenCode Go model id listed by the installed Pi SDK, such as `deepseek-v4-flash`, `deepseek-v4-pro`, `glm-5.1`, `kimi-k2.6`, `minimax-m2.7`, or `qwen3.6-plus`.
|
|
64
|
+
|
|
47
65
|
## Verification
|
|
48
66
|
|
|
49
67
|
```sh
|
|
@@ -16,6 +16,8 @@ Use this when the agent needs code, an API, live data, a write, or an external a
|
|
|
16
16
|
5. Keep secret names in `.env.schema`; values stay in ignored `.env` or hosted managed secrets.
|
|
17
17
|
6. Use `ctx.clock` for date-sensitive tool logic instead of calling `new Date()` directly.
|
|
18
18
|
7. Add eval guards for destructive or external side effects.
|
|
19
|
+
8. Add deterministic fixtures, fake branches, or direct tool inputs for important success and failure paths.
|
|
20
|
+
9. Add evals that assert the tool is called with safe inputs, or not called when confirmation/intake is missing.
|
|
19
21
|
|
|
20
22
|
## Examples
|
|
21
23
|
|