@andreprado/agentkit 0.1.0-alpha.6 → 0.1.0-alpha.7
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +4 -0
- package/docs/guides/connect-telegram.md +2 -0
- package/docs/guides/create-agent.md +9 -0
- package/docs/guides/run-evals.md +36 -1
- package/package.json +1 -1
- package/src/cli/cloud-client.ts +10 -2
- package/src/cli/commands/channels.ts +28 -2
- package/src/cli/deploy-chat-ui.ts +7 -0
- package/src/cli/help.ts +18 -2
- package/src/cli/index.ts +124 -4
- package/src/index.ts +44 -1
- package/src/providers/pi.ts +1 -1
- package/src/providers/test.ts +1 -1
- package/src/runtime/config.ts +8 -0
- package/src/runtime/database.ts +93 -2
- package/src/runtime/db-commands.ts +9 -0
- package/src/runtime/dev-server.ts +43 -2
- package/src/runtime/evals.ts +210 -20
- package/src/runtime/inspect.ts +5 -0
- package/src/runtime/spec.ts +152 -0
- package/src/runtime/sync.ts +144 -0
- package/src/runtime/targets/cloudflare/build.ts +45 -0
- package/src/runtime/tools.ts +121 -1
- package/src/runtime/traces.ts +41 -0
- package/src/storage/sqlite.ts +136 -0
- package/src/templates/blank.ts +8 -1
- package/src/templates/dentista.ts +14 -1
- package/src/templates/skills/agentkit-build-agent/SKILL.md +9 -7
- package/src/templates/skills/agentkit-database/SKILL.md +5 -2
- package/src/templates/skills/agentkit-evals/SKILL.md +32 -3
- package/src/templates/skills/agentkit-evals/templates/multi-turn.eval.md +22 -0
- package/src/templates/support.ts +7 -0
package/src/storage/sqlite.ts
CHANGED
|
@@ -42,12 +42,30 @@ export type StoredToolCall = {
|
|
|
42
42
|
visibility: ToolCallVisibility;
|
|
43
43
|
input: unknown;
|
|
44
44
|
output: unknown;
|
|
45
|
+
rendered: unknown;
|
|
45
46
|
status: ToolCallStatus;
|
|
46
47
|
startedAt: string;
|
|
47
48
|
completedAt: string | null;
|
|
48
49
|
errorMessage: string | null;
|
|
49
50
|
};
|
|
50
51
|
|
|
52
|
+
export type StoredRun = {
|
|
53
|
+
id: string;
|
|
54
|
+
conversationId: string | null;
|
|
55
|
+
provider: string;
|
|
56
|
+
model: string;
|
|
57
|
+
status: RunStatus;
|
|
58
|
+
usage: {
|
|
59
|
+
inputTokens: number | null;
|
|
60
|
+
outputTokens: number | null;
|
|
61
|
+
totalTokens: number | null;
|
|
62
|
+
};
|
|
63
|
+
startedAt: string;
|
|
64
|
+
completedAt: string | null;
|
|
65
|
+
errorCode: string | null;
|
|
66
|
+
errorMessage: string | null;
|
|
67
|
+
};
|
|
68
|
+
|
|
51
69
|
export type PersistedChatRun = {
|
|
52
70
|
conversationId: string;
|
|
53
71
|
runId: string;
|
|
@@ -158,6 +176,13 @@ const MIGRATIONS = [
|
|
|
158
176
|
id: KNOWLEDGE_MIGRATION_ID,
|
|
159
177
|
sql: KNOWLEDGE_SCHEMA_SQL,
|
|
160
178
|
},
|
|
179
|
+
{
|
|
180
|
+
id: "0007_add_tool_call_rendered",
|
|
181
|
+
sql: `
|
|
182
|
+
ALTER TABLE tool_calls
|
|
183
|
+
ADD COLUMN rendered_json TEXT;
|
|
184
|
+
`,
|
|
185
|
+
},
|
|
161
186
|
] as const;
|
|
162
187
|
|
|
163
188
|
export async function openCapsuleStore(capsule: LoadedAgentCapsule): Promise<SqliteAgentKitStore> {
|
|
@@ -462,6 +487,7 @@ export class SqliteAgentKitStore {
|
|
|
462
487
|
toolCallId: string;
|
|
463
488
|
status: ToolCallStatus;
|
|
464
489
|
output?: unknown;
|
|
490
|
+
rendered?: unknown;
|
|
465
491
|
completedAt?: string;
|
|
466
492
|
errorMessage?: string;
|
|
467
493
|
secretValues?: string[];
|
|
@@ -474,6 +500,7 @@ export class SqliteAgentKitStore {
|
|
|
474
500
|
SET
|
|
475
501
|
status = $status,
|
|
476
502
|
output_json = $outputJson,
|
|
503
|
+
rendered_json = $renderedJson,
|
|
477
504
|
completed_at = $completedAt,
|
|
478
505
|
error_message = $errorMessage
|
|
479
506
|
WHERE id = $toolCallId
|
|
@@ -483,6 +510,7 @@ export class SqliteAgentKitStore {
|
|
|
483
510
|
$toolCallId: input.toolCallId,
|
|
484
511
|
$status: input.status,
|
|
485
512
|
$outputJson: input.output === undefined ? null : serializeJson(input.output, input.secretValues),
|
|
513
|
+
$renderedJson: input.rendered === undefined ? null : serializeJson(input.rendered, input.secretValues),
|
|
486
514
|
$completedAt: input.completedAt ?? new Date().toISOString(),
|
|
487
515
|
$errorMessage: redactSecrets(input.errorMessage ?? null, input.secretValues),
|
|
488
516
|
});
|
|
@@ -555,6 +583,35 @@ export class SqliteAgentKitStore {
|
|
|
555
583
|
});
|
|
556
584
|
}
|
|
557
585
|
|
|
586
|
+
listRunsForConversation(conversationId: string): StoredRun[] {
|
|
587
|
+
return this.sqlite(() => {
|
|
588
|
+
const rows = this.db
|
|
589
|
+
.prepare(
|
|
590
|
+
`
|
|
591
|
+
SELECT
|
|
592
|
+
id,
|
|
593
|
+
conversation_id,
|
|
594
|
+
provider,
|
|
595
|
+
model,
|
|
596
|
+
status,
|
|
597
|
+
input_tokens,
|
|
598
|
+
output_tokens,
|
|
599
|
+
total_tokens,
|
|
600
|
+
started_at,
|
|
601
|
+
completed_at,
|
|
602
|
+
error_code,
|
|
603
|
+
error_message
|
|
604
|
+
FROM runs
|
|
605
|
+
WHERE conversation_id = $conversationId
|
|
606
|
+
ORDER BY started_at ASC
|
|
607
|
+
`,
|
|
608
|
+
)
|
|
609
|
+
.all({ $conversationId: conversationId }) as SqliteRow[];
|
|
610
|
+
|
|
611
|
+
return rows.map(storedRunFromRow);
|
|
612
|
+
});
|
|
613
|
+
}
|
|
614
|
+
|
|
558
615
|
listToolCallsForRun(runId: string): StoredToolCall[] {
|
|
559
616
|
return this.sqlite(() => {
|
|
560
617
|
const rows = this.db
|
|
@@ -568,6 +625,7 @@ export class SqliteAgentKitStore {
|
|
|
568
625
|
visibility,
|
|
569
626
|
input_json,
|
|
570
627
|
output_json,
|
|
628
|
+
rendered_json,
|
|
571
629
|
status,
|
|
572
630
|
started_at,
|
|
573
631
|
completed_at,
|
|
@@ -583,6 +641,54 @@ export class SqliteAgentKitStore {
|
|
|
583
641
|
});
|
|
584
642
|
}
|
|
585
643
|
|
|
644
|
+
applyApplicationMigrations(migrations: Array<{ id: string; sql: string }>): { applied: string[]; skipped: string[] } {
|
|
645
|
+
return this.sqlite(() => {
|
|
646
|
+
if (migrations.length === 0) {
|
|
647
|
+
return { applied: [], skipped: [] };
|
|
648
|
+
}
|
|
649
|
+
|
|
650
|
+
this.db.exec(`
|
|
651
|
+
CREATE TABLE IF NOT EXISTS agentkit_app_migrations (
|
|
652
|
+
id TEXT PRIMARY KEY,
|
|
653
|
+
applied_at TEXT NOT NULL
|
|
654
|
+
)
|
|
655
|
+
`);
|
|
656
|
+
|
|
657
|
+
const applied = new Set(
|
|
658
|
+
(
|
|
659
|
+
this.db.prepare("SELECT id FROM agentkit_app_migrations").all() as Array<{
|
|
660
|
+
id: string;
|
|
661
|
+
}>
|
|
662
|
+
).map((row) => row.id),
|
|
663
|
+
);
|
|
664
|
+
const result = {
|
|
665
|
+
applied: [] as string[],
|
|
666
|
+
skipped: [] as string[],
|
|
667
|
+
};
|
|
668
|
+
|
|
669
|
+
for (const migration of migrations) {
|
|
670
|
+
if (applied.has(migration.id)) {
|
|
671
|
+
result.skipped.push(migration.id);
|
|
672
|
+
continue;
|
|
673
|
+
}
|
|
674
|
+
|
|
675
|
+
const statements = splitSqlStatements(migration.sql);
|
|
676
|
+
this.transaction(() => {
|
|
677
|
+
for (const statement of statements) {
|
|
678
|
+
this.db.exec(statement);
|
|
679
|
+
}
|
|
680
|
+
|
|
681
|
+
this.db
|
|
682
|
+
.prepare("INSERT INTO agentkit_app_migrations (id, applied_at) VALUES ($id, $appliedAt)")
|
|
683
|
+
.run({ $id: migration.id, $appliedAt: new Date().toISOString() });
|
|
684
|
+
});
|
|
685
|
+
result.applied.push(migration.id);
|
|
686
|
+
}
|
|
687
|
+
|
|
688
|
+
return result;
|
|
689
|
+
});
|
|
690
|
+
}
|
|
691
|
+
|
|
586
692
|
applyApplicationSchema(source: string): void {
|
|
587
693
|
this.sqlite(() => {
|
|
588
694
|
const statements = splitSqlStatements(source);
|
|
@@ -790,6 +896,31 @@ function storedMessageFromRow(row: SqliteRow): StoredMessage {
|
|
|
790
896
|
};
|
|
791
897
|
}
|
|
792
898
|
|
|
899
|
+
function storedRunFromRow(row: SqliteRow): StoredRun {
|
|
900
|
+
const status = String(row.status);
|
|
901
|
+
|
|
902
|
+
if (status !== "running" && status !== "completed" && status !== "failed") {
|
|
903
|
+
throw new AgentKitError("storage_error", `Invalid stored run status "${status}".`);
|
|
904
|
+
}
|
|
905
|
+
|
|
906
|
+
return {
|
|
907
|
+
id: String(row.id),
|
|
908
|
+
conversationId: row.conversation_id === null ? null : String(row.conversation_id),
|
|
909
|
+
provider: String(row.provider),
|
|
910
|
+
model: String(row.model),
|
|
911
|
+
status,
|
|
912
|
+
usage: {
|
|
913
|
+
inputTokens: nullableNumber(row.input_tokens),
|
|
914
|
+
outputTokens: nullableNumber(row.output_tokens),
|
|
915
|
+
totalTokens: nullableNumber(row.total_tokens),
|
|
916
|
+
},
|
|
917
|
+
startedAt: String(row.started_at),
|
|
918
|
+
completedAt: row.completed_at === null ? null : String(row.completed_at),
|
|
919
|
+
errorCode: row.error_code === null ? null : String(row.error_code),
|
|
920
|
+
errorMessage: row.error_message === null ? null : String(row.error_message),
|
|
921
|
+
};
|
|
922
|
+
}
|
|
923
|
+
|
|
793
924
|
function storedToolCallFromRow(row: SqliteRow): StoredToolCall {
|
|
794
925
|
const status = String(row.status);
|
|
795
926
|
const visibility = String(row.visibility);
|
|
@@ -810,6 +941,7 @@ function storedToolCallFromRow(row: SqliteRow): StoredToolCall {
|
|
|
810
941
|
visibility,
|
|
811
942
|
input: parseStoredJson(row.input_json),
|
|
812
943
|
output: row.output_json === null ? undefined : parseStoredJson(row.output_json),
|
|
944
|
+
rendered: row.rendered_json === null ? undefined : parseStoredJson(row.rendered_json),
|
|
813
945
|
status,
|
|
814
946
|
startedAt: String(row.started_at),
|
|
815
947
|
completedAt: row.completed_at === null ? null : String(row.completed_at),
|
|
@@ -817,6 +949,10 @@ function storedToolCallFromRow(row: SqliteRow): StoredToolCall {
|
|
|
817
949
|
};
|
|
818
950
|
}
|
|
819
951
|
|
|
952
|
+
function nullableNumber(value: unknown): number | null {
|
|
953
|
+
return value === null || value === undefined ? null : Number(value);
|
|
954
|
+
}
|
|
955
|
+
|
|
820
956
|
function parseStoredJson(value: unknown): unknown {
|
|
821
957
|
if (typeof value !== "string") {
|
|
822
958
|
return value;
|
package/src/templates/blank.ts
CHANGED
|
@@ -148,10 +148,12 @@ When the owner opens this folder in Codex, Claude Code, or another coding agent
|
|
|
148
148
|
Start building immediately:
|
|
149
149
|
|
|
150
150
|
- Start with \`skills/agentkit-capsule/SKILL.md\`, then use \`npm run agentkit -- docs llms\` as the docs router.
|
|
151
|
+
- Create or update \`AGENT_SPEC.md\` from the owner's request with \`npm run agentkit -- spec init --brief "<owner request>"\`; the owner should not fill this file by hand before work starts.
|
|
151
152
|
- Infer the first useful version from the owner's request.
|
|
152
153
|
- Edit \`prompts/instructions.md\` for the agent behavior.
|
|
153
154
|
- Edit \`agentkit.config.ts\` for provider, tools, secrets, access, and storage.
|
|
154
155
|
- Add TypeScript tools under \`tools/\` when the requested agent needs actions or external data.
|
|
156
|
+
- Add \`sync.ts\`, \`seed.sql\`, and ordered \`migrations/*.sql\` when the requested agent depends on external catalogs or production-shaped data changes.
|
|
155
157
|
- Do not wait for a wizard or recipe. AgentKit provides the scaffold and contract; you decide the implementation from the owner's brief.
|
|
156
158
|
- Ask follow-up questions only when missing information blocks a safe local implementation.
|
|
157
159
|
- State assumptions in the final response.
|
|
@@ -162,6 +164,9 @@ Start building immediately:
|
|
|
162
164
|
- \`npm run dev\`: run the local Agent Capsule runtime.
|
|
163
165
|
- \`npm run chat -- --message "hello"\`: send one local chat message.
|
|
164
166
|
- \`npm run eval\`: run agent evals.
|
|
167
|
+
- \`npm run agentkit -- spec check\`: verify the local agent implementation contract exists.
|
|
168
|
+
- \`npm run agentkit -- eval from-conversation <conversation-id>\`: turn a real conversation into a regression eval.
|
|
169
|
+
- \`npm run agentkit -- conversations trace <conversation-id>\`: inspect messages, runs, tool calls, inputs, outputs, rendered output, and final responses.
|
|
165
170
|
- \`npm run agentkit -- tool <name> --input fixtures/input.json\`: run one registered tool directly.
|
|
166
171
|
- \`npm run agentkit -- db migrate\`: apply internal migrations and \`schema.sql\` locally.
|
|
167
172
|
- \`npm run agentkit -- db reset --yes\`: recreate the local SQLite database and reapply schema.
|
|
@@ -212,7 +217,7 @@ export const myTool = defineTool({
|
|
|
212
217
|
|
|
213
218
|
Use \`ctx.db\` as the canonical helper. \`ctx.database\` and \`ctx.storage.sql\` are aliases. Use \`ctx.db.batch([...])\` for atomic writes; local tools can also use \`ctx.db.transaction(async (tx) => ...)\`. Do not import local database drivers or Node-only APIs in tools. AgentKit owns local and hosted database routing.
|
|
214
219
|
|
|
215
|
-
\`schema.sql\` is an idempotent bootstrap file
|
|
220
|
+
\`schema.sql\` is an idempotent bootstrap file. Use \`CREATE TABLE IF NOT EXISTS\`, \`CREATE INDEX IF NOT EXISTS\`, and safe additive changes. Use ordered \`migrations/*.sql\` for production-shaped schema evolution; \`npm run agentkit -- db migrate\` applies unapplied local migrations before \`schema.sql\`.
|
|
216
221
|
|
|
217
222
|
## Hosted Deploy
|
|
218
223
|
|
|
@@ -252,8 +257,10 @@ Example owner request:
|
|
|
252
257
|
Turn the request into a working local capsule:
|
|
253
258
|
|
|
254
259
|
- Update \`prompts/instructions.md\` with domain-specific behavior, boundaries, intake questions, and escalation rules.
|
|
260
|
+
- Create or update \`AGENT_SPEC.md\` with \`npm run agentkit -- spec init --brief "<owner request>"\`. The owner gives the general idea; the coding agent turns it into the structured contract.
|
|
255
261
|
- Update \`agentkit.config.ts\` when tools, secrets, provider, or access rules change.
|
|
256
262
|
- Add TypeScript tools under \`tools/\` for real actions or external data.
|
|
263
|
+
- Use \`npm run agentkit -- sync init\` when the agent needs catalog sync, fixture seed data, or ordered migrations.
|
|
257
264
|
- Keep the first version runnable with \`test/fake\` unless the owner explicitly asks for a real provider.
|
|
258
265
|
- Do not use a wizard or recipe. Build the capsule directly from the scaffold, the AgentKit contract, and the owner's brief.
|
|
259
266
|
- Make practical assumptions and list them in your final response.
|
|
@@ -857,6 +857,8 @@ Esta é uma AgentKit Agent Capsule para a Clara, atendente em português de um c
|
|
|
857
857
|
|
|
858
858
|
A Clara conversa com clientes, coleta nome, email e telefone, consulta disponibilidade e agenda consultas em slots de 30 minutos. Ela também consulta e altera o próprio horário do cliente usando email e telefone como verificação mínima.
|
|
859
859
|
|
|
860
|
+
Quando o dono pedir mudanças em linguagem natural, crie ou atualize \`AGENT_SPEC.md\` com \`npm run agentkit -- spec init --brief "<pedido do dono>"\`. O dono dá a ideia geral; o agente de código transforma isso no contrato estruturado.
|
|
861
|
+
|
|
860
862
|
## Regras de Produto
|
|
861
863
|
|
|
862
864
|
- Persona: Clara, atendente humana, simpática e objetiva.
|
|
@@ -885,6 +887,9 @@ A Clara conversa com clientes, coleta nome, email e telefone, consulta disponibi
|
|
|
885
887
|
- \`npm run agentkit -- tool consultar_consulta --input '{"email":"ana@example.com","telefone":"11999999999"}'\`: consultar uma consulta.
|
|
886
888
|
- \`npm run agentkit -- tool alterar_consulta --input '{"email":"ana@example.com","telefone":"11999999999","novaData":"2099-01-06","novoHorario":"10:30","confirmadoPeloCliente":true}'\`: alterar uma consulta confirmada.
|
|
887
889
|
- \`npm run eval\`: rodar evals.
|
|
890
|
+
- \`npm run agentkit -- spec check\`: validar o contrato local da implementação.
|
|
891
|
+
- \`npm run agentkit -- eval from-conversation <conversation-id>\`: transformar uma conversa real em eval de regressão.
|
|
892
|
+
- \`npm run agentkit -- conversations trace <conversation-id>\`: inspecionar mensagens, runs, ferramentas, inputs, outputs, renderização e resposta final.
|
|
888
893
|
- \`npm run dev\`: rodar a runtime local.
|
|
889
894
|
- \`npm run agentkit -- inspect\`: imprimir o estado da cápsula.
|
|
890
895
|
- \`npm run agentkit -- docs llms\`: imprimir o roteador leve da documentação AgentKit.
|
|
@@ -900,7 +905,8 @@ A Clara conversa com clientes, coleta nome, email e telefone, consulta disponibi
|
|
|
900
905
|
## Regras de Implementação
|
|
901
906
|
|
|
902
907
|
- Use \`ctx.db\` nas ferramentas. Não importe drivers SQLite, Turso ou Node-only APIs.
|
|
903
|
-
- Mantenha \`schema.sql\` idempotente com \`CREATE TABLE IF NOT EXISTS\` e \`CREATE INDEX IF NOT EXISTS\`.
|
|
908
|
+
- Mantenha \`schema.sql\` idempotente com \`CREATE TABLE IF NOT EXISTS\` e \`CREATE INDEX IF NOT EXISTS\`. Use \`migrations/*.sql\` ordenadas para evolução de schema com cara de produção.
|
|
909
|
+
- Use \`npm run agentkit -- sync init\` quando a agente depender de catálogo externo, fixture local ou seed padronizado.
|
|
904
910
|
- Mantenha segredos em \`.env\`, nunca em arquivos versionados.
|
|
905
911
|
- Não espere wizard ou recipe. AgentKit fornece o scaffold e o contrato; implemente diretamente conforme o brief do dono.
|
|
906
912
|
- Para trocar para um provider real, o dono deve escolher OpenRouter, OpenAI, Anthropic ou outro provider suportado. Depois edite \`agentkit.config.ts\`, atualize \`.env.schema\` e configure secrets locais/hosted.
|
|
@@ -933,6 +939,13 @@ npm run chat -- --message "Oi, quero marcar uma consulta."
|
|
|
933
939
|
|
|
934
940
|
A cápsula começa com \`test/fake\`. Isso valida caminhos determinísticos, mas não prova qualidade de conversa natural.
|
|
935
941
|
|
|
942
|
+
Se o dono pedir uma mudança grande, crie ou atualize o contrato de implementação:
|
|
943
|
+
|
|
944
|
+
\`\`\`sh
|
|
945
|
+
npm run agentkit -- spec init --brief "<pedido do dono>"
|
|
946
|
+
npm run agentkit -- spec check
|
|
947
|
+
\`\`\`
|
|
948
|
+
|
|
936
949
|
## Ferramentas
|
|
937
950
|
|
|
938
951
|
- \`listar_horarios_disponiveis\`: recebe \`data\` em \`YYYY-MM-DD\` e retorna slots livres.
|
|
@@ -10,12 +10,14 @@ Use this when the owner asks for an agent in plain language.
|
|
|
10
10
|
## Workflow
|
|
11
11
|
|
|
12
12
|
1. Read `agentkit.config.ts`, `prompts/instructions.md`, `schema.sql`, `evals/`, and existing `tools/`.
|
|
13
|
-
2.
|
|
14
|
-
3.
|
|
15
|
-
4.
|
|
16
|
-
5. Add
|
|
17
|
-
6. Add or
|
|
18
|
-
7.
|
|
13
|
+
2. If `AGENT_SPEC.md` does not exist, create it from the owner's plain-language request with `npm run agentkit -- spec init --brief "<owner request>"`. If it exists, update it directly before changing behavior.
|
|
14
|
+
3. Infer the first useful local version from the owner's brief and the spec. Do not ask the owner to fill a form.
|
|
15
|
+
4. Edit `prompts/instructions.md` for behavior, boundaries, intake questions, escalation rules, and tool-use policy.
|
|
16
|
+
5. Add tools only when the agent needs action, live data, authorization-sensitive data, or durable writes.
|
|
17
|
+
6. Add database tables to `schema.sql` or ordered `migrations/*.sql` when the agent owns records.
|
|
18
|
+
7. Add `sync.ts` and `seed.sql` with `npm run agentkit -- sync init` when the agent depends on external catalogs or recurring imports.
|
|
19
|
+
8. Add or update evals for the main flow. Prefer multi-turn `turns` evals for real conversations.
|
|
20
|
+
9. Keep the capsule runnable on `test/fake` unless the owner has chosen a real provider.
|
|
19
21
|
|
|
20
22
|
## Templates
|
|
21
23
|
|
|
@@ -35,6 +37,7 @@ npm run typecheck
|
|
|
35
37
|
npm run agentkit -- inspect
|
|
36
38
|
npm run chat -- --message "hello"
|
|
37
39
|
npm run eval
|
|
40
|
+
npm run agentkit -- spec check
|
|
38
41
|
```
|
|
39
42
|
|
|
40
43
|
If a tool was added:
|
|
@@ -46,4 +49,3 @@ npm run agentkit -- tool <tool_name> --input '<json>'
|
|
|
46
49
|
## Final Response
|
|
47
50
|
|
|
48
51
|
Summarize the files changed, assumptions made, verification results, and whether the behavior was tested with `test/fake` or a real provider selected by the owner.
|
|
49
|
-
|
|
@@ -9,12 +9,14 @@ Use this when a capsule owns durable application records.
|
|
|
9
9
|
|
|
10
10
|
## Rules
|
|
11
11
|
|
|
12
|
-
- Put
|
|
12
|
+
- Put the first idempotent bootstrap schema in `schema.sql`.
|
|
13
|
+
- For production-shaped changes, prefer ordered `migrations/*.sql` files such as `migrations/0001_initial.sql`.
|
|
13
14
|
- Keep `schema.sql` idempotent with `CREATE TABLE IF NOT EXISTS`, `CREATE INDEX IF NOT EXISTS`, and safe additive changes.
|
|
14
15
|
- Keep deploy-ready capsules on `storage.driver: "agentkit"`.
|
|
15
16
|
- Use `ctx.db` inside tools. `ctx.database` and `ctx.storage.sql` are aliases.
|
|
16
17
|
- Do not import SQLite, Turso, or other database drivers from tools.
|
|
17
18
|
- Do not edit `.agentkit/agentkit.db` by hand.
|
|
19
|
+
- For external catalogs, run `npm run agentkit -- sync init`, then implement `sync.ts` and keep local fixtures in `seed.sql`.
|
|
18
20
|
|
|
19
21
|
## Templates
|
|
20
22
|
|
|
@@ -26,6 +28,8 @@ Use this when a capsule owns durable application records.
|
|
|
26
28
|
```sh
|
|
27
29
|
npm run agentkit -- db migrate
|
|
28
30
|
npm run agentkit -- db seed --file seed.sql
|
|
31
|
+
npm run agentkit -- sync init
|
|
32
|
+
npm run agentkit -- sync run
|
|
29
33
|
npm run agentkit -- db shell
|
|
30
34
|
npm run agentkit -- db reset --yes
|
|
31
35
|
```
|
|
@@ -39,4 +43,3 @@ npm run agentkit -- tool <tool_name> --input '<json>'
|
|
|
39
43
|
```
|
|
40
44
|
|
|
41
45
|
Hosted deploy applies AgentKit-managed storage internally. The user should not create hosted databases or buckets by hand.
|
|
42
|
-
|
|
@@ -11,14 +11,43 @@ Use evals after chat works and before claiming behavior is stable.
|
|
|
11
11
|
|
|
12
12
|
1. Create or edit `evals/<name>.eval.ts`.
|
|
13
13
|
2. Keep assertions small and deterministic.
|
|
14
|
-
3. Use `
|
|
15
|
-
4.
|
|
16
|
-
5.
|
|
14
|
+
3. Use `turns` for full conversation flows, such as user asks, agent calls a tool, then the answer follows the required format.
|
|
15
|
+
4. Use `persisted_tool_call` for tool behavior stored in local SQLite.
|
|
16
|
+
5. Convert real failures into regression tests with `npm run agentkit -- eval from-conversation <conversation-id>`.
|
|
17
|
+
6. Do not put secrets or real client PII in evals.
|
|
18
|
+
7. For tools that write externally, delete, charge money, send email, or call real customer systems, branch on `ctx.runtime.environment === "eval"` inside the registered tool.
|
|
19
|
+
|
|
20
|
+
## Multi-turn Example
|
|
21
|
+
|
|
22
|
+
```ts
|
|
23
|
+
export default {
|
|
24
|
+
name: "buyer under budget",
|
|
25
|
+
turns: [
|
|
26
|
+
{
|
|
27
|
+
input: "I want a house up to 600k near Pinheiros.",
|
|
28
|
+
expect: {
|
|
29
|
+
persisted_tool_call: {
|
|
30
|
+
name: "buscar_imoveis",
|
|
31
|
+
status: "completed",
|
|
32
|
+
input: { maxPrice: 600000 },
|
|
33
|
+
},
|
|
34
|
+
},
|
|
35
|
+
},
|
|
36
|
+
{
|
|
37
|
+
input: "Show me the best two.",
|
|
38
|
+
expect: {
|
|
39
|
+
contains: ["R$", "Pinheiros"],
|
|
40
|
+
},
|
|
41
|
+
},
|
|
42
|
+
],
|
|
43
|
+
};
|
|
44
|
+
```
|
|
17
45
|
|
|
18
46
|
## Templates
|
|
19
47
|
|
|
20
48
|
- `templates/smoke.eval.md`
|
|
21
49
|
- `templates/tool-call.eval.md`
|
|
50
|
+
- `templates/multi-turn.eval.md`
|
|
22
51
|
- `templates/no-leak.eval.md`
|
|
23
52
|
|
|
24
53
|
## Verification
|
|
@@ -0,0 +1,22 @@
|
|
|
1
|
+
```ts
|
|
2
|
+
export default {
|
|
3
|
+
name: "main conversation flow",
|
|
4
|
+
turns: [
|
|
5
|
+
{
|
|
6
|
+
input: "I need help finding an option under my budget.",
|
|
7
|
+
expect: {
|
|
8
|
+
contains: "budget",
|
|
9
|
+
},
|
|
10
|
+
},
|
|
11
|
+
{
|
|
12
|
+
input: "Show me the best match.",
|
|
13
|
+
expect: {
|
|
14
|
+
persisted_tool_call: {
|
|
15
|
+
name: "replace_with_tool_name",
|
|
16
|
+
status: "completed",
|
|
17
|
+
},
|
|
18
|
+
},
|
|
19
|
+
},
|
|
20
|
+
],
|
|
21
|
+
};
|
|
22
|
+
```
|
package/src/templates/support.ts
CHANGED
|
@@ -191,10 +191,12 @@ When the owner opens this folder in Codex, Claude Code, or another coding agent
|
|
|
191
191
|
Start building immediately:
|
|
192
192
|
|
|
193
193
|
- Start with \`skills/agentkit-capsule/SKILL.md\`, then use \`npm run agentkit -- docs llms\` as the docs router.
|
|
194
|
+
- Create or update \`AGENT_SPEC.md\` from the owner's request with \`npm run agentkit -- spec init --brief "<owner request>"\`; the owner should not fill this file by hand before work starts.
|
|
194
195
|
- Infer the first useful version from the owner's request.
|
|
195
196
|
- Edit \`prompts/instructions.md\` for support behavior.
|
|
196
197
|
- Edit \`agentkit.config.ts\` for provider, tools, secrets, access, and storage.
|
|
197
198
|
- Add or replace TypeScript tools under \`tools/\` when the requested support agent needs actions or external data.
|
|
199
|
+
- Add \`sync.ts\`, \`seed.sql\`, and ordered \`migrations/*.sql\` when the support agent depends on external catalogs or production-shaped data changes.
|
|
198
200
|
- Do not wait for a wizard or recipe. AgentKit provides the scaffold and contract; you decide the implementation from the owner's brief.
|
|
199
201
|
- Ask follow-up questions only when missing information blocks a safe local implementation.
|
|
200
202
|
- State assumptions in the final response.
|
|
@@ -205,6 +207,9 @@ Start building immediately:
|
|
|
205
207
|
- \`npm run chat -- --message "hello"\`: send one local chat message.
|
|
206
208
|
- \`npm run agentkit -- tool lookup_order --input '{"orderId":"A100"}'\`: test the example tool directly.
|
|
207
209
|
- \`npm run eval\`: run agent evals.
|
|
210
|
+
- \`npm run agentkit -- spec check\`: verify the local agent implementation contract exists.
|
|
211
|
+
- \`npm run agentkit -- eval from-conversation <conversation-id>\`: turn a real conversation into a regression eval.
|
|
212
|
+
- \`npm run agentkit -- conversations trace <conversation-id>\`: inspect messages, runs, tool calls, inputs, outputs, rendered output, and final responses.
|
|
208
213
|
- \`npm run dev\`: run the local Agent Capsule runtime.
|
|
209
214
|
- \`npm run agentkit -- inspect\`: print machine-readable capsule state.
|
|
210
215
|
- \`printf %s "$VALUE" | npm run agentkit -- env set <NAME> --stdin\`: write a local secret value to ignored \`.env\` without putting it in shell history.
|
|
@@ -259,8 +264,10 @@ Example owner request:
|
|
|
259
264
|
Turn the request into a working local capsule:
|
|
260
265
|
|
|
261
266
|
- Update \`prompts/instructions.md\` with domain-specific behavior, boundaries, intake questions, and escalation rules.
|
|
267
|
+
- Create or update \`AGENT_SPEC.md\` with \`npm run agentkit -- spec init --brief "<owner request>"\`. The owner gives the general idea; the coding agent turns it into the structured contract.
|
|
262
268
|
- Update \`agentkit.config.ts\` when tools, secrets, provider, or access rules change.
|
|
263
269
|
- Add, replace, or remove TypeScript tools under \`tools/\` for real actions or external data.
|
|
270
|
+
- Use \`npm run agentkit -- sync init\` when the agent needs catalog sync, fixture seed data, or ordered migrations.
|
|
264
271
|
- Keep the first version runnable with \`test/fake\` unless the owner explicitly asks for a real provider.
|
|
265
272
|
- Do not use a wizard or recipe. Build the capsule directly from the scaffold, the AgentKit contract, and the owner's brief.
|
|
266
273
|
- Make practical assumptions and list them in your final response.
|