@andreprado/agentkit 0.1.0-alpha.6 → 0.1.0-alpha.7

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -42,12 +42,30 @@ export type StoredToolCall = {
42
42
  visibility: ToolCallVisibility;
43
43
  input: unknown;
44
44
  output: unknown;
45
+ rendered: unknown;
45
46
  status: ToolCallStatus;
46
47
  startedAt: string;
47
48
  completedAt: string | null;
48
49
  errorMessage: string | null;
49
50
  };
50
51
 
52
+ export type StoredRun = {
53
+ id: string;
54
+ conversationId: string | null;
55
+ provider: string;
56
+ model: string;
57
+ status: RunStatus;
58
+ usage: {
59
+ inputTokens: number | null;
60
+ outputTokens: number | null;
61
+ totalTokens: number | null;
62
+ };
63
+ startedAt: string;
64
+ completedAt: string | null;
65
+ errorCode: string | null;
66
+ errorMessage: string | null;
67
+ };
68
+
51
69
  export type PersistedChatRun = {
52
70
  conversationId: string;
53
71
  runId: string;
@@ -158,6 +176,13 @@ const MIGRATIONS = [
158
176
  id: KNOWLEDGE_MIGRATION_ID,
159
177
  sql: KNOWLEDGE_SCHEMA_SQL,
160
178
  },
179
+ {
180
+ id: "0007_add_tool_call_rendered",
181
+ sql: `
182
+ ALTER TABLE tool_calls
183
+ ADD COLUMN rendered_json TEXT;
184
+ `,
185
+ },
161
186
  ] as const;
162
187
 
163
188
  export async function openCapsuleStore(capsule: LoadedAgentCapsule): Promise<SqliteAgentKitStore> {
@@ -462,6 +487,7 @@ export class SqliteAgentKitStore {
462
487
  toolCallId: string;
463
488
  status: ToolCallStatus;
464
489
  output?: unknown;
490
+ rendered?: unknown;
465
491
  completedAt?: string;
466
492
  errorMessage?: string;
467
493
  secretValues?: string[];
@@ -474,6 +500,7 @@ export class SqliteAgentKitStore {
474
500
  SET
475
501
  status = $status,
476
502
  output_json = $outputJson,
503
+ rendered_json = $renderedJson,
477
504
  completed_at = $completedAt,
478
505
  error_message = $errorMessage
479
506
  WHERE id = $toolCallId
@@ -483,6 +510,7 @@ export class SqliteAgentKitStore {
483
510
  $toolCallId: input.toolCallId,
484
511
  $status: input.status,
485
512
  $outputJson: input.output === undefined ? null : serializeJson(input.output, input.secretValues),
513
+ $renderedJson: input.rendered === undefined ? null : serializeJson(input.rendered, input.secretValues),
486
514
  $completedAt: input.completedAt ?? new Date().toISOString(),
487
515
  $errorMessage: redactSecrets(input.errorMessage ?? null, input.secretValues),
488
516
  });
@@ -555,6 +583,35 @@ export class SqliteAgentKitStore {
555
583
  });
556
584
  }
557
585
 
586
+ listRunsForConversation(conversationId: string): StoredRun[] {
587
+ return this.sqlite(() => {
588
+ const rows = this.db
589
+ .prepare(
590
+ `
591
+ SELECT
592
+ id,
593
+ conversation_id,
594
+ provider,
595
+ model,
596
+ status,
597
+ input_tokens,
598
+ output_tokens,
599
+ total_tokens,
600
+ started_at,
601
+ completed_at,
602
+ error_code,
603
+ error_message
604
+ FROM runs
605
+ WHERE conversation_id = $conversationId
606
+ ORDER BY started_at ASC
607
+ `,
608
+ )
609
+ .all({ $conversationId: conversationId }) as SqliteRow[];
610
+
611
+ return rows.map(storedRunFromRow);
612
+ });
613
+ }
614
+
558
615
  listToolCallsForRun(runId: string): StoredToolCall[] {
559
616
  return this.sqlite(() => {
560
617
  const rows = this.db
@@ -568,6 +625,7 @@ export class SqliteAgentKitStore {
568
625
  visibility,
569
626
  input_json,
570
627
  output_json,
628
+ rendered_json,
571
629
  status,
572
630
  started_at,
573
631
  completed_at,
@@ -583,6 +641,54 @@ export class SqliteAgentKitStore {
583
641
  });
584
642
  }
585
643
 
644
+ applyApplicationMigrations(migrations: Array<{ id: string; sql: string }>): { applied: string[]; skipped: string[] } {
645
+ return this.sqlite(() => {
646
+ if (migrations.length === 0) {
647
+ return { applied: [], skipped: [] };
648
+ }
649
+
650
+ this.db.exec(`
651
+ CREATE TABLE IF NOT EXISTS agentkit_app_migrations (
652
+ id TEXT PRIMARY KEY,
653
+ applied_at TEXT NOT NULL
654
+ )
655
+ `);
656
+
657
+ const applied = new Set(
658
+ (
659
+ this.db.prepare("SELECT id FROM agentkit_app_migrations").all() as Array<{
660
+ id: string;
661
+ }>
662
+ ).map((row) => row.id),
663
+ );
664
+ const result = {
665
+ applied: [] as string[],
666
+ skipped: [] as string[],
667
+ };
668
+
669
+ for (const migration of migrations) {
670
+ if (applied.has(migration.id)) {
671
+ result.skipped.push(migration.id);
672
+ continue;
673
+ }
674
+
675
+ const statements = splitSqlStatements(migration.sql);
676
+ this.transaction(() => {
677
+ for (const statement of statements) {
678
+ this.db.exec(statement);
679
+ }
680
+
681
+ this.db
682
+ .prepare("INSERT INTO agentkit_app_migrations (id, applied_at) VALUES ($id, $appliedAt)")
683
+ .run({ $id: migration.id, $appliedAt: new Date().toISOString() });
684
+ });
685
+ result.applied.push(migration.id);
686
+ }
687
+
688
+ return result;
689
+ });
690
+ }
691
+
586
692
  applyApplicationSchema(source: string): void {
587
693
  this.sqlite(() => {
588
694
  const statements = splitSqlStatements(source);
@@ -790,6 +896,31 @@ function storedMessageFromRow(row: SqliteRow): StoredMessage {
790
896
  };
791
897
  }
792
898
 
899
+ function storedRunFromRow(row: SqliteRow): StoredRun {
900
+ const status = String(row.status);
901
+
902
+ if (status !== "running" && status !== "completed" && status !== "failed") {
903
+ throw new AgentKitError("storage_error", `Invalid stored run status "${status}".`);
904
+ }
905
+
906
+ return {
907
+ id: String(row.id),
908
+ conversationId: row.conversation_id === null ? null : String(row.conversation_id),
909
+ provider: String(row.provider),
910
+ model: String(row.model),
911
+ status,
912
+ usage: {
913
+ inputTokens: nullableNumber(row.input_tokens),
914
+ outputTokens: nullableNumber(row.output_tokens),
915
+ totalTokens: nullableNumber(row.total_tokens),
916
+ },
917
+ startedAt: String(row.started_at),
918
+ completedAt: row.completed_at === null ? null : String(row.completed_at),
919
+ errorCode: row.error_code === null ? null : String(row.error_code),
920
+ errorMessage: row.error_message === null ? null : String(row.error_message),
921
+ };
922
+ }
923
+
793
924
  function storedToolCallFromRow(row: SqliteRow): StoredToolCall {
794
925
  const status = String(row.status);
795
926
  const visibility = String(row.visibility);
@@ -810,6 +941,7 @@ function storedToolCallFromRow(row: SqliteRow): StoredToolCall {
810
941
  visibility,
811
942
  input: parseStoredJson(row.input_json),
812
943
  output: row.output_json === null ? undefined : parseStoredJson(row.output_json),
944
+ rendered: row.rendered_json === null ? undefined : parseStoredJson(row.rendered_json),
813
945
  status,
814
946
  startedAt: String(row.started_at),
815
947
  completedAt: row.completed_at === null ? null : String(row.completed_at),
@@ -817,6 +949,10 @@ function storedToolCallFromRow(row: SqliteRow): StoredToolCall {
817
949
  };
818
950
  }
819
951
 
952
+ function nullableNumber(value: unknown): number | null {
953
+ return value === null || value === undefined ? null : Number(value);
954
+ }
955
+
820
956
  function parseStoredJson(value: unknown): unknown {
821
957
  if (typeof value !== "string") {
822
958
  return value;
@@ -148,10 +148,12 @@ When the owner opens this folder in Codex, Claude Code, or another coding agent
148
148
  Start building immediately:
149
149
 
150
150
  - Start with \`skills/agentkit-capsule/SKILL.md\`, then use \`npm run agentkit -- docs llms\` as the docs router.
151
+ - Create or update \`AGENT_SPEC.md\` from the owner's request with \`npm run agentkit -- spec init --brief "<owner request>"\`; the owner should not fill this file by hand before work starts.
151
152
  - Infer the first useful version from the owner's request.
152
153
  - Edit \`prompts/instructions.md\` for the agent behavior.
153
154
  - Edit \`agentkit.config.ts\` for provider, tools, secrets, access, and storage.
154
155
  - Add TypeScript tools under \`tools/\` when the requested agent needs actions or external data.
156
+ - Add \`sync.ts\`, \`seed.sql\`, and ordered \`migrations/*.sql\` when the requested agent depends on external catalogs or production-shaped data changes.
155
157
  - Do not wait for a wizard or recipe. AgentKit provides the scaffold and contract; you decide the implementation from the owner's brief.
156
158
  - Ask follow-up questions only when missing information blocks a safe local implementation.
157
159
  - State assumptions in the final response.
@@ -162,6 +164,9 @@ Start building immediately:
162
164
  - \`npm run dev\`: run the local Agent Capsule runtime.
163
165
  - \`npm run chat -- --message "hello"\`: send one local chat message.
164
166
  - \`npm run eval\`: run agent evals.
167
+ - \`npm run agentkit -- spec check\`: verify the local agent implementation contract exists.
168
+ - \`npm run agentkit -- eval from-conversation <conversation-id>\`: turn a real conversation into a regression eval.
169
+ - \`npm run agentkit -- conversations trace <conversation-id>\`: inspect messages, runs, tool calls, inputs, outputs, rendered output, and final responses.
165
170
  - \`npm run agentkit -- tool <name> --input fixtures/input.json\`: run one registered tool directly.
166
171
  - \`npm run agentkit -- db migrate\`: apply internal migrations and \`schema.sql\` locally.
167
172
  - \`npm run agentkit -- db reset --yes\`: recreate the local SQLite database and reapply schema.
@@ -212,7 +217,7 @@ export const myTool = defineTool({
212
217
 
213
218
  Use \`ctx.db\` as the canonical helper. \`ctx.database\` and \`ctx.storage.sql\` are aliases. Use \`ctx.db.batch([...])\` for atomic writes; local tools can also use \`ctx.db.transaction(async (tx) => ...)\`. Do not import local database drivers or Node-only APIs in tools. AgentKit owns local and hosted database routing.
214
219
 
215
- \`schema.sql\` is an idempotent bootstrap file in v1. Use \`CREATE TABLE IF NOT EXISTS\`, \`CREATE INDEX IF NOT EXISTS\`, and safe additive changes. AgentKit does not run destructive schema changes or ordered \`migrations/*.sql\` automatically yet.
220
+ \`schema.sql\` is an idempotent bootstrap file. Use \`CREATE TABLE IF NOT EXISTS\`, \`CREATE INDEX IF NOT EXISTS\`, and safe additive changes. Use ordered \`migrations/*.sql\` for production-shaped schema evolution; \`npm run agentkit -- db migrate\` applies unapplied local migrations before \`schema.sql\`.
216
221
 
217
222
  ## Hosted Deploy
218
223
 
@@ -252,8 +257,10 @@ Example owner request:
252
257
  Turn the request into a working local capsule:
253
258
 
254
259
  - Update \`prompts/instructions.md\` with domain-specific behavior, boundaries, intake questions, and escalation rules.
260
+ - Create or update \`AGENT_SPEC.md\` with \`npm run agentkit -- spec init --brief "<owner request>"\`. The owner gives the general idea; the coding agent turns it into the structured contract.
255
261
  - Update \`agentkit.config.ts\` when tools, secrets, provider, or access rules change.
256
262
  - Add TypeScript tools under \`tools/\` for real actions or external data.
263
+ - Use \`npm run agentkit -- sync init\` when the agent needs catalog sync, fixture seed data, or ordered migrations.
257
264
  - Keep the first version runnable with \`test/fake\` unless the owner explicitly asks for a real provider.
258
265
  - Do not use a wizard or recipe. Build the capsule directly from the scaffold, the AgentKit contract, and the owner's brief.
259
266
  - Make practical assumptions and list them in your final response.
@@ -857,6 +857,8 @@ Esta é uma AgentKit Agent Capsule para a Clara, atendente em português de um c
857
857
 
858
858
  A Clara conversa com clientes, coleta nome, email e telefone, consulta disponibilidade e agenda consultas em slots de 30 minutos. Ela também consulta e altera o próprio horário do cliente usando email e telefone como verificação mínima.
859
859
 
860
+ Quando o dono pedir mudanças em linguagem natural, crie ou atualize \`AGENT_SPEC.md\` com \`npm run agentkit -- spec init --brief "<pedido do dono>"\`. O dono dá a ideia geral; o agente de código transforma isso no contrato estruturado.
861
+
860
862
  ## Regras de Produto
861
863
 
862
864
  - Persona: Clara, atendente humana, simpática e objetiva.
@@ -885,6 +887,9 @@ A Clara conversa com clientes, coleta nome, email e telefone, consulta disponibi
885
887
  - \`npm run agentkit -- tool consultar_consulta --input '{"email":"ana@example.com","telefone":"11999999999"}'\`: consultar uma consulta.
886
888
  - \`npm run agentkit -- tool alterar_consulta --input '{"email":"ana@example.com","telefone":"11999999999","novaData":"2099-01-06","novoHorario":"10:30","confirmadoPeloCliente":true}'\`: alterar uma consulta confirmada.
887
889
  - \`npm run eval\`: rodar evals.
890
+ - \`npm run agentkit -- spec check\`: validar o contrato local da implementação.
891
+ - \`npm run agentkit -- eval from-conversation <conversation-id>\`: transformar uma conversa real em eval de regressão.
892
+ - \`npm run agentkit -- conversations trace <conversation-id>\`: inspecionar mensagens, runs, ferramentas, inputs, outputs, renderização e resposta final.
888
893
  - \`npm run dev\`: rodar a runtime local.
889
894
  - \`npm run agentkit -- inspect\`: imprimir o estado da cápsula.
890
895
  - \`npm run agentkit -- docs llms\`: imprimir o roteador leve da documentação AgentKit.
@@ -900,7 +905,8 @@ A Clara conversa com clientes, coleta nome, email e telefone, consulta disponibi
900
905
  ## Regras de Implementação
901
906
 
902
907
  - Use \`ctx.db\` nas ferramentas. Não importe drivers SQLite, Turso ou Node-only APIs.
903
- - Mantenha \`schema.sql\` idempotente com \`CREATE TABLE IF NOT EXISTS\` e \`CREATE INDEX IF NOT EXISTS\`.
908
+ - Mantenha \`schema.sql\` idempotente com \`CREATE TABLE IF NOT EXISTS\` e \`CREATE INDEX IF NOT EXISTS\`. Use \`migrations/*.sql\` ordenadas para evolução de schema com cara de produção.
909
+ - Use \`npm run agentkit -- sync init\` quando a agente depender de catálogo externo, fixture local ou seed padronizado.
904
910
  - Mantenha segredos em \`.env\`, nunca em arquivos versionados.
905
911
  - Não espere wizard ou recipe. AgentKit fornece o scaffold e o contrato; implemente diretamente conforme o brief do dono.
906
912
  - Para trocar para um provider real, o dono deve escolher OpenRouter, OpenAI, Anthropic ou outro provider suportado. Depois edite \`agentkit.config.ts\`, atualize \`.env.schema\` e configure secrets locais/hosted.
@@ -933,6 +939,13 @@ npm run chat -- --message "Oi, quero marcar uma consulta."
933
939
 
934
940
  A cápsula começa com \`test/fake\`. Isso valida caminhos determinísticos, mas não prova qualidade de conversa natural.
935
941
 
942
+ Se o dono pedir uma mudança grande, crie ou atualize o contrato de implementação:
943
+
944
+ \`\`\`sh
945
+ npm run agentkit -- spec init --brief "<pedido do dono>"
946
+ npm run agentkit -- spec check
947
+ \`\`\`
948
+
936
949
  ## Ferramentas
937
950
 
938
951
  - \`listar_horarios_disponiveis\`: recebe \`data\` em \`YYYY-MM-DD\` e retorna slots livres.
@@ -10,12 +10,14 @@ Use this when the owner asks for an agent in plain language.
10
10
  ## Workflow
11
11
 
12
12
  1. Read `agentkit.config.ts`, `prompts/instructions.md`, `schema.sql`, `evals/`, and existing `tools/`.
13
- 2. Infer the first useful local version from the owner's brief.
14
- 3. Edit `prompts/instructions.md` for behavior, boundaries, intake questions, escalation rules, and tool-use policy.
15
- 4. Add tools only when the agent needs action, live data, authorization-sensitive data, or durable writes.
16
- 5. Add database tables to `schema.sql` when the agent owns records.
17
- 6. Add or update evals for the main flow.
18
- 7. Keep the capsule runnable on `test/fake` unless the owner has chosen a real provider.
13
+ 2. If `AGENT_SPEC.md` does not exist, create it from the owner's plain-language request with `npm run agentkit -- spec init --brief "<owner request>"`. If it exists, update it directly before changing behavior.
14
+ 3. Infer the first useful local version from the owner's brief and the spec. Do not ask the owner to fill a form.
15
+ 4. Edit `prompts/instructions.md` for behavior, boundaries, intake questions, escalation rules, and tool-use policy.
16
+ 5. Add tools only when the agent needs action, live data, authorization-sensitive data, or durable writes.
17
+ 6. Add database tables to `schema.sql` or ordered `migrations/*.sql` when the agent owns records.
18
+ 7. Add `sync.ts` and `seed.sql` with `npm run agentkit -- sync init` when the agent depends on external catalogs or recurring imports.
19
+ 8. Add or update evals for the main flow. Prefer multi-turn `turns` evals for real conversations.
20
+ 9. Keep the capsule runnable on `test/fake` unless the owner has chosen a real provider.
19
21
 
20
22
  ## Templates
21
23
 
@@ -35,6 +37,7 @@ npm run typecheck
35
37
  npm run agentkit -- inspect
36
38
  npm run chat -- --message "hello"
37
39
  npm run eval
40
+ npm run agentkit -- spec check
38
41
  ```
39
42
 
40
43
  If a tool was added:
@@ -46,4 +49,3 @@ npm run agentkit -- tool <tool_name> --input '<json>'
46
49
  ## Final Response
47
50
 
48
51
  Summarize the files changed, assumptions made, verification results, and whether the behavior was tested with `test/fake` or a real provider selected by the owner.
49
-
@@ -9,12 +9,14 @@ Use this when a capsule owns durable application records.
9
9
 
10
10
  ## Rules
11
11
 
12
- - Put agent-owned tables and indexes in `schema.sql`.
12
+ - Put the first idempotent bootstrap schema in `schema.sql`.
13
+ - For production-shaped changes, prefer ordered `migrations/*.sql` files such as `migrations/0001_initial.sql`.
13
14
  - Keep `schema.sql` idempotent with `CREATE TABLE IF NOT EXISTS`, `CREATE INDEX IF NOT EXISTS`, and safe additive changes.
14
15
  - Keep deploy-ready capsules on `storage.driver: "agentkit"`.
15
16
  - Use `ctx.db` inside tools. `ctx.database` and `ctx.storage.sql` are aliases.
16
17
  - Do not import SQLite, Turso, or other database drivers from tools.
17
18
  - Do not edit `.agentkit/agentkit.db` by hand.
19
+ - For external catalogs, run `npm run agentkit -- sync init`, then implement `sync.ts` and keep local fixtures in `seed.sql`.
18
20
 
19
21
  ## Templates
20
22
 
@@ -26,6 +28,8 @@ Use this when a capsule owns durable application records.
26
28
  ```sh
27
29
  npm run agentkit -- db migrate
28
30
  npm run agentkit -- db seed --file seed.sql
31
+ npm run agentkit -- sync init
32
+ npm run agentkit -- sync run
29
33
  npm run agentkit -- db shell
30
34
  npm run agentkit -- db reset --yes
31
35
  ```
@@ -39,4 +43,3 @@ npm run agentkit -- tool <tool_name> --input '<json>'
39
43
  ```
40
44
 
41
45
  Hosted deploy applies AgentKit-managed storage internally. The user should not create hosted databases or buckets by hand.
42
-
@@ -11,14 +11,43 @@ Use evals after chat works and before claiming behavior is stable.
11
11
 
12
12
  1. Create or edit `evals/<name>.eval.ts`.
13
13
  2. Keep assertions small and deterministic.
14
- 3. Use `persisted_tool_call` for tool behavior stored in local SQLite.
15
- 4. Do not put secrets or real client PII in evals.
16
- 5. For tools that write externally, delete, charge money, send email, or call real customer systems, branch on `ctx.runtime.environment === "eval"` inside the registered tool.
14
+ 3. Use `turns` for full conversation flows, such as user asks, agent calls a tool, then the answer follows the required format.
15
+ 4. Use `persisted_tool_call` for tool behavior stored in local SQLite.
16
+ 5. Convert real failures into regression tests with `npm run agentkit -- eval from-conversation <conversation-id>`.
17
+ 6. Do not put secrets or real client PII in evals.
18
+ 7. For tools that write externally, delete, charge money, send email, or call real customer systems, branch on `ctx.runtime.environment === "eval"` inside the registered tool.
19
+
20
+ ## Multi-turn Example
21
+
22
+ ```ts
23
+ export default {
24
+ name: "buyer under budget",
25
+ turns: [
26
+ {
27
+ input: "I want a house up to 600k near Pinheiros.",
28
+ expect: {
29
+ persisted_tool_call: {
30
+ name: "buscar_imoveis",
31
+ status: "completed",
32
+ input: { maxPrice: 600000 },
33
+ },
34
+ },
35
+ },
36
+ {
37
+ input: "Show me the best two.",
38
+ expect: {
39
+ contains: ["R$", "Pinheiros"],
40
+ },
41
+ },
42
+ ],
43
+ };
44
+ ```
17
45
 
18
46
  ## Templates
19
47
 
20
48
  - `templates/smoke.eval.md`
21
49
  - `templates/tool-call.eval.md`
50
+ - `templates/multi-turn.eval.md`
22
51
  - `templates/no-leak.eval.md`
23
52
 
24
53
  ## Verification
@@ -0,0 +1,22 @@
1
+ ```ts
2
+ export default {
3
+ name: "main conversation flow",
4
+ turns: [
5
+ {
6
+ input: "I need help finding an option under my budget.",
7
+ expect: {
8
+ contains: "budget",
9
+ },
10
+ },
11
+ {
12
+ input: "Show me the best match.",
13
+ expect: {
14
+ persisted_tool_call: {
15
+ name: "replace_with_tool_name",
16
+ status: "completed",
17
+ },
18
+ },
19
+ },
20
+ ],
21
+ };
22
+ ```
@@ -191,10 +191,12 @@ When the owner opens this folder in Codex, Claude Code, or another coding agent
191
191
  Start building immediately:
192
192
 
193
193
  - Start with \`skills/agentkit-capsule/SKILL.md\`, then use \`npm run agentkit -- docs llms\` as the docs router.
194
+ - Create or update \`AGENT_SPEC.md\` from the owner's request with \`npm run agentkit -- spec init --brief "<owner request>"\`; the owner should not fill this file by hand before work starts.
194
195
  - Infer the first useful version from the owner's request.
195
196
  - Edit \`prompts/instructions.md\` for support behavior.
196
197
  - Edit \`agentkit.config.ts\` for provider, tools, secrets, access, and storage.
197
198
  - Add or replace TypeScript tools under \`tools/\` when the requested support agent needs actions or external data.
199
+ - Add \`sync.ts\`, \`seed.sql\`, and ordered \`migrations/*.sql\` when the support agent depends on external catalogs or production-shaped data changes.
198
200
  - Do not wait for a wizard or recipe. AgentKit provides the scaffold and contract; you decide the implementation from the owner's brief.
199
201
  - Ask follow-up questions only when missing information blocks a safe local implementation.
200
202
  - State assumptions in the final response.
@@ -205,6 +207,9 @@ Start building immediately:
205
207
  - \`npm run chat -- --message "hello"\`: send one local chat message.
206
208
  - \`npm run agentkit -- tool lookup_order --input '{"orderId":"A100"}'\`: test the example tool directly.
207
209
  - \`npm run eval\`: run agent evals.
210
+ - \`npm run agentkit -- spec check\`: verify the local agent implementation contract exists.
211
+ - \`npm run agentkit -- eval from-conversation <conversation-id>\`: turn a real conversation into a regression eval.
212
+ - \`npm run agentkit -- conversations trace <conversation-id>\`: inspect messages, runs, tool calls, inputs, outputs, rendered output, and final responses.
208
213
  - \`npm run dev\`: run the local Agent Capsule runtime.
209
214
  - \`npm run agentkit -- inspect\`: print machine-readable capsule state.
210
215
  - \`printf %s "$VALUE" | npm run agentkit -- env set <NAME> --stdin\`: write a local secret value to ignored \`.env\` without putting it in shell history.
@@ -259,8 +264,10 @@ Example owner request:
259
264
  Turn the request into a working local capsule:
260
265
 
261
266
  - Update \`prompts/instructions.md\` with domain-specific behavior, boundaries, intake questions, and escalation rules.
267
+ - Create or update \`AGENT_SPEC.md\` with \`npm run agentkit -- spec init --brief "<owner request>"\`. The owner gives the general idea; the coding agent turns it into the structured contract.
262
268
  - Update \`agentkit.config.ts\` when tools, secrets, provider, or access rules change.
263
269
  - Add, replace, or remove TypeScript tools under \`tools/\` for real actions or external data.
270
+ - Use \`npm run agentkit -- sync init\` when the agent needs catalog sync, fixture seed data, or ordered migrations.
264
271
  - Keep the first version runnable with \`test/fake\` unless the owner explicitly asks for a real provider.
265
272
  - Do not use a wizard or recipe. Build the capsule directly from the scaffold, the AgentKit contract, and the owner's brief.
266
273
  - Make practical assumptions and list them in your final response.