@andreprado/agentkit 0.1.0-alpha.14 → 0.1.0-alpha.15

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (42) hide show
  1. package/README.md +2 -0
  2. package/docs/guides/run-evals.md +73 -25
  3. package/docs/llms-full.txt +24 -9
  4. package/docs/llms.txt +2 -0
  5. package/package.json +1 -1
  6. package/src/cli/cloud-client.ts +30 -10
  7. package/src/cli/deploy-readiness.ts +32 -11
  8. package/src/cli/index.ts +20 -6
  9. package/src/cloud/client.ts +4 -3
  10. package/src/cloud/contracts.ts +1 -1
  11. package/src/create-project.ts +1 -1
  12. package/src/index.ts +24 -1
  13. package/src/providers/pi.ts +14 -1
  14. package/src/providers/test.ts +36 -0
  15. package/src/runtime/chat.ts +59 -42
  16. package/src/runtime/config.ts +11 -0
  17. package/src/runtime/core/manifest.ts +3 -3
  18. package/src/runtime/deploy-readiness.ts +3 -3
  19. package/src/runtime/dev-server.ts +171 -13
  20. package/src/runtime/env.ts +8 -3
  21. package/src/runtime/evals.ts +404 -69
  22. package/src/runtime/inspect.ts +12 -0
  23. package/src/runtime/prompt-context.ts +141 -0
  24. package/src/runtime/runtime-contract.ts +17 -7
  25. package/src/runtime/targets/cloudflare/build.ts +23 -3
  26. package/src/runtime/targets/container/server.ts +1 -1
  27. package/src/runtime/targets/vps/deploy.ts +25 -8
  28. package/src/runtime/tool-runner.ts +7 -0
  29. package/src/runtime/tools.ts +8 -2
  30. package/src/templates/blank.ts +8 -3
  31. package/src/templates/dentista.ts +18 -10
  32. package/src/templates/skills/agentkit-build-agent/SKILL.md +6 -5
  33. package/src/templates/skills/agentkit-build-agent/templates/appointment-intake.instructions.md +2 -1
  34. package/src/templates/skills/agentkit-capsule/SKILL.md +1 -1
  35. package/src/templates/skills/agentkit-evals/SKILL.md +53 -13
  36. package/src/templates/skills/agentkit-evals/templates/multi-turn.eval.md +13 -6
  37. package/src/templates/skills/agentkit-evals/templates/no-leak.eval.md +8 -4
  38. package/src/templates/skills/agentkit-evals/templates/smoke.eval.md +8 -4
  39. package/src/templates/skills/agentkit-evals/templates/tool-call.eval.md +16 -7
  40. package/src/templates/skills/agentkit-prompts/SKILL.md +3 -1
  41. package/src/templates/skills/agentkit-tools/SKILL.md +2 -1
  42. package/src/templates/support.ts +8 -3
package/README.md CHANGED
@@ -48,6 +48,8 @@ agentkit conversations trace <conversation-id>
48
48
 
49
49
  Knowledge indexes local `.md`, `.txt`, and `.csv` sources into the capsule database and registers the internal `agentkit_search_knowledge` chat tool when `knowledge` is configured in `agentkit.config.ts`. `agentkit dev` and `agentkit chat` sync configured sources automatically; when embeddings are configured, local semantic search uses a libSQL vector sidecar. Hosted Cloudflare deploys package configured local Knowledge sources and sync them into the project Turso database automatically during `agentkit deploy`; when embeddings are configured, the deploy also creates and populates the hosted Turso vector index.
50
50
 
51
+ Every chat run receives dynamic runtime date context: current ISO timestamp, local date, weekday, local date/time, and timezone. Set `timeZone` in `agentkit.config.ts` for scheduling agents so relative dates like "today" and "next Friday" resolve in the right business/user timezone.
52
+
51
53
  ## Deploy Later
52
54
 
53
55
  ```sh
@@ -52,72 +52,120 @@ Do not edit `.agentkit/agentkit.db` by hand.
52
52
  Generated smoke eval:
53
53
 
54
54
  ```ts
55
- export default {
55
+ import { defineEval } from "@andreprado/agentkit";
56
+
57
+ export default defineEval({
56
58
  name: "smoke",
57
59
  input: "Say hello in one short sentence.",
58
60
  expect: {
59
- contains: "hello",
61
+ response: {
62
+ caseInsensitiveContains: "hello",
63
+ maxLength: 160,
64
+ },
60
65
  },
61
- };
66
+ });
62
67
  ```
63
68
 
64
69
  Supported assertion types:
65
70
 
66
71
  ```txt
67
- contains
68
- not_contains
69
- regex
70
- matches_regex
71
- persisted_tool_call
72
+ response.contains
73
+ response.containsAll
74
+ response.containsAny
75
+ response.caseInsensitiveContains
76
+ response.notContains
77
+ response.regex
78
+ response.matchesRegex
79
+ response.notRegex
80
+ response.maxLength
81
+ tools.called
82
+ tools.calledOnce
83
+ tools.count
84
+ tools.order
85
+ tools.persisted
86
+ ```
87
+
88
+ Older flat aliases still work, including `contains`, `not_contains`, `regex`, `matches_regex`, and `persisted_tool_call`.
89
+
90
+ For date-sensitive evals, set top-level `now` to an ISO timestamp with an explicit timezone designator such as `Z` or `-05:00`. AgentKit uses that fixed clock for every turn and tool call in the eval:
91
+
92
+ ```ts
93
+ export default defineEval({
94
+ name: "appointment relative date",
95
+ now: "2026-02-04T02:30:00.000Z",
96
+ input: "What is today's date?",
97
+ expect: {
98
+ response: {
99
+ containsAll: ["2026-02-03", "Tuesday"],
100
+ },
101
+ },
102
+ });
72
103
  ```
73
104
 
74
105
  Multi-turn conversation evals use `turns`:
75
106
 
76
107
  ```ts
77
- export default {
108
+ import { defineEval } from "@andreprado/agentkit";
109
+
110
+ export default defineEval({
78
111
  name: "buyer under budget",
79
112
  turns: [
80
113
  {
81
114
  input: "I want a house up to 600k near Pinheiros.",
82
115
  expect: {
83
- persisted_tool_call: {
84
- name: "buscar_imoveis",
85
- status: "completed",
86
- input: { maxPrice: 600000 },
116
+ tools: {
117
+ calledOnce: "buscar_imoveis",
118
+ persisted: {
119
+ name: "buscar_imoveis",
120
+ status: "completed",
121
+ input: { maxPrice: 600000 },
122
+ },
87
123
  },
88
124
  },
89
125
  },
90
126
  {
91
127
  input: "Show me the best two.",
92
128
  expect: {
93
- contains: ["Pinheiros", "R$"],
129
+ response: {
130
+ containsAll: ["Pinheiros", "R$"],
131
+ },
94
132
  },
95
133
  },
96
134
  ],
97
- };
135
+ });
98
136
  ```
99
137
 
100
- `persisted_tool_call` validates the tool call saved in local SQLite `tool_calls`, not a provider-specific raw response shape. It can be a tool name string or an object with `name`, `input`, `output`, `rendered`, `status`, and/or `visibility`.
138
+ `tools.persisted` validates the tool call saved in local SQLite `tool_calls`, not a provider-specific raw response shape. It can be a tool name string or an object with `name`, `input`, `output`, `rendered`, `status`, and/or `visibility`.
139
+
140
+ Use `tools.count` for the exact number of persisted calls in that turn, `tools.calledOnce` for exactly one call by name, and `tools.order` for required relative order. `tools.order` allows extra calls before, between, or after the named calls; pair it with `tools.count` when the exact call set matters.
101
141
 
102
142
  Use response assertions and persisted tool assertions together when internal operational output must not leak:
103
143
 
104
144
  ```ts
105
- export default {
145
+ import { defineEval } from "@andreprado/agentkit";
146
+
147
+ export default defineEval({
106
148
  name: "triage lead",
107
149
  input: '{"tool":"triage_real_estate_lead","input":{"email":"ada@example.com"}}',
108
150
  expect: {
109
- not_contains: ["hot", "score"],
110
- persisted_tool_call: {
111
- name: "triage_real_estate_lead",
112
- status: "completed",
113
- visibility: "internal",
114
- output: { status: "hot" },
151
+ response: {
152
+ notContains: ["hot", "score"],
153
+ notRegex: ["API_KEY|secret|token"],
154
+ },
155
+ tools: {
156
+ calledOnce: "triage_real_estate_lead",
157
+ persisted: {
158
+ name: "triage_real_estate_lead",
159
+ status: "completed",
160
+ visibility: "internal",
161
+ output: { status: "hot" },
162
+ },
115
163
  },
116
164
  },
117
- };
165
+ });
118
166
  ```
119
167
 
120
- `tool_call` remains accepted as a backwards-compatible alias, but new evals should use `persisted_tool_call`.
168
+ `tool_call` and `persisted_tool_call` remain accepted as backwards-compatible aliases, but new evals should use `tools.persisted`.
121
169
 
122
170
  Safe external-tool pattern:
123
171
 
@@ -96,7 +96,7 @@ Prefer `env set --stdin` or `--from-env` for local secret values, and prefer `se
96
96
  Planned commands described by the contract but not implemented yet:
97
97
 
98
98
  ```sh
99
- agentkit eval create-from-conversation <conversation-id>
99
+ agentkit eval from-conversation <conversation-id>
100
100
  ```
101
101
 
102
102
  ## Create And Test A Capsule
@@ -166,6 +166,7 @@ export default defineAgent({
166
166
  name: "test",
167
167
  model: "fake",
168
168
  },
169
+ timeZone: "America/New_York",
169
170
  instructions: "./prompts/instructions.md",
170
171
  secrets: [],
171
172
  tools: [],
@@ -178,6 +179,8 @@ export default defineAgent({
178
179
  });
179
180
  ```
180
181
 
182
+ `timeZone` is optional and must be an IANA time zone when set. AgentKit injects dynamic runtime context into every chat run: current ISO timestamp, local date, local weekday, local date/time, and timezone. Use `timeZone` for scheduling, appointments, reminders, deadlines, and any prompt behavior that interprets "today", "tomorrow", weekdays, or relative dates. Do not hardcode today's date in `prompts/instructions.md`. If `timeZone` is omitted, AgentKit falls back to `AGENTKIT_TIME_ZONE`, then valid `TZ`, then the runtime default timezone.
183
+
181
184
  Valid runtime values:
182
185
 
183
186
  ```txt
@@ -473,6 +476,7 @@ Tool runtime rules:
473
476
  - tools that need SQL use canonical `ctx.db`; `ctx.database` and `ctx.storage.sql` are supported aliases;
474
477
  - tools can use `ctx.db.batch([...])` for atomic writes; local tools can also use `ctx.db.transaction(async (tx) => ...)`;
475
478
  - tools can inspect `ctx.runtime` with `{ environment, invocation, target, database }`;
479
+ - tools can use `ctx.clock` for the same runtime clock injected into the agent prompt, including `now`, `isoTimestamp`, `timeZone`, `localDate`, `localWeekday`, and `localDateTime`;
476
480
  - tools must not import local database drivers or Node-only APIs. Use AgentKit runtime services instead.
477
481
 
478
482
  ## Database Tools And Dual Storage
@@ -717,14 +721,25 @@ npm run eval
717
721
  Supported assertion types:
718
722
 
719
723
  ```txt
720
- contains
721
- not_contains
722
- regex
723
- matches_regex
724
- persisted_tool_call
725
- ```
726
-
727
- `persisted_tool_call` validates the tool call saved in local SQLite `tool_calls`, not a provider-specific raw response shape. It can be a tool name string or an object with `name`, `input`, `output`, `status`, and/or `visibility`. `tool_call` remains accepted as a backwards-compatible alias.
724
+ response.contains
725
+ response.containsAll
726
+ response.containsAny
727
+ response.caseInsensitiveContains
728
+ response.notContains
729
+ response.regex
730
+ response.matchesRegex
731
+ response.notRegex
732
+ response.maxLength
733
+ tools.called
734
+ tools.calledOnce
735
+ tools.count
736
+ tools.order
737
+ tools.persisted
738
+ ```
739
+
740
+ Import `defineEval` from `@andreprado/agentkit` when writing new evals. `tools.persisted` validates the tool call saved in local SQLite `tool_calls`, not a provider-specific raw response shape. It can be a tool name string or an object with `name`, `input`, `output`, `rendered`, `status`, and/or `visibility`. `tool_call` and `persisted_tool_call` remain accepted as backwards-compatible aliases, but new evals should use `tools.persisted`.
741
+
742
+ For date-sensitive evals, set top-level `now` to an ISO timestamp with an explicit timezone designator such as `Z` or `-05:00`. AgentKit uses that fixed clock for every turn and tool call in the eval so "today", "tomorrow", and weekdays remain deterministic while normal chat continues to use the real current date.
728
743
 
729
744
  Evals run the normal capsule tools. If a tool would write externally, delete, charge money, send email, or call a real customer system, make its `execute` implementation branch on `ctx.runtime.environment === "eval"` and return deterministic non-destructive output for eval runs. Do not invent an eval-only mock API; keep the behavior inside the registered tool contract unless AgentKit adds a first-class mock facility later.
730
745
 
package/docs/llms.txt CHANGED
@@ -71,6 +71,8 @@ UI testing is part of the handoff. For local UI testing, run `agentkit dev`, ope
71
71
 
72
72
  `test/fake` is deterministic and validates scaffold, direct tool calls, and fake-provider evals. It does not validate natural conversation quality. Before claiming real conversation behavior has been tested, ask the owner which provider to use: OpenRouter, OpenAI, Anthropic, or another supported provider. Do not choose for them.
73
73
 
74
+ AgentKit injects the current ISO timestamp, local date, weekday, local date/time, and timezone dynamically into every chat run. Set `timeZone` in `agentkit.config.ts` for scheduling agents so "today", "tomorrow", and weekdays resolve in the business/user timezone; otherwise AgentKit falls back to `AGENTKIT_TIME_ZONE`, valid `TZ`, then the runtime default. Do not hardcode today's date in prompts.
75
+
74
76
  Current local endpoints from `agentkit dev`:
75
77
 
76
78
  ```txt
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@andreprado/agentkit",
3
- "version": "0.1.0-alpha.14",
3
+ "version": "0.1.0-alpha.15",
4
4
  "private": false,
5
5
  "type": "module",
6
6
  "repository": {
@@ -39,12 +39,24 @@ export type CloudLimitsResponse = {
39
39
  };
40
40
  };
41
41
 
42
+ export type CloudProjectResolveResponse = {
43
+ project?: {
44
+ id?: string;
45
+ name?: string;
46
+ owner_state?: "anonymous" | "claimed";
47
+ };
48
+ };
49
+
42
50
  export type DeployAccessTokenCreateResponse = {
43
51
  access_token?: {
44
52
  id?: string;
45
53
  deploy_id?: string;
54
+ account_id?: string;
46
55
  name?: string;
47
56
  token?: string;
57
+ created_at?: string;
58
+ expires_at?: string;
59
+ last_used_at?: string;
48
60
  };
49
61
  };
50
62
 
@@ -52,7 +64,11 @@ export type DeployAccessTokenListResponse = {
52
64
  access_tokens?: Array<{
53
65
  id?: string;
54
66
  deploy_id?: string;
67
+ account_id?: string;
55
68
  name?: string;
69
+ created_at?: string;
70
+ expires_at?: string;
71
+ last_used_at?: string;
56
72
  }>;
57
73
  };
58
74
 
@@ -94,6 +110,16 @@ export async function readCloudAuth(): Promise<CloudAuthConfig | null> {
94
110
  return null;
95
111
  }
96
112
 
113
+ export async function readCloudAuthForApiUrl(apiUrl: string): Promise<CloudAuthConfig | null> {
114
+ const auth = await readCloudAuth();
115
+
116
+ if (auth?.source === "file" && !cloudAuthMatchesApiUrl(auth, apiUrl)) {
117
+ return null;
118
+ }
119
+
120
+ return auth;
121
+ }
122
+
97
123
  export async function writeCloudAuth(config: CloudAuthConfig): Promise<void> {
98
124
  await mkdir(join(homedir(), ".agentkit"), { recursive: true });
99
125
  await writeFile(cloudAuthPath(), `${JSON.stringify(config, null, 2)}\n`, { mode: 0o600 });
@@ -156,11 +182,8 @@ export async function cloudPost<T>(apiUrl: string, path: string, body: unknown):
156
182
 
157
183
  export async function cloudFetch(apiUrl: string, path: string, init: RequestInit = {}): Promise<Response> {
158
184
  const headers = new Headers(init.headers);
159
- const auth = await readCloudAuth();
160
- const apiToken =
161
- auth?.source === "file" && !cloudAuthMatchesApiUrl(auth, apiUrl)
162
- ? undefined
163
- : auth?.token;
185
+ const auth = await readCloudAuthForApiUrl(apiUrl);
186
+ const apiToken = auth?.token;
164
187
 
165
188
  if (apiToken && !headers.has("Authorization")) {
166
189
  headers.set("Authorization", `Bearer ${apiToken}`);
@@ -177,12 +200,9 @@ function cloudAuthMatchesApiUrl(auth: CloudAuthConfig, apiUrl: string): boolean
177
200
  }
178
201
 
179
202
  export async function cloudApiRequest(apiUrl: string, path: string, init: RequestInit): Promise<unknown> {
180
- const auth = await readCloudAuth();
203
+ const auth = await readCloudAuthForApiUrl(apiUrl);
181
204
  const headers = new Headers(init.headers);
182
- const apiToken =
183
- auth?.source === "file" && !cloudAuthMatchesApiUrl(auth, apiUrl)
184
- ? undefined
185
- : auth?.token;
205
+ const apiToken = auth?.token;
186
206
 
187
207
  if (!headers.has("Content-Type") && init.body !== undefined) {
188
208
  headers.set("Content-Type", "application/json");
@@ -1,15 +1,15 @@
1
1
  import { findAgentCapsuleRoot, loadAgentCapsule } from "../runtime/config";
2
- import { readLocalDeployStateIfExists } from "../runtime/core/deploy-state";
3
- import { loadDeployReadinessContext, stableCloudProjectId, type DeployReadinessContext } from "../runtime/deploy-readiness";
2
+ import { loadDeployReadinessContext, type DeployReadinessContext } from "../runtime/deploy-readiness";
4
3
  import { loadCapsuleEnv } from "../runtime/env";
5
4
  import { isAgentKitError } from "../runtime/errors";
6
5
  import {
7
6
  cloudApiRequest,
8
7
  formatCloudAuthSource,
9
- readCloudAuth,
8
+ readCloudAuthForApiUrl,
10
9
  type CloudCapabilitiesResponse,
11
10
  type CloudLimitsResponse,
12
11
  type CloudMeResponse,
12
+ type CloudProjectResolveResponse,
13
13
  type CloudSecretsResponse,
14
14
  } from "./cloud-client";
15
15
 
@@ -34,7 +34,7 @@ export async function checkDeployReadiness(options: {
34
34
  requireLogin: boolean;
35
35
  }): Promise<DeployReadinessReport> {
36
36
  const context = await loadDeployReadinessContext(process.cwd(), process.env);
37
- const auth = await readCloudAuth();
37
+ const auth = options.requireLogin ? await readCloudAuthForApiUrl(options.apiUrl) : null;
38
38
  const checks: DeployReadinessCheck[] = [];
39
39
  let cloudAuthUsable = false;
40
40
 
@@ -82,6 +82,19 @@ export async function checkDeployReadiness(options: {
82
82
  }
83
83
  }
84
84
 
85
+ if (cloudAuthUsable) {
86
+ try {
87
+ context.projectId = await resolveCloudProjectId(options.apiUrl, context.projectName);
88
+ } catch (error) {
89
+ checks.push({
90
+ status: "fail",
91
+ title: "Project identity",
92
+ message: `Could not resolve the account-scoped AgentKit Cloud project. ${formatReadinessError(error)}`,
93
+ });
94
+ cloudAuthUsable = false;
95
+ }
96
+ }
97
+
85
98
  if (cloudAuthUsable) {
86
99
  try {
87
100
  const payload = (await cloudApiRequest(options.apiUrl, "/v1/me/capabilities", {
@@ -315,16 +328,24 @@ export function formatReadinessError(error: unknown): string {
315
328
  return error instanceof Error ? error.message : String(error);
316
329
  }
317
330
 
318
- export async function resolveProjectIdForSecretCommand(cwd: string): Promise<string> {
319
- const state = await readLocalDeployStateIfExists(cwd);
331
+ export async function resolveProjectIdForSecretCommand(cwd: string, apiUrl: string): Promise<string> {
332
+ const root = await findAgentCapsuleRoot(cwd);
333
+ const capsule = await loadAgentCapsule(root);
334
+ return resolveCloudProjectId(apiUrl, capsule.config.name);
335
+ }
320
336
 
321
- if (state?.project_id) {
322
- return state.project_id;
337
+ async function resolveCloudProjectId(apiUrl: string, projectName: string): Promise<string> {
338
+ const payload = (await cloudApiRequest(apiUrl, "/v1/projects/resolve", {
339
+ method: "POST",
340
+ body: JSON.stringify({ project_name: projectName }),
341
+ })) as CloudProjectResolveResponse;
342
+ const projectId = payload.project?.id;
343
+
344
+ if (!projectId) {
345
+ throw new Error("AgentKit Cloud did not return a project id.");
323
346
  }
324
347
 
325
- const root = await findAgentCapsuleRoot(cwd);
326
- const capsule = await loadAgentCapsule(root);
327
- return stableCloudProjectId(capsule.config.name);
348
+ return projectId;
328
349
  }
329
350
 
330
351
  function secretSetCommand(secret: string, source: DeployReadinessContext["localSecrets"][string] | undefined): string {
package/src/cli/index.ts CHANGED
@@ -1,6 +1,6 @@
1
1
  import { spawn } from "node:child_process";
2
2
  import { existsSync } from "node:fs";
3
- import { mkdir, readFile, writeFile } from "node:fs/promises";
3
+ import { chmod, mkdir, readFile, writeFile } from "node:fs/promises";
4
4
  import { dirname, join, relative, resolve } from "node:path";
5
5
 
6
6
  import { createProject } from "../create-project";
@@ -38,6 +38,7 @@ import {
38
38
  clearCloudAuth,
39
39
  cloudApiRequest,
40
40
  readCloudAuth,
41
+ readCloudAuthForApiUrl,
41
42
  readCloudError,
42
43
  resolveCloudApiUrl,
43
44
  verifyCloudLogin,
@@ -607,10 +608,10 @@ async function main() {
607
608
  }
608
609
 
609
610
  const apiUrl = await resolveCloudApiUrl(args.flags.api);
610
- const auth = await readCloudAuth();
611
+ const auth = anonymous ? null : await readCloudAuthForApiUrl(apiUrl);
611
612
  const deploy = await deployBuildToAgentKitCloud(result, {
612
613
  apiUrl,
613
- apiToken: auth?.token,
614
+ apiToken: anonymous ? null : auth?.token,
614
615
  anonymous: anonymous || !auth?.token,
615
616
  });
616
617
  const statePath = await writeLocalDeployState(result.root, deploy);
@@ -659,8 +660,8 @@ async function main() {
659
660
 
660
661
  if (args.command === "secret") {
661
662
  const [subcommand, name, ...valueParts] = args.positional;
662
- const projectId = await resolveProjectIdForSecretCommand(process.cwd());
663
663
  const apiUrl = await resolveCloudApiUrl(args.flags.api);
664
+ const projectId = await resolveProjectIdForSecretCommand(process.cwd(), apiUrl);
664
665
 
665
666
  if (subcommand === "list") {
666
667
  const payload = await cloudApiRequest(apiUrl, `/v1/projects/${encodeURIComponent(projectId)}/secrets`, {
@@ -949,9 +950,12 @@ async function ensureDeployChatAccessToken(options: {
949
950
  `/v1/deploys/${encodeURIComponent(options.deployId)}/access-tokens`,
950
951
  { method: "GET" },
951
952
  )) as DeployAccessTokenListResponse;
952
- const stillExists = (listed.access_tokens ?? []).some((token) => token.id === existing.id);
953
+ const stillUsable = (listed.access_tokens ?? []).some(
954
+ (token) => token.id === existing.id && deployAccessTokenMetadataIsUsable(token),
955
+ );
953
956
 
954
- if (stillExists) {
957
+ if (stillUsable) {
958
+ await chmod(path, 0o600);
955
959
  return { status: "ready", path, tokenId: existing.id, created: false };
956
960
  }
957
961
  }
@@ -977,6 +981,15 @@ async function ensureDeployChatAccessToken(options: {
977
981
  }
978
982
  }
979
983
 
984
+ function deployAccessTokenMetadataIsUsable(token: { expires_at?: string }): boolean {
985
+ if (!token.expires_at) {
986
+ return true;
987
+ }
988
+
989
+ const expiresAtMs = Date.parse(token.expires_at);
990
+ return Number.isFinite(expiresAtMs) && expiresAtMs > Date.now();
991
+ }
992
+
980
993
  async function readDeployAccessTokenFile(path: string): Promise<{ id?: string; deploy_id?: string } | null> {
981
994
  try {
982
995
  const parsed = JSON.parse(await readFile(path, "utf8")) as DeployAccessTokenCreateResponse;
@@ -989,6 +1002,7 @@ async function readDeployAccessTokenFile(path: string): Promise<{ id?: string; d
989
1002
  async function writeSecretJson(path: string, value: unknown): Promise<void> {
990
1003
  await mkdir(dirname(path), { recursive: true });
991
1004
  await writeFile(path, `${JSON.stringify(value, null, 2)}\n`, { mode: 0o600 });
1005
+ await chmod(path, 0o600);
992
1006
  }
993
1007
 
994
1008
  async function requireLocalDeployState(): Promise<LocalDeployState> {
@@ -1,4 +1,4 @@
1
- import { mkdir, writeFile } from "node:fs/promises";
1
+ import { chmod, mkdir, writeFile } from "node:fs/promises";
2
2
  import { join } from "node:path";
3
3
 
4
4
  import type { AgentBuildResult } from "../runtime/targets/cloudflare/build";
@@ -15,7 +15,7 @@ export async function deployBuildToAgentKitCloud(
15
15
  const headers = new Headers({
16
16
  "Content-Type": "application/json",
17
17
  });
18
- const apiToken = options.apiToken ?? process.env.AGENTKIT_CLOUD_API_TOKEN;
18
+ const apiToken = options.anonymous ? undefined : options.apiToken ?? process.env.AGENTKIT_CLOUD_API_TOKEN;
19
19
 
20
20
  if (apiToken) {
21
21
  headers.set("Authorization", `Bearer ${apiToken}`);
@@ -70,7 +70,8 @@ export async function writeLocalDeployState(root: string, deploy: CreateDeployRe
70
70
  };
71
71
 
72
72
  await mkdir(join(root, ".agentkit"), { recursive: true });
73
- await writeFile(statePath, `${JSON.stringify(state, null, 2)}\n`);
73
+ await writeFile(statePath, `${JSON.stringify(state, null, 2)}\n`, { mode: 0o600 });
74
+ await chmod(statePath, 0o600);
74
75
  return statePath;
75
76
  }
76
77
 
@@ -58,6 +58,6 @@ export type CreateDeployResponse = {
58
58
 
59
59
  export type AgentKitCloudDeployOptions = {
60
60
  apiUrl: string;
61
- apiToken?: string;
61
+ apiToken?: string | null;
62
62
  anonymous?: boolean;
63
63
  };
@@ -114,7 +114,7 @@ async function resolveAgentKitDependency(): Promise<string> {
114
114
 
115
115
  const packageRoot = resolve(moduleDir, "..");
116
116
 
117
- if (await directoryExists(resolve(packageRoot, "../..", ".git"))) {
117
+ if (await pathExists(resolve(packageRoot, "../..", ".git"))) {
118
118
  return `file:${packageRoot}`;
119
119
  }
120
120
 
package/src/index.ts CHANGED
@@ -210,6 +210,7 @@ export type ToolRenderResult = {
210
210
  export type ToolRenderContext<Input = unknown> = {
211
211
  input: Input;
212
212
  runtime: ToolRuntimeContext;
213
+ clock: ToolClockContext;
213
214
  };
214
215
 
215
216
  export type ToolRuntimeContext = {
@@ -219,10 +220,20 @@ export type ToolRuntimeContext = {
219
220
  database: "sqlite" | "turso" | "none";
220
221
  };
221
222
 
223
+ export type ToolClockContext = {
224
+ now: Date;
225
+ isoTimestamp: string;
226
+ timeZone: string;
227
+ localDate: string;
228
+ localWeekday: string;
229
+ localDateTime: string;
230
+ };
231
+
222
232
  export type ToolContext = {
223
233
  secrets: Record<string, string>;
224
234
  signal: AbortSignal;
225
235
  runtime: ToolRuntimeContext;
236
+ clock: ToolClockContext;
226
237
  database: DatabaseRunner;
227
238
  db: DatabaseRunner;
228
239
  storage: {
@@ -254,6 +265,7 @@ export type AgentConfig = {
254
265
  name: string;
255
266
  runtime: AgentRuntime;
256
267
  provider: AgentProvider;
268
+ timeZone?: string;
257
269
  instructions: string;
258
270
  secrets: string[];
259
271
  tools?: AgentTool[];
@@ -379,7 +391,18 @@ function validatePositiveInteger(value: number, field: string): void {
379
391
  export { loadAgentCapsule, findAgentCapsuleRoot } from "./runtime/config";
380
392
  export { runAgentMessage, runAgentMessageFromCwd } from "./runtime/chat";
381
393
  export { listConversationsFromCwd, getConversationFromCwd } from "./runtime/conversations";
382
- export { runEvalsFromCwd } from "./runtime/evals";
394
+ export { defineEval, runEvalsFromCwd } from "./runtime/evals";
395
+ export type {
396
+ EvalCase,
397
+ EvalExpect,
398
+ EvalResponseExpectation,
399
+ EvalRunSummary,
400
+ EvalResult,
401
+ EvalToolsContainerExpectation,
402
+ EvalToolsExpectation,
403
+ EvalTurn,
404
+ ToolExpectation,
405
+ } from "./runtime/evals";
383
406
  export { getConversationTraceFromCwd } from "./runtime/traces";
384
407
  export { runToolFromCwd } from "./runtime/tool-runner";
385
408
  export { startAgentDevServer } from "./runtime/dev-server";
@@ -97,7 +97,7 @@ async function runPiProvider(input: ProviderRunInput): Promise<ProviderRunResult
97
97
  content: [
98
98
  {
99
99
  type: "text",
100
- text: result.rendered?.text ?? JSON.stringify(result.output),
100
+ text: providerToolResultText(result),
101
101
  },
102
102
  ],
103
103
  isError: false,
@@ -109,6 +109,19 @@ async function runPiProvider(input: ProviderRunInput): Promise<ProviderRunResult
109
109
  throw new AgentKitError("runtime_error", `Pi provider exceeded ${MAX_TOOL_TURNS} tool turns.`);
110
110
  }
111
111
 
112
+ export function providerToolResultText(result: {
113
+ name: string;
114
+ visibility?: "user" | "internal";
115
+ rendered?: { text?: string };
116
+ output: unknown;
117
+ }): string {
118
+ if (result.visibility === "internal") {
119
+ return `Internal tool ${result.name} completed.`;
120
+ }
121
+
122
+ return result.rendered?.text ?? JSON.stringify(result.output);
123
+ }
124
+
112
125
  async function runPiTurn(
113
126
  model: Model<any>,
114
127
  context: Parameters<typeof stream>[1],
@@ -37,6 +37,17 @@ export const testProviderAdapter: ProviderAdapter = {
37
37
  };
38
38
  }
39
39
 
40
+ const currentDate = answerCurrentDate(input.instructions, userMessage.content);
41
+
42
+ if (currentDate) {
43
+ return {
44
+ message: {
45
+ role: "assistant",
46
+ content: currentDate,
47
+ },
48
+ };
49
+ }
50
+
40
51
  const rememberedName = answerRememberedName(input.messages);
41
52
 
42
53
  if (rememberedName) {
@@ -95,6 +106,31 @@ function answerRememberedName(messages: AgentMessage[]): string | null {
95
106
  return null;
96
107
  }
97
108
 
109
+ function answerCurrentDate(instructions: string, content: string): string | null {
110
+ if (!/(today|current date|what date|data de hoje|hoje)/i.test(content)) {
111
+ return null;
112
+ }
113
+
114
+ const localDate = readRuntimeContextValue(instructions, "Local date");
115
+
116
+ if (!localDate) {
117
+ return null;
118
+ }
119
+
120
+ const localWeekday = readRuntimeContextValue(instructions, "Local weekday");
121
+ const timeZone = readRuntimeContextValue(instructions, "Time zone");
122
+ const weekday = localWeekday ? ` (${localWeekday})` : "";
123
+ const zone = timeZone ? ` in ${timeZone}` : "";
124
+
125
+ return `Today is ${localDate}${weekday}${zone}.`;
126
+ }
127
+
128
+ function readRuntimeContextValue(instructions: string, label: string): string | null {
129
+ const escapedLabel = label.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
130
+ const match = instructions.match(new RegExp(`^- ${escapedLabel}: (.+)$`, "m"));
131
+ return match?.[1]?.trim() || null;
132
+ }
133
+
98
134
  function asksForName(content: string): boolean {
99
135
  return /what(?:'s| is) my name\??/i.test(content) || /(como|qual)\s+(é|e)\s*(o\s+)?meu nome\??/i.test(content);
100
136
  }