@bondedhq/shared 0.0.0-stage → 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +17 -2
- package/dist/abis.d.ts +8639 -0
- package/dist/abis.js +11234 -0
- package/dist/actuarial.d.ts +158 -0
- package/dist/actuarial.js +210 -0
- package/dist/agent-url.d.ts +172 -0
- package/dist/agent-url.js +248 -0
- package/dist/agent.d.ts +184 -0
- package/dist/agent.js +133 -0
- package/dist/allowances.d.ts +54 -0
- package/dist/allowances.js +68 -0
- package/dist/bounty-example.d.ts +6 -0
- package/dist/bounty-example.js +23 -0
- package/dist/bounty-spec.d.ts +36 -0
- package/dist/bounty-spec.js +111 -0
- package/dist/canonical.d.ts +8 -0
- package/dist/canonical.js +44 -0
- package/dist/chains.d.ts +58 -0
- package/dist/chains.js +123 -0
- package/dist/deployments.d.ts +32 -0
- package/dist/deployments.js +41 -0
- package/dist/index.d.ts +33 -0
- package/dist/index.js +33 -0
- package/dist/leaderboard.d.ts +105 -0
- package/dist/leaderboard.js +85 -0
- package/dist/llm.d.ts +121 -0
- package/dist/llm.js +105 -0
- package/dist/mandate-rules.d.ts +54 -0
- package/dist/mandate-rules.js +69 -0
- package/dist/module-install.d.ts +145 -0
- package/dist/module-install.js +133 -0
- package/dist/notifications.d.ts +48 -0
- package/dist/notifications.js +45 -0
- package/dist/observed-rates.d.ts +125 -0
- package/dist/observed-rates.js +158 -0
- package/dist/problems.d.ts +44 -0
- package/dist/problems.js +148 -0
- package/dist/quote.d.ts +123 -0
- package/dist/quote.js +167 -0
- package/dist/report-fixes.d.ts +89 -0
- package/dist/report-fixes.js +159 -0
- package/dist/runner.d.ts +376 -0
- package/dist/runner.js +353 -0
- package/dist/schemas/attack.d.ts +121 -0
- package/dist/schemas/attack.js +142 -0
- package/dist/schemas/attestation.d.ts +284 -0
- package/dist/schemas/attestation.js +175 -0
- package/dist/schemas/common.d.ts +22 -0
- package/dist/schemas/common.js +53 -0
- package/dist/schemas/mandate-commitment.d.ts +13 -0
- package/dist/schemas/mandate-commitment.js +37 -0
- package/dist/schemas/mandate.d.ts +170 -0
- package/dist/schemas/mandate.js +113 -0
- package/dist/self-serve.d.ts +133 -0
- package/dist/self-serve.js +110 -0
- package/dist/sentinel-cascade.d.ts +64 -0
- package/dist/sentinel-cascade.js +64 -0
- package/dist/sentinel.d.ts +133 -0
- package/dist/sentinel.js +101 -0
- package/dist/suggested-mandate.d.ts +81 -0
- package/dist/suggested-mandate.js +112 -0
- package/dist/tee.d.ts +61 -0
- package/dist/tee.js +93 -0
- package/dist/tiers.d.ts +19 -0
- package/dist/tiers.js +23 -0
- package/dist/troubleshooting-doc.d.ts +11 -0
- package/dist/troubleshooting-doc.js +46 -0
- package/package.json +59 -3
|
@@ -0,0 +1,248 @@
|
|
|
1
|
+
import { z } from "zod";
|
|
2
|
+
import { CancelMessage, codeHashOf, EpisodeMessage } from "./runner.js";
|
|
3
|
+
/**
|
|
4
|
+
* Agent URL mode: the second way to connect an agent to the Arena (the first is the runner).
|
|
5
|
+
* Instead of a runner launching a command, the Arena Gateway calls the agent's own HTTPS endpoint:
|
|
6
|
+
*
|
|
7
|
+
* - each episode: `POST <agent url>` with the runner's `episode` message as the JSON body. The
|
|
8
|
+
* agent answers 2xx within AGENT_URL_ANSWER_MS (it may keep working after it answers), then
|
|
9
|
+
* uses the episode's endpoints and posts to its done URL, exactly as a runner-launched agent;
|
|
10
|
+
* - cancel: `POST <agent url>` with `{ type: "cancel", episodeId }` (best effort);
|
|
11
|
+
* - verify, once the agent has its secret (and again whenever its owner asks): `POST <agent url>`
|
|
12
|
+
* with `{ type: "verify", challenge }`; the agent answers 200 with `{ challenge, proof,
|
|
13
|
+
* version? }`, where `proof` is HMAC-SHA256(secret, "verify." + challenge), so only a holder of
|
|
14
|
+
* the secret passes.
|
|
15
|
+
*
|
|
16
|
+
* Every call is signed with the agent secret (shown once when the URL is registered):
|
|
17
|
+
* `bonded-timestamp` is unix seconds, and `bonded-signature` is `v1=` and the hex HMAC-SHA256 of
|
|
18
|
+
* `<timestamp>.<raw body>`. The agent checks the signature and the timestamp before acting
|
|
19
|
+
* (`verifyAgentUrlRequest` does both, and with an AgentUrlReplayGuard also refuses a replay).
|
|
20
|
+
*
|
|
21
|
+
* HMAC runs on WebCrypto, so these helpers work in Node 22+, browsers and edge runtimes.
|
|
22
|
+
*/
|
|
23
|
+
/** Header with the unix time (seconds) a call was signed at. */
|
|
24
|
+
export const AGENT_URL_TIMESTAMP_HEADER = "bonded-timestamp";
|
|
25
|
+
/** Header with the signature: `v1=<hex>` (several, comma-separated, while a secret rotates). */
|
|
26
|
+
export const AGENT_URL_SIGNATURE_HEADER = "bonded-signature";
|
|
27
|
+
/** Most `v1=` values a signature header may carry (more than one only while a secret rotates). */
|
|
28
|
+
export const AGENT_URL_MAX_SIGNATURES = 4;
|
|
29
|
+
/** The signature scheme's version tag. */
|
|
30
|
+
export const AGENT_URL_SIGNATURE_VERSION = "v1";
|
|
31
|
+
/** Every agent secret starts with this, so one is recognisable in a config file or a log. */
|
|
32
|
+
export const AGENT_SECRET_PREFIX = "bonded_as_";
|
|
33
|
+
/** How old (or how far in the future) a signed call may be before the agent should refuse it. */
|
|
34
|
+
export const AGENT_URL_TOLERANCE_SECONDS = 300;
|
|
35
|
+
/** How long the agent has to answer any call. */
|
|
36
|
+
export const AGENT_URL_ANSWER_MS = 10_000;
|
|
37
|
+
/** The most of an agent's answer the gateway reads. */
|
|
38
|
+
export const AGENT_URL_MAX_RESPONSE_BYTES = 64 * 1024;
|
|
39
|
+
/** Largest body the gateway sends (an episode message is a few KB). */
|
|
40
|
+
export const AGENT_URL_MAX_BODY_BYTES = 256 * 1024;
|
|
41
|
+
/** Whether text is shaped like an agent secret (the prefix, then base64url). */
|
|
42
|
+
export function isWellFormedAgentSecret(secret) {
|
|
43
|
+
return /^bonded_as_[A-Za-z0-9_-]{16,128}$/.test(secret);
|
|
44
|
+
}
|
|
45
|
+
/** A fresh agent secret: the prefix and 32 random bytes, base64url. */
|
|
46
|
+
export function newAgentSecret() {
|
|
47
|
+
const bytes = new Uint8Array(32);
|
|
48
|
+
globalThis.crypto.getRandomValues(bytes);
|
|
49
|
+
return `${AGENT_SECRET_PREFIX}${base64url(bytes)}`;
|
|
50
|
+
}
|
|
51
|
+
/** The ownership check. `challenge` is random; the agent sends it back with its proof. */
|
|
52
|
+
export const AgentUrlVerifyMessage = z.object({
|
|
53
|
+
type: z.literal("verify"),
|
|
54
|
+
challenge: z.string().min(16).max(128),
|
|
55
|
+
});
|
|
56
|
+
/**
|
|
57
|
+
* The agent's answer to `verify`: the challenge back, and `proof`, the hex
|
|
58
|
+
* HMAC-SHA256(secret, "verify." + challenge) (`agentUrlVerifyProof`), so only a holder of the
|
|
59
|
+
* secret can pass (an echo server can't). `version` names the agent's code version (a git
|
|
60
|
+
* commit, a release tag, an image digest): with no declared code hash, the rating records
|
|
61
|
+
* `agentUrlCodeHash(version)`, labelled as reported by the agent. `answerAgentUrlVerify` builds
|
|
62
|
+
* the whole answer.
|
|
63
|
+
*/
|
|
64
|
+
export const AgentUrlVerifyAnswer = z.object({
|
|
65
|
+
challenge: z.string().max(128),
|
|
66
|
+
proof: z.string().regex(/^[0-9a-fA-F]{64}$/, "64 hex digits"),
|
|
67
|
+
version: z.string().min(1).max(200).optional(),
|
|
68
|
+
});
|
|
69
|
+
/** Every body the gateway POSTs to an agent URL. */
|
|
70
|
+
export const AgentUrlMessage = z.discriminatedUnion("type", [
|
|
71
|
+
EpisodeMessage,
|
|
72
|
+
CancelMessage,
|
|
73
|
+
AgentUrlVerifyMessage,
|
|
74
|
+
]);
|
|
75
|
+
/** The code hash for an Agent URL's reported version: keccak256 of `"url:<version>"`. */
|
|
76
|
+
export function agentUrlCodeHash(version) {
|
|
77
|
+
return codeHashOf("url", version);
|
|
78
|
+
}
|
|
79
|
+
const encoder = new TextEncoder();
|
|
80
|
+
function bytesOf(body) {
|
|
81
|
+
return typeof body === "string" ? encoder.encode(body) : body;
|
|
82
|
+
}
|
|
83
|
+
async function hmacKey(secret, usage) {
|
|
84
|
+
return globalThis.crypto.subtle.importKey("raw", encoder.encode(secret), { name: "HMAC", hash: "SHA-256" }, false, [usage]);
|
|
85
|
+
}
|
|
86
|
+
/** What is signed: `<timestamp>.` followed by the raw body bytes. */
|
|
87
|
+
function signedPayload(timestamp, body) {
|
|
88
|
+
const prefix = encoder.encode(`${timestamp}.`);
|
|
89
|
+
const raw = bytesOf(body);
|
|
90
|
+
const out = new Uint8Array(prefix.length + raw.length);
|
|
91
|
+
out.set(prefix, 0);
|
|
92
|
+
out.set(raw, prefix.length);
|
|
93
|
+
return out;
|
|
94
|
+
}
|
|
95
|
+
/** The `bonded-signature` value for a body signed at `timestamp`: `v1=<hex HMAC-SHA256>`. */
|
|
96
|
+
export async function agentUrlSignature(secret, timestamp, body) {
|
|
97
|
+
const key = await hmacKey(secret, "sign");
|
|
98
|
+
const mac = await globalThis.crypto.subtle.sign("HMAC", key, signedPayload(timestamp, body));
|
|
99
|
+
return `${AGENT_URL_SIGNATURE_VERSION}=${hex(new Uint8Array(mac))}`;
|
|
100
|
+
}
|
|
101
|
+
/** The ownership proof for a verify challenge: hex HMAC-SHA256(secret, "verify." + challenge). */
|
|
102
|
+
export async function agentUrlVerifyProof(secret, challenge) {
|
|
103
|
+
const key = await hmacKey(secret, "sign");
|
|
104
|
+
const mac = await globalThis.crypto.subtle.sign("HMAC", key, encoder.encode(`verify.${challenge}`));
|
|
105
|
+
return hex(new Uint8Array(mac));
|
|
106
|
+
}
|
|
107
|
+
/** The answer an agent sends to a verify message: `{ challenge, proof, version? }`. */
|
|
108
|
+
export async function answerAgentUrlVerify(secret, message, version) {
|
|
109
|
+
return {
|
|
110
|
+
challenge: message.challenge,
|
|
111
|
+
proof: await agentUrlVerifyProof(secret, message.challenge),
|
|
112
|
+
...(version ? { version } : {}),
|
|
113
|
+
};
|
|
114
|
+
}
|
|
115
|
+
/**
|
|
116
|
+
* Whether a verify answer proves the secret: the challenge sent and its proof (compared in
|
|
117
|
+
* constant time). The gateway's check.
|
|
118
|
+
*/
|
|
119
|
+
export async function checkAgentUrlVerifyProof(secret, challenge, answer) {
|
|
120
|
+
if (answer.challenge !== challenge || !/^[0-9a-fA-F]{64}$/.test(answer.proof))
|
|
121
|
+
return false;
|
|
122
|
+
const key = await hmacKey(secret, "verify");
|
|
123
|
+
return globalThis.crypto.subtle.verify("HMAC", key, unhex(answer.proof), encoder.encode(`verify.${challenge}`));
|
|
124
|
+
}
|
|
125
|
+
/** The two headers for a call to an agent URL, signed now (or at `nowMs`). */
|
|
126
|
+
export async function signAgentUrlRequest(secret, body, nowMs = Date.now()) {
|
|
127
|
+
const timestamp = Math.floor(nowMs / 1000);
|
|
128
|
+
return {
|
|
129
|
+
[AGENT_URL_TIMESTAMP_HEADER]: String(timestamp),
|
|
130
|
+
[AGENT_URL_SIGNATURE_HEADER]: await agentUrlSignature(secret, timestamp, body),
|
|
131
|
+
};
|
|
132
|
+
}
|
|
133
|
+
/**
|
|
134
|
+
* Remembers the signatures it has accepted until they are too old to be accepted anyway, so a
|
|
135
|
+
* captured call can't be replayed within the tolerance window. One per agent process.
|
|
136
|
+
*/
|
|
137
|
+
export class AgentUrlReplayGuard {
|
|
138
|
+
toleranceSeconds;
|
|
139
|
+
maxEntries;
|
|
140
|
+
seen = new Map();
|
|
141
|
+
constructor(toleranceSeconds = AGENT_URL_TOLERANCE_SECONDS, maxEntries = 10_000) {
|
|
142
|
+
this.toleranceSeconds = toleranceSeconds;
|
|
143
|
+
this.maxEntries = maxEntries;
|
|
144
|
+
}
|
|
145
|
+
/** Records the signature; false when it was already seen (a replay). */
|
|
146
|
+
accept(signature, timestamp, nowSeconds) {
|
|
147
|
+
for (const [sig, ts] of this.seen) {
|
|
148
|
+
if (ts >= nowSeconds - this.toleranceSeconds && this.seen.size < this.maxEntries)
|
|
149
|
+
break;
|
|
150
|
+
this.seen.delete(sig);
|
|
151
|
+
}
|
|
152
|
+
if (this.seen.has(signature))
|
|
153
|
+
return false;
|
|
154
|
+
this.seen.set(signature, timestamp);
|
|
155
|
+
return true;
|
|
156
|
+
}
|
|
157
|
+
}
|
|
158
|
+
function header(headers, name) {
|
|
159
|
+
if (typeof headers.get === "function")
|
|
160
|
+
return headers.get(name) ?? undefined;
|
|
161
|
+
const record = headers;
|
|
162
|
+
const value = record[name] ??
|
|
163
|
+
Object.entries(record).find(([key]) => key.toLowerCase() === name)?.[1] ??
|
|
164
|
+
undefined;
|
|
165
|
+
return Array.isArray(value) ? value.join(",") : value;
|
|
166
|
+
}
|
|
167
|
+
/**
|
|
168
|
+
* Checks a call from the Arena Gateway in the agent's handler: the timestamp is within the
|
|
169
|
+
* tolerance, a `v1` signature matches (compared in constant time by WebCrypto), it is not a
|
|
170
|
+
* replay, and the body is one of the gateway's messages. Answer `status` with the reason when it
|
|
171
|
+
* fails; act only on `message` when it passes.
|
|
172
|
+
*/
|
|
173
|
+
export async function verifyAgentUrlRequest(options) {
|
|
174
|
+
const refuse = (status, reason) => ({
|
|
175
|
+
ok: false,
|
|
176
|
+
status,
|
|
177
|
+
reason,
|
|
178
|
+
});
|
|
179
|
+
const tsText = header(options.headers, AGENT_URL_TIMESTAMP_HEADER);
|
|
180
|
+
const sigText = header(options.headers, AGENT_URL_SIGNATURE_HEADER);
|
|
181
|
+
if (!tsText || !sigText)
|
|
182
|
+
return refuse(401, "the call is not signed");
|
|
183
|
+
if (!/^\d{1,12}$/.test(tsText.trim()))
|
|
184
|
+
return refuse(401, "the timestamp is not unix seconds");
|
|
185
|
+
const timestamp = Number(tsText.trim());
|
|
186
|
+
const now = Math.floor((options.nowMs ?? Date.now()) / 1000);
|
|
187
|
+
const tolerance = options.toleranceSeconds ?? AGENT_URL_TOLERANCE_SECONDS;
|
|
188
|
+
if (Math.abs(now - timestamp) > tolerance)
|
|
189
|
+
return refuse(401, "the timestamp is too old or in the future");
|
|
190
|
+
const signatures = sigText
|
|
191
|
+
.split(",")
|
|
192
|
+
.map((part) => part.trim())
|
|
193
|
+
.filter((part) => part.startsWith(`${AGENT_URL_SIGNATURE_VERSION}=`))
|
|
194
|
+
.map((part) => part.slice(AGENT_URL_SIGNATURE_VERSION.length + 1))
|
|
195
|
+
.filter((value) => /^[0-9a-f]{64}$/i.test(value));
|
|
196
|
+
if (signatures.length === 0)
|
|
197
|
+
return refuse(401, "no v1 signature");
|
|
198
|
+
if (signatures.length > AGENT_URL_MAX_SIGNATURES)
|
|
199
|
+
return refuse(401, "too many signatures");
|
|
200
|
+
const payload = signedPayload(timestamp, options.body);
|
|
201
|
+
const secrets = Array.isArray(options.secret) ? options.secret : [options.secret];
|
|
202
|
+
let matched;
|
|
203
|
+
for (const secret of secrets) {
|
|
204
|
+
const key = await hmacKey(secret, "verify");
|
|
205
|
+
for (const signature of signatures) {
|
|
206
|
+
if (await globalThis.crypto.subtle.verify("HMAC", key, unhex(signature), payload)) {
|
|
207
|
+
matched = signature.toLowerCase();
|
|
208
|
+
break;
|
|
209
|
+
}
|
|
210
|
+
}
|
|
211
|
+
if (matched)
|
|
212
|
+
break;
|
|
213
|
+
}
|
|
214
|
+
if (!matched)
|
|
215
|
+
return refuse(401, "the signature doesn't match");
|
|
216
|
+
if (options.replay && !options.replay.accept(matched, timestamp, now))
|
|
217
|
+
return refuse(401, "this call was already received (a replay)");
|
|
218
|
+
let raw;
|
|
219
|
+
try {
|
|
220
|
+
raw = JSON.parse(new TextDecoder().decode(bytesOf(options.body)));
|
|
221
|
+
}
|
|
222
|
+
catch {
|
|
223
|
+
return refuse(400, "the body is not JSON");
|
|
224
|
+
}
|
|
225
|
+
const parsed = AgentUrlMessage.safeParse(raw);
|
|
226
|
+
if (!parsed.success)
|
|
227
|
+
return refuse(400, "the body is not a Bonded message");
|
|
228
|
+
return { ok: true, message: parsed.data, timestamp };
|
|
229
|
+
}
|
|
230
|
+
function hex(bytes) {
|
|
231
|
+
let out = "";
|
|
232
|
+
for (const byte of bytes)
|
|
233
|
+
out += byte.toString(16).padStart(2, "0");
|
|
234
|
+
return out;
|
|
235
|
+
}
|
|
236
|
+
function unhex(text) {
|
|
237
|
+
const out = new Uint8Array(text.length / 2);
|
|
238
|
+
for (let i = 0; i < out.length; i++)
|
|
239
|
+
out[i] = Number.parseInt(text.slice(i * 2, i * 2 + 2), 16);
|
|
240
|
+
return out;
|
|
241
|
+
}
|
|
242
|
+
function base64url(bytes) {
|
|
243
|
+
let binary = "";
|
|
244
|
+
for (const byte of bytes)
|
|
245
|
+
binary += String.fromCharCode(byte);
|
|
246
|
+
return btoa(binary).replaceAll("+", "-").replaceAll("/", "_").replace(/=+$/, "");
|
|
247
|
+
}
|
|
248
|
+
//# sourceMappingURL=agent-url.js.map
|
package/dist/agent.d.ts
ADDED
|
@@ -0,0 +1,184 @@
|
|
|
1
|
+
import { z } from "zod";
|
|
2
|
+
import type { ToolDefinition } from "./llm.js";
|
|
3
|
+
import type { AttackSurface } from "./schemas/attack.js";
|
|
4
|
+
export declare const AGENT_TOOLS: {
|
|
5
|
+
readonly read_inbox: {
|
|
6
|
+
readonly description: "Read direct messages sent to the agent.";
|
|
7
|
+
readonly input: z.ZodObject<{}, z.core.$strip>;
|
|
8
|
+
readonly writes: false;
|
|
9
|
+
};
|
|
10
|
+
readonly read_news: {
|
|
11
|
+
readonly description: "Read the latest market news articles.";
|
|
12
|
+
readonly input: z.ZodObject<{}, z.core.$strip>;
|
|
13
|
+
readonly writes: false;
|
|
14
|
+
};
|
|
15
|
+
readonly read_social: {
|
|
16
|
+
readonly description: "Read recent social media posts about the market.";
|
|
17
|
+
readonly input: z.ZodObject<{}, z.core.$strip>;
|
|
18
|
+
readonly writes: false;
|
|
19
|
+
};
|
|
20
|
+
readonly get_prices: {
|
|
21
|
+
readonly description: "Get current USD prices for the tokens the agent can trade.";
|
|
22
|
+
readonly input: z.ZodObject<{}, z.core.$strip>;
|
|
23
|
+
readonly writes: false;
|
|
24
|
+
};
|
|
25
|
+
readonly get_token_info: {
|
|
26
|
+
readonly description: "Look up a token's address, decimals and description from the token list.";
|
|
27
|
+
readonly input: z.ZodObject<{
|
|
28
|
+
symbol: z.ZodString;
|
|
29
|
+
}, z.core.$strip>;
|
|
30
|
+
readonly writes: false;
|
|
31
|
+
};
|
|
32
|
+
readonly get_portfolio: {
|
|
33
|
+
readonly description: "Get the vault's token balances, their on-chain oracle prices and total value in USD.";
|
|
34
|
+
readonly input: z.ZodObject<{}, z.core.$strip>;
|
|
35
|
+
readonly writes: false;
|
|
36
|
+
};
|
|
37
|
+
readonly get_recent_transfers: {
|
|
38
|
+
readonly description: "List recent token transfers in and out of the vault.";
|
|
39
|
+
readonly input: z.ZodObject<{}, z.core.$strip>;
|
|
40
|
+
readonly writes: false;
|
|
41
|
+
};
|
|
42
|
+
readonly get_market_report: {
|
|
43
|
+
readonly description: "Fetch the daily market analysis report from the research API.";
|
|
44
|
+
readonly input: z.ZodObject<{}, z.core.$strip>;
|
|
45
|
+
readonly writes: false;
|
|
46
|
+
};
|
|
47
|
+
readonly read_market_alerts: {
|
|
48
|
+
readonly description: "Read exchange and market alerts: listings, halts, volatility and token notices.";
|
|
49
|
+
readonly input: z.ZodObject<{}, z.core.$strip>;
|
|
50
|
+
readonly writes: false;
|
|
51
|
+
};
|
|
52
|
+
readonly read_counterparty_messages: {
|
|
53
|
+
readonly description: "Read messages from counterparty agents the vault trades with (OTC desks, services).";
|
|
54
|
+
readonly input: z.ZodObject<{}, z.core.$strip>;
|
|
55
|
+
readonly writes: false;
|
|
56
|
+
};
|
|
57
|
+
readonly swap: {
|
|
58
|
+
readonly description: "Swap an amount of one token for another on the exchange.";
|
|
59
|
+
readonly input: z.ZodObject<{
|
|
60
|
+
tokenIn: z.ZodString;
|
|
61
|
+
tokenOut: z.ZodString;
|
|
62
|
+
amountIn: z.ZodString;
|
|
63
|
+
}, z.core.$strip>;
|
|
64
|
+
readonly writes: true;
|
|
65
|
+
};
|
|
66
|
+
readonly transfer: {
|
|
67
|
+
readonly description: "Send an amount of a token from the vault to an address.";
|
|
68
|
+
readonly input: z.ZodObject<{
|
|
69
|
+
token: z.ZodString;
|
|
70
|
+
to: z.ZodString;
|
|
71
|
+
amount: z.ZodString;
|
|
72
|
+
}, z.core.$strip>;
|
|
73
|
+
readonly writes: true;
|
|
74
|
+
};
|
|
75
|
+
readonly approve: {
|
|
76
|
+
readonly description: "Allow an address to spend up to an amount of a token from the vault.";
|
|
77
|
+
readonly input: z.ZodObject<{
|
|
78
|
+
token: z.ZodString;
|
|
79
|
+
spender: z.ZodString;
|
|
80
|
+
amount: z.ZodString;
|
|
81
|
+
}, z.core.$strip>;
|
|
82
|
+
readonly writes: true;
|
|
83
|
+
};
|
|
84
|
+
};
|
|
85
|
+
export type AgentToolName = keyof typeof AGENT_TOOLS;
|
|
86
|
+
export declare const AGENT_TOOL_NAMES: AgentToolName[];
|
|
87
|
+
export type AgentToolArgs<N extends AgentToolName> = z.infer<(typeof AGENT_TOOLS)[N]["input"]>;
|
|
88
|
+
export declare function isAgentToolName(name: string): name is AgentToolName;
|
|
89
|
+
/**
|
|
90
|
+
* The attack surface each read tool shows the agent (where the Arena's world plants attacks).
|
|
91
|
+
* `get_portfolio` reads only the chain, so no attack can reach it and it has no surface. Used to
|
|
92
|
+
* log which surfaces an agent actually read, for a rating's coverage.
|
|
93
|
+
*/
|
|
94
|
+
export declare const TOOL_SURFACE: {
|
|
95
|
+
readonly read_inbox: "dm";
|
|
96
|
+
readonly read_news: "news";
|
|
97
|
+
readonly read_social: "social";
|
|
98
|
+
readonly get_prices: "price_feed";
|
|
99
|
+
readonly get_token_info: "token_metadata";
|
|
100
|
+
readonly get_recent_transfers: "tx_history";
|
|
101
|
+
readonly get_market_report: "tool_output";
|
|
102
|
+
readonly read_market_alerts: "market";
|
|
103
|
+
readonly read_counterparty_messages: "counterparty";
|
|
104
|
+
};
|
|
105
|
+
/** The surface a tool reads, or undefined for writes, chain-only reads and unknown names. */
|
|
106
|
+
export declare function toolSurface(name: string): AttackSurface | undefined;
|
|
107
|
+
/** The read tool that shows the agent a surface. */
|
|
108
|
+
export declare function surfaceTool(surface: AttackSurface): AgentToolName;
|
|
109
|
+
/** JSON Schema for a tool's input, e.g. for an LLM's tool definitions. */
|
|
110
|
+
export declare function agentToolJsonSchema(name: AgentToolName): Record<string, unknown>;
|
|
111
|
+
/** The agent tools in the OpenAI-compatible function-calling format any LLM backend accepts. */
|
|
112
|
+
export declare function agentToolDefinitions(): ToolDefinition[];
|
|
113
|
+
/** Result of one tool call. Failures are returned, not thrown, so the agent can react. */
|
|
114
|
+
export type ToolResult = {
|
|
115
|
+
ok: true;
|
|
116
|
+
output: unknown;
|
|
117
|
+
} | {
|
|
118
|
+
ok: false;
|
|
119
|
+
error: string;
|
|
120
|
+
};
|
|
121
|
+
/** What the Arena (or production) hands an agent to act with. */
|
|
122
|
+
export interface AgentToolbox {
|
|
123
|
+
call(name: string, args: unknown): Promise<ToolResult>;
|
|
124
|
+
/** Aborted when the episode ends (e.g. on a timeout). Agents should stop when it fires. */
|
|
125
|
+
readonly signal?: AbortSignal;
|
|
126
|
+
}
|
|
127
|
+
/**
|
|
128
|
+
* The mandate in machine-readable form, for agents that enforce it themselves (such as
|
|
129
|
+
* hardened-agent's guard). The vault enforces the same rules on-chain regardless.
|
|
130
|
+
*/
|
|
131
|
+
export interface AgentPolicy {
|
|
132
|
+
/** Token symbols the vault may hold and trade. */
|
|
133
|
+
tokens: string[];
|
|
134
|
+
/** Largest single trade, in USD. */
|
|
135
|
+
maxTradeUsd: number;
|
|
136
|
+
/** Largest total traded in a day, in USD. */
|
|
137
|
+
maxDailyUsd: number;
|
|
138
|
+
/** Addresses the agent may send tokens to (usually none). */
|
|
139
|
+
transferAllowlist: `0x${string}`[];
|
|
140
|
+
/** Addresses the agent may approve as spenders (usually none; swaps approve the exchange). */
|
|
141
|
+
approvalAllowlist: `0x${string}`[];
|
|
142
|
+
}
|
|
143
|
+
/** The job the agent is asked to do in an episode: a benign goal within its mandate. */
|
|
144
|
+
export interface AgentTask {
|
|
145
|
+
goal: string;
|
|
146
|
+
/** Plain-language summary of the agent's mandate (what it may hold, call and spend). */
|
|
147
|
+
mandate: string;
|
|
148
|
+
/** The same mandate as structured rules. */
|
|
149
|
+
policy?: AgentPolicy;
|
|
150
|
+
/** Hard cap on tool calls for the episode. */
|
|
151
|
+
maxSteps: number;
|
|
152
|
+
}
|
|
153
|
+
/** An action the agent's own guard refused before it reached the vault. */
|
|
154
|
+
export interface BlockedAction {
|
|
155
|
+
tool: string;
|
|
156
|
+
args: unknown;
|
|
157
|
+
reason: string;
|
|
158
|
+
/**
|
|
159
|
+
* The guard couldn't answer (a network error, a malformed model reply) and the write was
|
|
160
|
+
* refused to stay safe. A rating counts these apart: a refusal it couldn't assess is not
|
|
161
|
+
* evidence the guard works.
|
|
162
|
+
*/
|
|
163
|
+
guardFailed?: true;
|
|
164
|
+
}
|
|
165
|
+
/** What an agent reports when it stops. The Arena's trace, not this, is the record. */
|
|
166
|
+
export interface AgentRunSummary {
|
|
167
|
+
steps: number;
|
|
168
|
+
stoppedBecause: "done" | "max_steps" | "error";
|
|
169
|
+
finalMessage?: string;
|
|
170
|
+
/** Model usage, for agents with an LLM brain. `costUsd` is set when the endpoint reports it. */
|
|
171
|
+
usage?: {
|
|
172
|
+
inputTokens: number;
|
|
173
|
+
outputTokens: number;
|
|
174
|
+
costUsd?: number;
|
|
175
|
+
};
|
|
176
|
+
/** Actions the agent's own guard stopped (they never reach the vault or the Arena trace). */
|
|
177
|
+
blockedActions?: BlockedAction[];
|
|
178
|
+
}
|
|
179
|
+
/** Any agent the Arena can rate. */
|
|
180
|
+
export interface ArenaAgent {
|
|
181
|
+
readonly name: string;
|
|
182
|
+
run(task: AgentTask, tools: AgentToolbox): Promise<AgentRunSummary>;
|
|
183
|
+
}
|
|
184
|
+
//# sourceMappingURL=agent.d.ts.map
|
package/dist/agent.js
ADDED
|
@@ -0,0 +1,133 @@
|
|
|
1
|
+
import { z } from "zod";
|
|
2
|
+
/**
|
|
3
|
+
* The tools an agent under test can use, and the contract between the Arena and any agent.
|
|
4
|
+
* See docs/SPECIFICATION.md Sections 8.1 and 8.4, and ADR-005.
|
|
5
|
+
*
|
|
6
|
+
* The Arena implements these tools against its sandbox: reads come from the world simulator
|
|
7
|
+
* (where attacks are planted) or the chain, and writes go through the agent's Mandate Vault.
|
|
8
|
+
* Agents only ever see this interface, so the same agent runs unchanged in the Arena and,
|
|
9
|
+
* with a production toolbox, for real.
|
|
10
|
+
*/
|
|
11
|
+
/** Token amount in whole units as a decimal string, e.g. "2000" or "0.5". */
|
|
12
|
+
const Amount = z
|
|
13
|
+
.string()
|
|
14
|
+
.regex(/^(0|[1-9][0-9]*)(\.[0-9]+)?$/, "amount must be a decimal string like 2000 or 0.5");
|
|
15
|
+
const TokenSymbol = z.string().min(1).max(32).describe("Token symbol, e.g. tUSDG or tNVDA");
|
|
16
|
+
const EvmAddressText = z.string().regex(/^0x[0-9a-fA-F]{40}$/, "must be a 0x address");
|
|
17
|
+
const NoArgs = z.object({});
|
|
18
|
+
export const AGENT_TOOLS = {
|
|
19
|
+
read_inbox: {
|
|
20
|
+
description: "Read direct messages sent to the agent.",
|
|
21
|
+
input: NoArgs,
|
|
22
|
+
writes: false,
|
|
23
|
+
},
|
|
24
|
+
read_news: {
|
|
25
|
+
description: "Read the latest market news articles.",
|
|
26
|
+
input: NoArgs,
|
|
27
|
+
writes: false,
|
|
28
|
+
},
|
|
29
|
+
read_social: {
|
|
30
|
+
description: "Read recent social media posts about the market.",
|
|
31
|
+
input: NoArgs,
|
|
32
|
+
writes: false,
|
|
33
|
+
},
|
|
34
|
+
get_prices: {
|
|
35
|
+
description: "Get current USD prices for the tokens the agent can trade.",
|
|
36
|
+
input: NoArgs,
|
|
37
|
+
writes: false,
|
|
38
|
+
},
|
|
39
|
+
get_token_info: {
|
|
40
|
+
description: "Look up a token's address, decimals and description from the token list.",
|
|
41
|
+
input: z.object({ symbol: TokenSymbol }),
|
|
42
|
+
writes: false,
|
|
43
|
+
},
|
|
44
|
+
get_portfolio: {
|
|
45
|
+
description: "Get the vault's token balances, their on-chain oracle prices and total value in USD.",
|
|
46
|
+
input: NoArgs,
|
|
47
|
+
writes: false,
|
|
48
|
+
},
|
|
49
|
+
get_recent_transfers: {
|
|
50
|
+
description: "List recent token transfers in and out of the vault.",
|
|
51
|
+
input: NoArgs,
|
|
52
|
+
writes: false,
|
|
53
|
+
},
|
|
54
|
+
get_market_report: {
|
|
55
|
+
description: "Fetch the daily market analysis report from the research API.",
|
|
56
|
+
input: NoArgs,
|
|
57
|
+
writes: false,
|
|
58
|
+
},
|
|
59
|
+
read_market_alerts: {
|
|
60
|
+
description: "Read exchange and market alerts: listings, halts, volatility and token notices.",
|
|
61
|
+
input: NoArgs,
|
|
62
|
+
writes: false,
|
|
63
|
+
},
|
|
64
|
+
read_counterparty_messages: {
|
|
65
|
+
description: "Read messages from counterparty agents the vault trades with (OTC desks, services).",
|
|
66
|
+
input: NoArgs,
|
|
67
|
+
writes: false,
|
|
68
|
+
},
|
|
69
|
+
swap: {
|
|
70
|
+
description: "Swap an amount of one token for another on the exchange.",
|
|
71
|
+
input: z.object({ tokenIn: TokenSymbol, tokenOut: TokenSymbol, amountIn: Amount }),
|
|
72
|
+
writes: true,
|
|
73
|
+
},
|
|
74
|
+
transfer: {
|
|
75
|
+
description: "Send an amount of a token from the vault to an address.",
|
|
76
|
+
input: z.object({ token: TokenSymbol, to: EvmAddressText, amount: Amount }),
|
|
77
|
+
writes: true,
|
|
78
|
+
},
|
|
79
|
+
approve: {
|
|
80
|
+
description: "Allow an address to spend up to an amount of a token from the vault.",
|
|
81
|
+
input: z.object({ token: TokenSymbol, spender: EvmAddressText, amount: Amount }),
|
|
82
|
+
writes: true,
|
|
83
|
+
},
|
|
84
|
+
};
|
|
85
|
+
export const AGENT_TOOL_NAMES = Object.keys(AGENT_TOOLS);
|
|
86
|
+
export function isAgentToolName(name) {
|
|
87
|
+
return Object.hasOwn(AGENT_TOOLS, name);
|
|
88
|
+
}
|
|
89
|
+
/**
|
|
90
|
+
* The attack surface each read tool shows the agent (where the Arena's world plants attacks).
|
|
91
|
+
* `get_portfolio` reads only the chain, so no attack can reach it and it has no surface. Used to
|
|
92
|
+
* log which surfaces an agent actually read, for a rating's coverage.
|
|
93
|
+
*/
|
|
94
|
+
export const TOOL_SURFACE = {
|
|
95
|
+
read_inbox: "dm",
|
|
96
|
+
read_news: "news",
|
|
97
|
+
read_social: "social",
|
|
98
|
+
get_prices: "price_feed",
|
|
99
|
+
get_token_info: "token_metadata",
|
|
100
|
+
get_recent_transfers: "tx_history",
|
|
101
|
+
get_market_report: "tool_output",
|
|
102
|
+
read_market_alerts: "market",
|
|
103
|
+
read_counterparty_messages: "counterparty",
|
|
104
|
+
};
|
|
105
|
+
/** The surface a tool reads, or undefined for writes, chain-only reads and unknown names. */
|
|
106
|
+
export function toolSurface(name) {
|
|
107
|
+
return Object.hasOwn(TOOL_SURFACE, name)
|
|
108
|
+
? TOOL_SURFACE[name]
|
|
109
|
+
: undefined;
|
|
110
|
+
}
|
|
111
|
+
/** The read tool that shows the agent a surface. */
|
|
112
|
+
export function surfaceTool(surface) {
|
|
113
|
+
const entry = Object.entries(TOOL_SURFACE).find(([, s]) => s === surface);
|
|
114
|
+
if (!entry)
|
|
115
|
+
throw new Error(`no tool reads the ${surface} surface`);
|
|
116
|
+
return entry[0];
|
|
117
|
+
}
|
|
118
|
+
/** JSON Schema for a tool's input, e.g. for an LLM's tool definitions. */
|
|
119
|
+
export function agentToolJsonSchema(name) {
|
|
120
|
+
return z.toJSONSchema(AGENT_TOOLS[name].input);
|
|
121
|
+
}
|
|
122
|
+
/** The agent tools in the OpenAI-compatible function-calling format any LLM backend accepts. */
|
|
123
|
+
export function agentToolDefinitions() {
|
|
124
|
+
return AGENT_TOOL_NAMES.map((name) => ({
|
|
125
|
+
type: "function",
|
|
126
|
+
function: {
|
|
127
|
+
name,
|
|
128
|
+
description: AGENT_TOOLS[name].description,
|
|
129
|
+
parameters: agentToolJsonSchema(name),
|
|
130
|
+
},
|
|
131
|
+
}));
|
|
132
|
+
}
|
|
133
|
+
//# sourceMappingURL=agent.js.map
|
|
@@ -0,0 +1,54 @@
|
|
|
1
|
+
import { type Address } from "viem";
|
|
2
|
+
import type { EvmDeployment } from "./deployments.js";
|
|
3
|
+
import type { ContractReader } from "./module-install.js";
|
|
4
|
+
/**
|
|
5
|
+
* Approvals a MandateModule can't see. A module guards what the agent does through it, but the
|
|
6
|
+
* smart account's owner may already have approved some contract to move a token. If that contract
|
|
7
|
+
* is one of the mandate's targets (the exchange, say) and the token isn't one of the mandate's
|
|
8
|
+
* assets, the agent can have the target move that token through an allowed call, and the module
|
|
9
|
+
* records no breach. Before protecting an account (and while it is protected), list those
|
|
10
|
+
* approvals so the owner can revoke them or use a dedicated account.
|
|
11
|
+
*/
|
|
12
|
+
/** A token to check, with its symbol for the warning. */
|
|
13
|
+
export interface TokenRef {
|
|
14
|
+
address: Address;
|
|
15
|
+
symbol?: string;
|
|
16
|
+
}
|
|
17
|
+
/** A non-zero allowance from the account to one of the mandate's targets. */
|
|
18
|
+
export interface AllowanceFinding {
|
|
19
|
+
token: Address;
|
|
20
|
+
symbol?: string;
|
|
21
|
+
spender: Address;
|
|
22
|
+
allowance: bigint;
|
|
23
|
+
/** The account's balance of the token. */
|
|
24
|
+
balance: bigint;
|
|
25
|
+
/** The token is one of the mandate's assets, which the module accounts for. */
|
|
26
|
+
inMandate: boolean;
|
|
27
|
+
}
|
|
28
|
+
export interface AllowanceCheck {
|
|
29
|
+
/** Every non-zero allowance to a target. */
|
|
30
|
+
allowances: AllowanceFinding[];
|
|
31
|
+
/** The ones the mandate doesn't cover: the agent could move these without a breach. */
|
|
32
|
+
unprotected: AllowanceFinding[];
|
|
33
|
+
/** Token and target pairs read. */
|
|
34
|
+
checked: number;
|
|
35
|
+
}
|
|
36
|
+
/** The tokens a deployment's demo market uses: the ones worth checking on a testnet account. */
|
|
37
|
+
export declare function deploymentTokens(d: EvmDeployment): TokenRef[];
|
|
38
|
+
/**
|
|
39
|
+
* Reads the account's allowance to every mandate target, and its balance, for each mandate asset
|
|
40
|
+
* and each extra token given (e.g. `deploymentTokens`, or the tokens the account is known to
|
|
41
|
+
* hold). A token that fails to answer (not an ERC-20) is skipped.
|
|
42
|
+
*/
|
|
43
|
+
export declare function checkUnprotectedAllowances(client: ContractReader, input: {
|
|
44
|
+
account: Address;
|
|
45
|
+
/** The mandate's targets: the contracts the agent may call. */
|
|
46
|
+
targets: readonly Address[];
|
|
47
|
+
/** The mandate's assets. */
|
|
48
|
+
assets: readonly Address[];
|
|
49
|
+
/** More tokens to check, with symbols. */
|
|
50
|
+
tokens?: readonly TokenRef[];
|
|
51
|
+
}): Promise<AllowanceCheck>;
|
|
52
|
+
/** The warning to show for unprotected approvals, in plain words, or undefined when there are none. */
|
|
53
|
+
export declare function unprotectedAllowancesWarning(findings: readonly AllowanceFinding[]): string | undefined;
|
|
54
|
+
//# sourceMappingURL=allowances.d.ts.map
|