@evolvingmachines/sdk 0.0.51 → 0.0.52-launch-round-1.20260803.cb1be5b
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/chunk-VGAGB5J5.js +6 -0
- package/dist/hosted/cli.cjs +19 -0
- package/dist/hosted/cli.d.cts +105 -0
- package/dist/hosted/cli.d.ts +105 -0
- package/dist/hosted/cli.js +14 -0
- package/dist/index.cjs +440 -82
- package/dist/index.d.cts +1207 -478
- package/dist/index.d.ts +1207 -478
- package/dist/index.js +429 -79
- package/dist/managed-modal-CRZFDIRG.js +4 -0
- package/dist/tar-PPIAXEGM.js +1 -0
- package/dist/types-B8wdtDC4.d.cts +1602 -0
- package/dist/types-B8wdtDC4.d.ts +1602 -0
- package/harness-capabilities.json +345 -0
- package/hosted-error-codes.json +57 -0
- package/package.json +35 -10
- package/spec/openapi.yaml +4050 -0
|
@@ -0,0 +1,345 @@
|
|
|
1
|
+
{
|
|
2
|
+
"$comment": "Per-harness models and reasoning efforts, checked in so two repos that share no test runner can be held to one table. The evolve SDK owns it: AGENT_REGISTRY in packages/sdk-ts/src/registry.ts is the only place models, efforts, or defaults may change, and scripts/generate-harness-capabilities.ts regenerates this file from it (npm run generate:capabilities; the build fails when it is stale). The hosted-evals dashboard consumes this file for its capability document and its parity test diffs the served document against it, the TypeScript SDK test regenerates these exact bytes from the registry, and the Python SDK test asserts its own Literals cover them. Changing a model or an effort therefore means: registry.ts, regenerate this file, and nothing else.",
|
|
3
|
+
"reasoningEfforts": [
|
|
4
|
+
"off",
|
|
5
|
+
"minimal",
|
|
6
|
+
"low",
|
|
7
|
+
"medium",
|
|
8
|
+
"high",
|
|
9
|
+
"xhigh",
|
|
10
|
+
"max",
|
|
11
|
+
"thinking"
|
|
12
|
+
],
|
|
13
|
+
"binaryEffortValues": [
|
|
14
|
+
"off",
|
|
15
|
+
"minimal",
|
|
16
|
+
"medium",
|
|
17
|
+
"thinking"
|
|
18
|
+
],
|
|
19
|
+
"defaultReasoningEffort": "medium",
|
|
20
|
+
"harnesses": {
|
|
21
|
+
"claude": {
|
|
22
|
+
"defaultModel": "opus",
|
|
23
|
+
"models": [
|
|
24
|
+
{
|
|
25
|
+
"alias": "fable",
|
|
26
|
+
"modelId": "claude-fable-5",
|
|
27
|
+
"description": "Highest capability, long-horizon agentic work"
|
|
28
|
+
},
|
|
29
|
+
{
|
|
30
|
+
"alias": "opus",
|
|
31
|
+
"modelId": "claude-opus-5",
|
|
32
|
+
"description": "Complex reasoning, R&D, architecting"
|
|
33
|
+
},
|
|
34
|
+
{
|
|
35
|
+
"alias": "sonnet",
|
|
36
|
+
"modelId": "claude-sonnet-5",
|
|
37
|
+
"description": "Daily coding, features, tests"
|
|
38
|
+
},
|
|
39
|
+
{
|
|
40
|
+
"alias": "haiku",
|
|
41
|
+
"modelId": "claude-haiku-4-5-20251001",
|
|
42
|
+
"description": "Quick tasks, syntax correction"
|
|
43
|
+
},
|
|
44
|
+
{
|
|
45
|
+
"alias": "opus[1m]",
|
|
46
|
+
"modelId": "opus[1m]",
|
|
47
|
+
"description": "Complex reasoning with 1M context window"
|
|
48
|
+
},
|
|
49
|
+
{
|
|
50
|
+
"alias": "sonnet[1m]",
|
|
51
|
+
"modelId": "sonnet[1m]",
|
|
52
|
+
"description": "Daily coding with 1M context window"
|
|
53
|
+
}
|
|
54
|
+
],
|
|
55
|
+
"effortSupport": "level",
|
|
56
|
+
"efforts": [
|
|
57
|
+
"off",
|
|
58
|
+
"minimal",
|
|
59
|
+
"low",
|
|
60
|
+
"medium",
|
|
61
|
+
"high",
|
|
62
|
+
"xhigh",
|
|
63
|
+
"max",
|
|
64
|
+
"thinking"
|
|
65
|
+
],
|
|
66
|
+
"defaultEffort": "high"
|
|
67
|
+
},
|
|
68
|
+
"codex": {
|
|
69
|
+
"defaultModel": "gpt-5.6-sol",
|
|
70
|
+
"models": [
|
|
71
|
+
{
|
|
72
|
+
"alias": "gpt-5.6-sol",
|
|
73
|
+
"modelId": "gpt-5.6-sol",
|
|
74
|
+
"description": "Newest frontier flagship"
|
|
75
|
+
},
|
|
76
|
+
{
|
|
77
|
+
"alias": "gpt-5.6-terra",
|
|
78
|
+
"modelId": "gpt-5.6-terra",
|
|
79
|
+
"description": "Balances intelligence and cost"
|
|
80
|
+
},
|
|
81
|
+
{
|
|
82
|
+
"alias": "gpt-5.6-luna",
|
|
83
|
+
"modelId": "gpt-5.6-luna",
|
|
84
|
+
"description": "High-volume, cost-sensitive tier"
|
|
85
|
+
},
|
|
86
|
+
{
|
|
87
|
+
"alias": "gpt-5.5",
|
|
88
|
+
"modelId": "gpt-5.5",
|
|
89
|
+
"description": "Previous frontier model"
|
|
90
|
+
},
|
|
91
|
+
{
|
|
92
|
+
"alias": "gpt-5.3-codex",
|
|
93
|
+
"modelId": "gpt-5.3-codex",
|
|
94
|
+
"description": "Industry-leading code-optimized"
|
|
95
|
+
}
|
|
96
|
+
],
|
|
97
|
+
"effortSupport": "level",
|
|
98
|
+
"efforts": [
|
|
99
|
+
"off",
|
|
100
|
+
"minimal",
|
|
101
|
+
"low",
|
|
102
|
+
"medium",
|
|
103
|
+
"high",
|
|
104
|
+
"xhigh",
|
|
105
|
+
"max",
|
|
106
|
+
"thinking"
|
|
107
|
+
],
|
|
108
|
+
"defaultEffort": "high"
|
|
109
|
+
},
|
|
110
|
+
"droid": {
|
|
111
|
+
"defaultModel": "claude-opus-5",
|
|
112
|
+
"models": [
|
|
113
|
+
{
|
|
114
|
+
"alias": "claude-fable-5",
|
|
115
|
+
"modelId": "claude-fable-5",
|
|
116
|
+
"description": "Factory-managed Claude Fable 5"
|
|
117
|
+
},
|
|
118
|
+
{
|
|
119
|
+
"alias": "claude-opus-5",
|
|
120
|
+
"modelId": "claude-opus-5",
|
|
121
|
+
"description": "Factory-managed Claude Opus 5"
|
|
122
|
+
},
|
|
123
|
+
{
|
|
124
|
+
"alias": "claude-sonnet-5",
|
|
125
|
+
"modelId": "claude-sonnet-5",
|
|
126
|
+
"description": "Factory-managed Claude Sonnet 5"
|
|
127
|
+
},
|
|
128
|
+
{
|
|
129
|
+
"alias": "claude-haiku-4-5",
|
|
130
|
+
"modelId": "claude-haiku-4-5-20251001",
|
|
131
|
+
"description": "Factory-managed Claude Haiku 4.5"
|
|
132
|
+
},
|
|
133
|
+
{
|
|
134
|
+
"alias": "gpt-5.6-sol",
|
|
135
|
+
"modelId": "gpt-5.6-sol",
|
|
136
|
+
"description": "Factory-managed GPT-5.6 Sol"
|
|
137
|
+
},
|
|
138
|
+
{
|
|
139
|
+
"alias": "gpt-5.6-terra",
|
|
140
|
+
"modelId": "gpt-5.6-terra",
|
|
141
|
+
"description": "Factory-managed GPT-5.6 Terra"
|
|
142
|
+
},
|
|
143
|
+
{
|
|
144
|
+
"alias": "gpt-5.6-luna",
|
|
145
|
+
"modelId": "gpt-5.6-luna",
|
|
146
|
+
"description": "Factory-managed GPT-5.6 Luna"
|
|
147
|
+
},
|
|
148
|
+
{
|
|
149
|
+
"alias": "gemini-3.6-flash",
|
|
150
|
+
"modelId": "gemini-3.6-flash",
|
|
151
|
+
"description": "Factory-managed Gemini 3.6 Flash"
|
|
152
|
+
},
|
|
153
|
+
{
|
|
154
|
+
"alias": "qwen3.7-max",
|
|
155
|
+
"modelId": "qwen3.7-max",
|
|
156
|
+
"description": "Qwen 3.7 Max via the Evolve gateway"
|
|
157
|
+
},
|
|
158
|
+
{
|
|
159
|
+
"alias": "kimi-k3",
|
|
160
|
+
"modelId": "kimi-k3",
|
|
161
|
+
"description": "Factory-managed Droid Core Kimi K3"
|
|
162
|
+
},
|
|
163
|
+
{
|
|
164
|
+
"alias": "glm-5.2",
|
|
165
|
+
"modelId": "glm-5.2",
|
|
166
|
+
"description": "Factory-managed Droid Core GLM-5.2"
|
|
167
|
+
}
|
|
168
|
+
],
|
|
169
|
+
"effortSupport": "level",
|
|
170
|
+
"efforts": [
|
|
171
|
+
"off",
|
|
172
|
+
"minimal",
|
|
173
|
+
"low",
|
|
174
|
+
"medium",
|
|
175
|
+
"high",
|
|
176
|
+
"xhigh",
|
|
177
|
+
"max",
|
|
178
|
+
"thinking"
|
|
179
|
+
],
|
|
180
|
+
"defaultEffort": "high"
|
|
181
|
+
},
|
|
182
|
+
"gemini": {
|
|
183
|
+
"defaultModel": "gemini-3.6-flash",
|
|
184
|
+
"models": [
|
|
185
|
+
{
|
|
186
|
+
"alias": "gemini-3.6-flash",
|
|
187
|
+
"modelId": "gemini-3.6-flash",
|
|
188
|
+
"description": "Newest stable workhorse: coding + agentic planning"
|
|
189
|
+
},
|
|
190
|
+
{
|
|
191
|
+
"alias": "gemini-3.5-flash-lite",
|
|
192
|
+
"modelId": "gemini-3.5-flash-lite",
|
|
193
|
+
"description": "Most cost-effective 3.5-class model"
|
|
194
|
+
},
|
|
195
|
+
{
|
|
196
|
+
"alias": "gemini-3.1-pro-preview",
|
|
197
|
+
"modelId": "gemini-3.1-pro-preview",
|
|
198
|
+
"description": "Latest pro, complex agentic + coding"
|
|
199
|
+
}
|
|
200
|
+
],
|
|
201
|
+
"effortSupport": "none",
|
|
202
|
+
"efforts": [],
|
|
203
|
+
"defaultEffort": null
|
|
204
|
+
},
|
|
205
|
+
"kimi": {
|
|
206
|
+
"defaultModel": "kimi-k3",
|
|
207
|
+
"models": [
|
|
208
|
+
{
|
|
209
|
+
"alias": "kimi-k3",
|
|
210
|
+
"modelId": "moonshot/kimi-k3",
|
|
211
|
+
"description": "Latest flagship: 1M context, always-on thinking",
|
|
212
|
+
"maxContextSize": 1048576
|
|
213
|
+
},
|
|
214
|
+
{
|
|
215
|
+
"alias": "kimi-k2.7-code",
|
|
216
|
+
"modelId": "moonshot/kimi-k2.7-code",
|
|
217
|
+
"description": "Latest coding-specialized standard model"
|
|
218
|
+
},
|
|
219
|
+
{
|
|
220
|
+
"alias": "kimi-k3-raptor",
|
|
221
|
+
"modelId": "kimi-k3-raptor",
|
|
222
|
+
"description": "Evolve-managed Kimi K3 Raptor route for latency-sensitive agent runs",
|
|
223
|
+
"maxContextSize": 1048576
|
|
224
|
+
},
|
|
225
|
+
{
|
|
226
|
+
"alias": "kimi-k2p7-code-raptor",
|
|
227
|
+
"modelId": "kimi-k2p7-code-raptor",
|
|
228
|
+
"description": "Evolve-managed Kimi K2.7 Code Raptor route for latency-sensitive agent runs"
|
|
229
|
+
}
|
|
230
|
+
],
|
|
231
|
+
"effortSupport": "level",
|
|
232
|
+
"efforts": [
|
|
233
|
+
"off",
|
|
234
|
+
"minimal",
|
|
235
|
+
"low",
|
|
236
|
+
"medium",
|
|
237
|
+
"high",
|
|
238
|
+
"xhigh",
|
|
239
|
+
"max",
|
|
240
|
+
"thinking"
|
|
241
|
+
],
|
|
242
|
+
"defaultEffort": "max"
|
|
243
|
+
},
|
|
244
|
+
"opencode": {
|
|
245
|
+
"defaultModel": "openrouter/anthropic/claude-opus-5",
|
|
246
|
+
"models": [
|
|
247
|
+
{
|
|
248
|
+
"alias": "openrouter/anthropic/claude-fable-5",
|
|
249
|
+
"modelId": "openrouter/anthropic/claude-fable-5",
|
|
250
|
+
"description": "Anthropic Fable via OpenRouter"
|
|
251
|
+
},
|
|
252
|
+
{
|
|
253
|
+
"alias": "openrouter/anthropic/claude-opus-5",
|
|
254
|
+
"modelId": "openrouter/anthropic/claude-opus-5",
|
|
255
|
+
"description": "Anthropic Opus 5 via OpenRouter"
|
|
256
|
+
},
|
|
257
|
+
{
|
|
258
|
+
"alias": "openrouter/anthropic/claude-sonnet-5",
|
|
259
|
+
"modelId": "openrouter/anthropic/claude-sonnet-5",
|
|
260
|
+
"description": "Anthropic Sonnet 5 via OpenRouter"
|
|
261
|
+
},
|
|
262
|
+
{
|
|
263
|
+
"alias": "openrouter/anthropic/claude-haiku-4.5",
|
|
264
|
+
"modelId": "openrouter/anthropic/claude-haiku-4.5",
|
|
265
|
+
"description": "Anthropic Haiku via OpenRouter"
|
|
266
|
+
},
|
|
267
|
+
{
|
|
268
|
+
"alias": "openrouter/openai/gpt-5.6-sol",
|
|
269
|
+
"modelId": "openrouter/openai/gpt-5.6-sol",
|
|
270
|
+
"description": "OpenAI GPT-5.6 Sol via OpenRouter"
|
|
271
|
+
},
|
|
272
|
+
{
|
|
273
|
+
"alias": "openrouter/openai/gpt-5.6-terra",
|
|
274
|
+
"modelId": "openrouter/openai/gpt-5.6-terra",
|
|
275
|
+
"description": "OpenAI GPT-5.6 Terra via OpenRouter"
|
|
276
|
+
},
|
|
277
|
+
{
|
|
278
|
+
"alias": "openrouter/openai/gpt-5.6-luna",
|
|
279
|
+
"modelId": "openrouter/openai/gpt-5.6-luna",
|
|
280
|
+
"description": "OpenAI GPT-5.6 Luna via OpenRouter"
|
|
281
|
+
},
|
|
282
|
+
{
|
|
283
|
+
"alias": "openrouter/google/gemini-3.6-flash",
|
|
284
|
+
"modelId": "openrouter/google/gemini-3.6-flash",
|
|
285
|
+
"description": "Gemini 3.6 Flash via OpenRouter"
|
|
286
|
+
},
|
|
287
|
+
{
|
|
288
|
+
"alias": "openrouter/qwen/qwen3.7-max",
|
|
289
|
+
"modelId": "openrouter/qwen/qwen3.7-max",
|
|
290
|
+
"description": "Qwen 3.7 Max via OpenRouter"
|
|
291
|
+
},
|
|
292
|
+
{
|
|
293
|
+
"alias": "openrouter/moonshotai/kimi-k3",
|
|
294
|
+
"modelId": "openrouter/moonshotai/kimi-k3",
|
|
295
|
+
"description": "Kimi K3 via OpenRouter"
|
|
296
|
+
},
|
|
297
|
+
{
|
|
298
|
+
"alias": "openrouter/z-ai/glm-5.2",
|
|
299
|
+
"modelId": "openrouter/z-ai/glm-5.2",
|
|
300
|
+
"description": "Zhipu GLM-5.2 via OpenRouter"
|
|
301
|
+
}
|
|
302
|
+
],
|
|
303
|
+
"effortSupport": "level",
|
|
304
|
+
"efforts": [
|
|
305
|
+
"off",
|
|
306
|
+
"minimal",
|
|
307
|
+
"low",
|
|
308
|
+
"medium",
|
|
309
|
+
"high",
|
|
310
|
+
"xhigh",
|
|
311
|
+
"max",
|
|
312
|
+
"thinking"
|
|
313
|
+
],
|
|
314
|
+
"defaultEffort": "high"
|
|
315
|
+
},
|
|
316
|
+
"qwen": {
|
|
317
|
+
"defaultModel": "qwen3.7-max",
|
|
318
|
+
"models": [
|
|
319
|
+
{
|
|
320
|
+
"alias": "qwen3.7-max",
|
|
321
|
+
"modelId": "qwen3.7-max",
|
|
322
|
+
"description": "Strongest reasoning and coding option"
|
|
323
|
+
},
|
|
324
|
+
{
|
|
325
|
+
"alias": "qwen3.7-plus",
|
|
326
|
+
"modelId": "qwen3.7-plus",
|
|
327
|
+
"description": "Latest balanced Qwen Cloud recommendation"
|
|
328
|
+
},
|
|
329
|
+
{
|
|
330
|
+
"alias": "qwen3.6-flash",
|
|
331
|
+
"modelId": "qwen3.6-flash",
|
|
332
|
+
"description": "Fast and cost-effective option"
|
|
333
|
+
}
|
|
334
|
+
],
|
|
335
|
+
"effortSupport": "binary",
|
|
336
|
+
"efforts": [
|
|
337
|
+
"off",
|
|
338
|
+
"minimal",
|
|
339
|
+
"medium",
|
|
340
|
+
"thinking"
|
|
341
|
+
],
|
|
342
|
+
"defaultEffort": "thinking"
|
|
343
|
+
}
|
|
344
|
+
}
|
|
345
|
+
}
|
|
@@ -0,0 +1,57 @@
|
|
|
1
|
+
{
|
|
2
|
+
"$comment": "The hosted evals API's error-code vocabulary, checked in so two repos that share no test runner can be held to one list. The contract owns it: the ErrorCode enum in spec/openapi.yaml is the only place a code may be added. This file is that enum's shadow, and the tests keep everyone honest — the server regenerates these exact bytes and diffs them, the TypeScript SDK asserts HOSTED_ERROR_CODES equals codes[], and the Python SDK asserts both its tuple and its Literal do. Adding a code therefore means: spec enum, this file, both SDK lists.",
|
|
3
|
+
"codes": [
|
|
4
|
+
"missing_authorization",
|
|
5
|
+
"invalid_api_key",
|
|
6
|
+
"credential_service_unavailable",
|
|
7
|
+
"rate_limited",
|
|
8
|
+
"insufficient_credits",
|
|
9
|
+
"invalid_json",
|
|
10
|
+
"invalid_input",
|
|
11
|
+
"invalid_limit",
|
|
12
|
+
"invalid_status",
|
|
13
|
+
"invalid_cursor",
|
|
14
|
+
"invalid_after",
|
|
15
|
+
"invalid_format",
|
|
16
|
+
"invalid_ids",
|
|
17
|
+
"invalid_multipart",
|
|
18
|
+
"idempotency_key_reused",
|
|
19
|
+
"dataset_not_found",
|
|
20
|
+
"dataset_version_not_found",
|
|
21
|
+
"dataset_name_taken",
|
|
22
|
+
"dataset_in_use",
|
|
23
|
+
"dataset_not_owned",
|
|
24
|
+
"upstream_not_watchable",
|
|
25
|
+
"no_active_version",
|
|
26
|
+
"version_not_ready",
|
|
27
|
+
"version_not_activatable",
|
|
28
|
+
"unknown_task_names",
|
|
29
|
+
"no_tasks",
|
|
30
|
+
"agent_not_found",
|
|
31
|
+
"agent_name_taken",
|
|
32
|
+
"agent_name_reserved",
|
|
33
|
+
"agent_invalid_name",
|
|
34
|
+
"agent_source_required",
|
|
35
|
+
"agent_source_conflict",
|
|
36
|
+
"agent_invalid_env",
|
|
37
|
+
"agent_too_large",
|
|
38
|
+
"agent_limit_reached",
|
|
39
|
+
"agent_version_not_found",
|
|
40
|
+
"job_too_large",
|
|
41
|
+
"provider_unsupported",
|
|
42
|
+
"job_not_found",
|
|
43
|
+
"job_not_terminal",
|
|
44
|
+
"no_failed_trials",
|
|
45
|
+
"trial_not_found",
|
|
46
|
+
"concurrent_update",
|
|
47
|
+
"regrade_source_ineligible",
|
|
48
|
+
"no_regradable_trials",
|
|
49
|
+
"import_not_found",
|
|
50
|
+
"import_too_large",
|
|
51
|
+
"invalid_archive",
|
|
52
|
+
"package_not_retained",
|
|
53
|
+
"package_corrupt",
|
|
54
|
+
"package_missing",
|
|
55
|
+
"internal_error"
|
|
56
|
+
]
|
|
57
|
+
}
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@evolvingmachines/sdk",
|
|
3
|
-
"version": "0.0.
|
|
3
|
+
"version": "0.0.52-launch-round-1.20260803.cb1be5b",
|
|
4
4
|
"keywords": [
|
|
5
5
|
"ai",
|
|
6
6
|
"agents",
|
|
@@ -23,27 +23,36 @@
|
|
|
23
23
|
"url": "git+https://github.com/evolving-machines-lab/evolve.git"
|
|
24
24
|
},
|
|
25
25
|
"type": "module",
|
|
26
|
+
"bin": {
|
|
27
|
+
"evolve-evals": "dist/hosted/cli.js"
|
|
28
|
+
},
|
|
26
29
|
"main": "dist/index.cjs",
|
|
27
30
|
"types": "dist/index.d.ts",
|
|
28
31
|
"license": "Apache-2.0",
|
|
29
32
|
"files": [
|
|
30
33
|
"dist",
|
|
31
|
-
"LICENSE"
|
|
34
|
+
"LICENSE",
|
|
35
|
+
"spec",
|
|
36
|
+
"hosted-error-codes.json",
|
|
37
|
+
"harness-capabilities.json"
|
|
32
38
|
],
|
|
33
39
|
"exports": {
|
|
34
40
|
".": {
|
|
35
41
|
"types": "./dist/index.d.ts",
|
|
36
42
|
"import": "./dist/index.js",
|
|
37
43
|
"require": "./dist/index.cjs"
|
|
38
|
-
}
|
|
44
|
+
},
|
|
45
|
+
"./harness-capabilities.json": "./harness-capabilities.json"
|
|
39
46
|
},
|
|
40
47
|
"scripts": {
|
|
41
|
-
"build": "tsup --minify",
|
|
48
|
+
"build": "tsx scripts/generate-harness-capabilities.ts --check && tsup --minify && node scripts/copy-spec.mjs",
|
|
49
|
+
"generate:capabilities": "tsx scripts/generate-harness-capabilities.ts",
|
|
42
50
|
"dev": "tsup src/index.ts --watch",
|
|
43
|
-
"type-check": "tsc --noEmit",
|
|
51
|
+
"type-check": "tsc --noEmit && tsc -p tsconfig.types-test.json",
|
|
44
52
|
"test": "npm run test:unit && npm run test:integration",
|
|
45
53
|
"pretest:unit": "npm run build",
|
|
46
|
-
"test:unit": "tsx tests/unit/semaphore.test.ts && tsx tests/unit/swarm-concurrency.test.ts && tsx tests/unit/swarm-verify.test.ts && tsx tests/unit/swarm-retry-verify.test.ts && tsx tests/unit/prompt-construction.test.ts && tsx tests/unit/observability-metadata.test.ts && tsx tests/unit/skills-integration.test.ts && tsx tests/unit/auth-config.test.ts && tsx tests/unit/browser-config.test.ts && tsx tests/unit/integrations-config.test.ts && tsx tests/unit/plugins-config.test.ts && tsx tests/unit/session-runtime.test.ts && tsx tests/unit/cost-api.test.ts && tsx tests/unit/storage-config.test.ts && tsx tests/unit/checkpoint-tar.test.ts && tsx tests/unit/checkpoint-errors.test.ts && tsx tests/unit/checkpoint-flows.test.ts && tsx tests/unit/checkpoint-dx.test.ts && tsx tests/unit/checkpoint-edge-cases.test.ts && tsx tests/unit/kimi-parser.test.ts && tsx tests/unit/kimi-mcp.test.ts && tsx tests/unit/opencode-parser.test.ts && tsx tests/unit/droid-parser.test.ts && tsx tests/unit/droid-mcp.test.ts && tsx tests/unit/provider-parity.test.ts && tsx tests/unit/storage-client.test.ts && tsx tests/unit/sessions-client.test.ts",
|
|
54
|
+
"test:unit": "tsx tests/unit/semaphore.test.ts && tsx tests/unit/swarm-concurrency.test.ts && tsx tests/unit/swarm-verify.test.ts && tsx tests/unit/swarm-retry-verify.test.ts && tsx tests/unit/prompt-construction.test.ts && tsx tests/unit/observability-metadata.test.ts && tsx tests/unit/observability-identity.test.ts && tsx tests/unit/skills-integration.test.ts && tsx tests/unit/auth-config.test.ts && tsx tests/unit/config-validation.test.ts && tsx tests/unit/browser-config.test.ts && tsx tests/unit/integrations-config.test.ts && tsx tests/unit/managed-secrets.test.ts && tsx tests/unit/managed-modal.test.ts && tsx tests/unit/plugins-config.test.ts && tsx tests/unit/session-runtime.test.ts && tsx tests/unit/cost-api.test.ts && tsx tests/unit/storage-config.test.ts && tsx tests/unit/checkpoint-tar.test.ts && tsx tests/unit/checkpoint-errors.test.ts && tsx tests/unit/checkpoint-flows.test.ts && tsx tests/unit/checkpoint-dx.test.ts && tsx tests/unit/checkpoint-edge-cases.test.ts && tsx tests/unit/codex-parser-errors.test.ts && tsx tests/unit/codex-toml.test.ts && tsx tests/unit/kimi-parser.test.ts && tsx tests/unit/kimi-mcp.test.ts && tsx tests/unit/opencode-parser.test.ts && tsx tests/unit/droid-parser.test.ts && tsx tests/unit/droid-mcp.test.ts && tsx tests/unit/provider-parity.test.ts && tsx tests/unit/storage-client.test.ts && tsx tests/unit/sessions-client.test.ts && tsx tests/unit/hosted-client.test.ts && tsx tests/unit/hosted-ergonomics.test.ts && tsx tests/unit/hosted-cli.test.ts && tsx tests/unit/hosted-cli-bin.test.ts && tsx tests/unit/hosted-types.test.ts && tsx tests/unit/hosted-error-codes.test.ts && tsx tests/unit/hosted-spec-gate.test.ts && tsx tests/unit/harness-capabilities.test.ts && tsx tests/unit/sandbox-artifacts.test.ts && tsx tests/unit/upload-file-from-path.test.ts && tsx tests/unit/file-utils.test.ts && tsx --expose-gc tests/unit/hosted-tar.test.ts",
|
|
55
|
+
"test:unit:upload-from-path": "tsx tests/unit/upload-file-from-path.test.ts",
|
|
47
56
|
"test:unit:parity": "tsx tests/unit/provider-parity.test.ts",
|
|
48
57
|
"test:unit:semaphore": "tsx tests/unit/semaphore.test.ts",
|
|
49
58
|
"test:unit:swarm": "tsx tests/unit/swarm-concurrency.test.ts",
|
|
@@ -51,12 +60,17 @@
|
|
|
51
60
|
"test:unit:retry": "tsx tests/unit/swarm-retry-verify.test.ts",
|
|
52
61
|
"test:unit:prompts": "tsx tests/unit/prompt-construction.test.ts",
|
|
53
62
|
"test:unit:observability": "tsx tests/unit/observability-metadata.test.ts",
|
|
63
|
+
"test:unit:observability-identity": "tsx tests/unit/observability-identity.test.ts",
|
|
54
64
|
"test:unit:skills": "tsx tests/unit/skills-integration.test.ts",
|
|
55
65
|
"test:unit:auth": "tsx tests/unit/auth-config.test.ts",
|
|
66
|
+
"test:unit:config-validation": "tsx tests/unit/config-validation.test.ts",
|
|
56
67
|
"test:unit:browser": "tsx tests/unit/browser-config.test.ts",
|
|
57
68
|
"test:unit:integrations": "tsx tests/unit/integrations-config.test.ts",
|
|
69
|
+
"test:unit:managed-secrets": "tsx tests/unit/managed-secrets.test.ts",
|
|
58
70
|
"test:unit:plugins": "tsx tests/unit/plugins-config.test.ts",
|
|
59
71
|
"test:unit:storage-config": "tsx tests/unit/storage-config.test.ts",
|
|
72
|
+
"test:unit:error-codes": "tsx tests/unit/hosted-error-codes.test.ts",
|
|
73
|
+
"test:unit:capabilities": "tsx tests/unit/harness-capabilities.test.ts",
|
|
60
74
|
"test:unit:checkpoint-tar": "tsx tests/unit/checkpoint-tar.test.ts",
|
|
61
75
|
"test:unit:checkpoint-errors": "tsx tests/unit/checkpoint-errors.test.ts",
|
|
62
76
|
"test:unit:checkpoint-flows": "tsx tests/unit/checkpoint-flows.test.ts",
|
|
@@ -64,8 +78,10 @@
|
|
|
64
78
|
"test:unit:checkpoint-edge-cases": "tsx tests/unit/checkpoint-edge-cases.test.ts",
|
|
65
79
|
"test:unit:storage-client": "tsx tests/unit/storage-client.test.ts",
|
|
66
80
|
"test:unit:sessions-client": "tsx tests/unit/sessions-client.test.ts",
|
|
81
|
+
"test:unit:hosted-client": "tsx tests/unit/hosted-client.test.ts",
|
|
67
82
|
"test:unit:kimi-parser": "tsx tests/unit/kimi-parser.test.ts",
|
|
68
83
|
"test:unit:kimi-mcp": "tsx tests/unit/kimi-mcp.test.ts",
|
|
84
|
+
"test:unit:codex-toml": "tsx tests/unit/codex-toml.test.ts",
|
|
69
85
|
"pretest:integration": "npm run build",
|
|
70
86
|
"test:integration": "npm run test:01",
|
|
71
87
|
"test:integration:all": "npm run test:01 && npm run test:02 && npm run test:03 && npm run test:04 && npm run test:05 && npm run test:06 && npm run test:07 && npm run test:08 && npm run test:09 && npm run test:10 && npm run test:11 && npm run test:12 && npm run test:13 && npm run test:14 && npm run test:15 && npm run test:24",
|
|
@@ -93,15 +109,23 @@
|
|
|
93
109
|
"test:claude": "tsx tests/integration/01-all-agents-parallel.ts claude",
|
|
94
110
|
"test:codex": "tsx tests/integration/01-all-agents-parallel.ts codex",
|
|
95
111
|
"test:gemini": "tsx tests/integration/01-all-agents-parallel.ts gemini",
|
|
96
|
-
"test:qwen": "tsx tests/integration/01-all-agents-parallel.ts qwen"
|
|
112
|
+
"test:qwen": "tsx tests/integration/01-all-agents-parallel.ts qwen",
|
|
113
|
+
"test:unit:hosted-cli": "tsx tests/unit/hosted-cli.test.ts",
|
|
114
|
+
"test:unit:hosted-cli-bin": "tsx tests/unit/hosted-cli-bin.test.ts",
|
|
115
|
+
"test:unit:sandbox-artifacts": "tsx tests/unit/sandbox-artifacts.test.ts",
|
|
116
|
+
"test:unit:hosted-tar": "tsx --expose-gc tests/unit/hosted-tar.test.ts",
|
|
117
|
+
"prepack": "node scripts/copy-spec.mjs"
|
|
97
118
|
},
|
|
98
119
|
"dependencies": {
|
|
99
120
|
"@agentclientprotocol/sdk": "^0.5.1",
|
|
100
|
-
"@evolvingmachines/
|
|
101
|
-
"@evolvingmachines/
|
|
102
|
-
"@evolvingmachines/modal": "^0.0.
|
|
121
|
+
"@evolvingmachines/daytona": "^0.0.52-launch-round-1.20260803.cb1be5b",
|
|
122
|
+
"@evolvingmachines/e2b": "^0.0.52-launch-round-1.20260803.cb1be5b",
|
|
123
|
+
"@evolvingmachines/modal": "^0.0.52-launch-round-1.20260803.cb1be5b",
|
|
103
124
|
"ajv": "^8.17.1",
|
|
104
125
|
"p-map": "^7.0.2",
|
|
126
|
+
"smol-toml": "^1.7.0",
|
|
127
|
+
"tar-stream": "^3.1.7",
|
|
128
|
+
"yaml": "^2.9.0",
|
|
105
129
|
"zod": "^3.24.0",
|
|
106
130
|
"zod-to-json-schema": "^3.25.0"
|
|
107
131
|
},
|
|
@@ -119,6 +143,7 @@
|
|
|
119
143
|
},
|
|
120
144
|
"devDependencies": {
|
|
121
145
|
"@types/node": "^22.15.18",
|
|
146
|
+
"@types/tar-stream": "^3.1.4",
|
|
122
147
|
"tsup": "^8.4.0",
|
|
123
148
|
"typescript": "^5.8.3"
|
|
124
149
|
}
|