@meistrari/tela-skills 1.6.2 → 1.7.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@meistrari/tela-skills",
3
- "version": "1.6.2",
3
+ "version": "1.7.0",
4
4
  "description": "Tela API skills for Claude Code",
5
5
  "type": "module",
6
6
  "bin": {
@@ -91,6 +91,30 @@ bun --preload ~/.claude/skills/tela/preload.ts -e "console.log(await tela.listPr
91
91
  - `tela.waitForTestCase(testCaseId)` - Wait for test case to complete
92
92
  - `tela.abortTestCase(testCaseId)` - Abort a running test case
93
93
 
94
+ ### Agent Test Cases
95
+
96
+ Test cases for agents (FCC). Runs are keyed by `(testCaseId, commitHash)` — one run per test case per agent version. Run/continue/run-all return `202 { executionId }` immediately; poll with `waitForAgentTestCaseRun`. Derived statuses: `new`, `running`, `executed`, `success`, `failed`, `error` (terminal: `success`/`failed`/`error`).
97
+
98
+ - `tela.listAgentTestCases(agentId, options?)` - List test cases (pass `commitHash` to attach runs/stats)
99
+ - `tela.createAgentTestCase(agentId, payload)` - Create a test case
100
+ - `tela.updateAgentTestCase(agentId, testId, payload)` - Update a test case
101
+ - `tela.deleteAgentTestCase(agentId, testId)` - Delete a test case
102
+ - `tela.runAgentTestCase(agentId, testId, commitHash)` - Run against an agent version
103
+ - `tela.continueAgentTestCase(agentId, testId, commitHash, message)` - Continue session (multiturn)
104
+ - `tela.runAllAgentTestCases(agentId, { commitHash, testCaseIds? | filters? })` - Batch run (returns `skipped` reasons)
105
+ - `tela.getAgentTestCaseRun(agentId, testId, commitHash)` - Get the run for a commit
106
+ - `tela.listAgentTestCaseRuns(agentId, testId, options?)` - Run history across commits (cursor pagination)
107
+ - `tela.waitForAgentTestCaseRun(agentId, testId, commitHash, options?)` - Poll until terminal status
108
+ - `tela.updateAgentTestCaseAttributeFeedback(agentId, testId, commitHash, attributes)` - Thumbs per attribute (feeds the answer bank)
109
+ - `tela.getAgentTestCaseStats(agentId, commitHash, options?)` - Aggregated stats for a commit
110
+ - `tela.listAgentTestMetrics(agentId)` / `createAgentTestMetric` / `updateAgentTestMetric` / `deleteAgentTestMetric` - Custom metrics over output attributes
111
+ - `tela.listAgentTestCaseTags(agentId)` / `createAgentTestCaseTag` / `updateAgentTestCaseTag` / `deleteAgentTestCaseTag` - Tag CRUD
112
+ - `tela.getAgentTestCaseTags(agentId, testId)` / `addAgentTestCaseTags` / `removeAgentTestCaseTag` - Tag assignment
113
+ - `tela.bulkAddAgentTestCaseTags(agentId, testCaseIds, tagIds)` / `bulkRemoveAgentTestCaseTags(...)` - Bulk tagging
114
+ - `tela.getAgentHistory(agentId, options?)` / `tela.getLatestAgentCommit(agentId)` - Resolve agent versions (commitHash)
115
+ - `tela.buildAgentTestInputs(variables, options?)` / `tela.createAgentTestCasePayload(variables, options?)` - Payload helpers (vault:// values become file inputs)
116
+ - `tela.getAgentTestCaseRunStatus(run)` - Derive run status locally
117
+
94
118
  ### Tasks
95
119
 
96
120
  - `tela.listTasks(options?)` - List tasks with pagination and filtering
@@ -192,6 +216,31 @@ console.log('Variables:', version?.variables?.map(v => v.name))
192
216
  "
193
217
  ```
194
218
 
219
+ ### Create and Run an Agent Test Case
220
+ ```bash
221
+ bun --preload ~/.claude/skills/tela/preload.ts -e "
222
+ const agentId = 'AGENT_ID'
223
+ const vaultRef = await tela.uploadFile('./contract.pdf')
224
+
225
+ const testCase = await tela.createAgentTestCase(agentId, tela.createAgentTestCasePayload({
226
+ document: vaultRef,
227
+ instructions: 'Extract the parties and total amount',
228
+ }, {
229
+ title: 'Contract extraction',
230
+ evaluationInstructions: 'The output must list both parties and the correct total.',
231
+ fileNames: { document: 'contract.pdf' },
232
+ }))
233
+
234
+ const commitHash = await tela.getLatestAgentCommit(agentId)
235
+ await tela.runAgentTestCase(agentId, testCase.id, commitHash)
236
+
237
+ const { run, status } = await tela.waitForAgentTestCaseRun(agentId, testCase.id, commitHash)
238
+ console.log('Status:', status)
239
+ console.log('Output:', run.output)
240
+ console.log('Validation:', run.validationSummary)
241
+ "
242
+ ```
243
+
195
244
  ### List Tasks
196
245
  ```bash
197
246
  bun --preload ~/.claude/skills/tela/preload.ts -e "
@@ -332,7 +381,8 @@ bun localhost.ts "console.log(await tela.listProjects())"
332
381
  ├── prompts.ts # Prompt/Canvas list functions
333
382
  ├── canvas.ts # Canvas create/update functions
334
383
  ├── workflows.ts # Workflow create/update/run/validation functions
335
- ├── test-cases.ts # Test case functions
384
+ ├── test-cases.ts # Test case functions (prompt/canvas)
385
+ ├── agent-test-cases.ts # Agent test case functions (FCC)
336
386
  ├── tasks.ts # Task management functions
337
387
  ├── vault.ts # Vault file management functions
338
388
  └── localhost-utils.ts # Docker discovery, JWT signing, workspace fetching (internal, not on tela global)
@@ -0,0 +1,316 @@
1
+ import { beforeEach, describe, expect, mock, test } from 'bun:test'
2
+
3
+ const apiRequest = mock(async (_endpoint: string, _options?: RequestInit): Promise<unknown> => ({ data: {} }))
4
+
5
+ class ApiRequestError extends Error {
6
+ constructor(public readonly status: number, message: string) {
7
+ super(message)
8
+ this.name = 'ApiRequestError'
9
+ }
10
+ }
11
+
12
+ void mock.module('./common.ts', () => ({
13
+ apiRequest,
14
+ ApiRequestError,
15
+ }))
16
+
17
+ describe('agent-test-cases', () => {
18
+ beforeEach(() => {
19
+ apiRequest.mockClear()
20
+ apiRequest.mockImplementation(async () => ({ data: {} }))
21
+ })
22
+
23
+ test('lists test cases with repeated array query params', async () => {
24
+ const { listAgentTestCases } = await import('./agent-test-cases.ts')
25
+ apiRequest.mockImplementation(async () => ({ data: [] }))
26
+
27
+ await listAgentTestCases('agent-1', {
28
+ commitHash: 'abc123',
29
+ title: 'search',
30
+ status: ['failed', 'error'],
31
+ tagIds: ['t1', 't2'],
32
+ createdBy: ['u1'],
33
+ createdAtSince: '2026-01-01T00:00:00Z',
34
+ })
35
+
36
+ const [endpoint] = apiRequest.mock.calls[0]!
37
+ const query = new URLSearchParams(String(endpoint).split('?')[1])
38
+
39
+ expect(String(endpoint).startsWith('/agent/agent-1/tests?')).toBe(true)
40
+ expect(query.get('commitHash')).toBe('abc123')
41
+ expect(query.get('title')).toBe('search')
42
+ expect(query.getAll('status')).toEqual(['failed', 'error'])
43
+ expect(query.getAll('tagIds')).toEqual(['t1', 't2'])
44
+ expect(query.getAll('createdBy')).toEqual(['u1'])
45
+ expect(query.get('createdAtSince')).toBe('2026-01-01T00:00:00Z')
46
+ })
47
+
48
+ test('omits the query string when no options are given', async () => {
49
+ const { listAgentTestCases } = await import('./agent-test-cases.ts')
50
+ apiRequest.mockImplementation(async () => ({ data: [] }))
51
+
52
+ await listAgentTestCases('agent-1')
53
+
54
+ expect(apiRequest.mock.calls[0]![0]).toBe('/agent/agent-1/tests')
55
+ })
56
+
57
+ test('pre-serializes bodies so vault refs are not rewritten by parsePayload', async () => {
58
+ const { createAgentTestCase } = await import('./agent-test-cases.ts')
59
+
60
+ await createAgentTestCase('agent-1', {
61
+ inputs: [{ type: 'file', name: 'doc', vaultRef: 'vault://abc', filename: 'doc.pdf' }],
62
+ })
63
+
64
+ const [endpoint, options] = apiRequest.mock.calls[0]!
65
+ expect(endpoint).toBe('/agent/agent-1/tests')
66
+ expect(options?.method).toBe('POST')
67
+ expect(typeof options?.body).toBe('string')
68
+
69
+ const body = JSON.parse(String(options?.body))
70
+ expect(body.inputs[0].vaultRef).toBe('vault://abc')
71
+ })
72
+
73
+ test('unwraps the { data } envelope', async () => {
74
+ const { createAgentTestCase } = await import('./agent-test-cases.ts')
75
+ apiRequest.mockImplementation(async () => ({ data: { id: 'tc-1', title: 'Test' } }))
76
+
77
+ const testCase = await createAgentTestCase('agent-1', { title: 'Test' })
78
+
79
+ expect(testCase).toEqual({ id: 'tc-1', title: 'Test' } as any)
80
+ })
81
+
82
+ test('runs a test case against a commit', async () => {
83
+ const { runAgentTestCase } = await import('./agent-test-cases.ts')
84
+ apiRequest.mockImplementation(async () => ({ data: { executionId: 'exec-1' } }))
85
+
86
+ const result = await runAgentTestCase('agent-1', 'tc-1', 'abc123')
87
+
88
+ const [endpoint, options] = apiRequest.mock.calls[0]!
89
+ expect(endpoint).toBe('/agent/agent-1/tests/tc-1/run')
90
+ expect(JSON.parse(String(options?.body))).toEqual({ commitHash: 'abc123' })
91
+ expect(result.executionId).toBe('exec-1')
92
+ })
93
+
94
+ test('continues a session with a message', async () => {
95
+ const { continueAgentTestCase } = await import('./agent-test-cases.ts')
96
+ apiRequest.mockImplementation(async () => ({ data: { executionId: 'exec-2' } }))
97
+
98
+ await continueAgentTestCase('agent-1', 'tc-1', 'abc123', 'next turn')
99
+
100
+ const [endpoint, options] = apiRequest.mock.calls[0]!
101
+ expect(endpoint).toBe('/agent/agent-1/tests/tc-1/continue')
102
+ expect(JSON.parse(String(options?.body))).toEqual({ commitHash: 'abc123', message: 'next turn' })
103
+ })
104
+
105
+ test('sends attribute feedback wrapped in attributes', async () => {
106
+ const { updateAgentTestCaseAttributeFeedback } = await import('./agent-test-cases.ts')
107
+
108
+ await updateAgentTestCaseAttributeFeedback('agent-1', 'tc-1', 'abc123', [
109
+ { path: 'summary', feedback: 1 },
110
+ { path: 'total', feedback: 0, message: 'wrong' },
111
+ ])
112
+
113
+ const [endpoint, options] = apiRequest.mock.calls[0]!
114
+ expect(endpoint).toBe('/agent/agent-1/tests/tc-1/runs/abc123/attribute-feedback')
115
+ expect(options?.method).toBe('PATCH')
116
+
117
+ const body = JSON.parse(String(options?.body))
118
+ expect(body.attributes).toHaveLength(2)
119
+ expect(body.attributes[1]).toEqual({ path: 'total', feedback: 0, message: 'wrong' })
120
+ })
121
+
122
+ test('requires commitHash on the stats endpoint query', async () => {
123
+ const { getAgentTestCaseStats } = await import('./agent-test-cases.ts')
124
+
125
+ await getAgentTestCaseStats('agent-1', 'abc123', { tagIds: ['t1'] })
126
+
127
+ const [endpoint] = apiRequest.mock.calls[0]!
128
+ const query = new URLSearchParams(String(endpoint).split('?')[1])
129
+
130
+ expect(String(endpoint).startsWith('/agent/agent-1/tests/stats?')).toBe(true)
131
+ expect(query.get('commitHash')).toBe('abc123')
132
+ expect(query.getAll('tagIds')).toEqual(['t1'])
133
+ })
134
+
135
+ test('run history keeps pagination fields from the response envelope', async () => {
136
+ const { listAgentTestCaseRuns } = await import('./agent-test-cases.ts')
137
+ // This endpoint's envelope is { data: entries, nextCursor, totalRuns } —
138
+ // pagination fields live beside data, not inside it
139
+ apiRequest.mockImplementation(async () => ({
140
+ data: [{ commitHash: 'abc123', score: 1 }],
141
+ nextCursor: 'cursor-1',
142
+ totalRuns: 5,
143
+ }))
144
+
145
+ const result = await listAgentTestCaseRuns('agent-1', 'tc-1', { limit: 1 })
146
+
147
+ expect(apiRequest.mock.calls[0]![0]).toBe('/agent/agent-1/tests/tc-1/runs?limit=1')
148
+ expect(result.data).toHaveLength(1)
149
+ expect(result.nextCursor).toBe('cursor-1')
150
+ expect(result.totalRuns).toBe(5)
151
+ })
152
+
153
+ describe('buildAgentTestInputs', () => {
154
+ test('splits text and vault file values', async () => {
155
+ const { buildAgentTestInputs } = await import('./agent-test-cases.ts')
156
+
157
+ const inputs = buildAgentTestInputs({
158
+ context: 'Some text',
159
+ document: 'vault://abc123',
160
+ }, { fileNames: { document: 'report.pdf' } })
161
+
162
+ expect(inputs).toEqual([
163
+ { type: 'text', name: 'context', content: 'Some text' },
164
+ { type: 'file', name: 'document', vaultRef: 'vault://abc123', filename: 'report.pdf' },
165
+ ])
166
+ })
167
+
168
+ test('defaults the filename to the input name', async () => {
169
+ const { buildAgentTestInputs } = await import('./agent-test-cases.ts')
170
+
171
+ const inputs = buildAgentTestInputs({ document: 'vault://abc123' })
172
+
173
+ expect(inputs[0]).toEqual({ type: 'file', name: 'document', vaultRef: 'vault://abc123', filename: 'document' })
174
+ })
175
+ })
176
+
177
+ describe('createAgentTestCasePayload', () => {
178
+ test('builds payload with expectations and derived title', async () => {
179
+ const { createAgentTestCasePayload } = await import('./agent-test-cases.ts')
180
+
181
+ const payload = createAgentTestCasePayload({ input: 'Hello world, this is a long input value' }, {
182
+ evaluationInstructions: 'Must greet back',
183
+ })
184
+
185
+ expect(payload.title).toBe(`Test: ${'Hello world, this is a long input value'.slice(0, 30)}...`)
186
+ expect(payload.evaluationInstructions).toBe('Must greet back')
187
+ expect(payload.inputs).toHaveLength(1)
188
+ })
189
+
190
+ test('falls back to the first filename for file-only cases', async () => {
191
+ const { createAgentTestCasePayload } = await import('./agent-test-cases.ts')
192
+
193
+ const payload = createAgentTestCasePayload({ doc: 'vault://abc' }, { fileNames: { doc: 'contract.pdf' } })
194
+
195
+ expect(payload.title).toBe('Test: contract.pdf')
196
+ })
197
+ })
198
+
199
+ describe('getAgentTestCaseRunStatus', () => {
200
+ const baseRun = {
201
+ executionSessionId: null as string | null,
202
+ executionStatus: null as any,
203
+ evaluationSessionId: null as string | null,
204
+ evaluationStatus: null as any,
205
+ feedback: null as 0 | 1 | null,
206
+ }
207
+
208
+ test('derives every status branch', async () => {
209
+ const { getAgentTestCaseRunStatus } = await import('./agent-test-cases.ts')
210
+
211
+ expect(getAgentTestCaseRunStatus(null)).toBe('new')
212
+ expect(getAgentTestCaseRunStatus({ ...baseRun, executionStatus: 'error' })).toBe('error')
213
+ expect(getAgentTestCaseRunStatus({ ...baseRun, executionStatus: 'completed', evaluationStatus: 'error' })).toBe('error')
214
+ expect(getAgentTestCaseRunStatus({ ...baseRun, executionStatus: 'running' })).toBe('running')
215
+ expect(getAgentTestCaseRunStatus({ ...baseRun, executionStatus: 'completed', evaluationSessionId: 's', evaluationStatus: 'running' })).toBe('running')
216
+ expect(getAgentTestCaseRunStatus({ ...baseRun, executionStatus: 'completed' })).toBe('executed')
217
+ expect(getAgentTestCaseRunStatus({ ...baseRun, executionStatus: 'completed', evaluationSessionId: 's', evaluationStatus: 'completed', feedback: 1 })).toBe('success')
218
+ expect(getAgentTestCaseRunStatus({ ...baseRun, executionStatus: 'completed', evaluationSessionId: 's', evaluationStatus: 'completed', feedback: 0 })).toBe('failed')
219
+ expect(getAgentTestCaseRunStatus({ ...baseRun, executionStatus: 'completed', evaluationSessionId: 's', evaluationStatus: 'failed' })).toBe('failed')
220
+ expect(getAgentTestCaseRunStatus({ ...baseRun, executionSessionId: 's' })).toBe('executed')
221
+ })
222
+ })
223
+
224
+ describe('waitForAgentTestCaseRun', () => {
225
+ test('retries 404s until the run appears, then resolves on terminal status', async () => {
226
+ const { waitForAgentTestCaseRun } = await import('./agent-test-cases.ts')
227
+
228
+ let calls = 0
229
+ apiRequest.mockImplementation(async () => {
230
+ calls++
231
+ if (calls === 1)
232
+ throw new ApiRequestError(404, 'API request failed: 404 Not Found - {}')
233
+ if (calls === 2) {
234
+ return { data: { executionStatus: 'running', executionSessionId: 's', evaluationSessionId: null, evaluationStatus: null, feedback: null } }
235
+ }
236
+ return { data: { executionStatus: 'completed', executionSessionId: 's', evaluationSessionId: 'e', evaluationStatus: 'completed', feedback: 1 } }
237
+ })
238
+
239
+ const { status } = await waitForAgentTestCaseRun('agent-1', 'tc-1', 'abc123', { intervalMs: 1 })
240
+
241
+ expect(status).toBe('success')
242
+ expect(calls).toBe(3)
243
+ })
244
+
245
+ test('stops at executed when until is executed', async () => {
246
+ const { waitForAgentTestCaseRun } = await import('./agent-test-cases.ts')
247
+
248
+ apiRequest.mockImplementation(async () => ({
249
+ data: { executionStatus: 'completed', executionSessionId: 's', evaluationSessionId: null, evaluationStatus: null, feedback: null },
250
+ }))
251
+
252
+ const { status } = await waitForAgentTestCaseRun('agent-1', 'tc-1', 'abc123', { intervalMs: 1, until: 'executed' })
253
+
254
+ expect(status).toBe('executed')
255
+ })
256
+
257
+ test('throws after maxAttempts without a terminal status', async () => {
258
+ const { waitForAgentTestCaseRun } = await import('./agent-test-cases.ts')
259
+
260
+ apiRequest.mockImplementation(async () => ({
261
+ data: { executionStatus: 'running', executionSessionId: 's', evaluationSessionId: null, evaluationStatus: null, feedback: null },
262
+ }))
263
+
264
+ expect(waitForAgentTestCaseRun('agent-1', 'tc-1', 'abc123', { intervalMs: 1, maxAttempts: 2 }))
265
+ .rejects
266
+ .toThrow('did not complete within timeout')
267
+ })
268
+
269
+ test('rethrows non-404 API errors', async () => {
270
+ const { waitForAgentTestCaseRun } = await import('./agent-test-cases.ts')
271
+
272
+ apiRequest.mockImplementation(async () => {
273
+ throw new ApiRequestError(500, 'API request failed: 500 Internal Server Error - {}')
274
+ })
275
+
276
+ expect(waitForAgentTestCaseRun('agent-1', 'tc-1', 'abc123', { intervalMs: 1 }))
277
+ .rejects
278
+ .toThrow('500')
279
+ })
280
+
281
+ test('rethrows untyped errors even when the message mentions 404', async () => {
282
+ const { waitForAgentTestCaseRun } = await import('./agent-test-cases.ts')
283
+
284
+ apiRequest.mockImplementation(async () => {
285
+ throw new Error('fetch failed with 404 somewhere')
286
+ })
287
+
288
+ expect(waitForAgentTestCaseRun('agent-1', 'tc-1', 'abc123', { intervalMs: 1 }))
289
+ .rejects
290
+ .toThrow('fetch failed')
291
+ })
292
+ })
293
+
294
+ describe('getLatestAgentCommit', () => {
295
+ test('returns the newest commit hash', async () => {
296
+ const { getLatestAgentCommit } = await import('./agent-test-cases.ts')
297
+ apiRequest.mockImplementation(async () => ({
298
+ data: { commits: [{ commitHash: 'abc123' }], branch: 'main', pagination: { page: 1, limit: 1, totalCount: 5, hasMore: true } },
299
+ }))
300
+
301
+ const hash = await getLatestAgentCommit('agent-1')
302
+
303
+ expect(apiRequest.mock.calls[0]![0]).toBe('/agent/agent-1/history?limit=1')
304
+ expect(hash).toBe('abc123')
305
+ })
306
+
307
+ test('throws when the agent has no commits', async () => {
308
+ const { getLatestAgentCommit } = await import('./agent-test-cases.ts')
309
+ apiRequest.mockImplementation(async () => ({
310
+ data: { commits: [], branch: 'main', pagination: { page: 1, limit: 1, totalCount: 0, hasMore: false } },
311
+ }))
312
+
313
+ expect(getLatestAgentCommit('agent-1')).rejects.toThrow('has no commits')
314
+ })
315
+ })
316
+ })
@@ -0,0 +1,831 @@
1
+ import { ApiRequestError, apiRequest } from './common.ts'
2
+
3
+ // ============================================================================
4
+ // Types
5
+ // ============================================================================
6
+
7
+ /**
8
+ * A single named input for an agent test case.
9
+ * Text inputs carry inline content; file inputs reference a Vault file.
10
+ */
11
+ export type AgentTestInput
12
+ = | { type: 'text', name: string, content: string }
13
+ | { type: 'file', name: string, vaultRef: string, filename: string, metadata?: string }
14
+
15
+ export interface AgentTestCaseValidationReferenceFile {
16
+ index: number
17
+ name: string
18
+ mimeType?: string | null
19
+ vaultRef: string
20
+ url?: string
21
+ }
22
+
23
+ export interface AgentTestCaseAnswers {
24
+ good: unknown[]
25
+ bad: unknown[]
26
+ }
27
+
28
+ export interface AgentTestCaseTag {
29
+ testCaseTagAssignmentId?: string
30
+ id: string
31
+ name: string
32
+ color: string
33
+ promptId: string | null
34
+ agentId: string | null
35
+ workspaceId: string
36
+ createdAt: string
37
+ updatedAt: string
38
+ deletedAt: string | null
39
+ }
40
+
41
+ export interface AgentTestCase {
42
+ id: string
43
+ agentId: string
44
+ workspaceId: string
45
+ title: string
46
+ inputs: AgentTestInput[]
47
+ expectedOutput: Record<string, unknown> | string | null
48
+ validationReferenceFiles: AgentTestCaseValidationReferenceFile[]
49
+ evaluationInstructions: string | null
50
+ answers: Record<string, AgentTestCaseAnswers>
51
+ metadata: Record<string, unknown>
52
+ tags: AgentTestCaseTag[]
53
+ createdBy: string
54
+ createdAt: string
55
+ updatedAt: string
56
+ }
57
+
58
+ export type AgentTestCaseFeedback = 0 | 1 | null
59
+ export type AgentTestCaseValidationType = 'good' | 'bad' | 'warning' | 'missing'
60
+
61
+ export interface AgentTestCaseValidationSummaryEntry {
62
+ type: AgentTestCaseValidationType
63
+ source?: 'llm-judge' | 'manual' | 'answer-match'
64
+ message?: string
65
+ matchedAnswer?: unknown
66
+ }
67
+
68
+ export interface AgentTestCaseRun {
69
+ id: string
70
+ testId: string
71
+ commitHash: string
72
+ executionSessionId: string | null
73
+ executionStatus: 'pending' | 'running' | 'completed' | 'error' | null
74
+ executionError: string | null
75
+ savedInputs: AgentTestInput[]
76
+ savedChatMessages: Array<{
77
+ content: string
78
+ stepEndIndex: number
79
+ attachments?: Array<{ vaultRef: string, fileName: string, fileType: string }>
80
+ output?: unknown
81
+ duration?: number
82
+ }>
83
+ output: Record<string, unknown> | string | null
84
+ executionUsage: Record<string, unknown> | null
85
+ evaluationSessionId: string | null
86
+ evaluationInputSnapshot: unknown
87
+ evaluationRequirements: string
88
+ evaluationStatus: 'pending' | 'running' | 'completed' | 'failed' | 'error' | null
89
+ evaluationError: string | null
90
+ feedback: AgentTestCaseFeedback
91
+ attributeFeedback: Record<string, 0 | 1>
92
+ validationSummary: Record<string, AgentTestCaseValidationSummaryEntry>
93
+ evaluatedAt: string | null
94
+ createdAt: string
95
+ updatedAt: string
96
+ }
97
+
98
+ /**
99
+ * Derived run status — not a stored column. Computed from execution/evaluation
100
+ * state by getAgentTestCaseRunStatus. Terminal: success, failed, error.
101
+ */
102
+ export type AgentTestCaseStatus = 'new' | 'running' | 'executed' | 'success' | 'failed' | 'error'
103
+
104
+ export interface AgentTestCaseCounters {
105
+ total: number
106
+ good: number
107
+ bad: number
108
+ pending: number
109
+ }
110
+
111
+ export interface AgentTestMetric {
112
+ id: string
113
+ agentId: string
114
+ workspaceId: string
115
+ name: string
116
+ attributeKeys: string[]
117
+ position: number
118
+ createdBy: string
119
+ updatedBy: string | null
120
+ createdAt: string
121
+ updatedAt: string
122
+ }
123
+
124
+ export type AgentTestMetricStatus = 'completed' | 'pending' | 'partial' | 'unavailable'
125
+
126
+ export interface AgentTestMetricStats {
127
+ id: string
128
+ name: string
129
+ attributeKeys: string[]
130
+ availableAttributeKeys: string[]
131
+ missingAttributeKeys: string[]
132
+ stats: AgentTestCaseCounters
133
+ reviewedStats?: { total: number, good: number, bad: number }
134
+ score: number | null
135
+ status: AgentTestMetricStatus
136
+ }
137
+
138
+ export interface AgentTestCaseWithRun {
139
+ test: AgentTestCase
140
+ run: AgentTestCaseRun | null
141
+ stats?: AgentTestCaseCounters
142
+ metricStats?: AgentTestMetricStats[]
143
+ }
144
+
145
+ export interface AgentTestCaseStats {
146
+ agentId: string
147
+ commitHash: string
148
+ status: 'pending' | 'completed'
149
+ stats: AgentTestCaseCounters
150
+ reviewedStats: { total: number, good: number, bad: number }
151
+ testCases: {
152
+ total: number
153
+ executed: number
154
+ evaluated: number
155
+ passed: number
156
+ failed: number
157
+ pendingExecution: number
158
+ pendingEvaluation: number
159
+ running: number
160
+ error: number
161
+ }
162
+ metrics: AgentTestMetricStats[]
163
+ }
164
+
165
+ export interface AgentTestCaseRunHistoryEntry {
166
+ commitHash: string
167
+ createdAt: string
168
+ score: number
169
+ total: number
170
+ good: number
171
+ bad: number
172
+ pending: number
173
+ attributeOutcomes: Record<string, unknown>
174
+ }
175
+
176
+ export interface ListAgentTestCaseRunsResult {
177
+ data: AgentTestCaseRunHistoryEntry[]
178
+ nextCursor: string | null
179
+ totalRuns: number
180
+ }
181
+
182
+ export interface AgentCommit {
183
+ commitHash: string
184
+ shortHash: string
185
+ message: string
186
+ createdAt: string
187
+ author: { name: string, email: string, avatarUrl?: string }
188
+ }
189
+
190
+ export interface AgentHistory {
191
+ commits: AgentCommit[]
192
+ branch: string
193
+ pagination: { page: number, limit: number, totalCount: number, hasMore: boolean }
194
+ }
195
+
196
+ // ============================================================================
197
+ // Payloads & Options
198
+ // ============================================================================
199
+
200
+ export interface CreateAgentTestCasePayload {
201
+ title?: string
202
+ inputs?: AgentTestInput[]
203
+ expectedOutput?: Record<string, unknown> | string | null
204
+ validationReferenceFiles?: AgentTestCaseValidationReferenceFile[]
205
+ evaluationInstructions?: string | null
206
+ answers?: Record<string, AgentTestCaseAnswers>
207
+ metadata?: Record<string, unknown>
208
+ }
209
+
210
+ export type UpdateAgentTestCasePayload = CreateAgentTestCasePayload
211
+
212
+ export interface ListAgentTestCasesOptions {
213
+ /** Attach the run for this agent commit to each test case (enables stats) */
214
+ commitHash?: string
215
+ /** Search by test case title */
216
+ title?: string
217
+ /** Filter by creator user IDs */
218
+ createdBy?: string[]
219
+ createdAtSince?: string
220
+ createdAtUntil?: string
221
+ updatedAtSince?: string
222
+ updatedAtUntil?: string
223
+ /** Filter by derived run status (requires commitHash) */
224
+ status?: AgentTestCaseStatus[]
225
+ /** Filter by tag IDs */
226
+ tagIds?: string[]
227
+ }
228
+
229
+ export interface RunAllAgentTestCasesPayload {
230
+ commitHash: string
231
+ /** Explicit test case IDs — mutually exclusive with filters */
232
+ testCaseIds?: string[]
233
+ /** Filter-based selection — mutually exclusive with testCaseIds */
234
+ filters?: {
235
+ title?: string
236
+ createdBy?: string[]
237
+ createdAtSince?: string
238
+ createdAtUntil?: string
239
+ updatedAtSince?: string
240
+ updatedAtUntil?: string
241
+ tagIds?: string[]
242
+ }
243
+ }
244
+
245
+ export type AgentTestCaseSkipReason
246
+ = | 'already-running'
247
+ | 'already-passed'
248
+ | 'already-failed'
249
+ | 'missing-input'
250
+ | 'unknown-input'
251
+ | 'invalid-input'
252
+ | 'not-accessible'
253
+ | 'usage-limit'
254
+
255
+ export interface RunAllAgentTestCasesResult {
256
+ executionId: string
257
+ executionIds: string[]
258
+ testCaseIds: string[]
259
+ skipped: Array<{ testCaseId: string, reason: AgentTestCaseSkipReason }>
260
+ }
261
+
262
+ export interface AgentTestCaseAttributeFeedback {
263
+ path: string
264
+ feedback: AgentTestCaseFeedback
265
+ type?: AgentTestCaseValidationType
266
+ message?: string
267
+ }
268
+
269
+ export interface CreateAgentTestMetricPayload {
270
+ name: string
271
+ attributeKeys: string[]
272
+ position?: number
273
+ }
274
+
275
+ export type UpdateAgentTestMetricPayload = Partial<CreateAgentTestMetricPayload>
276
+
277
+ // ============================================================================
278
+ // Internal helpers
279
+ // ============================================================================
280
+
281
+ // Agent test case endpoints use `vaultRef` fields that must NOT be rewritten
282
+ // by parsePayload's vault:// → { file_url } expansion, so every body is
283
+ // pre-serialized with JSON.stringify (string bodies skip parsePayload).
284
+ function jsonBody(payload: unknown): string {
285
+ return JSON.stringify(payload)
286
+ }
287
+
288
+ function buildListQuery(options?: ListAgentTestCasesOptions): string {
289
+ const params = new URLSearchParams()
290
+
291
+ if (options?.commitHash)
292
+ params.set('commitHash', options.commitHash)
293
+ if (options?.title)
294
+ params.set('title', options.title)
295
+ if (options?.createdAtSince)
296
+ params.set('createdAtSince', options.createdAtSince)
297
+ if (options?.createdAtUntil)
298
+ params.set('createdAtUntil', options.createdAtUntil)
299
+ if (options?.updatedAtSince)
300
+ params.set('updatedAtSince', options.updatedAtSince)
301
+ if (options?.updatedAtUntil)
302
+ params.set('updatedAtUntil', options.updatedAtUntil)
303
+ // Arrays use repeated keys (Elysia query array format)
304
+ for (const v of options?.createdBy ?? []) params.append('createdBy', v)
305
+ for (const v of options?.status ?? []) params.append('status', v)
306
+ for (const v of options?.tagIds ?? []) params.append('tagIds', v)
307
+
308
+ const qs = params.toString()
309
+ return qs ? `?${qs}` : ''
310
+ }
311
+
312
+ // ============================================================================
313
+ // Test Cases (CRUD)
314
+ // ============================================================================
315
+
316
+ /**
317
+ * List agent test cases with optional filters.
318
+ * Pass commitHash to attach that commit's run (and stats) to each test case.
319
+ * Returns all matching cases (no pagination on this endpoint).
320
+ */
321
+ export async function listAgentTestCases(
322
+ agentId: string,
323
+ options?: ListAgentTestCasesOptions,
324
+ ): Promise<AgentTestCaseWithRun[]> {
325
+ const { data } = await apiRequest<{ data: AgentTestCaseWithRun[] }>(
326
+ `/agent/${agentId}/tests${buildListQuery(options)}`,
327
+ )
328
+ return data
329
+ }
330
+
331
+ /**
332
+ * Create a new test case for an agent.
333
+ * Use createAgentTestCasePayload/buildAgentTestInputs to build inputs.
334
+ */
335
+ export async function createAgentTestCase(
336
+ agentId: string,
337
+ payload: CreateAgentTestCasePayload = {},
338
+ ): Promise<AgentTestCase> {
339
+ const { data } = await apiRequest<{ data: AgentTestCase }>(`/agent/${agentId}/tests`, {
340
+ method: 'POST',
341
+ body: jsonBody(payload),
342
+ })
343
+ return data
344
+ }
345
+
346
+ /**
347
+ * Update an existing agent test case (title, inputs, expectations, etc.)
348
+ */
349
+ export async function updateAgentTestCase(
350
+ agentId: string,
351
+ testId: string,
352
+ payload: UpdateAgentTestCasePayload,
353
+ ): Promise<AgentTestCase> {
354
+ const { data } = await apiRequest<{ data: AgentTestCase }>(`/agent/${agentId}/tests/${testId}`, {
355
+ method: 'PATCH',
356
+ body: jsonBody(payload),
357
+ })
358
+ return data
359
+ }
360
+
361
+ /**
362
+ * Delete an agent test case (soft delete)
363
+ */
364
+ export async function deleteAgentTestCase(agentId: string, testId: string): Promise<void> {
365
+ await apiRequest(`/agent/${agentId}/tests/${testId}`, { method: 'DELETE' })
366
+ }
367
+
368
+ // ============================================================================
369
+ // Execution & Evaluation
370
+ // ============================================================================
371
+
372
+ /**
373
+ * Run a test case against an agent version (commitHash).
374
+ * Returns 202 immediately — poll getAgentTestCaseRun / waitForAgentTestCaseRun.
375
+ * One run exists per (testId, commitHash); re-running replaces it.
376
+ */
377
+ export async function runAgentTestCase(
378
+ agentId: string,
379
+ testId: string,
380
+ commitHash: string,
381
+ ): Promise<{ executionId: string }> {
382
+ const { data } = await apiRequest<{ data: { executionId: string } }>(
383
+ `/agent/${agentId}/tests/${testId}/run`,
384
+ { method: 'POST', body: jsonBody({ commitHash }) },
385
+ )
386
+ return data
387
+ }
388
+
389
+ /**
390
+ * Continue a test case session with a new user message (multiturn).
391
+ * Requires an existing, non-running run for the commit. Re-evaluates after the turn.
392
+ */
393
+ export async function continueAgentTestCase(
394
+ agentId: string,
395
+ testId: string,
396
+ commitHash: string,
397
+ message: string,
398
+ ): Promise<{ executionId: string }> {
399
+ const { data } = await apiRequest<{ data: { executionId: string } }>(
400
+ `/agent/${agentId}/tests/${testId}/continue`,
401
+ { method: 'POST', body: jsonBody({ commitHash, message }) },
402
+ )
403
+ return data
404
+ }
405
+
406
+ /**
407
+ * Run all (or a subset of) test cases against an agent version in batch.
408
+ * Select via explicit testCaseIds OR filters (mutually exclusive).
409
+ * Already-passed/failed/running cases are skipped with a reason.
410
+ */
411
+ export async function runAllAgentTestCases(
412
+ agentId: string,
413
+ payload: RunAllAgentTestCasesPayload,
414
+ ): Promise<RunAllAgentTestCasesResult> {
415
+ const { data } = await apiRequest<{ data: RunAllAgentTestCasesResult }>(
416
+ `/agent/${agentId}/tests/run-all`,
417
+ { method: 'POST', body: jsonBody(payload) },
418
+ )
419
+ return data
420
+ }
421
+
422
+ /**
423
+ * Get the run of a test case for a specific agent commit
424
+ */
425
+ export async function getAgentTestCaseRun(
426
+ agentId: string,
427
+ testId: string,
428
+ commitHash: string,
429
+ ): Promise<AgentTestCaseRun> {
430
+ const { data } = await apiRequest<{ data: AgentTestCaseRun }>(
431
+ `/agent/${agentId}/tests/${testId}/runs/${commitHash}`,
432
+ )
433
+ return data
434
+ }
435
+
436
+ /**
437
+ * List a test case's run history across commits (keyset pagination)
438
+ */
439
+ export async function listAgentTestCaseRuns(
440
+ agentId: string,
441
+ testId: string,
442
+ options?: { limit?: number, cursor?: string },
443
+ ): Promise<ListAgentTestCaseRunsResult> {
444
+ const params = new URLSearchParams()
445
+ if (options?.limit !== undefined)
446
+ params.set('limit', String(options.limit))
447
+ if (options?.cursor)
448
+ params.set('cursor', options.cursor)
449
+ const qs = params.toString()
450
+
451
+ // Unlike the other endpoints, pagination fields live on the envelope itself:
452
+ // the API returns { data: entries, nextCursor, totalRuns }
453
+ return await apiRequest<ListAgentTestCaseRunsResult>(
454
+ `/agent/${agentId}/tests/${testId}/runs${qs ? `?${qs}` : ''}`,
455
+ )
456
+ }
457
+
458
+ /**
459
+ * Record manual thumbs up/down per output attribute path on a run.
460
+ * Also feeds the test case's answer bank (answers[path].good/bad), which
461
+ * auto-classifies future runs via answer-match. feedback: 1=good, 0=bad, null=clear.
462
+ */
463
+ export async function updateAgentTestCaseAttributeFeedback(
464
+ agentId: string,
465
+ testId: string,
466
+ commitHash: string,
467
+ attributes: AgentTestCaseAttributeFeedback[],
468
+ ): Promise<AgentTestCaseRun> {
469
+ const { data } = await apiRequest<{ data: AgentTestCaseRun }>(
470
+ `/agent/${agentId}/tests/${testId}/runs/${commitHash}/attribute-feedback`,
471
+ { method: 'PATCH', body: jsonBody({ attributes }) },
472
+ )
473
+ return data
474
+ }
475
+
476
+ // ============================================================================
477
+ // Stats & Metrics
478
+ // ============================================================================
479
+
480
+ /**
481
+ * Aggregated test case stats for an agent commit (scores per output attribute,
482
+ * counters, custom metrics). Accepts the same filters as listAgentTestCases.
483
+ */
484
+ export async function getAgentTestCaseStats(
485
+ agentId: string,
486
+ commitHash: string,
487
+ options?: Omit<ListAgentTestCasesOptions, 'commitHash'>,
488
+ ): Promise<AgentTestCaseStats> {
489
+ const { data } = await apiRequest<{ data: AgentTestCaseStats }>(
490
+ `/agent/${agentId}/tests/stats${buildListQuery({ ...options, commitHash })}`,
491
+ )
492
+ return data
493
+ }
494
+
495
+ /**
496
+ * List custom test metrics for an agent
497
+ */
498
+ export async function listAgentTestMetrics(agentId: string): Promise<AgentTestMetric[]> {
499
+ const { data } = await apiRequest<{ data: AgentTestMetric[] }>(`/agent/${agentId}/test-metrics`)
500
+ return data
501
+ }
502
+
503
+ /**
504
+ * Create a custom test metric grouping output attributes.
505
+ * attributeKeys must exist in the agent's output-format.json.
506
+ */
507
+ export async function createAgentTestMetric(
508
+ agentId: string,
509
+ payload: CreateAgentTestMetricPayload,
510
+ ): Promise<AgentTestMetric> {
511
+ const { data } = await apiRequest<{ data: AgentTestMetric }>(`/agent/${agentId}/test-metrics`, {
512
+ method: 'POST',
513
+ body: jsonBody(payload),
514
+ })
515
+ return data
516
+ }
517
+
518
+ /**
519
+ * Update a custom test metric
520
+ */
521
+ export async function updateAgentTestMetric(
522
+ agentId: string,
523
+ metricId: string,
524
+ payload: UpdateAgentTestMetricPayload,
525
+ ): Promise<AgentTestMetric> {
526
+ const { data } = await apiRequest<{ data: AgentTestMetric }>(
527
+ `/agent/${agentId}/test-metrics/${metricId}`,
528
+ { method: 'PATCH', body: jsonBody(payload) },
529
+ )
530
+ return data
531
+ }
532
+
533
+ /**
534
+ * Delete a custom test metric (soft delete)
535
+ */
536
+ export async function deleteAgentTestMetric(agentId: string, metricId: string): Promise<void> {
537
+ await apiRequest(`/agent/${agentId}/test-metrics/${metricId}`, { method: 'DELETE' })
538
+ }
539
+
540
+ // ============================================================================
541
+ // Tags
542
+ // ============================================================================
543
+
544
+ /**
545
+ * List an agent's test case tags
546
+ */
547
+ export async function listAgentTestCaseTags(agentId: string): Promise<AgentTestCaseTag[]> {
548
+ const { data } = await apiRequest<{ data: AgentTestCaseTag[] }>(`/agent/${agentId}/tags`)
549
+ return data
550
+ }
551
+
552
+ /**
553
+ * Create a tag scoped to an agent
554
+ */
555
+ export async function createAgentTestCaseTag(
556
+ agentId: string,
557
+ payload: { name: string, color: string },
558
+ ): Promise<AgentTestCaseTag> {
559
+ const { data } = await apiRequest<{ data: AgentTestCaseTag }>(`/agent/${agentId}/tags`, {
560
+ method: 'POST',
561
+ body: jsonBody(payload),
562
+ })
563
+ return data
564
+ }
565
+
566
+ /**
567
+ * Update a tag's name/color
568
+ */
569
+ export async function updateAgentTestCaseTag(
570
+ agentId: string,
571
+ tagId: string,
572
+ payload: { name?: string, color?: string },
573
+ ): Promise<AgentTestCaseTag> {
574
+ const { data } = await apiRequest<{ data: AgentTestCaseTag }>(`/agent/${agentId}/tags/${tagId}`, {
575
+ method: 'PATCH',
576
+ body: jsonBody(payload),
577
+ })
578
+ return data
579
+ }
580
+
581
+ /**
582
+ * Delete a tag (removes all its assignments)
583
+ */
584
+ export async function deleteAgentTestCaseTag(agentId: string, tagId: string): Promise<void> {
585
+ await apiRequest(`/agent/${agentId}/tags/${tagId}`, { method: 'DELETE' })
586
+ }
587
+
588
+ /**
589
+ * List tags assigned to a test case
590
+ */
591
+ export async function getAgentTestCaseTags(
592
+ agentId: string,
593
+ testId: string,
594
+ ): Promise<AgentTestCaseTag[]> {
595
+ const { data } = await apiRequest<{ data: AgentTestCaseTag[] }>(
596
+ `/agent/${agentId}/tests/${testId}/tags`,
597
+ )
598
+ return data
599
+ }
600
+
601
+ /**
602
+ * Assign tags to a test case
603
+ */
604
+ export async function addAgentTestCaseTags(
605
+ agentId: string,
606
+ testId: string,
607
+ tagIds: string[],
608
+ ): Promise<void> {
609
+ await apiRequest(`/agent/${agentId}/tests/${testId}/tags`, {
610
+ method: 'POST',
611
+ body: jsonBody({ tagIds }),
612
+ })
613
+ }
614
+
615
+ /**
616
+ * Remove one tag assignment from a test case
617
+ */
618
+ export async function removeAgentTestCaseTag(
619
+ agentId: string,
620
+ testId: string,
621
+ tagId: string,
622
+ ): Promise<void> {
623
+ await apiRequest(`/agent/${agentId}/tests/${testId}/tags/${tagId}`, { method: 'DELETE' })
624
+ }
625
+
626
+ /**
627
+ * Assign tags to many test cases at once
628
+ */
629
+ export async function bulkAddAgentTestCaseTags(
630
+ agentId: string,
631
+ testCaseIds: string[],
632
+ tagIds: string[],
633
+ ): Promise<void> {
634
+ await apiRequest(`/agent/${agentId}/tests/bulk-add-tags`, {
635
+ method: 'POST',
636
+ body: jsonBody({ testCaseIds, tagIds }),
637
+ })
638
+ }
639
+
640
+ /**
641
+ * Remove tags from many test cases at once
642
+ */
643
+ export async function bulkRemoveAgentTestCaseTags(
644
+ agentId: string,
645
+ testCaseIds: string[],
646
+ tagIds: string[],
647
+ ): Promise<void> {
648
+ await apiRequest(`/agent/${agentId}/tests/bulk-remove-tags`, {
649
+ method: 'POST',
650
+ body: jsonBody({ testCaseIds, tagIds }),
651
+ })
652
+ }
653
+
654
+ // ============================================================================
655
+ // Agent version helpers
656
+ // ============================================================================
657
+
658
+ /**
659
+ * Get an agent's commit history (agent versions)
660
+ */
661
+ export async function getAgentHistory(
662
+ agentId: string,
663
+ options?: { branch?: string, page?: number, limit?: number },
664
+ ): Promise<AgentHistory> {
665
+ const params = new URLSearchParams()
666
+ if (options?.branch)
667
+ params.set('branch', options.branch)
668
+ if (options?.page !== undefined)
669
+ params.set('page', String(options.page))
670
+ if (options?.limit !== undefined)
671
+ params.set('limit', String(options.limit))
672
+ const qs = params.toString()
673
+
674
+ const { data } = await apiRequest<{ data: AgentHistory }>(
675
+ `/agent/${agentId}/history${qs ? `?${qs}` : ''}`,
676
+ )
677
+ return data
678
+ }
679
+
680
+ /**
681
+ * Resolve the agent's latest commitHash — the version test cases run against
682
+ */
683
+ export async function getLatestAgentCommit(agentId: string): Promise<string> {
684
+ const history = await getAgentHistory(agentId, { limit: 1 })
685
+ const commitHash = history.commits[0]?.commitHash
686
+ if (!commitHash) {
687
+ throw new Error(`Agent ${agentId} has no commits`)
688
+ }
689
+ return commitHash
690
+ }
691
+
692
+ // ============================================================================
693
+ // Helper Functions
694
+ // ============================================================================
695
+
696
+ /**
697
+ * Build the inputs array from a simple Record of variable values.
698
+ * Values starting with vault:// become file inputs; everything else is text.
699
+ * fileNames maps input names to display filenames (defaults to the input name).
700
+ */
701
+ export function buildAgentTestInputs(
702
+ variables: Record<string, string>,
703
+ options?: { fileNames?: Record<string, string> },
704
+ ): AgentTestInput[] {
705
+ return Object.entries(variables).map(([name, value]) => {
706
+ if (typeof value === 'string' && value.startsWith('vault://')) {
707
+ return {
708
+ type: 'file' as const,
709
+ name,
710
+ vaultRef: value,
711
+ filename: options?.fileNames?.[name] ?? name,
712
+ }
713
+ }
714
+ return { type: 'text' as const, name, content: value }
715
+ })
716
+ }
717
+
718
+ /**
719
+ * Create an agent test case payload from variable values.
720
+ * The evaluator needs at least one expectation source to run:
721
+ * expectedOutput, validationReferenceFiles, or evaluationInstructions.
722
+ */
723
+ export function createAgentTestCasePayload(
724
+ variables: Record<string, string>,
725
+ options?: {
726
+ title?: string
727
+ expectedOutput?: Record<string, unknown> | string
728
+ evaluationInstructions?: string
729
+ validationReferenceFiles?: AgentTestCaseValidationReferenceFile[]
730
+ metadata?: Record<string, unknown>
731
+ fileNames?: Record<string, string>
732
+ },
733
+ ): CreateAgentTestCasePayload {
734
+ const inputs = buildAgentTestInputs(variables, { fileNames: options?.fileNames })
735
+
736
+ const firstText = inputs.find(i => i.type === 'text' && i.content)
737
+ const firstFile = inputs.find(i => i.type === 'file')
738
+ const fallbackTitle = firstText?.type === 'text'
739
+ ? `Test: ${firstText.content.slice(0, 30)}...`
740
+ : firstFile?.type === 'file'
741
+ ? `Test: ${firstFile.filename}`
742
+ : 'Untitled Test'
743
+
744
+ return {
745
+ title: options?.title ?? fallbackTitle,
746
+ inputs,
747
+ expectedOutput: options?.expectedOutput,
748
+ evaluationInstructions: options?.evaluationInstructions,
749
+ validationReferenceFiles: options?.validationReferenceFiles,
750
+ metadata: options?.metadata,
751
+ }
752
+ }
753
+
754
+ /**
755
+ * Derive the run status. Intentionally a verbatim copy of the backend's
756
+ * getAgentTestCaseRunStatus (agent-test-case.controller.ts) so client-side
757
+ * status always matches API filtering and the UI — including completed
758
+ * evaluations without positive feedback counting as failed.
759
+ * Terminal: success, failed, error. `executed` = execution done, not evaluated.
760
+ */
761
+ export function getAgentTestCaseRunStatus(
762
+ run: Pick<
763
+ AgentTestCaseRun,
764
+ 'executionSessionId' | 'executionStatus' | 'evaluationSessionId' | 'evaluationStatus' | 'feedback'
765
+ > | null | undefined,
766
+ ): AgentTestCaseStatus {
767
+ if (!run)
768
+ return 'new'
769
+ if (run.executionStatus === 'error' || run.evaluationStatus === 'error')
770
+ return 'error'
771
+ if (
772
+ run.executionStatus === 'pending'
773
+ || run.executionStatus === 'running'
774
+ || (run.evaluationSessionId && (run.evaluationStatus === 'pending' || run.evaluationStatus === 'running'))
775
+ ) {
776
+ return 'running'
777
+ }
778
+ if (run.executionStatus === 'completed' && !run.evaluationSessionId)
779
+ return 'executed'
780
+ if (run.evaluationStatus === 'completed')
781
+ return run.feedback === 1 ? 'success' : 'failed'
782
+ if (run.evaluationStatus === 'failed' || run.feedback === 0)
783
+ return 'failed'
784
+ if (run.executionSessionId)
785
+ return 'executed'
786
+ return 'new'
787
+ }
788
+
789
+ /**
790
+ * Poll a test case run until it reaches a terminal status.
791
+ * The run row may not exist yet right after triggering — 404s are retried.
792
+ * By default waits for evaluation (success/failed/error); pass
793
+ * until: 'executed' to stop as soon as execution completes.
794
+ */
795
+ export async function waitForAgentTestCaseRun(
796
+ agentId: string,
797
+ testId: string,
798
+ commitHash: string,
799
+ options?: { intervalMs?: number, maxAttempts?: number, until?: 'evaluated' | 'executed' },
800
+ ): Promise<{ run: AgentTestCaseRun, status: AgentTestCaseStatus }> {
801
+ const interval = options?.intervalMs ?? 2000
802
+ const maxAttempts = options?.maxAttempts ?? 300 // 10 minutes default
803
+ const until = options?.until ?? 'evaluated'
804
+
805
+ for (let i = 0; i < maxAttempts; i++) {
806
+ let run: AgentTestCaseRun | null = null
807
+ try {
808
+ run = await getAgentTestCaseRun(agentId, testId, commitHash)
809
+ }
810
+ catch (error) {
811
+ // Run row is created asynchronously by the trigger task
812
+ if (!(error instanceof ApiRequestError) || error.status !== 404) {
813
+ throw error
814
+ }
815
+ }
816
+
817
+ if (run) {
818
+ const status = getAgentTestCaseRunStatus(run)
819
+ if (status === 'success' || status === 'failed' || status === 'error') {
820
+ return { run, status }
821
+ }
822
+ if (status === 'executed' && until === 'executed') {
823
+ return { run, status }
824
+ }
825
+ }
826
+
827
+ await new Promise(resolve => setTimeout(resolve, interval))
828
+ }
829
+
830
+ throw new Error(`Agent test case run ${testId}@${commitHash} did not complete within timeout`)
831
+ }
package/skill/common.ts CHANGED
@@ -139,12 +139,19 @@ export async function apiRequestRaw(endpoint: string, options: Omit<RequestInit,
139
139
  })
140
140
  }
141
141
 
142
+ export class ApiRequestError extends Error {
143
+ constructor(public readonly status: number, message: string) {
144
+ super(message)
145
+ this.name = 'ApiRequestError'
146
+ }
147
+ }
148
+
142
149
  export async function apiRequest<T>(endpoint: string, options: Omit<RequestInit, 'body'> & { body?: unknown } = {}): Promise<T> {
143
150
  const response = await apiRequestRaw(endpoint, options)
144
151
 
145
152
  if (!response.ok) {
146
153
  const error = await response.text()
147
- throw new Error(`API request failed: ${response.status} ${response.statusText} - ${error}`)
154
+ throw new ApiRequestError(response.status, `API request failed: ${response.status} ${response.statusText} - ${error}`)
148
155
  }
149
156
 
150
157
  return await (response.json() as Promise<T>)
package/skill/index.ts CHANGED
@@ -1,3 +1,66 @@
1
+ export {
2
+ addAgentTestCaseTags,
3
+ type AgentCommit,
4
+ type AgentHistory,
5
+ type AgentTestCase,
6
+ type AgentTestCaseAnswers,
7
+ type AgentTestCaseAttributeFeedback,
8
+ type AgentTestCaseCounters,
9
+ type AgentTestCaseFeedback,
10
+ type AgentTestCaseRun,
11
+ type AgentTestCaseRunHistoryEntry,
12
+ type AgentTestCaseSkipReason,
13
+ type AgentTestCaseStats,
14
+ type AgentTestCaseStatus,
15
+ type AgentTestCaseTag,
16
+ type AgentTestCaseValidationReferenceFile,
17
+ type AgentTestCaseValidationSummaryEntry,
18
+ type AgentTestCaseValidationType,
19
+ type AgentTestCaseWithRun,
20
+ type AgentTestInput,
21
+ type AgentTestMetric,
22
+ type AgentTestMetricStats,
23
+ type AgentTestMetricStatus,
24
+ // Helpers
25
+ buildAgentTestInputs,
26
+ bulkAddAgentTestCaseTags,
27
+ bulkRemoveAgentTestCaseTags,
28
+ continueAgentTestCase,
29
+ createAgentTestCase,
30
+ createAgentTestCasePayload,
31
+ type CreateAgentTestCasePayload,
32
+ createAgentTestCaseTag,
33
+ createAgentTestMetric,
34
+ type CreateAgentTestMetricPayload,
35
+ deleteAgentTestCase,
36
+ deleteAgentTestCaseTag,
37
+ deleteAgentTestMetric,
38
+ getAgentHistory,
39
+ getAgentTestCaseRun,
40
+ getAgentTestCaseRunStatus,
41
+ getAgentTestCaseStats,
42
+ getAgentTestCaseTags,
43
+ getLatestAgentCommit,
44
+ // Functions
45
+ listAgentTestCaseRuns,
46
+ type ListAgentTestCaseRunsResult,
47
+ listAgentTestCases,
48
+ type ListAgentTestCasesOptions,
49
+ listAgentTestCaseTags,
50
+ listAgentTestMetrics,
51
+ removeAgentTestCaseTag,
52
+ runAgentTestCase,
53
+ runAllAgentTestCases,
54
+ type RunAllAgentTestCasesPayload,
55
+ type RunAllAgentTestCasesResult,
56
+ updateAgentTestCase,
57
+ updateAgentTestCaseAttributeFeedback,
58
+ type UpdateAgentTestCasePayload,
59
+ updateAgentTestCaseTag,
60
+ updateAgentTestMetric,
61
+ type UpdateAgentTestMetricPayload,
62
+ waitForAgentTestCaseRun,
63
+ } from './agent-test-cases.ts'
1
64
  export {
2
65
  type Canvas,
3
66
  type CanvasMessage,
@@ -27,7 +90,7 @@ export {
27
90
  // Types
28
91
  type Variable,
29
92
  } from './canvas.ts'
30
- export { apiRequest, getApiBaseUrl, getAppBaseUrl, getCanvasUrl, getProjectUrl, loadApiKey, parsePayload, SESSION_PATH } from './common.ts'
93
+ export { apiRequest, ApiRequestError, getApiBaseUrl, getAppBaseUrl, getCanvasUrl, getProjectUrl, loadApiKey, parsePayload, SESSION_PATH } from './common.ts'
31
94
  export { createProject, type CreateProjectPayload, listProjects, type Project } from './projects.ts'
32
95
  export {
33
96
  getPromotedVersion,