@meistrari/tela-skills 1.6.2 → 1.8.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@meistrari/tela-skills",
3
- "version": "1.6.2",
3
+ "version": "1.8.0",
4
4
  "description": "Tela API skills for Claude Code",
5
5
  "type": "module",
6
6
  "bin": {
@@ -91,6 +91,63 @@ bun --preload ~/.claude/skills/tela/preload.ts -e "console.log(await tela.listPr
91
91
  - `tela.waitForTestCase(testCaseId)` - Wait for test case to complete
92
92
  - `tela.abortTestCase(testCaseId)` - Abort a running test case
93
93
 
94
+ ### Agents (FCC)
95
+
96
+ Create and iterate on Tela agents 100% via API — never clone the agent repository. An agent is a versioned git repo behind the API: every edit creates a commit (`commitHash` = agent version), `main` HEAD is the draft, and the published commit is what production runs.
97
+
98
+ **Golden rule — instructions**: agent instructions live in the *Purpose* block inside `CLAUDE.md`. The rest of `CLAUDE.md` belongs to the agent template: it is not shown in the Tela UI and gets overwritten on template syncs. Always use `updateAgentPurpose` (writes to `putAgentFile('CLAUDE.md')` are refused). For extra reusable behavior, use skills (`createAgentCustomSkill`/`addAgentSkill`) and subagents — both survive template syncs.
99
+
100
+ **Referencing things in the Purpose** (agents are NOT canvas — `{var}` syntax does not apply). Exact forms, written as plain text: text inputs as `<inputName>`, file inputs as `input/<inputName>/`, skills as `./.claude/skills/<name>/SKILL.md`, subagents as `./.claude/agents/<name>.md`, repo files by path. The Tela UI renders these as interactive chips — but only when they are plain text surrounded by spaces/punctuation and the target already exists (declare inputs / add skills first). **Never wrap references in backticks or code formatting** — the UI stops recognizing them. Never describe the output shape in the Purpose — that belongs in `output-format.json`.
101
+
102
+ - `tela.listAgents(options?)` / `tela.getAgent(agentId)` - List/get agents (inputs synced with the repo)
103
+ - `tela.createAgent({ title, projectId, description?, templateId? })` - Create an agent (provisions the repo)
104
+ - `tela.updateAgent(agentId, payload)` / `tela.deleteAgent(agentId)` - Update metadata / soft delete
105
+ - `tela.listAgentTemplates()` - Templates for `createAgent({ templateId })`
106
+ - `tela.getAgentPurpose(agentId, options?)` / `tela.updateAgentPurpose(agentId, content, options?)` - Read/write the agent's instructions (the ONLY way to edit them)
107
+ - `tela.getAgentFileTree(agentId, options?)` / `tela.getAgentFile(agentId, path, options?)` - Browse the repo (pass `branch`/`commitHash` for other versions)
108
+ - `tela.putAgentFile(agentId, path, content, options?)` - Create/update a file (commits; refuses managed files — CLAUDE.md/AGENTS.md, `agent.config.json`, `output-format.json`, `.claude/settings.json` — which go through their dedicated functions)
109
+ - `tela.deleteAgentFile` / `tela.renameAgentFile` - Remove/move files (protected runtime files are refused)
110
+ - `tela.uploadAgentFiles(agentId, files, options?)` - Multipart upload for binaries/batches. Layout conventions: static supporting material goes in `references/`; never write into `input/` (creating `input/<x>/` implicitly declares variable `x`); `output/` is runtime-only. Execution `attachments` mount at `input/references/` in the sandbox
111
+ - `tela.getAgentOutputSchema(agentId)` / `tela.updateAgentOutputSchema(agentId, schema)` - Read/write `output-format.json`. Flat map `attributeName -> { type, description, ... }` (no JSON Schema wrapper). Non-empty → structured JSON result; `{}` → markdown. These attribute paths are what test case attribute feedback and metrics reference
112
+ - `tela.getAgentConfig(agentId)` / `tela.updateAgentConfig(agentId, partial)` - Read/merge `agent.config.json` (model/harness rejected — use the dedicated functions)
113
+ - `tela.updateAgentModel(agentId, model)` / `tela.updateAgentHarness(agentId, harness)` - Change model/harness (server keeps template + layout in sync)
114
+ - `tela.listAgentInputs` / `createAgentInput` / `updateAgentInput` / `deleteAgentInput` - Declared input variables. Name: `^[a-zA-Z0-9_-]+$` (`references` is reserved). Declaring commits `input/<name>/.gitkeep` to the repo — the folder is the declaration. Executions are validated against declarations (unknown name, type mismatch, or missing required input → 400)
115
+ - `tela.discoverAgentSkill(ref)` / `tela.addAgentSkill(agentId, ref)` / `tela.listAgentSkillsStore()` - Skills from GitHub (`gh:owner/repo/path[@branch]`)
116
+ - `tela.createAgentCustomSkill(agentId, name, skillMd)` - Author a skill directly in the repo
117
+ - `tela.createAgentSubagent(agentId, name)` / `tela.updateAgentSubagent(agentId, name, content)` - Subagents (`.claude/agents/<name>.md`)
118
+ - `tela.addAgentCapability(agentId, { kind, sourceId })` - Attach a published canvas/workflow/agent as a tool
119
+ - `tela.restoreAgentVersion(agentId, commitHash)` / `tela.publishAgent(agentId, commitHash)` - Rollback / publish
120
+ - `tela.testAgent(agentId, payload)` / `tela.runAgent(agentId, payload)` - Execute draft (`main`) / production (published). Async: 202 + `sessionId`
121
+ - `tela.waitForAgentSession(sessionId)` / `tela.runAgentAndWait(agentId, payload, options?)` - Poll to completion / execute+wait in one call
122
+ - `tela.getAgentSessionThread` / `getAgentSessionTimeline` / `getAgentSessionFiles` / `getAgentSessionFile` - Inspect a session
123
+ - `tela.cancelAgentSession` / `tela.endAgentSession` - Stop sessions
124
+ - `tela.buildAgentExecutionInputs(variables, options?)` - Payload helper (vault:// values become file inputs)
125
+ - `tela.getAgentUrl(agentId)` - Link to the agent in the Tela app
126
+
127
+ ### Agent Test Cases
128
+
129
+ Test cases for agents (FCC). Runs are keyed by `(testCaseId, commitHash)` — one run per test case per agent version. Run/continue/run-all return `202 { executionId }` immediately; poll with `waitForAgentTestCaseRun`. Derived statuses: `new`, `running`, `executed`, `success`, `failed`, `error` (terminal: `success`/`failed`/`error`).
130
+
131
+ - `tela.listAgentTestCases(agentId, options?)` - List test cases (pass `commitHash` to attach runs/stats)
132
+ - `tela.createAgentTestCase(agentId, payload)` - Create a test case
133
+ - `tela.updateAgentTestCase(agentId, testId, payload)` - Update a test case
134
+ - `tela.deleteAgentTestCase(agentId, testId)` - Delete a test case
135
+ - `tela.runAgentTestCase(agentId, testId, commitHash)` - Run against an agent version
136
+ - `tela.continueAgentTestCase(agentId, testId, commitHash, message)` - Continue session (multiturn)
137
+ - `tela.runAllAgentTestCases(agentId, { commitHash, testCaseIds? | filters? })` - Batch run (returns `skipped` reasons)
138
+ - `tela.getAgentTestCaseRun(agentId, testId, commitHash)` - Get the run for a commit
139
+ - `tela.listAgentTestCaseRuns(agentId, testId, options?)` - Run history across commits (cursor pagination)
140
+ - `tela.waitForAgentTestCaseRun(agentId, testId, commitHash, options?)` - Poll until terminal status
141
+ - `tela.updateAgentTestCaseAttributeFeedback(agentId, testId, commitHash, attributes)` - Thumbs per attribute (feeds the answer bank)
142
+ - `tela.getAgentTestCaseStats(agentId, commitHash, options?)` - Aggregated stats for a commit
143
+ - `tela.listAgentTestMetrics(agentId)` / `createAgentTestMetric` / `updateAgentTestMetric` / `deleteAgentTestMetric` - Custom metrics over output attributes
144
+ - `tela.listAgentTestCaseTags(agentId)` / `createAgentTestCaseTag` / `updateAgentTestCaseTag` / `deleteAgentTestCaseTag` - Tag CRUD
145
+ - `tela.getAgentTestCaseTags(agentId, testId)` / `addAgentTestCaseTags` / `removeAgentTestCaseTag` - Tag assignment
146
+ - `tela.bulkAddAgentTestCaseTags(agentId, testCaseIds, tagIds)` / `bulkRemoveAgentTestCaseTags(...)` - Bulk tagging
147
+ - `tela.getAgentHistory(agentId, options?)` / `tela.getLatestAgentCommit(agentId)` - Resolve agent versions (commitHash)
148
+ - `tela.buildAgentTestInputs(variables, options?)` / `tela.createAgentTestCasePayload(variables, options?)` - Payload helpers (vault:// values become file inputs)
149
+ - `tela.getAgentTestCaseRunStatus(run)` - Derive run status locally
150
+
94
151
  ### Tasks
95
152
 
96
153
  - `tela.listTasks(options?)` - List tasks with pagination and filtering
@@ -192,6 +249,66 @@ console.log('Variables:', version?.variables?.map(v => v.name))
192
249
  "
193
250
  ```
194
251
 
252
+ ### Create an Agent and Iterate via API (no repo cloning)
253
+ ```bash
254
+ bun --preload ~/.claude/skills/tela/preload.ts -e "
255
+ const agent = await tela.createAgent({ title: 'Contract Analyzer', projectId: 'PROJECT_ID' })
256
+
257
+ await tela.updateAgentPurpose(agent.id, [
258
+ 'You analyze contracts and extract the parties and total amount.',
259
+ 'The contract PDF is at input/contract/.',
260
+ 'Always answer in Brazilian Portuguese.',
261
+ ].join('\n'))
262
+
263
+ await tela.createAgentInput(agent.id, { name: 'contract', type: 'file', required: true })
264
+ await tela.updateAgentOutputSchema(agent.id, {
265
+ parties: { type: 'array', items: { type: 'string' }, description: 'All contract parties' },
266
+ total: { type: 'number', description: 'Total amount' },
267
+ })
268
+
269
+ const { status, output } = await tela.runAgentAndWait(agent.id, {
270
+ message: 'Analyze the attached contract',
271
+ inputSchema: tela.buildAgentExecutionInputs({ contract: await tela.uploadFile('./contract.pdf') }),
272
+ })
273
+ console.log(status, output)
274
+ console.log('Edit in UI:', tela.getAgentUrl(agent.id))
275
+ "
276
+ ```
277
+
278
+ ### Publish an Agent Version
279
+ ```bash
280
+ bun --preload ~/.claude/skills/tela/preload.ts -e "
281
+ const commitHash = await tela.getLatestAgentCommit('AGENT_ID')
282
+ await tela.publishAgent('AGENT_ID', commitHash)
283
+ console.log('Published', commitHash)
284
+ "
285
+ ```
286
+
287
+ ### Create and Run an Agent Test Case
288
+ ```bash
289
+ bun --preload ~/.claude/skills/tela/preload.ts -e "
290
+ const agentId = 'AGENT_ID'
291
+ const vaultRef = await tela.uploadFile('./contract.pdf')
292
+
293
+ const testCase = await tela.createAgentTestCase(agentId, tela.createAgentTestCasePayload({
294
+ document: vaultRef,
295
+ instructions: 'Extract the parties and total amount',
296
+ }, {
297
+ title: 'Contract extraction',
298
+ evaluationInstructions: 'The output must list both parties and the correct total.',
299
+ fileNames: { document: 'contract.pdf' },
300
+ }))
301
+
302
+ const commitHash = await tela.getLatestAgentCommit(agentId)
303
+ await tela.runAgentTestCase(agentId, testCase.id, commitHash)
304
+
305
+ const { run, status } = await tela.waitForAgentTestCaseRun(agentId, testCase.id, commitHash)
306
+ console.log('Status:', status)
307
+ console.log('Output:', run.output)
308
+ console.log('Validation:', run.validationSummary)
309
+ "
310
+ ```
311
+
195
312
  ### List Tasks
196
313
  ```bash
197
314
  bun --preload ~/.claude/skills/tela/preload.ts -e "
@@ -332,7 +449,9 @@ bun localhost.ts "console.log(await tela.listProjects())"
332
449
  ├── prompts.ts # Prompt/Canvas list functions
333
450
  ├── canvas.ts # Canvas create/update functions
334
451
  ├── workflows.ts # Workflow create/update/run/validation functions
335
- ├── test-cases.ts # Test case functions
452
+ ├── test-cases.ts # Test case functions (prompt/canvas)
453
+ ├── agents.ts # Agent editing/execution functions (FCC)
454
+ ├── agent-test-cases.ts # Agent test case functions (FCC)
336
455
  ├── tasks.ts # Task management functions
337
456
  ├── vault.ts # Vault file management functions
338
457
  └── localhost-utils.ts # Docker discovery, JWT signing, workspace fetching (internal, not on tela global)
@@ -0,0 +1,316 @@
1
+ import { beforeEach, describe, expect, mock, test } from 'bun:test'
2
+
3
+ const apiRequest = mock(async (_endpoint: string, _options?: RequestInit): Promise<unknown> => ({ data: {} }))
4
+
5
+ class ApiRequestError extends Error {
6
+ constructor(public readonly status: number, message: string) {
7
+ super(message)
8
+ this.name = 'ApiRequestError'
9
+ }
10
+ }
11
+
12
+ void mock.module('./common.ts', () => ({
13
+ apiRequest,
14
+ ApiRequestError,
15
+ }))
16
+
17
+ describe('agent-test-cases', () => {
18
+ beforeEach(() => {
19
+ apiRequest.mockClear()
20
+ apiRequest.mockImplementation(async () => ({ data: {} }))
21
+ })
22
+
23
+ test('lists test cases with repeated array query params', async () => {
24
+ const { listAgentTestCases } = await import('./agent-test-cases.ts')
25
+ apiRequest.mockImplementation(async () => ({ data: [] }))
26
+
27
+ await listAgentTestCases('agent-1', {
28
+ commitHash: 'abc123',
29
+ title: 'search',
30
+ status: ['failed', 'error'],
31
+ tagIds: ['t1', 't2'],
32
+ createdBy: ['u1'],
33
+ createdAtSince: '2026-01-01T00:00:00Z',
34
+ })
35
+
36
+ const [endpoint] = apiRequest.mock.calls[0]!
37
+ const query = new URLSearchParams(String(endpoint).split('?')[1])
38
+
39
+ expect(String(endpoint).startsWith('/agent/agent-1/tests?')).toBe(true)
40
+ expect(query.get('commitHash')).toBe('abc123')
41
+ expect(query.get('title')).toBe('search')
42
+ expect(query.getAll('status')).toEqual(['failed', 'error'])
43
+ expect(query.getAll('tagIds')).toEqual(['t1', 't2'])
44
+ expect(query.getAll('createdBy')).toEqual(['u1'])
45
+ expect(query.get('createdAtSince')).toBe('2026-01-01T00:00:00Z')
46
+ })
47
+
48
+ test('omits the query string when no options are given', async () => {
49
+ const { listAgentTestCases } = await import('./agent-test-cases.ts')
50
+ apiRequest.mockImplementation(async () => ({ data: [] }))
51
+
52
+ await listAgentTestCases('agent-1')
53
+
54
+ expect(apiRequest.mock.calls[0]![0]).toBe('/agent/agent-1/tests')
55
+ })
56
+
57
+ test('pre-serializes bodies so vault refs are not rewritten by parsePayload', async () => {
58
+ const { createAgentTestCase } = await import('./agent-test-cases.ts')
59
+
60
+ await createAgentTestCase('agent-1', {
61
+ inputs: [{ type: 'file', name: 'doc', vaultRef: 'vault://abc', filename: 'doc.pdf' }],
62
+ })
63
+
64
+ const [endpoint, options] = apiRequest.mock.calls[0]!
65
+ expect(endpoint).toBe('/agent/agent-1/tests')
66
+ expect(options?.method).toBe('POST')
67
+ expect(typeof options?.body).toBe('string')
68
+
69
+ const body = JSON.parse(String(options?.body))
70
+ expect(body.inputs[0].vaultRef).toBe('vault://abc')
71
+ })
72
+
73
+ test('unwraps the { data } envelope', async () => {
74
+ const { createAgentTestCase } = await import('./agent-test-cases.ts')
75
+ apiRequest.mockImplementation(async () => ({ data: { id: 'tc-1', title: 'Test' } }))
76
+
77
+ const testCase = await createAgentTestCase('agent-1', { title: 'Test' })
78
+
79
+ expect(testCase).toEqual({ id: 'tc-1', title: 'Test' } as any)
80
+ })
81
+
82
+ test('runs a test case against a commit', async () => {
83
+ const { runAgentTestCase } = await import('./agent-test-cases.ts')
84
+ apiRequest.mockImplementation(async () => ({ data: { executionId: 'exec-1' } }))
85
+
86
+ const result = await runAgentTestCase('agent-1', 'tc-1', 'abc123')
87
+
88
+ const [endpoint, options] = apiRequest.mock.calls[0]!
89
+ expect(endpoint).toBe('/agent/agent-1/tests/tc-1/run')
90
+ expect(JSON.parse(String(options?.body))).toEqual({ commitHash: 'abc123' })
91
+ expect(result.executionId).toBe('exec-1')
92
+ })
93
+
94
+ test('continues a session with a message', async () => {
95
+ const { continueAgentTestCase } = await import('./agent-test-cases.ts')
96
+ apiRequest.mockImplementation(async () => ({ data: { executionId: 'exec-2' } }))
97
+
98
+ await continueAgentTestCase('agent-1', 'tc-1', 'abc123', 'next turn')
99
+
100
+ const [endpoint, options] = apiRequest.mock.calls[0]!
101
+ expect(endpoint).toBe('/agent/agent-1/tests/tc-1/continue')
102
+ expect(JSON.parse(String(options?.body))).toEqual({ commitHash: 'abc123', message: 'next turn' })
103
+ })
104
+
105
+ test('sends attribute feedback wrapped in attributes', async () => {
106
+ const { updateAgentTestCaseAttributeFeedback } = await import('./agent-test-cases.ts')
107
+
108
+ await updateAgentTestCaseAttributeFeedback('agent-1', 'tc-1', 'abc123', [
109
+ { path: 'summary', feedback: 1 },
110
+ { path: 'total', feedback: 0, message: 'wrong' },
111
+ ])
112
+
113
+ const [endpoint, options] = apiRequest.mock.calls[0]!
114
+ expect(endpoint).toBe('/agent/agent-1/tests/tc-1/runs/abc123/attribute-feedback')
115
+ expect(options?.method).toBe('PATCH')
116
+
117
+ const body = JSON.parse(String(options?.body))
118
+ expect(body.attributes).toHaveLength(2)
119
+ expect(body.attributes[1]).toEqual({ path: 'total', feedback: 0, message: 'wrong' })
120
+ })
121
+
122
+ test('requires commitHash on the stats endpoint query', async () => {
123
+ const { getAgentTestCaseStats } = await import('./agent-test-cases.ts')
124
+
125
+ await getAgentTestCaseStats('agent-1', 'abc123', { tagIds: ['t1'] })
126
+
127
+ const [endpoint] = apiRequest.mock.calls[0]!
128
+ const query = new URLSearchParams(String(endpoint).split('?')[1])
129
+
130
+ expect(String(endpoint).startsWith('/agent/agent-1/tests/stats?')).toBe(true)
131
+ expect(query.get('commitHash')).toBe('abc123')
132
+ expect(query.getAll('tagIds')).toEqual(['t1'])
133
+ })
134
+
135
+ test('run history keeps pagination fields from the response envelope', async () => {
136
+ const { listAgentTestCaseRuns } = await import('./agent-test-cases.ts')
137
+ // This endpoint's envelope is { data: entries, nextCursor, totalRuns } —
138
+ // pagination fields live beside data, not inside it
139
+ apiRequest.mockImplementation(async () => ({
140
+ data: [{ commitHash: 'abc123', score: 1 }],
141
+ nextCursor: 'cursor-1',
142
+ totalRuns: 5,
143
+ }))
144
+
145
+ const result = await listAgentTestCaseRuns('agent-1', 'tc-1', { limit: 1 })
146
+
147
+ expect(apiRequest.mock.calls[0]![0]).toBe('/agent/agent-1/tests/tc-1/runs?limit=1')
148
+ expect(result.data).toHaveLength(1)
149
+ expect(result.nextCursor).toBe('cursor-1')
150
+ expect(result.totalRuns).toBe(5)
151
+ })
152
+
153
+ describe('buildAgentTestInputs', () => {
154
+ test('splits text and vault file values', async () => {
155
+ const { buildAgentTestInputs } = await import('./agent-test-cases.ts')
156
+
157
+ const inputs = buildAgentTestInputs({
158
+ context: 'Some text',
159
+ document: 'vault://abc123',
160
+ }, { fileNames: { document: 'report.pdf' } })
161
+
162
+ expect(inputs).toEqual([
163
+ { type: 'text', name: 'context', content: 'Some text' },
164
+ { type: 'file', name: 'document', vaultRef: 'vault://abc123', filename: 'report.pdf' },
165
+ ])
166
+ })
167
+
168
+ test('defaults the filename to the input name', async () => {
169
+ const { buildAgentTestInputs } = await import('./agent-test-cases.ts')
170
+
171
+ const inputs = buildAgentTestInputs({ document: 'vault://abc123' })
172
+
173
+ expect(inputs[0]).toEqual({ type: 'file', name: 'document', vaultRef: 'vault://abc123', filename: 'document' })
174
+ })
175
+ })
176
+
177
+ describe('createAgentTestCasePayload', () => {
178
+ test('builds payload with expectations and derived title', async () => {
179
+ const { createAgentTestCasePayload } = await import('./agent-test-cases.ts')
180
+
181
+ const payload = createAgentTestCasePayload({ input: 'Hello world, this is a long input value' }, {
182
+ evaluationInstructions: 'Must greet back',
183
+ })
184
+
185
+ expect(payload.title).toBe(`Test: ${'Hello world, this is a long input value'.slice(0, 30)}...`)
186
+ expect(payload.evaluationInstructions).toBe('Must greet back')
187
+ expect(payload.inputs).toHaveLength(1)
188
+ })
189
+
190
+ test('falls back to the first filename for file-only cases', async () => {
191
+ const { createAgentTestCasePayload } = await import('./agent-test-cases.ts')
192
+
193
+ const payload = createAgentTestCasePayload({ doc: 'vault://abc' }, { fileNames: { doc: 'contract.pdf' } })
194
+
195
+ expect(payload.title).toBe('Test: contract.pdf')
196
+ })
197
+ })
198
+
199
+ describe('getAgentTestCaseRunStatus', () => {
200
+ const baseRun = {
201
+ executionSessionId: null as string | null,
202
+ executionStatus: null as any,
203
+ evaluationSessionId: null as string | null,
204
+ evaluationStatus: null as any,
205
+ feedback: null as 0 | 1 | null,
206
+ }
207
+
208
+ test('derives every status branch', async () => {
209
+ const { getAgentTestCaseRunStatus } = await import('./agent-test-cases.ts')
210
+
211
+ expect(getAgentTestCaseRunStatus(null)).toBe('new')
212
+ expect(getAgentTestCaseRunStatus({ ...baseRun, executionStatus: 'error' })).toBe('error')
213
+ expect(getAgentTestCaseRunStatus({ ...baseRun, executionStatus: 'completed', evaluationStatus: 'error' })).toBe('error')
214
+ expect(getAgentTestCaseRunStatus({ ...baseRun, executionStatus: 'running' })).toBe('running')
215
+ expect(getAgentTestCaseRunStatus({ ...baseRun, executionStatus: 'completed', evaluationSessionId: 's', evaluationStatus: 'running' })).toBe('running')
216
+ expect(getAgentTestCaseRunStatus({ ...baseRun, executionStatus: 'completed' })).toBe('executed')
217
+ expect(getAgentTestCaseRunStatus({ ...baseRun, executionStatus: 'completed', evaluationSessionId: 's', evaluationStatus: 'completed', feedback: 1 })).toBe('success')
218
+ expect(getAgentTestCaseRunStatus({ ...baseRun, executionStatus: 'completed', evaluationSessionId: 's', evaluationStatus: 'completed', feedback: 0 })).toBe('failed')
219
+ expect(getAgentTestCaseRunStatus({ ...baseRun, executionStatus: 'completed', evaluationSessionId: 's', evaluationStatus: 'failed' })).toBe('failed')
220
+ expect(getAgentTestCaseRunStatus({ ...baseRun, executionSessionId: 's' })).toBe('executed')
221
+ })
222
+ })
223
+
224
+ describe('waitForAgentTestCaseRun', () => {
225
+ test('retries 404s until the run appears, then resolves on terminal status', async () => {
226
+ const { waitForAgentTestCaseRun } = await import('./agent-test-cases.ts')
227
+
228
+ let calls = 0
229
+ apiRequest.mockImplementation(async () => {
230
+ calls++
231
+ if (calls === 1)
232
+ throw new ApiRequestError(404, 'API request failed: 404 Not Found - {}')
233
+ if (calls === 2) {
234
+ return { data: { executionStatus: 'running', executionSessionId: 's', evaluationSessionId: null, evaluationStatus: null, feedback: null } }
235
+ }
236
+ return { data: { executionStatus: 'completed', executionSessionId: 's', evaluationSessionId: 'e', evaluationStatus: 'completed', feedback: 1 } }
237
+ })
238
+
239
+ const { status } = await waitForAgentTestCaseRun('agent-1', 'tc-1', 'abc123', { intervalMs: 1 })
240
+
241
+ expect(status).toBe('success')
242
+ expect(calls).toBe(3)
243
+ })
244
+
245
+ test('stops at executed when until is executed', async () => {
246
+ const { waitForAgentTestCaseRun } = await import('./agent-test-cases.ts')
247
+
248
+ apiRequest.mockImplementation(async () => ({
249
+ data: { executionStatus: 'completed', executionSessionId: 's', evaluationSessionId: null, evaluationStatus: null, feedback: null },
250
+ }))
251
+
252
+ const { status } = await waitForAgentTestCaseRun('agent-1', 'tc-1', 'abc123', { intervalMs: 1, until: 'executed' })
253
+
254
+ expect(status).toBe('executed')
255
+ })
256
+
257
+ test('throws after maxAttempts without a terminal status', async () => {
258
+ const { waitForAgentTestCaseRun } = await import('./agent-test-cases.ts')
259
+
260
+ apiRequest.mockImplementation(async () => ({
261
+ data: { executionStatus: 'running', executionSessionId: 's', evaluationSessionId: null, evaluationStatus: null, feedback: null },
262
+ }))
263
+
264
+ expect(waitForAgentTestCaseRun('agent-1', 'tc-1', 'abc123', { intervalMs: 1, maxAttempts: 2 }))
265
+ .rejects
266
+ .toThrow('did not complete within timeout')
267
+ })
268
+
269
+ test('rethrows non-404 API errors', async () => {
270
+ const { waitForAgentTestCaseRun } = await import('./agent-test-cases.ts')
271
+
272
+ apiRequest.mockImplementation(async () => {
273
+ throw new ApiRequestError(500, 'API request failed: 500 Internal Server Error - {}')
274
+ })
275
+
276
+ expect(waitForAgentTestCaseRun('agent-1', 'tc-1', 'abc123', { intervalMs: 1 }))
277
+ .rejects
278
+ .toThrow('500')
279
+ })
280
+
281
+ test('rethrows untyped errors even when the message mentions 404', async () => {
282
+ const { waitForAgentTestCaseRun } = await import('./agent-test-cases.ts')
283
+
284
+ apiRequest.mockImplementation(async () => {
285
+ throw new Error('fetch failed with 404 somewhere')
286
+ })
287
+
288
+ expect(waitForAgentTestCaseRun('agent-1', 'tc-1', 'abc123', { intervalMs: 1 }))
289
+ .rejects
290
+ .toThrow('fetch failed')
291
+ })
292
+ })
293
+
294
+ describe('getLatestAgentCommit', () => {
295
+ test('returns the newest commit hash', async () => {
296
+ const { getLatestAgentCommit } = await import('./agent-test-cases.ts')
297
+ apiRequest.mockImplementation(async () => ({
298
+ data: { commits: [{ commitHash: 'abc123' }], branch: 'main', pagination: { page: 1, limit: 1, totalCount: 5, hasMore: true } },
299
+ }))
300
+
301
+ const hash = await getLatestAgentCommit('agent-1')
302
+
303
+ expect(apiRequest.mock.calls[0]![0]).toBe('/agent/agent-1/history?limit=1')
304
+ expect(hash).toBe('abc123')
305
+ })
306
+
307
+ test('throws when the agent has no commits', async () => {
308
+ const { getLatestAgentCommit } = await import('./agent-test-cases.ts')
309
+ apiRequest.mockImplementation(async () => ({
310
+ data: { commits: [], branch: 'main', pagination: { page: 1, limit: 1, totalCount: 0, hasMore: false } },
311
+ }))
312
+
313
+ expect(getLatestAgentCommit('agent-1')).rejects.toThrow('has no commits')
314
+ })
315
+ })
316
+ })