@meistrari/tela-skills 1.6.2 → 1.8.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +1 -1
- package/skill/TELA_SKILL.md +120 -1
- package/skill/agent-test-cases.test.ts +316 -0
- package/skill/agent-test-cases.ts +831 -0
- package/skill/agents.test.ts +539 -0
- package/skill/agents.ts +1289 -0
- package/skill/common.ts +8 -1
- package/skill/index.ts +154 -1
package/package.json
CHANGED
package/skill/TELA_SKILL.md
CHANGED
|
@@ -91,6 +91,63 @@ bun --preload ~/.claude/skills/tela/preload.ts -e "console.log(await tela.listPr
|
|
|
91
91
|
- `tela.waitForTestCase(testCaseId)` - Wait for test case to complete
|
|
92
92
|
- `tela.abortTestCase(testCaseId)` - Abort a running test case
|
|
93
93
|
|
|
94
|
+
### Agents (FCC)
|
|
95
|
+
|
|
96
|
+
Create and iterate on Tela agents 100% via API — never clone the agent repository. An agent is a versioned git repo behind the API: every edit creates a commit (`commitHash` = agent version), `main` HEAD is the draft, and the published commit is what production runs.
|
|
97
|
+
|
|
98
|
+
**Golden rule — instructions**: agent instructions live in the *Purpose* block inside `CLAUDE.md`. The rest of `CLAUDE.md` belongs to the agent template: it is not shown in the Tela UI and gets overwritten on template syncs. Always use `updateAgentPurpose` (writes to `putAgentFile('CLAUDE.md')` are refused). For extra reusable behavior, use skills (`createAgentCustomSkill`/`addAgentSkill`) and subagents — both survive template syncs.
|
|
99
|
+
|
|
100
|
+
**Referencing things in the Purpose** (agents are NOT canvas — `{var}` syntax does not apply). Exact forms, written as plain text: text inputs as `<inputName>`, file inputs as `input/<inputName>/`, skills as `./.claude/skills/<name>/SKILL.md`, subagents as `./.claude/agents/<name>.md`, repo files by path. The Tela UI renders these as interactive chips — but only when they are plain text surrounded by spaces/punctuation and the target already exists (declare inputs / add skills first). **Never wrap references in backticks or code formatting** — the UI stops recognizing them. Never describe the output shape in the Purpose — that belongs in `output-format.json`.
|
|
101
|
+
|
|
102
|
+
- `tela.listAgents(options?)` / `tela.getAgent(agentId)` - List/get agents (inputs synced with the repo)
|
|
103
|
+
- `tela.createAgent({ title, projectId, description?, templateId? })` - Create an agent (provisions the repo)
|
|
104
|
+
- `tela.updateAgent(agentId, payload)` / `tela.deleteAgent(agentId)` - Update metadata / soft delete
|
|
105
|
+
- `tela.listAgentTemplates()` - Templates for `createAgent({ templateId })`
|
|
106
|
+
- `tela.getAgentPurpose(agentId, options?)` / `tela.updateAgentPurpose(agentId, content, options?)` - Read/write the agent's instructions (the ONLY way to edit them)
|
|
107
|
+
- `tela.getAgentFileTree(agentId, options?)` / `tela.getAgentFile(agentId, path, options?)` - Browse the repo (pass `branch`/`commitHash` for other versions)
|
|
108
|
+
- `tela.putAgentFile(agentId, path, content, options?)` - Create/update a file (commits; refuses managed files — CLAUDE.md/AGENTS.md, `agent.config.json`, `output-format.json`, `.claude/settings.json` — which go through their dedicated functions)
|
|
109
|
+
- `tela.deleteAgentFile` / `tela.renameAgentFile` - Remove/move files (protected runtime files are refused)
|
|
110
|
+
- `tela.uploadAgentFiles(agentId, files, options?)` - Multipart upload for binaries/batches. Layout conventions: static supporting material goes in `references/`; never write into `input/` (creating `input/<x>/` implicitly declares variable `x`); `output/` is runtime-only. Execution `attachments` mount at `input/references/` in the sandbox
|
|
111
|
+
- `tela.getAgentOutputSchema(agentId)` / `tela.updateAgentOutputSchema(agentId, schema)` - Read/write `output-format.json`. Flat map `attributeName -> { type, description, ... }` (no JSON Schema wrapper). Non-empty → structured JSON result; `{}` → markdown. These attribute paths are what test case attribute feedback and metrics reference
|
|
112
|
+
- `tela.getAgentConfig(agentId)` / `tela.updateAgentConfig(agentId, partial)` - Read/merge `agent.config.json` (model/harness rejected — use the dedicated functions)
|
|
113
|
+
- `tela.updateAgentModel(agentId, model)` / `tela.updateAgentHarness(agentId, harness)` - Change model/harness (server keeps template + layout in sync)
|
|
114
|
+
- `tela.listAgentInputs` / `createAgentInput` / `updateAgentInput` / `deleteAgentInput` - Declared input variables. Name: `^[a-zA-Z0-9_-]+$` (`references` is reserved). Declaring commits `input/<name>/.gitkeep` to the repo — the folder is the declaration. Executions are validated against declarations (unknown name, type mismatch, or missing required input → 400)
|
|
115
|
+
- `tela.discoverAgentSkill(ref)` / `tela.addAgentSkill(agentId, ref)` / `tela.listAgentSkillsStore()` - Skills from GitHub (`gh:owner/repo/path[@branch]`)
|
|
116
|
+
- `tela.createAgentCustomSkill(agentId, name, skillMd)` - Author a skill directly in the repo
|
|
117
|
+
- `tela.createAgentSubagent(agentId, name)` / `tela.updateAgentSubagent(agentId, name, content)` - Subagents (`.claude/agents/<name>.md`)
|
|
118
|
+
- `tela.addAgentCapability(agentId, { kind, sourceId })` - Attach a published canvas/workflow/agent as a tool
|
|
119
|
+
- `tela.restoreAgentVersion(agentId, commitHash)` / `tela.publishAgent(agentId, commitHash)` - Rollback / publish
|
|
120
|
+
- `tela.testAgent(agentId, payload)` / `tela.runAgent(agentId, payload)` - Execute draft (`main`) / production (published). Async: 202 + `sessionId`
|
|
121
|
+
- `tela.waitForAgentSession(sessionId)` / `tela.runAgentAndWait(agentId, payload, options?)` - Poll to completion / execute+wait in one call
|
|
122
|
+
- `tela.getAgentSessionThread` / `getAgentSessionTimeline` / `getAgentSessionFiles` / `getAgentSessionFile` - Inspect a session
|
|
123
|
+
- `tela.cancelAgentSession` / `tela.endAgentSession` - Stop sessions
|
|
124
|
+
- `tela.buildAgentExecutionInputs(variables, options?)` - Payload helper (vault:// values become file inputs)
|
|
125
|
+
- `tela.getAgentUrl(agentId)` - Link to the agent in the Tela app
|
|
126
|
+
|
|
127
|
+
### Agent Test Cases
|
|
128
|
+
|
|
129
|
+
Test cases for agents (FCC). Runs are keyed by `(testCaseId, commitHash)` — one run per test case per agent version. Run/continue/run-all return `202 { executionId }` immediately; poll with `waitForAgentTestCaseRun`. Derived statuses: `new`, `running`, `executed`, `success`, `failed`, `error` (terminal: `success`/`failed`/`error`).
|
|
130
|
+
|
|
131
|
+
- `tela.listAgentTestCases(agentId, options?)` - List test cases (pass `commitHash` to attach runs/stats)
|
|
132
|
+
- `tela.createAgentTestCase(agentId, payload)` - Create a test case
|
|
133
|
+
- `tela.updateAgentTestCase(agentId, testId, payload)` - Update a test case
|
|
134
|
+
- `tela.deleteAgentTestCase(agentId, testId)` - Delete a test case
|
|
135
|
+
- `tela.runAgentTestCase(agentId, testId, commitHash)` - Run against an agent version
|
|
136
|
+
- `tela.continueAgentTestCase(agentId, testId, commitHash, message)` - Continue session (multiturn)
|
|
137
|
+
- `tela.runAllAgentTestCases(agentId, { commitHash, testCaseIds? | filters? })` - Batch run (returns `skipped` reasons)
|
|
138
|
+
- `tela.getAgentTestCaseRun(agentId, testId, commitHash)` - Get the run for a commit
|
|
139
|
+
- `tela.listAgentTestCaseRuns(agentId, testId, options?)` - Run history across commits (cursor pagination)
|
|
140
|
+
- `tela.waitForAgentTestCaseRun(agentId, testId, commitHash, options?)` - Poll until terminal status
|
|
141
|
+
- `tela.updateAgentTestCaseAttributeFeedback(agentId, testId, commitHash, attributes)` - Thumbs per attribute (feeds the answer bank)
|
|
142
|
+
- `tela.getAgentTestCaseStats(agentId, commitHash, options?)` - Aggregated stats for a commit
|
|
143
|
+
- `tela.listAgentTestMetrics(agentId)` / `createAgentTestMetric` / `updateAgentTestMetric` / `deleteAgentTestMetric` - Custom metrics over output attributes
|
|
144
|
+
- `tela.listAgentTestCaseTags(agentId)` / `createAgentTestCaseTag` / `updateAgentTestCaseTag` / `deleteAgentTestCaseTag` - Tag CRUD
|
|
145
|
+
- `tela.getAgentTestCaseTags(agentId, testId)` / `addAgentTestCaseTags` / `removeAgentTestCaseTag` - Tag assignment
|
|
146
|
+
- `tela.bulkAddAgentTestCaseTags(agentId, testCaseIds, tagIds)` / `bulkRemoveAgentTestCaseTags(...)` - Bulk tagging
|
|
147
|
+
- `tela.getAgentHistory(agentId, options?)` / `tela.getLatestAgentCommit(agentId)` - Resolve agent versions (commitHash)
|
|
148
|
+
- `tela.buildAgentTestInputs(variables, options?)` / `tela.createAgentTestCasePayload(variables, options?)` - Payload helpers (vault:// values become file inputs)
|
|
149
|
+
- `tela.getAgentTestCaseRunStatus(run)` - Derive run status locally
|
|
150
|
+
|
|
94
151
|
### Tasks
|
|
95
152
|
|
|
96
153
|
- `tela.listTasks(options?)` - List tasks with pagination and filtering
|
|
@@ -192,6 +249,66 @@ console.log('Variables:', version?.variables?.map(v => v.name))
|
|
|
192
249
|
"
|
|
193
250
|
```
|
|
194
251
|
|
|
252
|
+
### Create an Agent and Iterate via API (no repo cloning)
|
|
253
|
+
```bash
|
|
254
|
+
bun --preload ~/.claude/skills/tela/preload.ts -e "
|
|
255
|
+
const agent = await tela.createAgent({ title: 'Contract Analyzer', projectId: 'PROJECT_ID' })
|
|
256
|
+
|
|
257
|
+
await tela.updateAgentPurpose(agent.id, [
|
|
258
|
+
'You analyze contracts and extract the parties and total amount.',
|
|
259
|
+
'The contract PDF is at input/contract/.',
|
|
260
|
+
'Always answer in Brazilian Portuguese.',
|
|
261
|
+
].join('\n'))
|
|
262
|
+
|
|
263
|
+
await tela.createAgentInput(agent.id, { name: 'contract', type: 'file', required: true })
|
|
264
|
+
await tela.updateAgentOutputSchema(agent.id, {
|
|
265
|
+
parties: { type: 'array', items: { type: 'string' }, description: 'All contract parties' },
|
|
266
|
+
total: { type: 'number', description: 'Total amount' },
|
|
267
|
+
})
|
|
268
|
+
|
|
269
|
+
const { status, output } = await tela.runAgentAndWait(agent.id, {
|
|
270
|
+
message: 'Analyze the attached contract',
|
|
271
|
+
inputSchema: tela.buildAgentExecutionInputs({ contract: await tela.uploadFile('./contract.pdf') }),
|
|
272
|
+
})
|
|
273
|
+
console.log(status, output)
|
|
274
|
+
console.log('Edit in UI:', tela.getAgentUrl(agent.id))
|
|
275
|
+
"
|
|
276
|
+
```
|
|
277
|
+
|
|
278
|
+
### Publish an Agent Version
|
|
279
|
+
```bash
|
|
280
|
+
bun --preload ~/.claude/skills/tela/preload.ts -e "
|
|
281
|
+
const commitHash = await tela.getLatestAgentCommit('AGENT_ID')
|
|
282
|
+
await tela.publishAgent('AGENT_ID', commitHash)
|
|
283
|
+
console.log('Published', commitHash)
|
|
284
|
+
"
|
|
285
|
+
```
|
|
286
|
+
|
|
287
|
+
### Create and Run an Agent Test Case
|
|
288
|
+
```bash
|
|
289
|
+
bun --preload ~/.claude/skills/tela/preload.ts -e "
|
|
290
|
+
const agentId = 'AGENT_ID'
|
|
291
|
+
const vaultRef = await tela.uploadFile('./contract.pdf')
|
|
292
|
+
|
|
293
|
+
const testCase = await tela.createAgentTestCase(agentId, tela.createAgentTestCasePayload({
|
|
294
|
+
document: vaultRef,
|
|
295
|
+
instructions: 'Extract the parties and total amount',
|
|
296
|
+
}, {
|
|
297
|
+
title: 'Contract extraction',
|
|
298
|
+
evaluationInstructions: 'The output must list both parties and the correct total.',
|
|
299
|
+
fileNames: { document: 'contract.pdf' },
|
|
300
|
+
}))
|
|
301
|
+
|
|
302
|
+
const commitHash = await tela.getLatestAgentCommit(agentId)
|
|
303
|
+
await tela.runAgentTestCase(agentId, testCase.id, commitHash)
|
|
304
|
+
|
|
305
|
+
const { run, status } = await tela.waitForAgentTestCaseRun(agentId, testCase.id, commitHash)
|
|
306
|
+
console.log('Status:', status)
|
|
307
|
+
console.log('Output:', run.output)
|
|
308
|
+
console.log('Validation:', run.validationSummary)
|
|
309
|
+
"
|
|
310
|
+
```
|
|
311
|
+
|
|
195
312
|
### List Tasks
|
|
196
313
|
```bash
|
|
197
314
|
bun --preload ~/.claude/skills/tela/preload.ts -e "
|
|
@@ -332,7 +449,9 @@ bun localhost.ts "console.log(await tela.listProjects())"
|
|
|
332
449
|
├── prompts.ts # Prompt/Canvas list functions
|
|
333
450
|
├── canvas.ts # Canvas create/update functions
|
|
334
451
|
├── workflows.ts # Workflow create/update/run/validation functions
|
|
335
|
-
├── test-cases.ts # Test case functions
|
|
452
|
+
├── test-cases.ts # Test case functions (prompt/canvas)
|
|
453
|
+
├── agents.ts # Agent editing/execution functions (FCC)
|
|
454
|
+
├── agent-test-cases.ts # Agent test case functions (FCC)
|
|
336
455
|
├── tasks.ts # Task management functions
|
|
337
456
|
├── vault.ts # Vault file management functions
|
|
338
457
|
└── localhost-utils.ts # Docker discovery, JWT signing, workspace fetching (internal, not on tela global)
|
|
@@ -0,0 +1,316 @@
|
|
|
1
|
+
import { beforeEach, describe, expect, mock, test } from 'bun:test'
|
|
2
|
+
|
|
3
|
+
const apiRequest = mock(async (_endpoint: string, _options?: RequestInit): Promise<unknown> => ({ data: {} }))
|
|
4
|
+
|
|
5
|
+
class ApiRequestError extends Error {
|
|
6
|
+
constructor(public readonly status: number, message: string) {
|
|
7
|
+
super(message)
|
|
8
|
+
this.name = 'ApiRequestError'
|
|
9
|
+
}
|
|
10
|
+
}
|
|
11
|
+
|
|
12
|
+
void mock.module('./common.ts', () => ({
|
|
13
|
+
apiRequest,
|
|
14
|
+
ApiRequestError,
|
|
15
|
+
}))
|
|
16
|
+
|
|
17
|
+
describe('agent-test-cases', () => {
|
|
18
|
+
beforeEach(() => {
|
|
19
|
+
apiRequest.mockClear()
|
|
20
|
+
apiRequest.mockImplementation(async () => ({ data: {} }))
|
|
21
|
+
})
|
|
22
|
+
|
|
23
|
+
test('lists test cases with repeated array query params', async () => {
|
|
24
|
+
const { listAgentTestCases } = await import('./agent-test-cases.ts')
|
|
25
|
+
apiRequest.mockImplementation(async () => ({ data: [] }))
|
|
26
|
+
|
|
27
|
+
await listAgentTestCases('agent-1', {
|
|
28
|
+
commitHash: 'abc123',
|
|
29
|
+
title: 'search',
|
|
30
|
+
status: ['failed', 'error'],
|
|
31
|
+
tagIds: ['t1', 't2'],
|
|
32
|
+
createdBy: ['u1'],
|
|
33
|
+
createdAtSince: '2026-01-01T00:00:00Z',
|
|
34
|
+
})
|
|
35
|
+
|
|
36
|
+
const [endpoint] = apiRequest.mock.calls[0]!
|
|
37
|
+
const query = new URLSearchParams(String(endpoint).split('?')[1])
|
|
38
|
+
|
|
39
|
+
expect(String(endpoint).startsWith('/agent/agent-1/tests?')).toBe(true)
|
|
40
|
+
expect(query.get('commitHash')).toBe('abc123')
|
|
41
|
+
expect(query.get('title')).toBe('search')
|
|
42
|
+
expect(query.getAll('status')).toEqual(['failed', 'error'])
|
|
43
|
+
expect(query.getAll('tagIds')).toEqual(['t1', 't2'])
|
|
44
|
+
expect(query.getAll('createdBy')).toEqual(['u1'])
|
|
45
|
+
expect(query.get('createdAtSince')).toBe('2026-01-01T00:00:00Z')
|
|
46
|
+
})
|
|
47
|
+
|
|
48
|
+
test('omits the query string when no options are given', async () => {
|
|
49
|
+
const { listAgentTestCases } = await import('./agent-test-cases.ts')
|
|
50
|
+
apiRequest.mockImplementation(async () => ({ data: [] }))
|
|
51
|
+
|
|
52
|
+
await listAgentTestCases('agent-1')
|
|
53
|
+
|
|
54
|
+
expect(apiRequest.mock.calls[0]![0]).toBe('/agent/agent-1/tests')
|
|
55
|
+
})
|
|
56
|
+
|
|
57
|
+
test('pre-serializes bodies so vault refs are not rewritten by parsePayload', async () => {
|
|
58
|
+
const { createAgentTestCase } = await import('./agent-test-cases.ts')
|
|
59
|
+
|
|
60
|
+
await createAgentTestCase('agent-1', {
|
|
61
|
+
inputs: [{ type: 'file', name: 'doc', vaultRef: 'vault://abc', filename: 'doc.pdf' }],
|
|
62
|
+
})
|
|
63
|
+
|
|
64
|
+
const [endpoint, options] = apiRequest.mock.calls[0]!
|
|
65
|
+
expect(endpoint).toBe('/agent/agent-1/tests')
|
|
66
|
+
expect(options?.method).toBe('POST')
|
|
67
|
+
expect(typeof options?.body).toBe('string')
|
|
68
|
+
|
|
69
|
+
const body = JSON.parse(String(options?.body))
|
|
70
|
+
expect(body.inputs[0].vaultRef).toBe('vault://abc')
|
|
71
|
+
})
|
|
72
|
+
|
|
73
|
+
test('unwraps the { data } envelope', async () => {
|
|
74
|
+
const { createAgentTestCase } = await import('./agent-test-cases.ts')
|
|
75
|
+
apiRequest.mockImplementation(async () => ({ data: { id: 'tc-1', title: 'Test' } }))
|
|
76
|
+
|
|
77
|
+
const testCase = await createAgentTestCase('agent-1', { title: 'Test' })
|
|
78
|
+
|
|
79
|
+
expect(testCase).toEqual({ id: 'tc-1', title: 'Test' } as any)
|
|
80
|
+
})
|
|
81
|
+
|
|
82
|
+
test('runs a test case against a commit', async () => {
|
|
83
|
+
const { runAgentTestCase } = await import('./agent-test-cases.ts')
|
|
84
|
+
apiRequest.mockImplementation(async () => ({ data: { executionId: 'exec-1' } }))
|
|
85
|
+
|
|
86
|
+
const result = await runAgentTestCase('agent-1', 'tc-1', 'abc123')
|
|
87
|
+
|
|
88
|
+
const [endpoint, options] = apiRequest.mock.calls[0]!
|
|
89
|
+
expect(endpoint).toBe('/agent/agent-1/tests/tc-1/run')
|
|
90
|
+
expect(JSON.parse(String(options?.body))).toEqual({ commitHash: 'abc123' })
|
|
91
|
+
expect(result.executionId).toBe('exec-1')
|
|
92
|
+
})
|
|
93
|
+
|
|
94
|
+
test('continues a session with a message', async () => {
|
|
95
|
+
const { continueAgentTestCase } = await import('./agent-test-cases.ts')
|
|
96
|
+
apiRequest.mockImplementation(async () => ({ data: { executionId: 'exec-2' } }))
|
|
97
|
+
|
|
98
|
+
await continueAgentTestCase('agent-1', 'tc-1', 'abc123', 'next turn')
|
|
99
|
+
|
|
100
|
+
const [endpoint, options] = apiRequest.mock.calls[0]!
|
|
101
|
+
expect(endpoint).toBe('/agent/agent-1/tests/tc-1/continue')
|
|
102
|
+
expect(JSON.parse(String(options?.body))).toEqual({ commitHash: 'abc123', message: 'next turn' })
|
|
103
|
+
})
|
|
104
|
+
|
|
105
|
+
test('sends attribute feedback wrapped in attributes', async () => {
|
|
106
|
+
const { updateAgentTestCaseAttributeFeedback } = await import('./agent-test-cases.ts')
|
|
107
|
+
|
|
108
|
+
await updateAgentTestCaseAttributeFeedback('agent-1', 'tc-1', 'abc123', [
|
|
109
|
+
{ path: 'summary', feedback: 1 },
|
|
110
|
+
{ path: 'total', feedback: 0, message: 'wrong' },
|
|
111
|
+
])
|
|
112
|
+
|
|
113
|
+
const [endpoint, options] = apiRequest.mock.calls[0]!
|
|
114
|
+
expect(endpoint).toBe('/agent/agent-1/tests/tc-1/runs/abc123/attribute-feedback')
|
|
115
|
+
expect(options?.method).toBe('PATCH')
|
|
116
|
+
|
|
117
|
+
const body = JSON.parse(String(options?.body))
|
|
118
|
+
expect(body.attributes).toHaveLength(2)
|
|
119
|
+
expect(body.attributes[1]).toEqual({ path: 'total', feedback: 0, message: 'wrong' })
|
|
120
|
+
})
|
|
121
|
+
|
|
122
|
+
test('requires commitHash on the stats endpoint query', async () => {
|
|
123
|
+
const { getAgentTestCaseStats } = await import('./agent-test-cases.ts')
|
|
124
|
+
|
|
125
|
+
await getAgentTestCaseStats('agent-1', 'abc123', { tagIds: ['t1'] })
|
|
126
|
+
|
|
127
|
+
const [endpoint] = apiRequest.mock.calls[0]!
|
|
128
|
+
const query = new URLSearchParams(String(endpoint).split('?')[1])
|
|
129
|
+
|
|
130
|
+
expect(String(endpoint).startsWith('/agent/agent-1/tests/stats?')).toBe(true)
|
|
131
|
+
expect(query.get('commitHash')).toBe('abc123')
|
|
132
|
+
expect(query.getAll('tagIds')).toEqual(['t1'])
|
|
133
|
+
})
|
|
134
|
+
|
|
135
|
+
test('run history keeps pagination fields from the response envelope', async () => {
|
|
136
|
+
const { listAgentTestCaseRuns } = await import('./agent-test-cases.ts')
|
|
137
|
+
// This endpoint's envelope is { data: entries, nextCursor, totalRuns } —
|
|
138
|
+
// pagination fields live beside data, not inside it
|
|
139
|
+
apiRequest.mockImplementation(async () => ({
|
|
140
|
+
data: [{ commitHash: 'abc123', score: 1 }],
|
|
141
|
+
nextCursor: 'cursor-1',
|
|
142
|
+
totalRuns: 5,
|
|
143
|
+
}))
|
|
144
|
+
|
|
145
|
+
const result = await listAgentTestCaseRuns('agent-1', 'tc-1', { limit: 1 })
|
|
146
|
+
|
|
147
|
+
expect(apiRequest.mock.calls[0]![0]).toBe('/agent/agent-1/tests/tc-1/runs?limit=1')
|
|
148
|
+
expect(result.data).toHaveLength(1)
|
|
149
|
+
expect(result.nextCursor).toBe('cursor-1')
|
|
150
|
+
expect(result.totalRuns).toBe(5)
|
|
151
|
+
})
|
|
152
|
+
|
|
153
|
+
describe('buildAgentTestInputs', () => {
|
|
154
|
+
test('splits text and vault file values', async () => {
|
|
155
|
+
const { buildAgentTestInputs } = await import('./agent-test-cases.ts')
|
|
156
|
+
|
|
157
|
+
const inputs = buildAgentTestInputs({
|
|
158
|
+
context: 'Some text',
|
|
159
|
+
document: 'vault://abc123',
|
|
160
|
+
}, { fileNames: { document: 'report.pdf' } })
|
|
161
|
+
|
|
162
|
+
expect(inputs).toEqual([
|
|
163
|
+
{ type: 'text', name: 'context', content: 'Some text' },
|
|
164
|
+
{ type: 'file', name: 'document', vaultRef: 'vault://abc123', filename: 'report.pdf' },
|
|
165
|
+
])
|
|
166
|
+
})
|
|
167
|
+
|
|
168
|
+
test('defaults the filename to the input name', async () => {
|
|
169
|
+
const { buildAgentTestInputs } = await import('./agent-test-cases.ts')
|
|
170
|
+
|
|
171
|
+
const inputs = buildAgentTestInputs({ document: 'vault://abc123' })
|
|
172
|
+
|
|
173
|
+
expect(inputs[0]).toEqual({ type: 'file', name: 'document', vaultRef: 'vault://abc123', filename: 'document' })
|
|
174
|
+
})
|
|
175
|
+
})
|
|
176
|
+
|
|
177
|
+
describe('createAgentTestCasePayload', () => {
|
|
178
|
+
test('builds payload with expectations and derived title', async () => {
|
|
179
|
+
const { createAgentTestCasePayload } = await import('./agent-test-cases.ts')
|
|
180
|
+
|
|
181
|
+
const payload = createAgentTestCasePayload({ input: 'Hello world, this is a long input value' }, {
|
|
182
|
+
evaluationInstructions: 'Must greet back',
|
|
183
|
+
})
|
|
184
|
+
|
|
185
|
+
expect(payload.title).toBe(`Test: ${'Hello world, this is a long input value'.slice(0, 30)}...`)
|
|
186
|
+
expect(payload.evaluationInstructions).toBe('Must greet back')
|
|
187
|
+
expect(payload.inputs).toHaveLength(1)
|
|
188
|
+
})
|
|
189
|
+
|
|
190
|
+
test('falls back to the first filename for file-only cases', async () => {
|
|
191
|
+
const { createAgentTestCasePayload } = await import('./agent-test-cases.ts')
|
|
192
|
+
|
|
193
|
+
const payload = createAgentTestCasePayload({ doc: 'vault://abc' }, { fileNames: { doc: 'contract.pdf' } })
|
|
194
|
+
|
|
195
|
+
expect(payload.title).toBe('Test: contract.pdf')
|
|
196
|
+
})
|
|
197
|
+
})
|
|
198
|
+
|
|
199
|
+
describe('getAgentTestCaseRunStatus', () => {
|
|
200
|
+
const baseRun = {
|
|
201
|
+
executionSessionId: null as string | null,
|
|
202
|
+
executionStatus: null as any,
|
|
203
|
+
evaluationSessionId: null as string | null,
|
|
204
|
+
evaluationStatus: null as any,
|
|
205
|
+
feedback: null as 0 | 1 | null,
|
|
206
|
+
}
|
|
207
|
+
|
|
208
|
+
test('derives every status branch', async () => {
|
|
209
|
+
const { getAgentTestCaseRunStatus } = await import('./agent-test-cases.ts')
|
|
210
|
+
|
|
211
|
+
expect(getAgentTestCaseRunStatus(null)).toBe('new')
|
|
212
|
+
expect(getAgentTestCaseRunStatus({ ...baseRun, executionStatus: 'error' })).toBe('error')
|
|
213
|
+
expect(getAgentTestCaseRunStatus({ ...baseRun, executionStatus: 'completed', evaluationStatus: 'error' })).toBe('error')
|
|
214
|
+
expect(getAgentTestCaseRunStatus({ ...baseRun, executionStatus: 'running' })).toBe('running')
|
|
215
|
+
expect(getAgentTestCaseRunStatus({ ...baseRun, executionStatus: 'completed', evaluationSessionId: 's', evaluationStatus: 'running' })).toBe('running')
|
|
216
|
+
expect(getAgentTestCaseRunStatus({ ...baseRun, executionStatus: 'completed' })).toBe('executed')
|
|
217
|
+
expect(getAgentTestCaseRunStatus({ ...baseRun, executionStatus: 'completed', evaluationSessionId: 's', evaluationStatus: 'completed', feedback: 1 })).toBe('success')
|
|
218
|
+
expect(getAgentTestCaseRunStatus({ ...baseRun, executionStatus: 'completed', evaluationSessionId: 's', evaluationStatus: 'completed', feedback: 0 })).toBe('failed')
|
|
219
|
+
expect(getAgentTestCaseRunStatus({ ...baseRun, executionStatus: 'completed', evaluationSessionId: 's', evaluationStatus: 'failed' })).toBe('failed')
|
|
220
|
+
expect(getAgentTestCaseRunStatus({ ...baseRun, executionSessionId: 's' })).toBe('executed')
|
|
221
|
+
})
|
|
222
|
+
})
|
|
223
|
+
|
|
224
|
+
describe('waitForAgentTestCaseRun', () => {
|
|
225
|
+
test('retries 404s until the run appears, then resolves on terminal status', async () => {
|
|
226
|
+
const { waitForAgentTestCaseRun } = await import('./agent-test-cases.ts')
|
|
227
|
+
|
|
228
|
+
let calls = 0
|
|
229
|
+
apiRequest.mockImplementation(async () => {
|
|
230
|
+
calls++
|
|
231
|
+
if (calls === 1)
|
|
232
|
+
throw new ApiRequestError(404, 'API request failed: 404 Not Found - {}')
|
|
233
|
+
if (calls === 2) {
|
|
234
|
+
return { data: { executionStatus: 'running', executionSessionId: 's', evaluationSessionId: null, evaluationStatus: null, feedback: null } }
|
|
235
|
+
}
|
|
236
|
+
return { data: { executionStatus: 'completed', executionSessionId: 's', evaluationSessionId: 'e', evaluationStatus: 'completed', feedback: 1 } }
|
|
237
|
+
})
|
|
238
|
+
|
|
239
|
+
const { status } = await waitForAgentTestCaseRun('agent-1', 'tc-1', 'abc123', { intervalMs: 1 })
|
|
240
|
+
|
|
241
|
+
expect(status).toBe('success')
|
|
242
|
+
expect(calls).toBe(3)
|
|
243
|
+
})
|
|
244
|
+
|
|
245
|
+
test('stops at executed when until is executed', async () => {
|
|
246
|
+
const { waitForAgentTestCaseRun } = await import('./agent-test-cases.ts')
|
|
247
|
+
|
|
248
|
+
apiRequest.mockImplementation(async () => ({
|
|
249
|
+
data: { executionStatus: 'completed', executionSessionId: 's', evaluationSessionId: null, evaluationStatus: null, feedback: null },
|
|
250
|
+
}))
|
|
251
|
+
|
|
252
|
+
const { status } = await waitForAgentTestCaseRun('agent-1', 'tc-1', 'abc123', { intervalMs: 1, until: 'executed' })
|
|
253
|
+
|
|
254
|
+
expect(status).toBe('executed')
|
|
255
|
+
})
|
|
256
|
+
|
|
257
|
+
test('throws after maxAttempts without a terminal status', async () => {
|
|
258
|
+
const { waitForAgentTestCaseRun } = await import('./agent-test-cases.ts')
|
|
259
|
+
|
|
260
|
+
apiRequest.mockImplementation(async () => ({
|
|
261
|
+
data: { executionStatus: 'running', executionSessionId: 's', evaluationSessionId: null, evaluationStatus: null, feedback: null },
|
|
262
|
+
}))
|
|
263
|
+
|
|
264
|
+
expect(waitForAgentTestCaseRun('agent-1', 'tc-1', 'abc123', { intervalMs: 1, maxAttempts: 2 }))
|
|
265
|
+
.rejects
|
|
266
|
+
.toThrow('did not complete within timeout')
|
|
267
|
+
})
|
|
268
|
+
|
|
269
|
+
test('rethrows non-404 API errors', async () => {
|
|
270
|
+
const { waitForAgentTestCaseRun } = await import('./agent-test-cases.ts')
|
|
271
|
+
|
|
272
|
+
apiRequest.mockImplementation(async () => {
|
|
273
|
+
throw new ApiRequestError(500, 'API request failed: 500 Internal Server Error - {}')
|
|
274
|
+
})
|
|
275
|
+
|
|
276
|
+
expect(waitForAgentTestCaseRun('agent-1', 'tc-1', 'abc123', { intervalMs: 1 }))
|
|
277
|
+
.rejects
|
|
278
|
+
.toThrow('500')
|
|
279
|
+
})
|
|
280
|
+
|
|
281
|
+
test('rethrows untyped errors even when the message mentions 404', async () => {
|
|
282
|
+
const { waitForAgentTestCaseRun } = await import('./agent-test-cases.ts')
|
|
283
|
+
|
|
284
|
+
apiRequest.mockImplementation(async () => {
|
|
285
|
+
throw new Error('fetch failed with 404 somewhere')
|
|
286
|
+
})
|
|
287
|
+
|
|
288
|
+
expect(waitForAgentTestCaseRun('agent-1', 'tc-1', 'abc123', { intervalMs: 1 }))
|
|
289
|
+
.rejects
|
|
290
|
+
.toThrow('fetch failed')
|
|
291
|
+
})
|
|
292
|
+
})
|
|
293
|
+
|
|
294
|
+
describe('getLatestAgentCommit', () => {
|
|
295
|
+
test('returns the newest commit hash', async () => {
|
|
296
|
+
const { getLatestAgentCommit } = await import('./agent-test-cases.ts')
|
|
297
|
+
apiRequest.mockImplementation(async () => ({
|
|
298
|
+
data: { commits: [{ commitHash: 'abc123' }], branch: 'main', pagination: { page: 1, limit: 1, totalCount: 5, hasMore: true } },
|
|
299
|
+
}))
|
|
300
|
+
|
|
301
|
+
const hash = await getLatestAgentCommit('agent-1')
|
|
302
|
+
|
|
303
|
+
expect(apiRequest.mock.calls[0]![0]).toBe('/agent/agent-1/history?limit=1')
|
|
304
|
+
expect(hash).toBe('abc123')
|
|
305
|
+
})
|
|
306
|
+
|
|
307
|
+
test('throws when the agent has no commits', async () => {
|
|
308
|
+
const { getLatestAgentCommit } = await import('./agent-test-cases.ts')
|
|
309
|
+
apiRequest.mockImplementation(async () => ({
|
|
310
|
+
data: { commits: [], branch: 'main', pagination: { page: 1, limit: 1, totalCount: 0, hasMore: false } },
|
|
311
|
+
}))
|
|
312
|
+
|
|
313
|
+
expect(getLatestAgentCommit('agent-1')).rejects.toThrow('has no commits')
|
|
314
|
+
})
|
|
315
|
+
})
|
|
316
|
+
})
|