@databricks/appkit 0.73.0 → 0.74.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CLAUDE.md +7 -0
- package/dist/appkit/package.js +1 -1
- package/dist/beta.d.ts +6 -6
- package/dist/beta.js +6 -6
- package/dist/cli/commands/agent/eval.js +82 -21
- package/dist/cli/commands/agent/eval.js.map +1 -1
- package/dist/evals/dataset.d.ts +14 -1
- package/dist/evals/dataset.d.ts.map +1 -1
- package/dist/evals/dataset.js +16 -1
- package/dist/evals/dataset.js.map +1 -1
- package/dist/evals/define-eval.d.ts +4 -2
- package/dist/evals/define-eval.d.ts.map +1 -1
- package/dist/evals/define-eval.js +5 -1
- package/dist/evals/define-eval.js.map +1 -1
- package/dist/evals/discover.d.ts +15 -1
- package/dist/evals/discover.d.ts.map +1 -1
- package/dist/evals/discover.js +26 -2
- package/dist/evals/discover.js.map +1 -1
- package/dist/evals/http-driver.d.ts.map +1 -1
- package/dist/evals/http-driver.js +31 -9
- package/dist/evals/http-driver.js.map +1 -1
- package/dist/evals/index.d.ts +5 -5
- package/dist/evals/index.js +5 -5
- package/dist/evals/report.d.ts +16 -1
- package/dist/evals/report.d.ts.map +1 -1
- package/dist/evals/report.js +64 -2
- package/dist/evals/report.js.map +1 -1
- package/dist/evals/run-eval.d.ts +5 -0
- package/dist/evals/run-eval.d.ts.map +1 -1
- package/dist/evals/run-eval.js +44 -3
- package/dist/evals/run-eval.js.map +1 -1
- package/dist/evals/run-evals.d.ts +32 -3
- package/dist/evals/run-evals.d.ts.map +1 -1
- package/dist/evals/run-evals.js +128 -26
- package/dist/evals/run-evals.js.map +1 -1
- package/dist/evals/types.d.ts +40 -2
- package/dist/evals/types.d.ts.map +1 -1
- package/dist/plugins/server/index.js +2 -2
- package/dist/plugins/server/index.js.map +1 -1
- package/dist/plugins/server/remote-tunnel/remote-tunnel-manager.js +3 -3
- package/dist/plugins/server/remote-tunnel/remote-tunnel-manager.js.map +1 -1
- package/dist/plugins/server/static-server.js +3 -3
- package/dist/plugins/server/static-server.js.map +1 -1
- package/dist/plugins/server/utils.js +3 -3
- package/dist/plugins/server/utils.js.map +1 -1
- package/dist/plugins/server/vite-dev-server.js +4 -4
- package/dist/plugins/server/vite-dev-server.js.map +1 -1
- package/dist/shared/src/schemas/manifest.d.ts +54 -54
- package/dist/type-generator/database/generate.js +3 -3
- package/dist/type-generator/database/generate.js.map +1 -1
- package/dist/type-generator/migration.js +2 -2
- package/dist/type-generator/migration.js.map +1 -1
- package/dist/type-generator/serving/server-file-extractor.js +3 -3
- package/dist/type-generator/serving/server-file-extractor.js.map +1 -1
- package/docs/api/appkit/Function.defineEvalConfig.md +18 -0
- package/docs/api/appkit/Function.discoverEvalConfigs.md +18 -0
- package/docs/api/appkit/Function.formatResultsJUnit.md +18 -0
- package/docs/api/appkit/Function.formatResultsJson.md +18 -0
- package/docs/api/appkit/Function.runWithRetries.md +28 -0
- package/docs/api/appkit/Function.userTurns.md +20 -0
- package/docs/api/appkit/Interface.DiscoveredEvalConfig.md +25 -0
- package/docs/api/appkit/Interface.DriveResult.md +28 -0
- package/docs/api/appkit/Interface.EvalDefinition.md +22 -0
- package/docs/api/appkit/Interface.EvalDriver.md +10 -4
- package/docs/api/appkit/Interface.EvalResult.md +11 -0
- package/docs/api/appkit/Interface.EvalSummary.md +11 -0
- package/docs/api/appkit/Interface.RunEvalOptions.md +11 -0
- package/docs/api/appkit/Interface.RunEvalsOptions.md +23 -1
- package/docs/api/appkit/Interface.TestContext.md +30 -8
- package/docs/api/appkit.md +7 -0
- package/llms.txt +7 -0
- package/package.json +1 -1
- package/sbom.cdx.json +1 -1
|
@@ -0,0 +1,18 @@
|
|
|
1
|
+
# Function: discoverEvalConfigs()
|
|
2
|
+
|
|
3
|
+
```ts
|
|
4
|
+
function discoverEvalConfigs(rootDir: string): DiscoveredEvalConfig[];
|
|
5
|
+
|
|
6
|
+
```
|
|
7
|
+
|
|
8
|
+
Discover the per-agent `evals.config.ts` (from [defineEvalConfig](./docs/api/appkit/Function.defineEvalConfig.md)) at `<rootDir>/server/agents/<agent>/evals/evals.config.ts`. Config is per-agent: each agent's config applies only to that agent's evals. Agents without a config file are omitted. Returns a stable, sorted list.
|
|
9
|
+
|
|
10
|
+
## Parameters[](#parameters "Direct link to Parameters")
|
|
11
|
+
|
|
12
|
+
| Parameter | Type |
|
|
13
|
+
| --------- | -------- |
|
|
14
|
+
| `rootDir` | `string` |
|
|
15
|
+
|
|
16
|
+
## Returns[](#returns "Direct link to Returns")
|
|
17
|
+
|
|
18
|
+
[`DiscoveredEvalConfig`](./docs/api/appkit/Interface.DiscoveredEvalConfig.md)\[]
|
|
@@ -0,0 +1,18 @@
|
|
|
1
|
+
# Function: formatResultsJUnit()
|
|
2
|
+
|
|
3
|
+
```ts
|
|
4
|
+
function formatResultsJUnit(results: EvalResult[]): string;
|
|
5
|
+
|
|
6
|
+
```
|
|
7
|
+
|
|
8
|
+
Render results as JUnit XML for standard CI test reporters: a single `<testsuite name="appkit-agent-evals">` with one `<testcase>` per result. Failures carry a `<failure>` (error or failing-gate summary); skips a `<skipped>`. All attribute/text values are XML-escaped.
|
|
9
|
+
|
|
10
|
+
## Parameters[](#parameters "Direct link to Parameters")
|
|
11
|
+
|
|
12
|
+
| Parameter | Type |
|
|
13
|
+
| --------- | ------------------------------------------------------------------ |
|
|
14
|
+
| `results` | [`EvalResult`](./docs/api/appkit/Interface.EvalResult.md)\[] |
|
|
15
|
+
|
|
16
|
+
## Returns[](#returns "Direct link to Returns")
|
|
17
|
+
|
|
18
|
+
`string`
|
|
@@ -0,0 +1,18 @@
|
|
|
1
|
+
# Function: formatResultsJson()
|
|
2
|
+
|
|
3
|
+
```ts
|
|
4
|
+
function formatResultsJson(results: EvalResult[]): string;
|
|
5
|
+
|
|
6
|
+
```
|
|
7
|
+
|
|
8
|
+
Render results as a machine-readable JSON report (2-space indented): `{ summary: EvalSummary, results: EvalResult[] }`. Faithful to the types — every field present on a result round-trips.
|
|
9
|
+
|
|
10
|
+
## Parameters[](#parameters "Direct link to Parameters")
|
|
11
|
+
|
|
12
|
+
| Parameter | Type |
|
|
13
|
+
| --------- | ------------------------------------------------------------------ |
|
|
14
|
+
| `results` | [`EvalResult`](./docs/api/appkit/Interface.EvalResult.md)\[] |
|
|
15
|
+
|
|
16
|
+
## Returns[](#returns "Direct link to Returns")
|
|
17
|
+
|
|
18
|
+
`string`
|
|
@@ -0,0 +1,28 @@
|
|
|
1
|
+
# Function: runWithRetries()
|
|
2
|
+
|
|
3
|
+
```ts
|
|
4
|
+
function runWithRetries(
|
|
5
|
+
retries: number,
|
|
6
|
+
attempt: (attemptNumber: number) => Promise<EvalResult>,
|
|
7
|
+
options: {
|
|
8
|
+
baseDelayMs?: number;
|
|
9
|
+
}): Promise<EvalResult>;
|
|
10
|
+
|
|
11
|
+
```
|
|
12
|
+
|
|
13
|
+
Run `attempt` up to `1 + retries` times, stopping as soon as it returns a result that is neither a thrown error / per-eval timeout (`error`) nor a transport/agent turn failure (`infraFailure`). Assertion failures set neither, so a failed-but-completed eval is returned on the first try and never retried. Returns the last result when every attempt failed on infra.
|
|
14
|
+
|
|
15
|
+
Between attempts it waits a full-jittered exponential backoff (infra flakes are overload-correlated). `retries` is coerced to a finite non-negative integer; `baseDelayMs: 0` disables the wait (tests).
|
|
16
|
+
|
|
17
|
+
## Parameters[](#parameters "Direct link to Parameters")
|
|
18
|
+
|
|
19
|
+
| Parameter | Type |
|
|
20
|
+
| ---------------------- | --------------------------------------------------------------------------------------------------------- |
|
|
21
|
+
| `retries` | `number` |
|
|
22
|
+
| `attempt` | (`attemptNumber`: `number`) => `Promise`<[`EvalResult`](./docs/api/appkit/Interface.EvalResult.md)> |
|
|
23
|
+
| `options` | { `baseDelayMs?`: `number`; } |
|
|
24
|
+
| `options.baseDelayMs?` | `number` |
|
|
25
|
+
|
|
26
|
+
## Returns[](#returns "Direct link to Returns")
|
|
27
|
+
|
|
28
|
+
`Promise`<[`EvalResult`](./docs/api/appkit/Interface.EvalResult.md)>
|
|
@@ -0,0 +1,20 @@
|
|
|
1
|
+
# Function: userTurns()
|
|
2
|
+
|
|
3
|
+
```ts
|
|
4
|
+
function userTurns(input: Record<string, unknown>): string[];
|
|
5
|
+
|
|
6
|
+
```
|
|
7
|
+
|
|
8
|
+
Extract every user-message content, in order, from an MLflow `{"messages":[{"role":"user","content":"..."}]}` input. A dataset row can carry a full multi-turn conversation; replaying these against one thread (one `t.send` per returned string) lets the agent see the accumulating history.
|
|
9
|
+
|
|
10
|
+
Only `role === "user"` turns are returned — any interleaved `assistant`/ `system` messages in the row are ignored, since the agent generates its own responses; you never inject the dataset's assistant turns. A single-user-turn row yields a one-element array (backward compatible); a row with no `messages` yields `[]`.
|
|
11
|
+
|
|
12
|
+
## Parameters[](#parameters "Direct link to Parameters")
|
|
13
|
+
|
|
14
|
+
| Parameter | Type |
|
|
15
|
+
| --------- | ----------------------------- |
|
|
16
|
+
| `input` | `Record`<`string`, `unknown`> |
|
|
17
|
+
|
|
18
|
+
## Returns[](#returns "Direct link to Returns")
|
|
19
|
+
|
|
20
|
+
`string`\[]
|
|
@@ -0,0 +1,25 @@
|
|
|
1
|
+
# Interface: DiscoveredEvalConfig
|
|
2
|
+
|
|
3
|
+
A per-agent `evals.config.ts` found under `server/agents/<agent>/evals/`.
|
|
4
|
+
|
|
5
|
+
## Properties[](#properties "Direct link to Properties")
|
|
6
|
+
|
|
7
|
+
### agent[](#agent "Direct link to agent")
|
|
8
|
+
|
|
9
|
+
```ts
|
|
10
|
+
agent: string;
|
|
11
|
+
|
|
12
|
+
```
|
|
13
|
+
|
|
14
|
+
The agent id whose evals this config applies to.
|
|
15
|
+
|
|
16
|
+
***
|
|
17
|
+
|
|
18
|
+
### file[](#file "Direct link to file")
|
|
19
|
+
|
|
20
|
+
```ts
|
|
21
|
+
file: string;
|
|
22
|
+
|
|
23
|
+
```
|
|
24
|
+
|
|
25
|
+
Absolute path to the `evals.config.ts` file.
|
|
@@ -37,6 +37,34 @@ Whether the turn completed without an agent/stream error.
|
|
|
37
37
|
|
|
38
38
|
***
|
|
39
39
|
|
|
40
|
+
### toolCallDetails[](#toolcalldetails "Direct link to toolCallDetails")
|
|
41
|
+
|
|
42
|
+
```ts
|
|
43
|
+
toolCallDetails: {
|
|
44
|
+
args: Record<string, unknown>;
|
|
45
|
+
name: string;
|
|
46
|
+
}[];
|
|
47
|
+
|
|
48
|
+
```
|
|
49
|
+
|
|
50
|
+
Tool calls with their parsed arguments, in call order.
|
|
51
|
+
|
|
52
|
+
#### args[](#args "Direct link to args")
|
|
53
|
+
|
|
54
|
+
```ts
|
|
55
|
+
args: Record<string, unknown>;
|
|
56
|
+
|
|
57
|
+
```
|
|
58
|
+
|
|
59
|
+
#### name[](#name "Direct link to name")
|
|
60
|
+
|
|
61
|
+
```ts
|
|
62
|
+
name: string;
|
|
63
|
+
|
|
64
|
+
```
|
|
65
|
+
|
|
66
|
+
***
|
|
67
|
+
|
|
40
68
|
### toolCalls[](#toolcalls "Direct link to toolCalls")
|
|
41
69
|
|
|
42
70
|
```ts
|
|
@@ -52,6 +52,28 @@ optional description: string;
|
|
|
52
52
|
|
|
53
53
|
Short human description, shown in reports.
|
|
54
54
|
|
|
55
|
+
***
|
|
56
|
+
|
|
57
|
+
### tags?[](#tags "Direct link to tags?")
|
|
58
|
+
|
|
59
|
+
```ts
|
|
60
|
+
optional tags: string[];
|
|
61
|
+
|
|
62
|
+
```
|
|
63
|
+
|
|
64
|
+
Free-form tags for filtering (see the runner's `tags` / `--tag` option).
|
|
65
|
+
|
|
66
|
+
***
|
|
67
|
+
|
|
68
|
+
### timeoutMs?[](#timeoutms "Direct link to timeoutMs?")
|
|
69
|
+
|
|
70
|
+
```ts
|
|
71
|
+
optional timeoutMs: number;
|
|
72
|
+
|
|
73
|
+
```
|
|
74
|
+
|
|
75
|
+
Per-eval timeout (ms): `runEval` races the test against it and records a non-passing result instead of hanging. Overrides the runner/CLI default.
|
|
76
|
+
|
|
55
77
|
## Methods[](#methods "Direct link to Methods")
|
|
56
78
|
|
|
57
79
|
### test()[](#test "Direct link to test()")
|
|
@@ -22,15 +22,21 @@ Drop the current conversation so the next `send` starts a fresh thread. Optional
|
|
|
22
22
|
### send()[](#send "Direct link to send()")
|
|
23
23
|
|
|
24
24
|
```ts
|
|
25
|
-
send(message: string
|
|
25
|
+
send(message: string, options?: {
|
|
26
|
+
signal?: AbortSignal;
|
|
27
|
+
}): Promise<DriveResult>;
|
|
26
28
|
|
|
27
29
|
```
|
|
28
30
|
|
|
31
|
+
Drive one turn. `options.signal`, when provided, aborts the in-flight turn: the runner passes its per-eval timeout signal so a timed-out eval cancels the request instead of leaking a live stream.
|
|
32
|
+
|
|
29
33
|
#### Parameters[](#parameters "Direct link to Parameters")
|
|
30
34
|
|
|
31
|
-
| Parameter
|
|
32
|
-
|
|
|
33
|
-
| `message`
|
|
35
|
+
| Parameter | Type |
|
|
36
|
+
| ----------------- | ----------------------------- |
|
|
37
|
+
| `message` | `string` |
|
|
38
|
+
| `options?` | { `signal?`: `AbortSignal`; } |
|
|
39
|
+
| `options.signal?` | `AbortSignal` |
|
|
34
40
|
|
|
35
41
|
#### Returns[](#returns-1 "Direct link to Returns")
|
|
36
42
|
|
|
@@ -42,6 +42,17 @@ id: string;
|
|
|
42
42
|
|
|
43
43
|
***
|
|
44
44
|
|
|
45
|
+
### infraFailure?[](#infrafailure "Direct link to infraFailure?")
|
|
46
|
+
|
|
47
|
+
```ts
|
|
48
|
+
optional infraFailure: boolean;
|
|
49
|
+
|
|
50
|
+
```
|
|
51
|
+
|
|
52
|
+
A turn failed at the transport/agent level (`succeeded: false`), not on an assertion — a retryable infra flake, distinct from `error`.
|
|
53
|
+
|
|
54
|
+
***
|
|
55
|
+
|
|
45
56
|
### passed[](#passed "Direct link to passed")
|
|
46
57
|
|
|
47
58
|
```ts
|
|
@@ -31,6 +31,17 @@ passed: number;
|
|
|
31
31
|
|
|
32
32
|
***
|
|
33
33
|
|
|
34
|
+
### passRate[](#passrate "Direct link to passRate")
|
|
35
|
+
|
|
36
|
+
```ts
|
|
37
|
+
passRate: number;
|
|
38
|
+
|
|
39
|
+
```
|
|
40
|
+
|
|
41
|
+
Fraction of scored (non-skipped) evals that passed, 0..1 (1 when none scored).
|
|
42
|
+
|
|
43
|
+
***
|
|
44
|
+
|
|
34
45
|
### skipped[](#skipped "Direct link to skipped")
|
|
35
46
|
|
|
36
47
|
```ts
|
|
@@ -43,3 +43,14 @@ optional strict: boolean;
|
|
|
43
43
|
```
|
|
44
44
|
|
|
45
45
|
When true, soft assertion failures also fail the eval.
|
|
46
|
+
|
|
47
|
+
***
|
|
48
|
+
|
|
49
|
+
### timeoutMs?[](#timeoutms "Direct link to timeoutMs?")
|
|
50
|
+
|
|
51
|
+
```ts
|
|
52
|
+
optional timeoutMs: number;
|
|
53
|
+
|
|
54
|
+
```
|
|
55
|
+
|
|
56
|
+
Runner-level default per-eval timeout (ms). `def.timeoutMs` wins over this; when both are unset the eval runs unbounded (current behavior).
|
|
@@ -160,6 +160,17 @@ Progress callback, invoked as evals are discovered, started, and finished.
|
|
|
160
160
|
|
|
161
161
|
***
|
|
162
162
|
|
|
163
|
+
### retries?[](#retries "Direct link to retries?")
|
|
164
|
+
|
|
165
|
+
```ts
|
|
166
|
+
optional retries: number;
|
|
167
|
+
|
|
168
|
+
```
|
|
169
|
+
|
|
170
|
+
Re-run an eval up to this many extra times when it fails on infrastructure — a thrown error/timeout (`result.error`) or a transport/agent turn failure (`result.infraFailure`). Assertion failures are never retried. Defaults to `0`.
|
|
171
|
+
|
|
172
|
+
***
|
|
173
|
+
|
|
163
174
|
### rootDir?[](#rootdir "Direct link to rootDir?")
|
|
164
175
|
|
|
165
176
|
```ts
|
|
@@ -182,6 +193,17 @@ Soft assertion failures also fail the eval.
|
|
|
182
193
|
|
|
183
194
|
***
|
|
184
195
|
|
|
196
|
+
### tags?[](#tags "Direct link to tags?")
|
|
197
|
+
|
|
198
|
+
```ts
|
|
199
|
+
optional tags: string[];
|
|
200
|
+
|
|
201
|
+
```
|
|
202
|
+
|
|
203
|
+
Only run evals whose `tags` intersect this list. Empty/undefined runs all. Tags live on the eval def, so filtering happens after each file is loaded.
|
|
204
|
+
|
|
205
|
+
***
|
|
206
|
+
|
|
185
207
|
### timeoutMs?[](#timeoutms "Direct link to timeoutMs?")
|
|
186
208
|
|
|
187
209
|
```ts
|
|
@@ -189,7 +211,7 @@ optional timeoutMs: number;
|
|
|
189
211
|
|
|
190
212
|
```
|
|
191
213
|
|
|
192
|
-
|
|
214
|
+
Default per-eval timeout (ms): `runEval` races the whole test against it and it also caps each driver turn. A per-eval `def.timeoutMs` overrides it, and it wins over an agent's `evals.config.ts` `timeoutMs`. Unbounded when unset.
|
|
193
215
|
|
|
194
216
|
***
|
|
195
217
|
|
|
@@ -152,6 +152,28 @@ Assert a tool was called during the run (gate by default).
|
|
|
152
152
|
|
|
153
153
|
***
|
|
154
154
|
|
|
155
|
+
### calledToolWith()[](#calledtoolwith "Direct link to calledToolWith()")
|
|
156
|
+
|
|
157
|
+
```ts
|
|
158
|
+
calledToolWith(name: string, expected: Record<string, unknown>): AssertionHandle;
|
|
159
|
+
|
|
160
|
+
```
|
|
161
|
+
|
|
162
|
+
Assert a tool was called with arguments that deep-contain `expected`: every key in `expected` must equal the actual argument (recursively for nested objects; arrays match element-for-element), so extra arguments are ignored. Gate by default.
|
|
163
|
+
|
|
164
|
+
#### Parameters[](#parameters-4 "Direct link to Parameters")
|
|
165
|
+
|
|
166
|
+
| Parameter | Type |
|
|
167
|
+
| ---------- | ----------------------------- |
|
|
168
|
+
| `name` | `string` |
|
|
169
|
+
| `expected` | `Record`<`string`, `unknown`> |
|
|
170
|
+
|
|
171
|
+
#### Returns[](#returns-4 "Direct link to Returns")
|
|
172
|
+
|
|
173
|
+
[`AssertionHandle`](./docs/api/appkit/Interface.AssertionHandle.md)
|
|
174
|
+
|
|
175
|
+
***
|
|
176
|
+
|
|
155
177
|
### check()[](#check "Direct link to check()")
|
|
156
178
|
|
|
157
179
|
```ts
|
|
@@ -161,14 +183,14 @@ check(value: string, matcher: Matcher): AssertionHandle;
|
|
|
161
183
|
|
|
162
184
|
Assert a value against a matcher, e.g. `t.check(t.reply, includes("Sunny"))`.
|
|
163
185
|
|
|
164
|
-
#### Parameters[](#parameters-
|
|
186
|
+
#### Parameters[](#parameters-5 "Direct link to Parameters")
|
|
165
187
|
|
|
166
188
|
| Parameter | Type |
|
|
167
189
|
| --------- | --------------------------------------------------------- |
|
|
168
190
|
| `value` | `string` |
|
|
169
191
|
| `matcher` | [`Matcher`](./docs/api/appkit/TypeAlias.Matcher.md) |
|
|
170
192
|
|
|
171
|
-
#### Returns[](#returns-
|
|
193
|
+
#### Returns[](#returns-5 "Direct link to Returns")
|
|
172
194
|
|
|
173
195
|
[`AssertionHandle`](./docs/api/appkit/Interface.AssertionHandle.md)
|
|
174
196
|
|
|
@@ -183,7 +205,7 @@ reset(): void;
|
|
|
183
205
|
|
|
184
206
|
Start a fresh conversation: the next `send` opens a new thread with no history. Use to run several independent one-shot checks in one test. Consecutive `send`s (without a `reset`) stay in one multi-turn conversation.
|
|
185
207
|
|
|
186
|
-
#### Returns[](#returns-
|
|
208
|
+
#### Returns[](#returns-6 "Direct link to Returns")
|
|
187
209
|
|
|
188
210
|
`void`
|
|
189
211
|
|
|
@@ -198,13 +220,13 @@ send(message: string): Promise<void>;
|
|
|
198
220
|
|
|
199
221
|
Send a user message to the agent and capture its response.
|
|
200
222
|
|
|
201
|
-
#### Parameters[](#parameters-
|
|
223
|
+
#### Parameters[](#parameters-6 "Direct link to Parameters")
|
|
202
224
|
|
|
203
225
|
| Parameter | Type |
|
|
204
226
|
| --------- | -------- |
|
|
205
227
|
| `message` | `string` |
|
|
206
228
|
|
|
207
|
-
#### Returns[](#returns-
|
|
229
|
+
#### Returns[](#returns-7 "Direct link to Returns")
|
|
208
230
|
|
|
209
231
|
`Promise`<`void`>
|
|
210
232
|
|
|
@@ -219,13 +241,13 @@ skip(reason?: string): never;
|
|
|
219
241
|
|
|
220
242
|
Skip this eval with an optional reason.
|
|
221
243
|
|
|
222
|
-
#### Parameters[](#parameters-
|
|
244
|
+
#### Parameters[](#parameters-7 "Direct link to Parameters")
|
|
223
245
|
|
|
224
246
|
| Parameter | Type |
|
|
225
247
|
| --------- | -------- |
|
|
226
248
|
| `reason?` | `string` |
|
|
227
249
|
|
|
228
|
-
#### Returns[](#returns-
|
|
250
|
+
#### Returns[](#returns-8 "Direct link to Returns")
|
|
229
251
|
|
|
230
252
|
`never`
|
|
231
253
|
|
|
@@ -240,6 +262,6 @@ succeeded(): AssertionHandle;
|
|
|
240
262
|
|
|
241
263
|
Assert the last turn completed successfully (gate by default).
|
|
242
264
|
|
|
243
|
-
#### Returns[](#returns-
|
|
265
|
+
#### Returns[](#returns-9 "Direct link to Returns")
|
|
244
266
|
|
|
245
267
|
[`AssertionHandle`](./docs/api/appkit/Interface.AssertionHandle.md)
|
package/docs/api/appkit.md
CHANGED
|
@@ -54,6 +54,7 @@ Documentation merge entry for Typedoc — combines the stable `@databricks/appki
|
|
|
54
54
|
| [DatabricksAuth](./docs/api/appkit/Interface.DatabricksAuth.md) | Resolved Databricks host + bearer token for the eval runner's REST calls. |
|
|
55
55
|
| [DatasetRow](./docs/api/appkit/Interface.DatasetRow.md) | One row of a managed evaluation dataset. `inputs` are the kwargs passed to the agent for the turn; `expectations` (when present) is the row's ground truth / guidelines. Mirrors the `{inputs, expectations}` shape of `mlflow.genai` datasets and of the Unity Catalog table backing a managed eval dataset. |
|
|
56
56
|
| [DiscoveredEval](./docs/api/appkit/Interface.DiscoveredEval.md) | An eval file found under `server/agents/<agent>/evals/`. |
|
|
57
|
+
| [DiscoveredEvalConfig](./docs/api/appkit/Interface.DiscoveredEvalConfig.md) | A per-agent `evals.config.ts` found under `server/agents/<agent>/evals/`. |
|
|
57
58
|
| [DriveResult](./docs/api/appkit/Interface.DriveResult.md) | What a driver returns for a single `t.send`. |
|
|
58
59
|
| [EndpointConfig](./docs/api/appkit/Interface.EndpointConfig.md) | - |
|
|
59
60
|
| [EntityMutationHooks](./docs/api/appkit/Interface.EntityMutationHooks.md) | Mutation lifecycle for one entity. A before hook may return a replacement payload, which is revalidated against the trusted schema before it is persisted. Every hook, the mutation, and any write a hook issues through `ctx.app.database` share one transaction, so a rejection anywhere rolls all of them back. Throw `DatabaseValidationError` to answer a generated route with `422`; any other failure stays an opaque server error. |
|
|
@@ -198,9 +199,11 @@ Documentation merge entry for Typedoc — combines the stable `@databricks/appki
|
|
|
198
199
|
| [createWorkspaceClient](./docs/api/appkit/Function.createWorkspaceClient.md) | Construct an AppKit workspace client. |
|
|
199
200
|
| [database](./docs/api/appkit/Function.database.md) | Create a typed database plugin registration for a finalized schema. |
|
|
200
201
|
| [defineEval](./docs/api/appkit/Function.defineEval.md) | Define an agent eval. Default-export the result from a `server/agents/<id>/evals/*.eval.ts` file. |
|
|
202
|
+
| [defineEvalConfig](./docs/api/appkit/Function.defineEvalConfig.md) | Define per-directory eval config. Default-export from `evals.config.ts`. |
|
|
201
203
|
| [defineManifest](./docs/api/appkit/Function.defineManifest.md) | Validates a raw manifest (typically a `manifest.json` import) against the canonical Zod schema and returns it as a strict [PluginManifest](./docs/api/appkit/Interface.PluginManifest.md). |
|
|
202
204
|
| [defineSchema](./docs/api/appkit/Function.defineSchema.md) | Compile one declared schema. The returned type keeps the table names the builder returned, so `api.tables` and `hooks` can name only real tables. |
|
|
203
205
|
| [defineTool](./docs/api/appkit/Function.defineTool.md) | Defines a single tool entry for a plugin's internal registry. |
|
|
206
|
+
| [discoverEvalConfigs](./docs/api/appkit/Function.discoverEvalConfigs.md) | Discover the per-agent `evals.config.ts` (from [defineEvalConfig](./docs/api/appkit/Function.defineEvalConfig.md)) at `<rootDir>/server/agents/<agent>/evals/evals.config.ts`. Config is per-agent: each agent's config applies only to that agent's evals. Agents without a config file are omitted. Returns a stable, sorted list. |
|
|
204
207
|
| [discoverEvalFiles](./docs/api/appkit/Function.discoverEvalFiles.md) | Discover evals under `<rootDir>/server/agents/<agent>/evals/` — co-located with each agent's `agent.{md,ts}` (same folder-per-agent layout the agents plugin discovers). The agent id is the folder name; the eval id is the file path relative to that evals dir with `.eval.ts` stripped. Sorted + stable. |
|
|
205
208
|
| [enumColumn](./docs/api/appkit/Function.enumColumn.md) | - |
|
|
206
209
|
| [equals](./docs/api/appkit/Function.equals.md) | Passes when the value equals `expected` exactly. |
|
|
@@ -212,6 +215,8 @@ Documentation merge entry for Typedoc — combines the stable `@databricks/appki
|
|
|
212
215
|
| [formatEvalDetail](./docs/api/appkit/Function.formatEvalDetail.md) | Indented detail lines for a failing eval (error + failing assertions). |
|
|
213
216
|
| [formatEvalHeadline](./docs/api/appkit/Function.formatEvalHeadline.md) | The one-line header for a single eval result (no failure detail). |
|
|
214
217
|
| [formatEvalResults](./docs/api/appkit/Function.formatEvalResults.md) | Render all results as a human-readable console report (non-streaming). |
|
|
218
|
+
| [formatResultsJson](./docs/api/appkit/Function.formatResultsJson.md) | Render results as a machine-readable JSON report (2-space indented): `{ summary: EvalSummary, results: EvalResult[] }`. Faithful to the types — every field present on a result round-trips. |
|
|
219
|
+
| [formatResultsJUnit](./docs/api/appkit/Function.formatResultsJUnit.md) | Render results as JUnit XML for standard CI test reporters: a single `<testsuite name="appkit-agent-evals">` with one `<testcase>` per result. Failures carry a `<failure>` (error or failing-gate summary); skips a `<skipped>`. All attribute/text values are XML-escaped. |
|
|
215
220
|
| [formatSummaryLine](./docs/api/appkit/Function.formatSummaryLine.md) | The final PASS/FAIL summary line. |
|
|
216
221
|
| [fromSupervisorApi](./docs/api/appkit/Function.fromSupervisorApi.md) | Creates an [AgentAdapter](./docs/api/appkit/Interface.AgentAdapter.md) backed by the Databricks AI Gateway Responses API (`/ai-gateway/mlflow/v1/responses`). |
|
|
217
222
|
| [functionToolToDefinition](./docs/api/appkit/Function.functionToolToDefinition.md) | - |
|
|
@@ -247,10 +252,12 @@ Documentation merge entry for Typedoc — combines the stable `@databricks/appki
|
|
|
247
252
|
| [runAgent](./docs/api/appkit/Function.runAgent.md) | Standalone agent execution without `createApp`. Resolves the adapter, binds inline tools, and drives the adapter's `run()` loop to completion. |
|
|
248
253
|
| [runEval](./docs/api/appkit/Function.runEval.md) | Run a single eval against a driver. Never throws for assertion or agent failures — those become a non-passing [EvalResult](./docs/api/appkit/Interface.EvalResult.md). Only a malformed eval definition surfaces as `result.error`. |
|
|
249
254
|
| [runEvalsInDir](./docs/api/appkit/Function.runEvalsInDir.md) | Discover, load, and run every eval under each agent's `evals/` dir, driving the agents on a running app. Never throws for an individual eval — load/run failures become non-passing [EvalResult](./docs/api/appkit/Interface.EvalResult.md)s. |
|
|
255
|
+
| [runWithRetries](./docs/api/appkit/Function.runWithRetries.md) | Run `attempt` up to `1 + retries` times, stopping as soon as it returns a result that is neither a thrown error / per-eval timeout (`error`) nor a transport/agent turn failure (`infraFailure`). Assertion failures set neither, so a failed-but-completed eval is returned on the first try and never retried. Returns the last result when every attempt failed on infra. |
|
|
250
256
|
| [summarize](./docs/api/appkit/Function.summarize.md) | - |
|
|
251
257
|
| [text](./docs/api/appkit/Function.text.md) | - |
|
|
252
258
|
| [timestamp](./docs/api/appkit/Function.timestamp.md) | - |
|
|
253
259
|
| [tool](./docs/api/appkit/Function.tool.md) | Factory for defining function tools with Zod schemas. |
|
|
254
260
|
| [toolsFromRegistry](./docs/api/appkit/Function.toolsFromRegistry.md) | Produces the `AgentToolDefinition[]` a ToolProvider exposes to the LLM, deriving `parameters` JSON Schema from each entry's Zod schema. |
|
|
261
|
+
| [userTurns](./docs/api/appkit/Function.userTurns.md) | Extract every user-message content, in order, from an MLflow `{"messages":[{"role":"user","content":"..."}]}` input. A dataset row can carry a full multi-turn conversation; replaying these against one thread (one `t.send` per returned string) lets the agent see the accumulating history. |
|
|
255
262
|
| [uuid](./docs/api/appkit/Function.uuid.md) | - |
|
|
256
263
|
| [varchar](./docs/api/appkit/Function.varchar.md) | - |
|
package/llms.txt
CHANGED
|
@@ -99,9 +99,11 @@ npx @databricks/appkit docs <query>
|
|
|
99
99
|
- [Function: createWorkspaceClient()](./docs/api/appkit/Function.createWorkspaceClient.md): Construct an AppKit workspace client.
|
|
100
100
|
- [Function: database()](./docs/api/appkit/Function.database.md): Create a typed database plugin registration for a finalized schema.
|
|
101
101
|
- [Function: defineEval()](./docs/api/appkit/Function.defineEval.md): Define an agent eval. Default-export the result from a
|
|
102
|
+
- [Function: defineEvalConfig()](./docs/api/appkit/Function.defineEvalConfig.md): Define per-directory eval config. Default-export from evals.config.ts.
|
|
102
103
|
- [Function: defineManifest()](./docs/api/appkit/Function.defineManifest.md): Validates a raw manifest (typically a manifest.json import) against the
|
|
103
104
|
- [Function: defineSchema()](./docs/api/appkit/Function.defineSchema.md): Compile one declared schema. The returned type keeps the table names the
|
|
104
105
|
- [Function: defineTool()](./docs/api/appkit/Function.defineTool.md): Defines a single tool entry for a plugin's internal registry.
|
|
106
|
+
- [Function: discoverEvalConfigs()](./docs/api/appkit/Function.discoverEvalConfigs.md): Discover the per-agent evals.config.ts (from defineEvalConfig) at
|
|
105
107
|
- [Function: discoverEvalFiles()](./docs/api/appkit/Function.discoverEvalFiles.md): Discover evals under /server/agents//evals/ — co-located
|
|
106
108
|
- [Function: enumColumn()](./docs/api/appkit/Function.enumColumn.md): Parameters
|
|
107
109
|
- [Function: equals()](./docs/api/appkit/Function.equals.md): Passes when the value equals expected exactly.
|
|
@@ -113,6 +115,8 @@ npx @databricks/appkit docs <query>
|
|
|
113
115
|
- [Function: formatEvalDetail()](./docs/api/appkit/Function.formatEvalDetail.md): Indented detail lines for a failing eval (error + failing assertions).
|
|
114
116
|
- [Function: formatEvalHeadline()](./docs/api/appkit/Function.formatEvalHeadline.md): The one-line header for a single eval result (no failure detail).
|
|
115
117
|
- [Function: formatEvalResults()](./docs/api/appkit/Function.formatEvalResults.md): Render all results as a human-readable console report (non-streaming).
|
|
118
|
+
- [Function: formatResultsJson()](./docs/api/appkit/Function.formatResultsJson.md): Render results as a machine-readable JSON report (2-space indented):
|
|
119
|
+
- [Function: formatResultsJUnit()](./docs/api/appkit/Function.formatResultsJUnit.md): Render results as JUnit XML for standard CI test reporters: a single
|
|
116
120
|
- [Function: formatSummaryLine()](./docs/api/appkit/Function.formatSummaryLine.md): The final PASS/FAIL summary line.
|
|
117
121
|
- [Function: fromSupervisorApi()](./docs/api/appkit/Function.fromSupervisorApi.md): Creates an AgentAdapter backed by the Databricks AI Gateway
|
|
118
122
|
- [Function: functionToolToDefinition()](./docs/api/appkit/Function.functionToolToDefinition.md): Parameters
|
|
@@ -148,11 +152,13 @@ npx @databricks/appkit docs <query>
|
|
|
148
152
|
- [Function: runAgent()](./docs/api/appkit/Function.runAgent.md): Standalone agent execution without createApp. Resolves the adapter, binds
|
|
149
153
|
- [Function: runEval()](./docs/api/appkit/Function.runEval.md): Run a single eval against a driver. Never throws for assertion or agent
|
|
150
154
|
- [Function: runEvalsInDir()](./docs/api/appkit/Function.runEvalsInDir.md): Discover, load, and run every eval under each agent's evals/ dir, driving
|
|
155
|
+
- [Function: runWithRetries()](./docs/api/appkit/Function.runWithRetries.md): Run attempt up to 1 + retries times, stopping as soon as it returns a
|
|
151
156
|
- [Function: summarize()](./docs/api/appkit/Function.summarize.md): Parameters
|
|
152
157
|
- [Function: text()](./docs/api/appkit/Function.text.md): Returns
|
|
153
158
|
- [Function: timestamp()](./docs/api/appkit/Function.timestamp.md): Parameters
|
|
154
159
|
- [Function: tool()](./docs/api/appkit/Function.tool.md): Factory for defining function tools with Zod schemas.
|
|
155
160
|
- [Function: toolsFromRegistry()](./docs/api/appkit/Function.toolsFromRegistry.md): Produces the AgentToolDefinition[] a ToolProvider exposes to the LLM,
|
|
161
|
+
- [Function: userTurns()](./docs/api/appkit/Function.userTurns.md): Extract every user-message content, in order, from an MLflow
|
|
156
162
|
- [Function: uuid()](./docs/api/appkit/Function.uuid.md): Returns
|
|
157
163
|
- [Function: varchar()](./docs/api/appkit/Function.varchar.md): Parameters
|
|
158
164
|
- [Interface: AgentAdapter](./docs/api/appkit/Interface.AgentAdapter.md): Properties
|
|
@@ -174,6 +180,7 @@ npx @databricks/appkit docs <query>
|
|
|
174
180
|
- [Interface: DatabricksAuth](./docs/api/appkit/Interface.DatabricksAuth.md): Resolved Databricks host + bearer token for the eval runner's REST calls.
|
|
175
181
|
- [Interface: DatasetRow](./docs/api/appkit/Interface.DatasetRow.md): One row of a managed evaluation dataset. inputs are the kwargs passed to the
|
|
176
182
|
- [Interface: DiscoveredEval](./docs/api/appkit/Interface.DiscoveredEval.md): An eval file found under server/agents//evals/.
|
|
183
|
+
- [Interface: DiscoveredEvalConfig](./docs/api/appkit/Interface.DiscoveredEvalConfig.md): A per-agent evals.config.ts found under server/agents//evals/.
|
|
177
184
|
- [Interface: DriveResult](./docs/api/appkit/Interface.DriveResult.md): What a driver returns for a single t.send.
|
|
178
185
|
- [Interface: EndpointConfig](./docs/api/appkit/Interface.EndpointConfig.md): Properties
|
|
179
186
|
- [Interface: EntityMutationHooks<TTable>](./docs/api/appkit/Interface.EntityMutationHooks.md): Mutation lifecycle for one entity. A before hook may return a replacement
|