@databricks/appkit 0.71.0 → 0.72.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CLAUDE.md +46 -0
- package/NOTICE.md +2 -2
- package/dist/appkit/package.js +1 -1
- package/dist/beta.d.ts +13 -1
- package/dist/beta.js +13 -1
- package/dist/cli/commands/agent/eval.js +112 -0
- package/dist/cli/commands/agent/eval.js.map +1 -0
- package/dist/cli/commands/agent/index.js +18 -0
- package/dist/cli/commands/agent/index.js.map +1 -0
- package/dist/cli/index.js +2 -0
- package/dist/cli/index.js.map +1 -1
- package/dist/connectors/index.js +2 -0
- package/dist/connectors/mlflow/auth.d.ts +18 -0
- package/dist/connectors/mlflow/auth.d.ts.map +1 -0
- package/dist/connectors/mlflow/auth.js +50 -0
- package/dist/connectors/mlflow/auth.js.map +1 -0
- package/dist/connectors/mlflow/client.d.ts +51 -0
- package/dist/connectors/mlflow/client.d.ts.map +1 -0
- package/dist/connectors/mlflow/client.js +93 -0
- package/dist/connectors/mlflow/client.js.map +1 -0
- package/dist/evals/define-eval.d.ts +26 -0
- package/dist/evals/define-eval.d.ts.map +1 -0
- package/dist/evals/define-eval.js +28 -0
- package/dist/evals/define-eval.js.map +1 -0
- package/dist/evals/discover.d.ts +20 -0
- package/dist/evals/discover.d.ts.map +1 -0
- package/dist/evals/discover.js +49 -0
- package/dist/evals/discover.js.map +1 -0
- package/dist/evals/http-driver.d.ts +33 -0
- package/dist/evals/http-driver.d.ts.map +1 -0
- package/dist/evals/http-driver.js +118 -0
- package/dist/evals/http-driver.js.map +1 -0
- package/dist/evals/index.js +13 -0
- package/dist/evals/judge.d.ts +26 -0
- package/dist/evals/judge.d.ts.map +1 -0
- package/dist/evals/judge.js +77 -0
- package/dist/evals/judge.js.map +1 -0
- package/dist/evals/matchers.d.ts +12 -0
- package/dist/evals/matchers.d.ts.map +1 -0
- package/dist/evals/matchers.js +26 -0
- package/dist/evals/matchers.js.map +1 -0
- package/dist/evals/mlflow-report.d.ts +36 -0
- package/dist/evals/mlflow-report.d.ts.map +1 -0
- package/dist/evals/mlflow-report.js +161 -0
- package/dist/evals/mlflow-report.js.map +1 -0
- package/dist/evals/mlflow-run.d.ts +11 -0
- package/dist/evals/mlflow-run.d.ts.map +1 -0
- package/dist/evals/mlflow-run.js +101 -0
- package/dist/evals/mlflow-run.js.map +1 -0
- package/dist/evals/pool.js +24 -0
- package/dist/evals/pool.js.map +1 -0
- package/dist/evals/report.d.ts +25 -0
- package/dist/evals/report.d.ts.map +1 -0
- package/dist/evals/report.js +57 -0
- package/dist/evals/report.js.map +1 -0
- package/dist/evals/run-eval.d.ts +20 -0
- package/dist/evals/run-eval.d.ts.map +1 -0
- package/dist/evals/run-eval.js +144 -0
- package/dist/evals/run-eval.js.map +1 -0
- package/dist/evals/run-evals.d.ts +85 -0
- package/dist/evals/run-evals.d.ts.map +1 -0
- package/dist/evals/run-evals.js +178 -0
- package/dist/evals/run-evals.js.map +1 -0
- package/dist/evals/types.d.ts +127 -0
- package/dist/evals/types.d.ts.map +1 -0
- package/dist/plugins/agents/agents.js +1 -1
- package/dist/shared/src/schemas/manifest.d.ts +33 -33
- package/docs/api/appkit/Class.MlflowClient.md +103 -0
- package/docs/api/appkit/Function.buildAssessments.md +16 -0
- package/docs/api/appkit/Function.configureJudge.md +18 -0
- package/docs/api/appkit/Function.createHttpDriver.md +18 -0
- package/docs/api/appkit/Function.defineEval.md +35 -0
- package/docs/api/appkit/Function.discoverEvalFiles.md +18 -0
- package/docs/api/appkit/Function.equals.md +18 -0
- package/docs/api/appkit/Function.evalGlyph.md +18 -0
- package/docs/api/appkit/Function.formatEvalDetail.md +18 -0
- package/docs/api/appkit/Function.formatEvalHeadline.md +18 -0
- package/docs/api/appkit/Function.formatEvalResults.md +18 -0
- package/docs/api/appkit/Function.formatSummaryLine.md +18 -0
- package/docs/api/appkit/Function.includes.md +18 -0
- package/docs/api/appkit/Function.isJudgeConfigured.md +10 -0
- package/docs/api/appkit/Function.matches.md +18 -0
- package/docs/api/appkit/Function.normalizeHost.md +18 -0
- package/docs/api/appkit/Function.reportToMlflow.md +23 -0
- package/docs/api/appkit/Function.resolveDatabricksAuth.md +16 -0
- package/docs/api/appkit/Function.runEval.md +19 -0
- package/docs/api/appkit/Function.runEvalsInDir.md +18 -0
- package/docs/api/appkit/Function.summarize.md +16 -0
- package/docs/api/appkit/Interface.AssertionHandle.md +54 -0
- package/docs/api/appkit/Interface.AssertionResult.md +48 -0
- package/docs/api/appkit/Interface.Assessment.md +83 -0
- package/docs/api/appkit/Interface.CustomJudgeSpec.md +30 -0
- package/docs/api/appkit/Interface.DatabricksAuth.md +21 -0
- package/docs/api/appkit/Interface.DiscoveredEval.md +36 -0
- package/docs/api/appkit/Interface.DriveResult.md +58 -0
- package/docs/api/appkit/Interface.EvalDefinition.md +46 -0
- package/docs/api/appkit/Interface.EvalDriver.md +22 -0
- package/docs/api/appkit/Interface.EvalResult.md +83 -0
- package/docs/api/appkit/Interface.EvalRunSummary.md +46 -0
- package/docs/api/appkit/Interface.EvalSummary.md +48 -0
- package/docs/api/appkit/Interface.HttpDriverOptions.md +67 -0
- package/docs/api/appkit/Interface.JudgeConfig.md +34 -0
- package/docs/api/appkit/Interface.JudgeScore.md +21 -0
- package/docs/api/appkit/Interface.MatchResult.md +34 -0
- package/docs/api/appkit/Interface.PostResult.md +30 -0
- package/docs/api/appkit/Interface.ReportOutcome.md +53 -0
- package/docs/api/appkit/Interface.ResolveDatabricksAuthOptions.md +34 -0
- package/docs/api/appkit/Interface.RunEvalOptions.md +34 -0
- package/docs/api/appkit/Interface.RunEvalsOptions.md +192 -0
- package/docs/api/appkit/Interface.TestContext.md +208 -0
- package/docs/api/appkit/TypeAlias.EvalProgress.md +26 -0
- package/docs/api/appkit/TypeAlias.Matcher.md +18 -0
- package/docs/api/appkit/TypeAlias.Severity.md +8 -0
- package/docs/api/appkit.md +63 -17
- package/llms.txt +46 -0
- package/package.json +2 -1
- package/sbom.cdx.json +1 -1
|
@@ -0,0 +1,18 @@
|
|
|
1
|
+
# Function: evalGlyph()
|
|
2
|
+
|
|
3
|
+
```ts
|
|
4
|
+
function evalGlyph(result: EvalResult): string;
|
|
5
|
+
|
|
6
|
+
```
|
|
7
|
+
|
|
8
|
+
Status glyph for a single eval result.
|
|
9
|
+
|
|
10
|
+
## Parameters[](#parameters "Direct link to Parameters")
|
|
11
|
+
|
|
12
|
+
| Parameter | Type |
|
|
13
|
+
| --------- | --------------------------------------------------------------- |
|
|
14
|
+
| `result` | [`EvalResult`](./docs/api/appkit/Interface.EvalResult.md) |
|
|
15
|
+
|
|
16
|
+
## Returns[](#returns "Direct link to Returns")
|
|
17
|
+
|
|
18
|
+
`string`
|
|
@@ -0,0 +1,18 @@
|
|
|
1
|
+
# Function: formatEvalDetail()
|
|
2
|
+
|
|
3
|
+
```ts
|
|
4
|
+
function formatEvalDetail(result: EvalResult): string[];
|
|
5
|
+
|
|
6
|
+
```
|
|
7
|
+
|
|
8
|
+
Indented detail lines for a failing eval (error + failing assertions).
|
|
9
|
+
|
|
10
|
+
## Parameters[](#parameters "Direct link to Parameters")
|
|
11
|
+
|
|
12
|
+
| Parameter | Type |
|
|
13
|
+
| --------- | --------------------------------------------------------------- |
|
|
14
|
+
| `result` | [`EvalResult`](./docs/api/appkit/Interface.EvalResult.md) |
|
|
15
|
+
|
|
16
|
+
## Returns[](#returns "Direct link to Returns")
|
|
17
|
+
|
|
18
|
+
`string`\[]
|
|
@@ -0,0 +1,18 @@
|
|
|
1
|
+
# Function: formatEvalHeadline()
|
|
2
|
+
|
|
3
|
+
```ts
|
|
4
|
+
function formatEvalHeadline(result: EvalResult): string;
|
|
5
|
+
|
|
6
|
+
```
|
|
7
|
+
|
|
8
|
+
The one-line header for a single eval result (no failure detail).
|
|
9
|
+
|
|
10
|
+
## Parameters[](#parameters "Direct link to Parameters")
|
|
11
|
+
|
|
12
|
+
| Parameter | Type |
|
|
13
|
+
| --------- | --------------------------------------------------------------- |
|
|
14
|
+
| `result` | [`EvalResult`](./docs/api/appkit/Interface.EvalResult.md) |
|
|
15
|
+
|
|
16
|
+
## Returns[](#returns "Direct link to Returns")
|
|
17
|
+
|
|
18
|
+
`string`
|
|
@@ -0,0 +1,18 @@
|
|
|
1
|
+
# Function: formatEvalResults()
|
|
2
|
+
|
|
3
|
+
```ts
|
|
4
|
+
function formatEvalResults(results: EvalResult[]): string;
|
|
5
|
+
|
|
6
|
+
```
|
|
7
|
+
|
|
8
|
+
Render all results as a human-readable console report (non-streaming).
|
|
9
|
+
|
|
10
|
+
## Parameters[](#parameters "Direct link to Parameters")
|
|
11
|
+
|
|
12
|
+
| Parameter | Type |
|
|
13
|
+
| --------- | ------------------------------------------------------------------ |
|
|
14
|
+
| `results` | [`EvalResult`](./docs/api/appkit/Interface.EvalResult.md)\[] |
|
|
15
|
+
|
|
16
|
+
## Returns[](#returns "Direct link to Returns")
|
|
17
|
+
|
|
18
|
+
`string`
|
|
@@ -0,0 +1,18 @@
|
|
|
1
|
+
# Function: formatSummaryLine()
|
|
2
|
+
|
|
3
|
+
```ts
|
|
4
|
+
function formatSummaryLine(results: EvalResult[]): string;
|
|
5
|
+
|
|
6
|
+
```
|
|
7
|
+
|
|
8
|
+
The final PASS/FAIL summary line.
|
|
9
|
+
|
|
10
|
+
## Parameters[](#parameters "Direct link to Parameters")
|
|
11
|
+
|
|
12
|
+
| Parameter | Type |
|
|
13
|
+
| --------- | ------------------------------------------------------------------ |
|
|
14
|
+
| `results` | [`EvalResult`](./docs/api/appkit/Interface.EvalResult.md)\[] |
|
|
15
|
+
|
|
16
|
+
## Returns[](#returns "Direct link to Returns")
|
|
17
|
+
|
|
18
|
+
`string`
|
|
@@ -0,0 +1,18 @@
|
|
|
1
|
+
# Function: includes()
|
|
2
|
+
|
|
3
|
+
```ts
|
|
4
|
+
function includes(substring: string): Matcher;
|
|
5
|
+
|
|
6
|
+
```
|
|
7
|
+
|
|
8
|
+
Passes when the value contains `substring`.
|
|
9
|
+
|
|
10
|
+
## Parameters[](#parameters "Direct link to Parameters")
|
|
11
|
+
|
|
12
|
+
| Parameter | Type |
|
|
13
|
+
| ----------- | -------- |
|
|
14
|
+
| `substring` | `string` |
|
|
15
|
+
|
|
16
|
+
## Returns[](#returns "Direct link to Returns")
|
|
17
|
+
|
|
18
|
+
[`Matcher`](./docs/api/appkit/TypeAlias.Matcher.md)
|
|
@@ -0,0 +1,18 @@
|
|
|
1
|
+
# Function: matches()
|
|
2
|
+
|
|
3
|
+
```ts
|
|
4
|
+
function matches(pattern: RegExp): Matcher;
|
|
5
|
+
|
|
6
|
+
```
|
|
7
|
+
|
|
8
|
+
Passes when the value matches `pattern`.
|
|
9
|
+
|
|
10
|
+
## Parameters[](#parameters "Direct link to Parameters")
|
|
11
|
+
|
|
12
|
+
| Parameter | Type |
|
|
13
|
+
| --------- | -------- |
|
|
14
|
+
| `pattern` | `RegExp` |
|
|
15
|
+
|
|
16
|
+
## Returns[](#returns "Direct link to Returns")
|
|
17
|
+
|
|
18
|
+
[`Matcher`](./docs/api/appkit/TypeAlias.Matcher.md)
|
|
@@ -0,0 +1,18 @@
|
|
|
1
|
+
# Function: normalizeHost()
|
|
2
|
+
|
|
3
|
+
```ts
|
|
4
|
+
function normalizeHost(raw: string): string;
|
|
5
|
+
|
|
6
|
+
```
|
|
7
|
+
|
|
8
|
+
Ensure the host has a scheme (Databricks env often lacks `https://`).
|
|
9
|
+
|
|
10
|
+
## Parameters[](#parameters "Direct link to Parameters")
|
|
11
|
+
|
|
12
|
+
| Parameter | Type |
|
|
13
|
+
| --------- | -------- |
|
|
14
|
+
| `raw` | `string` |
|
|
15
|
+
|
|
16
|
+
## Returns[](#returns "Direct link to Returns")
|
|
17
|
+
|
|
18
|
+
`string`
|
|
@@ -0,0 +1,23 @@
|
|
|
1
|
+
# Function: reportToMlflow()
|
|
2
|
+
|
|
3
|
+
```ts
|
|
4
|
+
function reportToMlflow(
|
|
5
|
+
client: MlflowClient,
|
|
6
|
+
results: EvalResult[],
|
|
7
|
+
sqlWarehouseId?: string): Promise<ReportOutcome>;
|
|
8
|
+
|
|
9
|
+
```
|
|
10
|
+
|
|
11
|
+
Write one pass/fail assessment per eval result to the Databricks MLflow REST API. Never throws — failures are collected so the run still reports.
|
|
12
|
+
|
|
13
|
+
## Parameters[](#parameters "Direct link to Parameters")
|
|
14
|
+
|
|
15
|
+
| Parameter | Type |
|
|
16
|
+
| ----------------- | ------------------------------------------------------------------ |
|
|
17
|
+
| `client` | [`MlflowClient`](./docs/api/appkit/Class.MlflowClient.md) |
|
|
18
|
+
| `results` | [`EvalResult`](./docs/api/appkit/Interface.EvalResult.md)\[] |
|
|
19
|
+
| `sqlWarehouseId?` | `string` |
|
|
20
|
+
|
|
21
|
+
## Returns[](#returns "Direct link to Returns")
|
|
22
|
+
|
|
23
|
+
`Promise`<[`ReportOutcome`](./docs/api/appkit/Interface.ReportOutcome.md)>
|
|
@@ -0,0 +1,16 @@
|
|
|
1
|
+
# Function: resolveDatabricksAuth()
|
|
2
|
+
|
|
3
|
+
```ts
|
|
4
|
+
function resolveDatabricksAuth(options: ResolveDatabricksAuthOptions): Promise<DatabricksAuth | undefined>;
|
|
5
|
+
|
|
6
|
+
```
|
|
7
|
+
|
|
8
|
+
## Parameters[](#parameters "Direct link to Parameters")
|
|
9
|
+
|
|
10
|
+
| Parameter | Type |
|
|
11
|
+
| --------- | --------------------------------------------------------------------------------------------------- |
|
|
12
|
+
| `options` | [`ResolveDatabricksAuthOptions`](./docs/api/appkit/Interface.ResolveDatabricksAuthOptions.md) |
|
|
13
|
+
|
|
14
|
+
## Returns[](#returns "Direct link to Returns")
|
|
15
|
+
|
|
16
|
+
`Promise`<[`DatabricksAuth`](./docs/api/appkit/Interface.DatabricksAuth.md) | `undefined`>
|
|
@@ -0,0 +1,19 @@
|
|
|
1
|
+
# Function: runEval()
|
|
2
|
+
|
|
3
|
+
```ts
|
|
4
|
+
function runEval(def: EvalDefinition, options: RunEvalOptions): Promise<EvalResult>;
|
|
5
|
+
|
|
6
|
+
```
|
|
7
|
+
|
|
8
|
+
Run a single eval against a driver. Never throws for assertion or agent failures — those become a non-passing [EvalResult](./docs/api/appkit/Interface.EvalResult.md). Only a malformed eval definition surfaces as `result.error`.
|
|
9
|
+
|
|
10
|
+
## Parameters[](#parameters "Direct link to Parameters")
|
|
11
|
+
|
|
12
|
+
| Parameter | Type |
|
|
13
|
+
| --------- | ----------------------------------------------------------------------- |
|
|
14
|
+
| `def` | [`EvalDefinition`](./docs/api/appkit/Interface.EvalDefinition.md) |
|
|
15
|
+
| `options` | [`RunEvalOptions`](./docs/api/appkit/Interface.RunEvalOptions.md) |
|
|
16
|
+
|
|
17
|
+
## Returns[](#returns "Direct link to Returns")
|
|
18
|
+
|
|
19
|
+
`Promise`<[`EvalResult`](./docs/api/appkit/Interface.EvalResult.md)>
|
|
@@ -0,0 +1,18 @@
|
|
|
1
|
+
# Function: runEvalsInDir()
|
|
2
|
+
|
|
3
|
+
```ts
|
|
4
|
+
function runEvalsInDir(options: RunEvalsOptions): Promise<EvalRunSummary>;
|
|
5
|
+
|
|
6
|
+
```
|
|
7
|
+
|
|
8
|
+
Discover, load, and run every eval under each agent's `evals/` dir, driving the agents on a running app. Never throws for an individual eval — load/run failures become non-passing [EvalResult](./docs/api/appkit/Interface.EvalResult.md)s.
|
|
9
|
+
|
|
10
|
+
## Parameters[](#parameters "Direct link to Parameters")
|
|
11
|
+
|
|
12
|
+
| Parameter | Type |
|
|
13
|
+
| --------- | ------------------------------------------------------------------------- |
|
|
14
|
+
| `options` | [`RunEvalsOptions`](./docs/api/appkit/Interface.RunEvalsOptions.md) |
|
|
15
|
+
|
|
16
|
+
## Returns[](#returns "Direct link to Returns")
|
|
17
|
+
|
|
18
|
+
`Promise`<[`EvalRunSummary`](./docs/api/appkit/Interface.EvalRunSummary.md)>
|
|
@@ -0,0 +1,16 @@
|
|
|
1
|
+
# Function: summarize()
|
|
2
|
+
|
|
3
|
+
```ts
|
|
4
|
+
function summarize(results: EvalResult[]): EvalSummary;
|
|
5
|
+
|
|
6
|
+
```
|
|
7
|
+
|
|
8
|
+
## Parameters[](#parameters "Direct link to Parameters")
|
|
9
|
+
|
|
10
|
+
| Parameter | Type |
|
|
11
|
+
| --------- | ------------------------------------------------------------------ |
|
|
12
|
+
| `results` | [`EvalResult`](./docs/api/appkit/Interface.EvalResult.md)\[] |
|
|
13
|
+
|
|
14
|
+
## Returns[](#returns "Direct link to Returns")
|
|
15
|
+
|
|
16
|
+
[`EvalSummary`](./docs/api/appkit/Interface.EvalSummary.md)
|
|
@@ -0,0 +1,54 @@
|
|
|
1
|
+
# Interface: AssertionHandle
|
|
2
|
+
|
|
3
|
+
Chainable handle returned by every assertion to control its severity. Mirrors eve: assertions are gates by default; `.soft()` demotes to a tracked metric; `.atLeast(n)` is a soft, score-thresholded assertion.
|
|
4
|
+
|
|
5
|
+
## Methods[](#methods "Direct link to Methods")
|
|
6
|
+
|
|
7
|
+
### atLeast()[](#atleast "Direct link to atLeast()")
|
|
8
|
+
|
|
9
|
+
```ts
|
|
10
|
+
atLeast(threshold: number): AssertionHandle;
|
|
11
|
+
|
|
12
|
+
```
|
|
13
|
+
|
|
14
|
+
Soft assertion that passes only when the score is at least `threshold`.
|
|
15
|
+
|
|
16
|
+
#### Parameters[](#parameters "Direct link to Parameters")
|
|
17
|
+
|
|
18
|
+
| Parameter | Type |
|
|
19
|
+
| ----------- | -------- |
|
|
20
|
+
| `threshold` | `number` |
|
|
21
|
+
|
|
22
|
+
#### Returns[](#returns "Direct link to Returns")
|
|
23
|
+
|
|
24
|
+
`AssertionHandle`
|
|
25
|
+
|
|
26
|
+
***
|
|
27
|
+
|
|
28
|
+
### gate()[](#gate "Direct link to gate()")
|
|
29
|
+
|
|
30
|
+
```ts
|
|
31
|
+
gate(): AssertionHandle;
|
|
32
|
+
|
|
33
|
+
```
|
|
34
|
+
|
|
35
|
+
Promote to a hard gate — failure fails the eval (non-zero exit).
|
|
36
|
+
|
|
37
|
+
#### Returns[](#returns-1 "Direct link to Returns")
|
|
38
|
+
|
|
39
|
+
`AssertionHandle`
|
|
40
|
+
|
|
41
|
+
***
|
|
42
|
+
|
|
43
|
+
### soft()[](#soft "Direct link to soft()")
|
|
44
|
+
|
|
45
|
+
```ts
|
|
46
|
+
soft(): AssertionHandle;
|
|
47
|
+
|
|
48
|
+
```
|
|
49
|
+
|
|
50
|
+
Demote to a tracked metric — doesn't fail unless running with `strict`.
|
|
51
|
+
|
|
52
|
+
#### Returns[](#returns-2 "Direct link to Returns")
|
|
53
|
+
|
|
54
|
+
`AssertionHandle`
|
|
@@ -0,0 +1,48 @@
|
|
|
1
|
+
# Interface: AssertionResult
|
|
2
|
+
|
|
3
|
+
A single recorded assertion outcome.
|
|
4
|
+
|
|
5
|
+
## Properties[](#properties "Direct link to Properties")
|
|
6
|
+
|
|
7
|
+
### detail?[](#detail "Direct link to detail?")
|
|
8
|
+
|
|
9
|
+
```ts
|
|
10
|
+
optional detail: string;
|
|
11
|
+
|
|
12
|
+
```
|
|
13
|
+
|
|
14
|
+
***
|
|
15
|
+
|
|
16
|
+
### label[](#label "Direct link to label")
|
|
17
|
+
|
|
18
|
+
```ts
|
|
19
|
+
label: string;
|
|
20
|
+
|
|
21
|
+
```
|
|
22
|
+
|
|
23
|
+
***
|
|
24
|
+
|
|
25
|
+
### pass[](#pass "Direct link to pass")
|
|
26
|
+
|
|
27
|
+
```ts
|
|
28
|
+
pass: boolean;
|
|
29
|
+
|
|
30
|
+
```
|
|
31
|
+
|
|
32
|
+
***
|
|
33
|
+
|
|
34
|
+
### score?[](#score "Direct link to score?")
|
|
35
|
+
|
|
36
|
+
```ts
|
|
37
|
+
optional score: number;
|
|
38
|
+
|
|
39
|
+
```
|
|
40
|
+
|
|
41
|
+
***
|
|
42
|
+
|
|
43
|
+
### severity[](#severity "Direct link to severity")
|
|
44
|
+
|
|
45
|
+
```ts
|
|
46
|
+
severity: Severity;
|
|
47
|
+
|
|
48
|
+
```
|
|
@@ -0,0 +1,83 @@
|
|
|
1
|
+
# Interface: Assessment
|
|
2
|
+
|
|
3
|
+
A Feedback assessment in the MLflow REST proto-JSON shape.
|
|
4
|
+
|
|
5
|
+
## Properties[](#properties "Direct link to Properties")
|
|
6
|
+
|
|
7
|
+
### assessment\_name[](#assessment_name "Direct link to assessment_name")
|
|
8
|
+
|
|
9
|
+
```ts
|
|
10
|
+
assessment_name: string;
|
|
11
|
+
|
|
12
|
+
```
|
|
13
|
+
|
|
14
|
+
***
|
|
15
|
+
|
|
16
|
+
### feedback[](#feedback "Direct link to feedback")
|
|
17
|
+
|
|
18
|
+
```ts
|
|
19
|
+
feedback: {
|
|
20
|
+
value: unknown;
|
|
21
|
+
};
|
|
22
|
+
|
|
23
|
+
```
|
|
24
|
+
|
|
25
|
+
#### value[](#value "Direct link to value")
|
|
26
|
+
|
|
27
|
+
```ts
|
|
28
|
+
value: unknown;
|
|
29
|
+
|
|
30
|
+
```
|
|
31
|
+
|
|
32
|
+
***
|
|
33
|
+
|
|
34
|
+
### metadata?[](#metadata "Direct link to metadata?")
|
|
35
|
+
|
|
36
|
+
```ts
|
|
37
|
+
optional metadata: Record<string, string>;
|
|
38
|
+
|
|
39
|
+
```
|
|
40
|
+
|
|
41
|
+
***
|
|
42
|
+
|
|
43
|
+
### rationale?[](#rationale "Direct link to rationale?")
|
|
44
|
+
|
|
45
|
+
```ts
|
|
46
|
+
optional rationale: string;
|
|
47
|
+
|
|
48
|
+
```
|
|
49
|
+
|
|
50
|
+
***
|
|
51
|
+
|
|
52
|
+
### source[](#source "Direct link to source")
|
|
53
|
+
|
|
54
|
+
```ts
|
|
55
|
+
source: {
|
|
56
|
+
source_id: string;
|
|
57
|
+
source_type: "CODE" | "HUMAN" | "LLM_JUDGE";
|
|
58
|
+
};
|
|
59
|
+
|
|
60
|
+
```
|
|
61
|
+
|
|
62
|
+
#### source\_id[](#source_id "Direct link to source_id")
|
|
63
|
+
|
|
64
|
+
```ts
|
|
65
|
+
source_id: string;
|
|
66
|
+
|
|
67
|
+
```
|
|
68
|
+
|
|
69
|
+
#### source\_type[](#source_type "Direct link to source_type")
|
|
70
|
+
|
|
71
|
+
```ts
|
|
72
|
+
source_type: "CODE" | "HUMAN" | "LLM_JUDGE";
|
|
73
|
+
|
|
74
|
+
```
|
|
75
|
+
|
|
76
|
+
***
|
|
77
|
+
|
|
78
|
+
### trace\_id[](#trace_id "Direct link to trace_id")
|
|
79
|
+
|
|
80
|
+
```ts
|
|
81
|
+
trace_id: string;
|
|
82
|
+
|
|
83
|
+
```
|
|
@@ -0,0 +1,30 @@
|
|
|
1
|
+
# Interface: CustomJudgeSpec
|
|
2
|
+
|
|
3
|
+
A custom LLM-judge definition: a prompt template and choice→score mapping.
|
|
4
|
+
|
|
5
|
+
## Properties[](#properties "Direct link to Properties")
|
|
6
|
+
|
|
7
|
+
### choiceScores[](#choicescores "Direct link to choiceScores")
|
|
8
|
+
|
|
9
|
+
```ts
|
|
10
|
+
choiceScores: Record<string, number>;
|
|
11
|
+
|
|
12
|
+
```
|
|
13
|
+
|
|
14
|
+
***
|
|
15
|
+
|
|
16
|
+
### name[](#name "Direct link to name")
|
|
17
|
+
|
|
18
|
+
```ts
|
|
19
|
+
name: string;
|
|
20
|
+
|
|
21
|
+
```
|
|
22
|
+
|
|
23
|
+
***
|
|
24
|
+
|
|
25
|
+
### promptTemplate[](#prompttemplate "Direct link to promptTemplate")
|
|
26
|
+
|
|
27
|
+
```ts
|
|
28
|
+
promptTemplate: string;
|
|
29
|
+
|
|
30
|
+
```
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
# Interface: DatabricksAuth
|
|
2
|
+
|
|
3
|
+
Resolved Databricks host + bearer token for the eval runner's REST calls.
|
|
4
|
+
|
|
5
|
+
## Properties[](#properties "Direct link to Properties")
|
|
6
|
+
|
|
7
|
+
### host[](#host "Direct link to host")
|
|
8
|
+
|
|
9
|
+
```ts
|
|
10
|
+
host: string;
|
|
11
|
+
|
|
12
|
+
```
|
|
13
|
+
|
|
14
|
+
***
|
|
15
|
+
|
|
16
|
+
### token[](#token "Direct link to token")
|
|
17
|
+
|
|
18
|
+
```ts
|
|
19
|
+
token: string;
|
|
20
|
+
|
|
21
|
+
```
|
|
@@ -0,0 +1,36 @@
|
|
|
1
|
+
# Interface: DiscoveredEval
|
|
2
|
+
|
|
3
|
+
An eval file found under `server/agents/<agent>/evals/`.
|
|
4
|
+
|
|
5
|
+
## Properties[](#properties "Direct link to Properties")
|
|
6
|
+
|
|
7
|
+
### agent[](#agent "Direct link to agent")
|
|
8
|
+
|
|
9
|
+
```ts
|
|
10
|
+
agent: string;
|
|
11
|
+
|
|
12
|
+
```
|
|
13
|
+
|
|
14
|
+
The agent id (the `server/agents/<agent>` directory name).
|
|
15
|
+
|
|
16
|
+
***
|
|
17
|
+
|
|
18
|
+
### file[](#file "Direct link to file")
|
|
19
|
+
|
|
20
|
+
```ts
|
|
21
|
+
file: string;
|
|
22
|
+
|
|
23
|
+
```
|
|
24
|
+
|
|
25
|
+
Absolute path to the `*.eval.ts` file.
|
|
26
|
+
|
|
27
|
+
***
|
|
28
|
+
|
|
29
|
+
### id[](#id "Direct link to id")
|
|
30
|
+
|
|
31
|
+
```ts
|
|
32
|
+
id: string;
|
|
33
|
+
|
|
34
|
+
```
|
|
35
|
+
|
|
36
|
+
Id relative to the agent's evals dir, without `.eval.ts` (e.g. `weather/basic`).
|
|
@@ -0,0 +1,58 @@
|
|
|
1
|
+
# Interface: DriveResult
|
|
2
|
+
|
|
3
|
+
What a driver returns for a single `t.send`.
|
|
4
|
+
|
|
5
|
+
## Properties[](#properties "Direct link to Properties")
|
|
6
|
+
|
|
7
|
+
### reply[](#reply "Direct link to reply")
|
|
8
|
+
|
|
9
|
+
```ts
|
|
10
|
+
reply: string;
|
|
11
|
+
|
|
12
|
+
```
|
|
13
|
+
|
|
14
|
+
The final assistant message text.
|
|
15
|
+
|
|
16
|
+
***
|
|
17
|
+
|
|
18
|
+
### sessionId?[](#sessionid "Direct link to sessionId?")
|
|
19
|
+
|
|
20
|
+
```ts
|
|
21
|
+
optional sessionId: string;
|
|
22
|
+
|
|
23
|
+
```
|
|
24
|
+
|
|
25
|
+
Thread/session id, when the driver exposes one.
|
|
26
|
+
|
|
27
|
+
***
|
|
28
|
+
|
|
29
|
+
### succeeded[](#succeeded "Direct link to succeeded")
|
|
30
|
+
|
|
31
|
+
```ts
|
|
32
|
+
succeeded: boolean;
|
|
33
|
+
|
|
34
|
+
```
|
|
35
|
+
|
|
36
|
+
Whether the turn completed without an agent/stream error.
|
|
37
|
+
|
|
38
|
+
***
|
|
39
|
+
|
|
40
|
+
### toolCalls[](#toolcalls "Direct link to toolCalls")
|
|
41
|
+
|
|
42
|
+
```ts
|
|
43
|
+
toolCalls: string[];
|
|
44
|
+
|
|
45
|
+
```
|
|
46
|
+
|
|
47
|
+
Names of tools the agent called during the turn.
|
|
48
|
+
|
|
49
|
+
***
|
|
50
|
+
|
|
51
|
+
### traceId?[](#traceid "Direct link to traceId?")
|
|
52
|
+
|
|
53
|
+
```ts
|
|
54
|
+
optional traceId: string;
|
|
55
|
+
|
|
56
|
+
```
|
|
57
|
+
|
|
58
|
+
MLflow trace id for the turn, when tracing is enabled on the app.
|
|
@@ -0,0 +1,46 @@
|
|
|
1
|
+
# Interface: EvalDefinition
|
|
2
|
+
|
|
3
|
+
A single eval, default-exported from a `*.eval.ts` file.
|
|
4
|
+
|
|
5
|
+
## Properties[](#properties "Direct link to Properties")
|
|
6
|
+
|
|
7
|
+
### agent?[](#agent "Direct link to agent?")
|
|
8
|
+
|
|
9
|
+
```ts
|
|
10
|
+
optional agent: string;
|
|
11
|
+
|
|
12
|
+
```
|
|
13
|
+
|
|
14
|
+
Target agent id. Defaults to the eval's parent `server/agents/<id>` dir.
|
|
15
|
+
|
|
16
|
+
***
|
|
17
|
+
|
|
18
|
+
### description?[](#description "Direct link to description?")
|
|
19
|
+
|
|
20
|
+
```ts
|
|
21
|
+
optional description: string;
|
|
22
|
+
|
|
23
|
+
```
|
|
24
|
+
|
|
25
|
+
Short human description, shown in reports.
|
|
26
|
+
|
|
27
|
+
## Methods[](#methods "Direct link to Methods")
|
|
28
|
+
|
|
29
|
+
### test()[](#test "Direct link to test()")
|
|
30
|
+
|
|
31
|
+
```ts
|
|
32
|
+
test(t: TestContext): void | Promise<void>;
|
|
33
|
+
|
|
34
|
+
```
|
|
35
|
+
|
|
36
|
+
The eval body: drive the agent and assert on its behavior.
|
|
37
|
+
|
|
38
|
+
#### Parameters[](#parameters "Direct link to Parameters")
|
|
39
|
+
|
|
40
|
+
| Parameter | Type |
|
|
41
|
+
| --------- | ----------------------------------------------------------------- |
|
|
42
|
+
| `t` | [`TestContext`](./docs/api/appkit/Interface.TestContext.md) |
|
|
43
|
+
|
|
44
|
+
#### Returns[](#returns "Direct link to Returns")
|
|
45
|
+
|
|
46
|
+
`void` | `Promise`<`void`>
|
|
@@ -0,0 +1,22 @@
|
|
|
1
|
+
# Interface: EvalDriver
|
|
2
|
+
|
|
3
|
+
Abstraction over how the agent is driven. The HTTP driver posts to a running app's agents endpoint; future drivers (in-process) implement the same shape.
|
|
4
|
+
|
|
5
|
+
## Methods[](#methods "Direct link to Methods")
|
|
6
|
+
|
|
7
|
+
### send()[](#send "Direct link to send()")
|
|
8
|
+
|
|
9
|
+
```ts
|
|
10
|
+
send(message: string): Promise<DriveResult>;
|
|
11
|
+
|
|
12
|
+
```
|
|
13
|
+
|
|
14
|
+
#### Parameters[](#parameters "Direct link to Parameters")
|
|
15
|
+
|
|
16
|
+
| Parameter | Type |
|
|
17
|
+
| --------- | -------- |
|
|
18
|
+
| `message` | `string` |
|
|
19
|
+
|
|
20
|
+
#### Returns[](#returns "Direct link to Returns")
|
|
21
|
+
|
|
22
|
+
`Promise`<[`DriveResult`](./docs/api/appkit/Interface.DriveResult.md)>
|