@databricks/appkit 0.70.0 → 0.72.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CLAUDE.md +46 -0
- package/NOTICE.md +2 -2
- package/dist/appkit/package.js +1 -1
- package/dist/beta.d.ts +13 -1
- package/dist/beta.js +13 -1
- package/dist/cli/commands/agent/eval.js +112 -0
- package/dist/cli/commands/agent/eval.js.map +1 -0
- package/dist/cli/commands/agent/index.js +18 -0
- package/dist/cli/commands/agent/index.js.map +1 -0
- package/dist/cli/index.js +2 -0
- package/dist/cli/index.js.map +1 -1
- package/dist/connectors/index.js +2 -0
- package/dist/connectors/mlflow/auth.d.ts +18 -0
- package/dist/connectors/mlflow/auth.d.ts.map +1 -0
- package/dist/connectors/mlflow/auth.js +50 -0
- package/dist/connectors/mlflow/auth.js.map +1 -0
- package/dist/connectors/mlflow/client.d.ts +51 -0
- package/dist/connectors/mlflow/client.d.ts.map +1 -0
- package/dist/connectors/mlflow/client.js +93 -0
- package/dist/connectors/mlflow/client.js.map +1 -0
- package/dist/core/appkit.d.ts.map +1 -1
- package/dist/core/appkit.js +1 -0
- package/dist/core/appkit.js.map +1 -1
- package/dist/evals/define-eval.d.ts +26 -0
- package/dist/evals/define-eval.d.ts.map +1 -0
- package/dist/evals/define-eval.js +28 -0
- package/dist/evals/define-eval.js.map +1 -0
- package/dist/evals/discover.d.ts +20 -0
- package/dist/evals/discover.d.ts.map +1 -0
- package/dist/evals/discover.js +49 -0
- package/dist/evals/discover.js.map +1 -0
- package/dist/evals/http-driver.d.ts +33 -0
- package/dist/evals/http-driver.d.ts.map +1 -0
- package/dist/evals/http-driver.js +118 -0
- package/dist/evals/http-driver.js.map +1 -0
- package/dist/evals/index.js +13 -0
- package/dist/evals/judge.d.ts +26 -0
- package/dist/evals/judge.d.ts.map +1 -0
- package/dist/evals/judge.js +77 -0
- package/dist/evals/judge.js.map +1 -0
- package/dist/evals/matchers.d.ts +12 -0
- package/dist/evals/matchers.d.ts.map +1 -0
- package/dist/evals/matchers.js +26 -0
- package/dist/evals/matchers.js.map +1 -0
- package/dist/evals/mlflow-report.d.ts +36 -0
- package/dist/evals/mlflow-report.d.ts.map +1 -0
- package/dist/evals/mlflow-report.js +161 -0
- package/dist/evals/mlflow-report.js.map +1 -0
- package/dist/evals/mlflow-run.d.ts +11 -0
- package/dist/evals/mlflow-run.d.ts.map +1 -0
- package/dist/evals/mlflow-run.js +101 -0
- package/dist/evals/mlflow-run.js.map +1 -0
- package/dist/evals/pool.js +24 -0
- package/dist/evals/pool.js.map +1 -0
- package/dist/evals/report.d.ts +25 -0
- package/dist/evals/report.d.ts.map +1 -0
- package/dist/evals/report.js +57 -0
- package/dist/evals/report.js.map +1 -0
- package/dist/evals/run-eval.d.ts +20 -0
- package/dist/evals/run-eval.d.ts.map +1 -0
- package/dist/evals/run-eval.js +144 -0
- package/dist/evals/run-eval.js.map +1 -0
- package/dist/evals/run-evals.d.ts +85 -0
- package/dist/evals/run-evals.d.ts.map +1 -0
- package/dist/evals/run-evals.js +178 -0
- package/dist/evals/run-evals.js.map +1 -0
- package/dist/evals/types.d.ts +127 -0
- package/dist/evals/types.d.ts.map +1 -0
- package/dist/plugins/agents/agents.d.ts.map +1 -1
- package/dist/plugins/agents/agents.js +5 -2
- package/dist/plugins/agents/agents.js.map +1 -1
- package/dist/plugins/agents/mlflow.js +267 -11
- package/dist/plugins/agents/mlflow.js.map +1 -1
- package/dist/plugins/server/index.js +2 -2
- package/dist/plugins/server/index.js.map +1 -1
- package/dist/plugins/server/remote-tunnel/remote-tunnel-manager.js +3 -3
- package/dist/plugins/server/remote-tunnel/remote-tunnel-manager.js.map +1 -1
- package/dist/plugins/server/static-server.js +3 -3
- package/dist/plugins/server/static-server.js.map +1 -1
- package/dist/plugins/server/utils.js +3 -3
- package/dist/plugins/server/utils.js.map +1 -1
- package/dist/plugins/server/vite-dev-server.js +4 -4
- package/dist/plugins/server/vite-dev-server.js.map +1 -1
- package/dist/shared/src/schemas/manifest.d.ts +33 -33
- package/dist/telemetry/index.js +2 -2
- package/dist/telemetry/telemetry-manager.d.ts +2 -1
- package/dist/telemetry/telemetry-manager.js +114 -26
- package/dist/telemetry/telemetry-manager.js.map +1 -1
- package/dist/type-generator/database/generate.js +3 -3
- package/dist/type-generator/database/generate.js.map +1 -1
- package/dist/type-generator/errors.js +45 -11
- package/dist/type-generator/errors.js.map +1 -1
- package/dist/type-generator/migration.js +2 -2
- package/dist/type-generator/migration.js.map +1 -1
- package/dist/type-generator/mv-registry/sync.js +3 -3
- package/dist/type-generator/mv-registry/sync.js.map +1 -1
- package/dist/type-generator/mv-registry/types.d.ts +5 -2
- package/dist/type-generator/mv-registry/types.d.ts.map +1 -1
- package/dist/type-generator/query-registry.js +3 -3
- package/dist/type-generator/query-registry.js.map +1 -1
- package/dist/type-generator/serving/server-file-extractor.js +3 -3
- package/dist/type-generator/serving/server-file-extractor.js.map +1 -1
- package/docs/api/appkit/Class.MlflowClient.md +103 -0
- package/docs/api/appkit/Function.buildAssessments.md +16 -0
- package/docs/api/appkit/Function.configureJudge.md +18 -0
- package/docs/api/appkit/Function.createHttpDriver.md +18 -0
- package/docs/api/appkit/Function.defineEval.md +35 -0
- package/docs/api/appkit/Function.discoverEvalFiles.md +18 -0
- package/docs/api/appkit/Function.equals.md +18 -0
- package/docs/api/appkit/Function.evalGlyph.md +18 -0
- package/docs/api/appkit/Function.formatEvalDetail.md +18 -0
- package/docs/api/appkit/Function.formatEvalHeadline.md +18 -0
- package/docs/api/appkit/Function.formatEvalResults.md +18 -0
- package/docs/api/appkit/Function.formatSummaryLine.md +18 -0
- package/docs/api/appkit/Function.includes.md +18 -0
- package/docs/api/appkit/Function.isJudgeConfigured.md +10 -0
- package/docs/api/appkit/Function.matches.md +18 -0
- package/docs/api/appkit/Function.normalizeHost.md +18 -0
- package/docs/api/appkit/Function.reportToMlflow.md +23 -0
- package/docs/api/appkit/Function.resolveDatabricksAuth.md +16 -0
- package/docs/api/appkit/Function.runEval.md +19 -0
- package/docs/api/appkit/Function.runEvalsInDir.md +18 -0
- package/docs/api/appkit/Function.summarize.md +16 -0
- package/docs/api/appkit/Interface.AssertionHandle.md +54 -0
- package/docs/api/appkit/Interface.AssertionResult.md +48 -0
- package/docs/api/appkit/Interface.Assessment.md +83 -0
- package/docs/api/appkit/Interface.CustomJudgeSpec.md +30 -0
- package/docs/api/appkit/Interface.DatabricksAuth.md +21 -0
- package/docs/api/appkit/Interface.DiscoveredEval.md +36 -0
- package/docs/api/appkit/Interface.DriveResult.md +58 -0
- package/docs/api/appkit/Interface.EvalDefinition.md +46 -0
- package/docs/api/appkit/Interface.EvalDriver.md +22 -0
- package/docs/api/appkit/Interface.EvalResult.md +83 -0
- package/docs/api/appkit/Interface.EvalRunSummary.md +46 -0
- package/docs/api/appkit/Interface.EvalSummary.md +48 -0
- package/docs/api/appkit/Interface.HttpDriverOptions.md +67 -0
- package/docs/api/appkit/Interface.JudgeConfig.md +34 -0
- package/docs/api/appkit/Interface.JudgeScore.md +21 -0
- package/docs/api/appkit/Interface.MatchResult.md +34 -0
- package/docs/api/appkit/Interface.PostResult.md +30 -0
- package/docs/api/appkit/Interface.ReportOutcome.md +53 -0
- package/docs/api/appkit/Interface.ResolveDatabricksAuthOptions.md +34 -0
- package/docs/api/appkit/Interface.RunEvalOptions.md +34 -0
- package/docs/api/appkit/Interface.RunEvalsOptions.md +192 -0
- package/docs/api/appkit/Interface.TestContext.md +208 -0
- package/docs/api/appkit/TypeAlias.EvalProgress.md +26 -0
- package/docs/api/appkit/TypeAlias.Matcher.md +18 -0
- package/docs/api/appkit/TypeAlias.Severity.md +8 -0
- package/docs/api/appkit.md +63 -17
- package/llms.txt +46 -0
- package/package.json +5 -4
- package/sbom.cdx.json +1 -1
|
@@ -0,0 +1,83 @@
|
|
|
1
|
+
# Interface: EvalResult
|
|
2
|
+
|
|
3
|
+
The outcome of running one eval.
|
|
4
|
+
|
|
5
|
+
## Properties[](#properties "Direct link to Properties")
|
|
6
|
+
|
|
7
|
+
### assertions[](#assertions "Direct link to assertions")
|
|
8
|
+
|
|
9
|
+
```ts
|
|
10
|
+
assertions: AssertionResult[];
|
|
11
|
+
|
|
12
|
+
```
|
|
13
|
+
|
|
14
|
+
***
|
|
15
|
+
|
|
16
|
+
### description?[](#description "Direct link to description?")
|
|
17
|
+
|
|
18
|
+
```ts
|
|
19
|
+
optional description: string;
|
|
20
|
+
|
|
21
|
+
```
|
|
22
|
+
|
|
23
|
+
***
|
|
24
|
+
|
|
25
|
+
### error?[](#error "Direct link to error?")
|
|
26
|
+
|
|
27
|
+
```ts
|
|
28
|
+
optional error: string;
|
|
29
|
+
|
|
30
|
+
```
|
|
31
|
+
|
|
32
|
+
Set when the eval threw before completing.
|
|
33
|
+
|
|
34
|
+
***
|
|
35
|
+
|
|
36
|
+
### id[](#id "Direct link to id")
|
|
37
|
+
|
|
38
|
+
```ts
|
|
39
|
+
id: string;
|
|
40
|
+
|
|
41
|
+
```
|
|
42
|
+
|
|
43
|
+
***
|
|
44
|
+
|
|
45
|
+
### passed[](#passed "Direct link to passed")
|
|
46
|
+
|
|
47
|
+
```ts
|
|
48
|
+
passed: boolean;
|
|
49
|
+
|
|
50
|
+
```
|
|
51
|
+
|
|
52
|
+
True when all gates passed (and, under strict, all soft assertions too).
|
|
53
|
+
|
|
54
|
+
***
|
|
55
|
+
|
|
56
|
+
### skipped?[](#skipped "Direct link to skipped?")
|
|
57
|
+
|
|
58
|
+
```ts
|
|
59
|
+
optional skipped: {
|
|
60
|
+
reason?: string;
|
|
61
|
+
};
|
|
62
|
+
|
|
63
|
+
```
|
|
64
|
+
|
|
65
|
+
Set when the eval called `t.skip`.
|
|
66
|
+
|
|
67
|
+
#### reason?[](#reason "Direct link to reason?")
|
|
68
|
+
|
|
69
|
+
```ts
|
|
70
|
+
optional reason: string;
|
|
71
|
+
|
|
72
|
+
```
|
|
73
|
+
|
|
74
|
+
***
|
|
75
|
+
|
|
76
|
+
### traceId?[](#traceid "Direct link to traceId?")
|
|
77
|
+
|
|
78
|
+
```ts
|
|
79
|
+
optional traceId: string;
|
|
80
|
+
|
|
81
|
+
```
|
|
82
|
+
|
|
83
|
+
MLflow trace id of the eval's last turn, for attaching assessments.
|
|
@@ -0,0 +1,46 @@
|
|
|
1
|
+
# Interface: EvalRunSummary
|
|
2
|
+
|
|
3
|
+
## Properties[](#properties "Direct link to Properties")
|
|
4
|
+
|
|
5
|
+
### mlflow?[](#mlflow "Direct link to mlflow?")
|
|
6
|
+
|
|
7
|
+
```ts
|
|
8
|
+
optional mlflow: {
|
|
9
|
+
finish: FinishOutcome;
|
|
10
|
+
report: ReportOutcome;
|
|
11
|
+
runId: string;
|
|
12
|
+
};
|
|
13
|
+
|
|
14
|
+
```
|
|
15
|
+
|
|
16
|
+
Present when an MLflow evaluation run was created.
|
|
17
|
+
|
|
18
|
+
#### finish[](#finish "Direct link to finish")
|
|
19
|
+
|
|
20
|
+
```ts
|
|
21
|
+
finish: FinishOutcome;
|
|
22
|
+
|
|
23
|
+
```
|
|
24
|
+
|
|
25
|
+
#### report[](#report "Direct link to report")
|
|
26
|
+
|
|
27
|
+
```ts
|
|
28
|
+
report: ReportOutcome;
|
|
29
|
+
|
|
30
|
+
```
|
|
31
|
+
|
|
32
|
+
#### runId[](#runid "Direct link to runId")
|
|
33
|
+
|
|
34
|
+
```ts
|
|
35
|
+
runId: string;
|
|
36
|
+
|
|
37
|
+
```
|
|
38
|
+
|
|
39
|
+
***
|
|
40
|
+
|
|
41
|
+
### results[](#results "Direct link to results")
|
|
42
|
+
|
|
43
|
+
```ts
|
|
44
|
+
results: EvalResult[];
|
|
45
|
+
|
|
46
|
+
```
|
|
@@ -0,0 +1,48 @@
|
|
|
1
|
+
# Interface: EvalSummary
|
|
2
|
+
|
|
3
|
+
## Properties[](#properties "Direct link to Properties")
|
|
4
|
+
|
|
5
|
+
### allPassed[](#allpassed "Direct link to allPassed")
|
|
6
|
+
|
|
7
|
+
```ts
|
|
8
|
+
allPassed: boolean;
|
|
9
|
+
|
|
10
|
+
```
|
|
11
|
+
|
|
12
|
+
True when no eval failed (skips don't count as failures).
|
|
13
|
+
|
|
14
|
+
***
|
|
15
|
+
|
|
16
|
+
### failed[](#failed "Direct link to failed")
|
|
17
|
+
|
|
18
|
+
```ts
|
|
19
|
+
failed: number;
|
|
20
|
+
|
|
21
|
+
```
|
|
22
|
+
|
|
23
|
+
***
|
|
24
|
+
|
|
25
|
+
### passed[](#passed "Direct link to passed")
|
|
26
|
+
|
|
27
|
+
```ts
|
|
28
|
+
passed: number;
|
|
29
|
+
|
|
30
|
+
```
|
|
31
|
+
|
|
32
|
+
***
|
|
33
|
+
|
|
34
|
+
### skipped[](#skipped "Direct link to skipped")
|
|
35
|
+
|
|
36
|
+
```ts
|
|
37
|
+
skipped: number;
|
|
38
|
+
|
|
39
|
+
```
|
|
40
|
+
|
|
41
|
+
***
|
|
42
|
+
|
|
43
|
+
### total[](#total "Direct link to total")
|
|
44
|
+
|
|
45
|
+
```ts
|
|
46
|
+
total: number;
|
|
47
|
+
|
|
48
|
+
```
|
|
@@ -0,0 +1,67 @@
|
|
|
1
|
+
# Interface: HttpDriverOptions
|
|
2
|
+
|
|
3
|
+
## Properties[](#properties "Direct link to Properties")
|
|
4
|
+
|
|
5
|
+
### agent?[](#agent "Direct link to agent?")
|
|
6
|
+
|
|
7
|
+
```ts
|
|
8
|
+
optional agent: string;
|
|
9
|
+
|
|
10
|
+
```
|
|
11
|
+
|
|
12
|
+
Agent alias to target. Omit to use the app's default agent.
|
|
13
|
+
|
|
14
|
+
***
|
|
15
|
+
|
|
16
|
+
### baseUrl[](#baseurl "Direct link to baseUrl")
|
|
17
|
+
|
|
18
|
+
```ts
|
|
19
|
+
baseUrl: string;
|
|
20
|
+
|
|
21
|
+
```
|
|
22
|
+
|
|
23
|
+
Base URL of the running app, e.g. `http://localhost:3000`.
|
|
24
|
+
|
|
25
|
+
***
|
|
26
|
+
|
|
27
|
+
### headers?[](#headers "Direct link to headers?")
|
|
28
|
+
|
|
29
|
+
```ts
|
|
30
|
+
optional headers: Record<string, string>;
|
|
31
|
+
|
|
32
|
+
```
|
|
33
|
+
|
|
34
|
+
Extra request headers (e.g. auth for a deployed app).
|
|
35
|
+
|
|
36
|
+
***
|
|
37
|
+
|
|
38
|
+
### mlflowRunId?[](#mlflowrunid "Direct link to mlflowRunId?")
|
|
39
|
+
|
|
40
|
+
```ts
|
|
41
|
+
optional mlflowRunId: string;
|
|
42
|
+
|
|
43
|
+
```
|
|
44
|
+
|
|
45
|
+
MLflow run id to link each turn's trace to (for evaluation runs).
|
|
46
|
+
|
|
47
|
+
***
|
|
48
|
+
|
|
49
|
+
### path?[](#path "Direct link to path?")
|
|
50
|
+
|
|
51
|
+
```ts
|
|
52
|
+
optional path: string;
|
|
53
|
+
|
|
54
|
+
```
|
|
55
|
+
|
|
56
|
+
Chat endpoint path. Defaults to `/api/agents/chat`.
|
|
57
|
+
|
|
58
|
+
***
|
|
59
|
+
|
|
60
|
+
### timeoutMs?[](#timeoutms "Direct link to timeoutMs?")
|
|
61
|
+
|
|
62
|
+
```ts
|
|
63
|
+
optional timeoutMs: number;
|
|
64
|
+
|
|
65
|
+
```
|
|
66
|
+
|
|
67
|
+
Max wall-clock time for a single turn before it is abandoned as a failed turn (`succeeded: false`). Without this a hung agent — a blocked tool, a stalled model — never ends the SSE stream (heartbeats keep it alive), so the read loop spins forever and wedges the whole sequential suite. Defaults to 120s.
|
|
@@ -0,0 +1,34 @@
|
|
|
1
|
+
# Interface: JudgeConfig
|
|
2
|
+
|
|
3
|
+
## Properties[](#properties "Direct link to Properties")
|
|
4
|
+
|
|
5
|
+
### client[](#client "Direct link to client")
|
|
6
|
+
|
|
7
|
+
```ts
|
|
8
|
+
client: MlflowClient;
|
|
9
|
+
|
|
10
|
+
```
|
|
11
|
+
|
|
12
|
+
Client for the workspace hosting the judge serving endpoint.
|
|
13
|
+
|
|
14
|
+
***
|
|
15
|
+
|
|
16
|
+
### model[](#model "Direct link to model")
|
|
17
|
+
|
|
18
|
+
```ts
|
|
19
|
+
model: string;
|
|
20
|
+
|
|
21
|
+
```
|
|
22
|
+
|
|
23
|
+
Serving endpoint name used as the judge model.
|
|
24
|
+
|
|
25
|
+
***
|
|
26
|
+
|
|
27
|
+
### token[](#token "Direct link to token")
|
|
28
|
+
|
|
29
|
+
```ts
|
|
30
|
+
token: string;
|
|
31
|
+
|
|
32
|
+
```
|
|
33
|
+
|
|
34
|
+
Bearer token for the serving endpoint.
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
# Interface: JudgeScore
|
|
2
|
+
|
|
3
|
+
A normalized judge result. `score` is 0..1.
|
|
4
|
+
|
|
5
|
+
## Properties[](#properties "Direct link to Properties")
|
|
6
|
+
|
|
7
|
+
### rationale?[](#rationale "Direct link to rationale?")
|
|
8
|
+
|
|
9
|
+
```ts
|
|
10
|
+
optional rationale: string;
|
|
11
|
+
|
|
12
|
+
```
|
|
13
|
+
|
|
14
|
+
***
|
|
15
|
+
|
|
16
|
+
### score[](#score "Direct link to score")
|
|
17
|
+
|
|
18
|
+
```ts
|
|
19
|
+
score: number;
|
|
20
|
+
|
|
21
|
+
```
|
|
@@ -0,0 +1,34 @@
|
|
|
1
|
+
# Interface: MatchResult
|
|
2
|
+
|
|
3
|
+
Result of a deterministic matcher run against a value.
|
|
4
|
+
|
|
5
|
+
## Properties[](#properties "Direct link to Properties")
|
|
6
|
+
|
|
7
|
+
### detail?[](#detail "Direct link to detail?")
|
|
8
|
+
|
|
9
|
+
```ts
|
|
10
|
+
optional detail: string;
|
|
11
|
+
|
|
12
|
+
```
|
|
13
|
+
|
|
14
|
+
Human-readable explanation, shown on failure.
|
|
15
|
+
|
|
16
|
+
***
|
|
17
|
+
|
|
18
|
+
### pass[](#pass "Direct link to pass")
|
|
19
|
+
|
|
20
|
+
```ts
|
|
21
|
+
pass: boolean;
|
|
22
|
+
|
|
23
|
+
```
|
|
24
|
+
|
|
25
|
+
***
|
|
26
|
+
|
|
27
|
+
### score?[](#score "Direct link to score?")
|
|
28
|
+
|
|
29
|
+
```ts
|
|
30
|
+
optional score: number;
|
|
31
|
+
|
|
32
|
+
```
|
|
33
|
+
|
|
34
|
+
Optional 0..1 score for scored matchers (similarity, judges).
|
|
@@ -0,0 +1,30 @@
|
|
|
1
|
+
# Interface: PostResult
|
|
2
|
+
|
|
3
|
+
Structured result for a best-effort POST that must not throw.
|
|
4
|
+
|
|
5
|
+
## Properties[](#properties "Direct link to Properties")
|
|
6
|
+
|
|
7
|
+
### error?[](#error "Direct link to error?")
|
|
8
|
+
|
|
9
|
+
```ts
|
|
10
|
+
optional error: string;
|
|
11
|
+
|
|
12
|
+
```
|
|
13
|
+
|
|
14
|
+
***
|
|
15
|
+
|
|
16
|
+
### ok[](#ok "Direct link to ok")
|
|
17
|
+
|
|
18
|
+
```ts
|
|
19
|
+
ok: boolean;
|
|
20
|
+
|
|
21
|
+
```
|
|
22
|
+
|
|
23
|
+
***
|
|
24
|
+
|
|
25
|
+
### status?[](#status "Direct link to status?")
|
|
26
|
+
|
|
27
|
+
```ts
|
|
28
|
+
optional status: number;
|
|
29
|
+
|
|
30
|
+
```
|
|
@@ -0,0 +1,53 @@
|
|
|
1
|
+
# Interface: ReportOutcome
|
|
2
|
+
|
|
3
|
+
## Properties[](#properties "Direct link to Properties")
|
|
4
|
+
|
|
5
|
+
### failures[](#failures "Direct link to failures")
|
|
6
|
+
|
|
7
|
+
```ts
|
|
8
|
+
failures: {
|
|
9
|
+
error?: string;
|
|
10
|
+
status?: number;
|
|
11
|
+
traceId: string;
|
|
12
|
+
}[];
|
|
13
|
+
|
|
14
|
+
```
|
|
15
|
+
|
|
16
|
+
#### error?[](#error "Direct link to error?")
|
|
17
|
+
|
|
18
|
+
```ts
|
|
19
|
+
optional error: string;
|
|
20
|
+
|
|
21
|
+
```
|
|
22
|
+
|
|
23
|
+
#### status?[](#status "Direct link to status?")
|
|
24
|
+
|
|
25
|
+
```ts
|
|
26
|
+
optional status: number;
|
|
27
|
+
|
|
28
|
+
```
|
|
29
|
+
|
|
30
|
+
#### traceId[](#traceid "Direct link to traceId")
|
|
31
|
+
|
|
32
|
+
```ts
|
|
33
|
+
traceId: string;
|
|
34
|
+
|
|
35
|
+
```
|
|
36
|
+
|
|
37
|
+
***
|
|
38
|
+
|
|
39
|
+
### skipped[](#skipped "Direct link to skipped")
|
|
40
|
+
|
|
41
|
+
```ts
|
|
42
|
+
skipped: number;
|
|
43
|
+
|
|
44
|
+
```
|
|
45
|
+
|
|
46
|
+
***
|
|
47
|
+
|
|
48
|
+
### written[](#written "Direct link to written")
|
|
49
|
+
|
|
50
|
+
```ts
|
|
51
|
+
written: number;
|
|
52
|
+
|
|
53
|
+
```
|
|
@@ -0,0 +1,34 @@
|
|
|
1
|
+
# Interface: ResolveDatabricksAuthOptions
|
|
2
|
+
|
|
3
|
+
## Properties[](#properties "Direct link to Properties")
|
|
4
|
+
|
|
5
|
+
### host?[](#host "Direct link to host?")
|
|
6
|
+
|
|
7
|
+
```ts
|
|
8
|
+
optional host: string;
|
|
9
|
+
|
|
10
|
+
```
|
|
11
|
+
|
|
12
|
+
Explicit host; wins over the profile/SDK-resolved host when set.
|
|
13
|
+
|
|
14
|
+
***
|
|
15
|
+
|
|
16
|
+
### profile?[](#profile "Direct link to profile?")
|
|
17
|
+
|
|
18
|
+
```ts
|
|
19
|
+
optional profile: string;
|
|
20
|
+
|
|
21
|
+
```
|
|
22
|
+
|
|
23
|
+
`~/.databrickscfg` profile to authenticate with (e.g. `dogfood`).
|
|
24
|
+
|
|
25
|
+
***
|
|
26
|
+
|
|
27
|
+
### token?[](#token "Direct link to token?")
|
|
28
|
+
|
|
29
|
+
```ts
|
|
30
|
+
optional token: string;
|
|
31
|
+
|
|
32
|
+
```
|
|
33
|
+
|
|
34
|
+
Explicit bearer token; when set, no OAuth is minted (PAT/CI path).
|
|
@@ -0,0 +1,34 @@
|
|
|
1
|
+
# Interface: RunEvalOptions
|
|
2
|
+
|
|
3
|
+
## Properties[](#properties "Direct link to Properties")
|
|
4
|
+
|
|
5
|
+
### driver[](#driver "Direct link to driver")
|
|
6
|
+
|
|
7
|
+
```ts
|
|
8
|
+
driver: EvalDriver;
|
|
9
|
+
|
|
10
|
+
```
|
|
11
|
+
|
|
12
|
+
Drives the agent and returns reply/tool-calls/success per `send`.
|
|
13
|
+
|
|
14
|
+
***
|
|
15
|
+
|
|
16
|
+
### id[](#id "Direct link to id")
|
|
17
|
+
|
|
18
|
+
```ts
|
|
19
|
+
id: string;
|
|
20
|
+
|
|
21
|
+
```
|
|
22
|
+
|
|
23
|
+
Stable id for the eval (e.g. its file path relative to the evals dir).
|
|
24
|
+
|
|
25
|
+
***
|
|
26
|
+
|
|
27
|
+
### strict?[](#strict "Direct link to strict?")
|
|
28
|
+
|
|
29
|
+
```ts
|
|
30
|
+
optional strict: boolean;
|
|
31
|
+
|
|
32
|
+
```
|
|
33
|
+
|
|
34
|
+
When true, soft assertion failures also fail the eval.
|
|
@@ -0,0 +1,192 @@
|
|
|
1
|
+
# Interface: RunEvalsOptions
|
|
2
|
+
|
|
3
|
+
## Properties[](#properties "Direct link to Properties")
|
|
4
|
+
|
|
5
|
+
### baseUrl[](#baseurl "Direct link to baseUrl")
|
|
6
|
+
|
|
7
|
+
```ts
|
|
8
|
+
baseUrl: string;
|
|
9
|
+
|
|
10
|
+
```
|
|
11
|
+
|
|
12
|
+
Base URL of the running app to drive, e.g. `http://localhost:3000`.
|
|
13
|
+
|
|
14
|
+
***
|
|
15
|
+
|
|
16
|
+
### concurrency?[](#concurrency "Direct link to concurrency?")
|
|
17
|
+
|
|
18
|
+
```ts
|
|
19
|
+
optional concurrency: number;
|
|
20
|
+
|
|
21
|
+
```
|
|
22
|
+
|
|
23
|
+
Max evals to drive concurrently. Each eval opens one stream to the app as the same user, so keep this at or below the app's `maxConcurrentStreamsPerUser` (default 5) or the surplus streams hit the 429 guard. Defaults to 4; clamped to `[1, total]`.
|
|
24
|
+
|
|
25
|
+
***
|
|
26
|
+
|
|
27
|
+
### filter?[](#filter "Direct link to filter?")
|
|
28
|
+
|
|
29
|
+
```ts
|
|
30
|
+
optional filter: string;
|
|
31
|
+
|
|
32
|
+
```
|
|
33
|
+
|
|
34
|
+
Substring filter on `<agent>/<id>` (or an exact agent id).
|
|
35
|
+
|
|
36
|
+
***
|
|
37
|
+
|
|
38
|
+
### headers?[](#headers "Direct link to headers?")
|
|
39
|
+
|
|
40
|
+
```ts
|
|
41
|
+
optional headers: Record<string, string>;
|
|
42
|
+
|
|
43
|
+
```
|
|
44
|
+
|
|
45
|
+
Extra request headers for the driver (e.g. auth for a deployed app).
|
|
46
|
+
|
|
47
|
+
***
|
|
48
|
+
|
|
49
|
+
### judge?[](#judge "Direct link to judge?")
|
|
50
|
+
|
|
51
|
+
```ts
|
|
52
|
+
optional judge: {
|
|
53
|
+
host: string;
|
|
54
|
+
model: string;
|
|
55
|
+
token: string;
|
|
56
|
+
};
|
|
57
|
+
|
|
58
|
+
```
|
|
59
|
+
|
|
60
|
+
When set, enable `t.judge.*` LLM-as-judge scoring via autoevals against a Databricks serving endpoint (`model`).
|
|
61
|
+
|
|
62
|
+
#### host[](#host "Direct link to host")
|
|
63
|
+
|
|
64
|
+
```ts
|
|
65
|
+
host: string;
|
|
66
|
+
|
|
67
|
+
```
|
|
68
|
+
|
|
69
|
+
#### model[](#model "Direct link to model")
|
|
70
|
+
|
|
71
|
+
```ts
|
|
72
|
+
model: string;
|
|
73
|
+
|
|
74
|
+
```
|
|
75
|
+
|
|
76
|
+
#### token[](#token "Direct link to token")
|
|
77
|
+
|
|
78
|
+
```ts
|
|
79
|
+
token: string;
|
|
80
|
+
|
|
81
|
+
```
|
|
82
|
+
|
|
83
|
+
***
|
|
84
|
+
|
|
85
|
+
### mlflow?[](#mlflow "Direct link to mlflow?")
|
|
86
|
+
|
|
87
|
+
```ts
|
|
88
|
+
optional mlflow: {
|
|
89
|
+
experimentId: string;
|
|
90
|
+
host: string;
|
|
91
|
+
sqlWarehouseId?: string;
|
|
92
|
+
token: string;
|
|
93
|
+
};
|
|
94
|
+
|
|
95
|
+
```
|
|
96
|
+
|
|
97
|
+
When set, create a native MLflow "Evaluation run": each eval's trace is linked to the run, pass/fail is written as feedback, and aggregate metrics are logged. Requires Databricks creds + the target experiment.
|
|
98
|
+
|
|
99
|
+
#### experimentId[](#experimentid "Direct link to experimentId")
|
|
100
|
+
|
|
101
|
+
```ts
|
|
102
|
+
experimentId: string;
|
|
103
|
+
|
|
104
|
+
```
|
|
105
|
+
|
|
106
|
+
#### host[](#host-1 "Direct link to host")
|
|
107
|
+
|
|
108
|
+
```ts
|
|
109
|
+
host: string;
|
|
110
|
+
|
|
111
|
+
```
|
|
112
|
+
|
|
113
|
+
#### sqlWarehouseId?[](#sqlwarehouseid "Direct link to sqlWarehouseId?")
|
|
114
|
+
|
|
115
|
+
```ts
|
|
116
|
+
optional sqlWarehouseId: string;
|
|
117
|
+
|
|
118
|
+
```
|
|
119
|
+
|
|
120
|
+
SQL warehouse id for writing assessments to UC-backed (V4) traces.
|
|
121
|
+
|
|
122
|
+
#### token[](#token-1 "Direct link to token")
|
|
123
|
+
|
|
124
|
+
```ts
|
|
125
|
+
token: string;
|
|
126
|
+
|
|
127
|
+
```
|
|
128
|
+
|
|
129
|
+
***
|
|
130
|
+
|
|
131
|
+
### now?[](#now "Direct link to now?")
|
|
132
|
+
|
|
133
|
+
```ts
|
|
134
|
+
optional now: number;
|
|
135
|
+
|
|
136
|
+
```
|
|
137
|
+
|
|
138
|
+
Wall-clock timestamp (ms) for run create/finish — pass `Date.now()`.
|
|
139
|
+
|
|
140
|
+
***
|
|
141
|
+
|
|
142
|
+
### onEvent()?[](#onevent "Direct link to onEvent()?")
|
|
143
|
+
|
|
144
|
+
```ts
|
|
145
|
+
optional onEvent: (event: EvalProgress) => void;
|
|
146
|
+
|
|
147
|
+
```
|
|
148
|
+
|
|
149
|
+
Progress callback, invoked as evals are discovered, started, and finished.
|
|
150
|
+
|
|
151
|
+
#### Parameters[](#parameters "Direct link to Parameters")
|
|
152
|
+
|
|
153
|
+
| Parameter | Type |
|
|
154
|
+
| --------- | ------------------------------------------------------------------- |
|
|
155
|
+
| `event` | [`EvalProgress`](./docs/api/appkit/TypeAlias.EvalProgress.md) |
|
|
156
|
+
|
|
157
|
+
#### Returns[](#returns "Direct link to Returns")
|
|
158
|
+
|
|
159
|
+
`void`
|
|
160
|
+
|
|
161
|
+
***
|
|
162
|
+
|
|
163
|
+
### rootDir?[](#rootdir "Direct link to rootDir?")
|
|
164
|
+
|
|
165
|
+
```ts
|
|
166
|
+
optional rootDir: string;
|
|
167
|
+
|
|
168
|
+
```
|
|
169
|
+
|
|
170
|
+
Project root containing `server/agents/`. Defaults to `process.cwd()`.
|
|
171
|
+
|
|
172
|
+
***
|
|
173
|
+
|
|
174
|
+
### strict?[](#strict "Direct link to strict?")
|
|
175
|
+
|
|
176
|
+
```ts
|
|
177
|
+
optional strict: boolean;
|
|
178
|
+
|
|
179
|
+
```
|
|
180
|
+
|
|
181
|
+
Soft assertion failures also fail the eval.
|
|
182
|
+
|
|
183
|
+
***
|
|
184
|
+
|
|
185
|
+
### timeoutMs?[](#timeoutms "Direct link to timeoutMs?")
|
|
186
|
+
|
|
187
|
+
```ts
|
|
188
|
+
optional timeoutMs: number;
|
|
189
|
+
|
|
190
|
+
```
|
|
191
|
+
|
|
192
|
+
Per-turn wall-clock timeout (ms) before a turn is failed. Defaults to 120s.
|