@databricks/appkit 0.71.0 → 0.73.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CLAUDE.md +63 -0
- package/NOTICE.md +3 -2
- package/dist/appkit/package.js +1 -1
- package/dist/beta.d.ts +18 -3
- package/dist/beta.js +14 -1
- package/dist/cli/commands/agent/eval.js +120 -0
- package/dist/cli/commands/agent/eval.js.map +1 -0
- package/dist/cli/commands/agent/index.js +18 -0
- package/dist/cli/commands/agent/index.js.map +1 -0
- package/dist/cli/index.js +2 -0
- package/dist/cli/index.js.map +1 -1
- package/dist/connectors/index.js +2 -0
- package/dist/connectors/mlflow/auth.d.ts +28 -0
- package/dist/connectors/mlflow/auth.d.ts.map +1 -0
- package/dist/connectors/mlflow/auth.js +70 -0
- package/dist/connectors/mlflow/auth.js.map +1 -0
- package/dist/connectors/mlflow/client.d.ts +51 -0
- package/dist/connectors/mlflow/client.d.ts.map +1 -0
- package/dist/connectors/mlflow/client.js +93 -0
- package/dist/connectors/mlflow/client.js.map +1 -0
- package/dist/connectors/mlflow/index.d.ts +2 -0
- package/dist/database/errors.js +15 -5
- package/dist/database/errors.js.map +1 -1
- package/dist/database/runtime/data-path.d.ts +7 -0
- package/dist/database/runtime/data-path.d.ts.map +1 -0
- package/dist/database/runtime/data-path.js.map +1 -1
- package/dist/database/runtime/engine/drizzle-data-path.js +7 -5
- package/dist/database/runtime/engine/drizzle-data-path.js.map +1 -1
- package/dist/database/schema-builder/define-schema.d.ts +1 -1
- package/dist/database/schema-builder/define-schema.js +1 -1
- package/dist/database/schema-builder/define-schema.js.map +1 -1
- package/dist/errors/database-validation.d.ts +23 -0
- package/dist/errors/database-validation.d.ts.map +1 -0
- package/dist/errors/database-validation.js +24 -0
- package/dist/errors/database-validation.js.map +1 -0
- package/dist/errors/index.js +1 -0
- package/dist/evals/dataset.d.ts +36 -0
- package/dist/evals/dataset.d.ts.map +1 -0
- package/dist/evals/dataset.js +36 -0
- package/dist/evals/dataset.js.map +1 -0
- package/dist/evals/define-eval.d.ts +26 -0
- package/dist/evals/define-eval.d.ts.map +1 -0
- package/dist/evals/define-eval.js +28 -0
- package/dist/evals/define-eval.js.map +1 -0
- package/dist/evals/discover.d.ts +20 -0
- package/dist/evals/discover.d.ts.map +1 -0
- package/dist/evals/discover.js +49 -0
- package/dist/evals/discover.js.map +1 -0
- package/dist/evals/http-driver.d.ts +33 -0
- package/dist/evals/http-driver.d.ts.map +1 -0
- package/dist/evals/http-driver.js +123 -0
- package/dist/evals/http-driver.js.map +1 -0
- package/dist/evals/index.d.ts +14 -0
- package/dist/evals/index.js +14 -0
- package/dist/evals/judge.d.ts +27 -0
- package/dist/evals/judge.d.ts.map +1 -0
- package/dist/evals/judge.js +77 -0
- package/dist/evals/judge.js.map +1 -0
- package/dist/evals/matchers.d.ts +12 -0
- package/dist/evals/matchers.d.ts.map +1 -0
- package/dist/evals/matchers.js +26 -0
- package/dist/evals/matchers.js.map +1 -0
- package/dist/evals/mlflow-report.d.ts +37 -0
- package/dist/evals/mlflow-report.d.ts.map +1 -0
- package/dist/evals/mlflow-report.js +161 -0
- package/dist/evals/mlflow-report.js.map +1 -0
- package/dist/evals/mlflow-run.d.ts +13 -0
- package/dist/evals/mlflow-run.d.ts.map +1 -0
- package/dist/evals/mlflow-run.js +101 -0
- package/dist/evals/mlflow-run.js.map +1 -0
- package/dist/evals/pool.js +24 -0
- package/dist/evals/pool.js.map +1 -0
- package/dist/evals/report.d.ts +25 -0
- package/dist/evals/report.d.ts.map +1 -0
- package/dist/evals/report.js +57 -0
- package/dist/evals/report.js.map +1 -0
- package/dist/evals/run-eval.d.ts +23 -0
- package/dist/evals/run-eval.d.ts.map +1 -0
- package/dist/evals/run-eval.js +152 -0
- package/dist/evals/run-eval.js.map +1 -0
- package/dist/evals/run-evals.d.ts +94 -0
- package/dist/evals/run-evals.d.ts.map +1 -0
- package/dist/evals/run-evals.js +257 -0
- package/dist/evals/run-evals.js.map +1 -0
- package/dist/evals/types.d.ts +163 -0
- package/dist/evals/types.d.ts.map +1 -0
- package/dist/index.d.ts +2 -1
- package/dist/index.js +2 -1
- package/dist/plugin/plugin.d.ts.map +1 -1
- package/dist/plugin/plugin.js +1 -1
- package/dist/plugin/plugin.js.map +1 -1
- package/dist/plugins/agents/agents.js +1 -1
- package/dist/plugins/database/crud/contract.js +17 -8
- package/dist/plugins/database/crud/contract.js.map +1 -1
- package/dist/plugins/database/crud/exposure.js +63 -22
- package/dist/plugins/database/crud/exposure.js.map +1 -1
- package/dist/plugins/database/crud/request.js +50 -0
- package/dist/plugins/database/crud/request.js.map +1 -0
- package/dist/plugins/database/crud/response.js +77 -0
- package/dist/plugins/database/crud/response.js.map +1 -0
- package/dist/plugins/database/crud/routes.js +71 -52
- package/dist/plugins/database/crud/routes.js.map +1 -1
- package/dist/plugins/database/database.d.ts +6 -4
- package/dist/plugins/database/database.d.ts.map +1 -1
- package/dist/plugins/database/database.js +46 -16
- package/dist/plugins/database/database.js.map +1 -1
- package/dist/plugins/database/defaults.js +5 -1
- package/dist/plugins/database/defaults.js.map +1 -1
- package/dist/plugins/database/entity-client.js +143 -10
- package/dist/plugins/database/entity-client.js.map +1 -1
- package/dist/plugins/database/entity-types.d.ts +1 -1
- package/dist/plugins/database/hooks.d.ts +38 -0
- package/dist/plugins/database/hooks.d.ts.map +1 -0
- package/dist/plugins/database/index.d.ts +3 -2
- package/dist/plugins/database/lifecycle.js +67 -28
- package/dist/plugins/database/lifecycle.js.map +1 -1
- package/dist/plugins/database/scope.js +58 -0
- package/dist/plugins/database/scope.js.map +1 -0
- package/dist/plugins/database/types.d.ts +40 -12
- package/dist/plugins/database/types.d.ts.map +1 -1
- package/docs/api/appkit/Class.AppKitError.md +1 -0
- package/docs/api/appkit/Class.DatabaseValidationError.md +191 -0
- package/docs/api/appkit/Class.MlflowClient.md +103 -0
- package/docs/api/appkit/Function.buildAssessments.md +16 -0
- package/docs/api/appkit/Function.configureJudge.md +18 -0
- package/docs/api/appkit/Function.createHttpDriver.md +18 -0
- package/docs/api/appkit/Function.defineEval.md +35 -0
- package/docs/api/appkit/Function.defineSchema.md +1 -1
- package/docs/api/appkit/Function.discoverEvalFiles.md +18 -0
- package/docs/api/appkit/Function.equals.md +18 -0
- package/docs/api/appkit/Function.evalGlyph.md +18 -0
- package/docs/api/appkit/Function.formatEvalDetail.md +18 -0
- package/docs/api/appkit/Function.formatEvalHeadline.md +18 -0
- package/docs/api/appkit/Function.formatEvalResults.md +18 -0
- package/docs/api/appkit/Function.formatSummaryLine.md +18 -0
- package/docs/api/appkit/Function.includes.md +18 -0
- package/docs/api/appkit/Function.isJudgeConfigured.md +10 -0
- package/docs/api/appkit/Function.matches.md +18 -0
- package/docs/api/appkit/Function.normalizeHost.md +18 -0
- package/docs/api/appkit/Function.readEvalDataset.md +21 -0
- package/docs/api/appkit/Function.reportToMlflow.md +23 -0
- package/docs/api/appkit/Function.resolveDatabricksAuth.md +16 -0
- package/docs/api/appkit/Function.resolveWorkspaceClient.md +18 -0
- package/docs/api/appkit/Function.runEval.md +19 -0
- package/docs/api/appkit/Function.runEvalsInDir.md +18 -0
- package/docs/api/appkit/Function.summarize.md +16 -0
- package/docs/api/appkit/Interface.AssertionHandle.md +54 -0
- package/docs/api/appkit/Interface.AssertionResult.md +48 -0
- package/docs/api/appkit/Interface.Assessment.md +83 -0
- package/docs/api/appkit/Interface.CustomJudgeSpec.md +30 -0
- package/docs/api/appkit/Interface.DatabaseValidationIssue.md +21 -0
- package/docs/api/appkit/Interface.DatabricksAuth.md +21 -0
- package/docs/api/appkit/Interface.DatasetRow.md +21 -0
- package/docs/api/appkit/Interface.DiscoveredEval.md +36 -0
- package/docs/api/appkit/Interface.DriveResult.md +58 -0
- package/docs/api/appkit/Interface.EntityMutationHooks.md +173 -0
- package/docs/api/appkit/Interface.EvalDefinition.md +74 -0
- package/docs/api/appkit/Interface.EvalDriver.md +37 -0
- package/docs/api/appkit/Interface.EvalResult.md +83 -0
- package/docs/api/appkit/Interface.EvalRunSummary.md +46 -0
- package/docs/api/appkit/Interface.EvalSummary.md +48 -0
- package/docs/api/appkit/Interface.HookApp.md +12 -0
- package/docs/api/appkit/Interface.HookContext.md +21 -0
- package/docs/api/appkit/Interface.HttpDriverOptions.md +67 -0
- package/docs/api/appkit/Interface.JudgeConfig.md +34 -0
- package/docs/api/appkit/Interface.JudgeScore.md +21 -0
- package/docs/api/appkit/Interface.MatchResult.md +34 -0
- package/docs/api/appkit/Interface.PostResult.md +30 -0
- package/docs/api/appkit/Interface.ReadEvalDatasetOptions.md +34 -0
- package/docs/api/appkit/Interface.ReadSerializerContext.md +21 -0
- package/docs/api/appkit/Interface.ReportOutcome.md +53 -0
- package/docs/api/appkit/Interface.ResolveDatabricksAuthOptions.md +34 -0
- package/docs/api/appkit/Interface.RunEvalOptions.md +45 -0
- package/docs/api/appkit/Interface.RunEvalsOptions.md +214 -0
- package/docs/api/appkit/Interface.TestContext.md +245 -0
- package/docs/api/appkit/TypeAlias.DatabaseApiConfig.md +53 -0
- package/docs/api/appkit/TypeAlias.DatabaseApiWriteOperation.md +8 -0
- package/docs/api/appkit/TypeAlias.DatabaseApiWritesConfig.md +49 -0
- package/docs/api/appkit/TypeAlias.DatabaseExports.md +3 -3
- package/docs/api/appkit/TypeAlias.EntityHooks.md +25 -0
- package/docs/api/appkit/TypeAlias.EvalProgress.md +26 -0
- package/docs/api/appkit/TypeAlias.IDatabaseConfig.md +16 -5
- package/docs/api/appkit/TypeAlias.Matcher.md +18 -0
- package/docs/api/appkit/TypeAlias.ReadSerializer.md +19 -0
- package/docs/api/appkit/TypeAlias.Severity.md +8 -0
- package/docs/api/appkit/TypeAlias.TransactionClient.md +19 -0
- package/docs/api/appkit.md +157 -95
- package/docs/plugins/database.md +144 -0
- package/llms.txt +63 -0
- package/package.json +3 -2
- package/sbom.cdx.json +1 -1
|
@@ -0,0 +1,27 @@
|
|
|
1
|
+
import { MlflowClient } from "../connectors/mlflow/client.js";
|
|
2
|
+
import "../connectors/mlflow/index.js";
|
|
3
|
+
|
|
4
|
+
//#region src/evals/judge.d.ts
|
|
5
|
+
interface JudgeConfig {
|
|
6
|
+
/** Client for the workspace hosting the judge serving endpoint. */
|
|
7
|
+
client: MlflowClient;
|
|
8
|
+
/** Bearer token for the serving endpoint. */
|
|
9
|
+
token: string;
|
|
10
|
+
/** Serving endpoint name used as the judge model. */
|
|
11
|
+
model: string;
|
|
12
|
+
}
|
|
13
|
+
/** A normalized judge result. `score` is 0..1. */
|
|
14
|
+
interface JudgeScore {
|
|
15
|
+
score: number;
|
|
16
|
+
rationale?: string;
|
|
17
|
+
}
|
|
18
|
+
/**
|
|
19
|
+
* Configure the judge once. Sets the OpenAI-compatible client env autoevals
|
|
20
|
+
* reads and the default judge model. No-op-safe: on failure, judging stays
|
|
21
|
+
* disabled and {@link isJudgeConfigured} returns false.
|
|
22
|
+
*/
|
|
23
|
+
declare function configureJudge(config: JudgeConfig): Promise<void>;
|
|
24
|
+
declare function isJudgeConfigured(): boolean;
|
|
25
|
+
//#endregion
|
|
26
|
+
export { JudgeConfig, JudgeScore, configureJudge, isJudgeConfigured };
|
|
27
|
+
//# sourceMappingURL=judge.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"judge.d.ts","names":[],"sources":["../../src/evals/judge.ts"],"mappings":";;;;UAuBiB,WAAA;;EAEf,MAAA,EAAQ,YAAA;EAFO;EAIf,KAAA;;EAEA,KAAA;AAAA;;UAIe,UAAA;EACf,KAAA;EACA,SAAA;AAAA;AAFF;;;;;AAAA,iBAUsB,cAAA,CAAe,MAAA,EAAQ,WAAA,GAAc,OAAA;AAAA,iBAe3C,iBAAA,CAAA"}
|
|
@@ -0,0 +1,77 @@
|
|
|
1
|
+
//#region src/evals/judge.ts
|
|
2
|
+
let mod;
|
|
3
|
+
let enabled = false;
|
|
4
|
+
let configured = false;
|
|
5
|
+
let prevBaseUrl;
|
|
6
|
+
let prevApiKey;
|
|
7
|
+
/**
|
|
8
|
+
* Configure the judge once. Sets the OpenAI-compatible client env autoevals
|
|
9
|
+
* reads and the default judge model. No-op-safe: on failure, judging stays
|
|
10
|
+
* disabled and {@link isJudgeConfigured} returns false.
|
|
11
|
+
*/
|
|
12
|
+
async function configureJudge(config) {
|
|
13
|
+
try {
|
|
14
|
+
mod = await import("autoevals");
|
|
15
|
+
prevBaseUrl = process.env.OPENAI_BASE_URL;
|
|
16
|
+
prevApiKey = process.env.OPENAI_API_KEY;
|
|
17
|
+
process.env.OPENAI_BASE_URL = config.client.servingEndpointsUrl();
|
|
18
|
+
process.env.OPENAI_API_KEY = config.token;
|
|
19
|
+
configured = true;
|
|
20
|
+
mod.init({ defaultModel: config.model });
|
|
21
|
+
enabled = true;
|
|
22
|
+
} catch {
|
|
23
|
+
enabled = false;
|
|
24
|
+
}
|
|
25
|
+
}
|
|
26
|
+
function isJudgeConfigured() {
|
|
27
|
+
return enabled;
|
|
28
|
+
}
|
|
29
|
+
/**
|
|
30
|
+
* Restore the `OPENAI_*` env vars {@link configureJudge} set, so the judge
|
|
31
|
+
* bearer doesn't linger in `process.env` (readable by any imported eval code)
|
|
32
|
+
* after the run. Call once the run is done; safe when the judge was never
|
|
33
|
+
* configured. Also disables judging so a late `t.judge.*` call fails cleanly
|
|
34
|
+
* rather than hitting a torn-down client.
|
|
35
|
+
*/
|
|
36
|
+
function teardownJudge() {
|
|
37
|
+
if (!configured) return;
|
|
38
|
+
restoreEnv("OPENAI_BASE_URL", prevBaseUrl);
|
|
39
|
+
restoreEnv("OPENAI_API_KEY", prevApiKey);
|
|
40
|
+
configured = false;
|
|
41
|
+
enabled = false;
|
|
42
|
+
}
|
|
43
|
+
function restoreEnv(key, prev) {
|
|
44
|
+
if (prev === void 0) delete process.env[key];
|
|
45
|
+
else process.env[key] = prev;
|
|
46
|
+
}
|
|
47
|
+
/** Normalize an autoevals `Score` into a `JudgeScore`. */
|
|
48
|
+
function toJudgeScore(s) {
|
|
49
|
+
const rationale = s.metadata?.rationale;
|
|
50
|
+
return {
|
|
51
|
+
score: typeof s.score === "number" ? s.score : 0,
|
|
52
|
+
rationale: typeof rationale === "string" ? rationale : void 0
|
|
53
|
+
};
|
|
54
|
+
}
|
|
55
|
+
function ensure() {
|
|
56
|
+
if (!enabled || !mod) throw new Error("LLM judge is not configured. Pass --judge-model and authenticate via --profile (or DATABRICKS_HOST/DATABRICKS_TOKEN) to use t.judge.*");
|
|
57
|
+
return mod;
|
|
58
|
+
}
|
|
59
|
+
/** Factuality of `output` vs an `expected` reference. */
|
|
60
|
+
async function judgeFactuality(args) {
|
|
61
|
+
return toJudgeScore(await ensure().Factuality(args));
|
|
62
|
+
}
|
|
63
|
+
/** Whether `output` answers the question in `input`, optionally constrained by `criteria`. */
|
|
64
|
+
async function judgeClosedQA(args) {
|
|
65
|
+
return toJudgeScore(await ensure().ClosedQA(args));
|
|
66
|
+
}
|
|
67
|
+
/**
|
|
68
|
+
* A custom LLM judge defined by a prompt template + choice→score map — the
|
|
69
|
+
* TypeScript analog of MLflow's custom `@scorer`.
|
|
70
|
+
*/
|
|
71
|
+
async function judgeCustom(spec, args) {
|
|
72
|
+
return toJudgeScore(await ensure().LLMClassifierFromTemplate(spec)(args));
|
|
73
|
+
}
|
|
74
|
+
|
|
75
|
+
//#endregion
|
|
76
|
+
export { configureJudge, isJudgeConfigured, judgeClosedQA, judgeCustom, judgeFactuality, teardownJudge };
|
|
77
|
+
//# sourceMappingURL=judge.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"judge.js","names":[],"sources":["../../src/evals/judge.ts"],"sourcesContent":["import type { MlflowClient } from \"../connectors/mlflow\";\n\n/**\n * LLM-as-judge scoring via the `autoevals` library (the same scorers eve uses),\n * pointed at a Databricks serving endpoint. autoevals talks to an\n * OpenAI-compatible API; Databricks Model Serving exposes one at\n * `<host>/serving-endpoints`, so we set `OPENAI_BASE_URL`/`OPENAI_API_KEY` and\n * use the judge endpoint name as the model.\n *\n * There is no public REST to call Databricks' built-in judges directly (they're\n * Python/SDK-only and the rubric prompts live in the mlflow package), so we run\n * autoevals' equivalent scorers against a Databricks judge model.\n */\ntype AutoEvals = typeof import(\"autoevals\");\n\nlet mod: AutoEvals | undefined;\nlet enabled = false;\n// Whether configureJudge overwrote the OPENAI_* env vars and they still need\n// restoring, plus the values to restore them to.\nlet configured = false;\nlet prevBaseUrl: string | undefined;\nlet prevApiKey: string | undefined;\n\nexport interface JudgeConfig {\n /** Client for the workspace hosting the judge serving endpoint. */\n client: MlflowClient;\n /** Bearer token for the serving endpoint. */\n token: string;\n /** Serving endpoint name used as the judge model. */\n model: string;\n}\n\n/** A normalized judge result. `score` is 0..1. */\nexport interface JudgeScore {\n score: number;\n rationale?: string;\n}\n\n/**\n * Configure the judge once. Sets the OpenAI-compatible client env autoevals\n * reads and the default judge model. No-op-safe: on failure, judging stays\n * disabled and {@link isJudgeConfigured} returns false.\n */\nexport async function configureJudge(config: JudgeConfig): Promise<void> {\n try {\n mod = await import(\"autoevals\");\n prevBaseUrl = process.env.OPENAI_BASE_URL;\n prevApiKey = process.env.OPENAI_API_KEY;\n process.env.OPENAI_BASE_URL = config.client.servingEndpointsUrl();\n process.env.OPENAI_API_KEY = config.token;\n configured = true;\n mod.init({ defaultModel: config.model });\n enabled = true;\n } catch {\n enabled = false;\n }\n}\n\nexport function isJudgeConfigured(): boolean {\n return enabled;\n}\n\n/**\n * Restore the `OPENAI_*` env vars {@link configureJudge} set, so the judge\n * bearer doesn't linger in `process.env` (readable by any imported eval code)\n * after the run. Call once the run is done; safe when the judge was never\n * configured. Also disables judging so a late `t.judge.*` call fails cleanly\n * rather than hitting a torn-down client.\n */\nexport function teardownJudge(): void {\n if (!configured) return;\n restoreEnv(\"OPENAI_BASE_URL\", prevBaseUrl);\n restoreEnv(\"OPENAI_API_KEY\", prevApiKey);\n configured = false;\n enabled = false;\n}\n\nfunction restoreEnv(key: string, prev: string | undefined): void {\n if (prev === undefined) delete process.env[key];\n else process.env[key] = prev;\n}\n\n/** Normalize an autoevals `Score` into a `JudgeScore`. */\nexport function toJudgeScore(s: {\n score?: number | null;\n metadata?: Record<string, unknown>;\n}): JudgeScore {\n const rationale = s.metadata?.rationale;\n return {\n score: typeof s.score === \"number\" ? s.score : 0,\n rationale: typeof rationale === \"string\" ? rationale : undefined,\n };\n}\n\nfunction ensure(): AutoEvals {\n if (!enabled || !mod) {\n throw new Error(\n \"LLM judge is not configured. Pass --judge-model and authenticate via --profile (or DATABRICKS_HOST/DATABRICKS_TOKEN) to use t.judge.*\",\n );\n }\n return mod;\n}\n\n/** Factuality of `output` vs an `expected` reference. */\nexport async function judgeFactuality(args: {\n input: string;\n output: string;\n expected: string;\n}): Promise<JudgeScore> {\n return toJudgeScore(await ensure().Factuality(args));\n}\n\n/** Whether `output` answers the question in `input`, optionally constrained by `criteria`. */\nexport async function judgeClosedQA(args: {\n input: string;\n output: string;\n criteria: string;\n}): Promise<JudgeScore> {\n return toJudgeScore(await ensure().ClosedQA(args));\n}\n\n/**\n * A custom LLM judge defined by a prompt template + choice→score map — the\n * TypeScript analog of MLflow's custom `@scorer`.\n */\nexport async function judgeCustom(\n spec: {\n name: string;\n promptTemplate: string;\n choiceScores: Record<string, number>;\n },\n args: { input: string; output: string },\n): Promise<JudgeScore> {\n const scorer = ensure().LLMClassifierFromTemplate(spec);\n return toJudgeScore(await scorer(args));\n}\n"],"mappings":";AAeA,IAAI;AACJ,IAAI,UAAU;AAGd,IAAI,aAAa;AACjB,IAAI;AACJ,IAAI;;;;;;AAsBJ,eAAsB,eAAe,QAAoC;AACvE,KAAI;AACF,QAAM,MAAM,OAAO;AACnB,gBAAc,QAAQ,IAAI;AAC1B,eAAa,QAAQ,IAAI;AACzB,UAAQ,IAAI,kBAAkB,OAAO,OAAO,qBAAqB;AACjE,UAAQ,IAAI,iBAAiB,OAAO;AACpC,eAAa;AACb,MAAI,KAAK,EAAE,cAAc,OAAO,OAAO,CAAC;AACxC,YAAU;SACJ;AACN,YAAU;;;AAId,SAAgB,oBAA6B;AAC3C,QAAO;;;;;;;;;AAUT,SAAgB,gBAAsB;AACpC,KAAI,CAAC,WAAY;AACjB,YAAW,mBAAmB,YAAY;AAC1C,YAAW,kBAAkB,WAAW;AACxC,cAAa;AACb,WAAU;;AAGZ,SAAS,WAAW,KAAa,MAAgC;AAC/D,KAAI,SAAS,OAAW,QAAO,QAAQ,IAAI;KACtC,SAAQ,IAAI,OAAO;;;AAI1B,SAAgB,aAAa,GAGd;CACb,MAAM,YAAY,EAAE,UAAU;AAC9B,QAAO;EACL,OAAO,OAAO,EAAE,UAAU,WAAW,EAAE,QAAQ;EAC/C,WAAW,OAAO,cAAc,WAAW,YAAY;EACxD;;AAGH,SAAS,SAAoB;AAC3B,KAAI,CAAC,WAAW,CAAC,IACf,OAAM,IAAI,MACR,wIACD;AAEH,QAAO;;;AAIT,eAAsB,gBAAgB,MAId;AACtB,QAAO,aAAa,MAAM,QAAQ,CAAC,WAAW,KAAK,CAAC;;;AAItD,eAAsB,cAAc,MAIZ;AACtB,QAAO,aAAa,MAAM,QAAQ,CAAC,SAAS,KAAK,CAAC;;;;;;AAOpD,eAAsB,YACpB,MAKA,MACqB;AAErB,QAAO,aAAa,MADL,QAAQ,CAAC,0BAA0B,KAAK,CACtB,KAAK,CAAC"}
|
|
@@ -0,0 +1,12 @@
|
|
|
1
|
+
import { Matcher } from "./types.js";
|
|
2
|
+
|
|
3
|
+
//#region src/evals/matchers.d.ts
|
|
4
|
+
/** Passes when the value contains `substring`. */
|
|
5
|
+
declare function includes(substring: string): Matcher;
|
|
6
|
+
/** Passes when the value equals `expected` exactly. */
|
|
7
|
+
declare function equals(expected: string): Matcher;
|
|
8
|
+
/** Passes when the value matches `pattern`. */
|
|
9
|
+
declare function matches(pattern: RegExp): Matcher;
|
|
10
|
+
//#endregion
|
|
11
|
+
export { equals, includes, matches };
|
|
12
|
+
//# sourceMappingURL=matchers.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"matchers.d.ts","names":[],"sources":["../../src/evals/matchers.ts"],"mappings":";;;;iBAGgB,QAAA,CAAS,SAAA,WAAoB,OAAA;AAA7C;AAAA,iBAQgB,MAAA,CAAO,QAAA,WAAmB,OAAA;;iBAQ1B,OAAA,CAAQ,OAAA,EAAS,MAAA,GAAS,OAAA"}
|
|
@@ -0,0 +1,26 @@
|
|
|
1
|
+
//#region src/evals/matchers.ts
|
|
2
|
+
/** Passes when the value contains `substring`. */
|
|
3
|
+
function includes(substring) {
|
|
4
|
+
return (value) => ({
|
|
5
|
+
pass: value.includes(substring),
|
|
6
|
+
detail: `expected to include ${JSON.stringify(substring)}`
|
|
7
|
+
});
|
|
8
|
+
}
|
|
9
|
+
/** Passes when the value equals `expected` exactly. */
|
|
10
|
+
function equals(expected) {
|
|
11
|
+
return (value) => ({
|
|
12
|
+
pass: value === expected,
|
|
13
|
+
detail: `expected to equal ${JSON.stringify(expected)}`
|
|
14
|
+
});
|
|
15
|
+
}
|
|
16
|
+
/** Passes when the value matches `pattern`. */
|
|
17
|
+
function matches(pattern) {
|
|
18
|
+
return (value) => ({
|
|
19
|
+
pass: pattern.test(value),
|
|
20
|
+
detail: `expected to match ${pattern}`
|
|
21
|
+
});
|
|
22
|
+
}
|
|
23
|
+
|
|
24
|
+
//#endregion
|
|
25
|
+
export { equals, includes, matches };
|
|
26
|
+
//# sourceMappingURL=matchers.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"matchers.js","names":[],"sources":["../../src/evals/matchers.ts"],"sourcesContent":["import type { Matcher } from \"./types\";\n\n/** Passes when the value contains `substring`. */\nexport function includes(substring: string): Matcher {\n return (value) => ({\n pass: value.includes(substring),\n detail: `expected to include ${JSON.stringify(substring)}`,\n });\n}\n\n/** Passes when the value equals `expected` exactly. */\nexport function equals(expected: string): Matcher {\n return (value) => ({\n pass: value === expected,\n detail: `expected to equal ${JSON.stringify(expected)}`,\n });\n}\n\n/** Passes when the value matches `pattern`. */\nexport function matches(pattern: RegExp): Matcher {\n return (value) => ({\n pass: pattern.test(value),\n detail: `expected to match ${pattern}`,\n });\n}\n"],"mappings":";;AAGA,SAAgB,SAAS,WAA4B;AACnD,SAAQ,WAAW;EACjB,MAAM,MAAM,SAAS,UAAU;EAC/B,QAAQ,uBAAuB,KAAK,UAAU,UAAU;EACzD;;;AAIH,SAAgB,OAAO,UAA2B;AAChD,SAAQ,WAAW;EACjB,MAAM,UAAU;EAChB,QAAQ,qBAAqB,KAAK,UAAU,SAAS;EACtD;;;AAIH,SAAgB,QAAQ,SAA0B;AAChD,SAAQ,WAAW;EACjB,MAAM,QAAQ,KAAK,MAAM;EACzB,QAAQ,qBAAqB;EAC9B"}
|
|
@@ -0,0 +1,37 @@
|
|
|
1
|
+
import { MlflowClient } from "../connectors/mlflow/client.js";
|
|
2
|
+
import "../connectors/mlflow/index.js";
|
|
3
|
+
import { EvalResult } from "./types.js";
|
|
4
|
+
|
|
5
|
+
//#region src/evals/mlflow-report.d.ts
|
|
6
|
+
/** A Feedback assessment in the MLflow REST proto-JSON shape. */
|
|
7
|
+
interface Assessment {
|
|
8
|
+
trace_id: string;
|
|
9
|
+
assessment_name: string;
|
|
10
|
+
source: {
|
|
11
|
+
source_type: "CODE" | "HUMAN" | "LLM_JUDGE";
|
|
12
|
+
source_id: string;
|
|
13
|
+
};
|
|
14
|
+
feedback: {
|
|
15
|
+
value: unknown;
|
|
16
|
+
};
|
|
17
|
+
rationale?: string;
|
|
18
|
+
metadata?: Record<string, string>;
|
|
19
|
+
}
|
|
20
|
+
interface ReportOutcome {
|
|
21
|
+
written: number;
|
|
22
|
+
skipped: number;
|
|
23
|
+
failures: Array<{
|
|
24
|
+
traceId: string;
|
|
25
|
+
status?: number;
|
|
26
|
+
error?: string;
|
|
27
|
+
}>;
|
|
28
|
+
}
|
|
29
|
+
declare function buildAssessments(result: EvalResult): Assessment[];
|
|
30
|
+
/**
|
|
31
|
+
* Write one pass/fail assessment per eval result to the Databricks MLflow REST
|
|
32
|
+
* API. Never throws — failures are collected so the run still reports.
|
|
33
|
+
*/
|
|
34
|
+
declare function reportToMlflow(client: MlflowClient, results: EvalResult[], sqlWarehouseId?: string): Promise<ReportOutcome>;
|
|
35
|
+
//#endregion
|
|
36
|
+
export { Assessment, ReportOutcome, buildAssessments, reportToMlflow };
|
|
37
|
+
//# sourceMappingURL=mlflow-report.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"mlflow-report.d.ts","names":[],"sources":["../../src/evals/mlflow-report.ts"],"mappings":";;;;;;UAqBiB,UAAA;EACf,QAAA;EACA,eAAA;EACA,MAAA;IAAU,WAAA;IAA6C,SAAA;EAAA;EACvD,QAAA;IAAY,KAAA;EAAA;EACZ,SAAA;EACA,QAAA,GAAW,MAAA;AAAA;AAAA,UAGI,aAAA;EACf,OAAA;EACA,OAAA;EACA,QAAA,EAAU,KAAA;IAAQ,OAAA;IAAiB,MAAA;IAAiB,KAAA;EAAA;AAAA;AAAA,iBA2DtC,gBAAA,CAAiB,MAAA,EAAQ,UAAA,GAAa,UAAA;;;;;iBA6FhC,cAAA,CACpB,MAAA,EAAQ,YAAA,EACR,OAAA,EAAS,UAAA,IACT,cAAA,YACC,OAAA,CAAQ,aAAA"}
|
|
@@ -0,0 +1,161 @@
|
|
|
1
|
+
import { mapPool } from "./pool.js";
|
|
2
|
+
|
|
3
|
+
//#region src/evals/mlflow-report.ts
|
|
4
|
+
/**
|
|
5
|
+
* Max assessment writes in flight. Independent per-trace REST POSTs to the
|
|
6
|
+
* Databricks MLflow API; bounded to stay well under its rate limits.
|
|
7
|
+
*/
|
|
8
|
+
const ASSESSMENT_WRITE_CONCURRENCY = 8;
|
|
9
|
+
/**
|
|
10
|
+
* Retry budget for a 404 on an assessment write. The app exports each turn's
|
|
11
|
+
* trace asynchronously, so a write issued right after the run can beat the
|
|
12
|
+
* trace into the store (404 "trace not found"); linear backoff rides out that
|
|
13
|
+
* ingestion lag. Only 404s retry — a 4xx/5xx won't resolve by waiting.
|
|
14
|
+
*/
|
|
15
|
+
const ASSESSMENT_WRITE_RETRIES = 5;
|
|
16
|
+
const ASSESSMENT_RETRY_BASE_MS = 500;
|
|
17
|
+
const TRACE_NOT_FOUND = 404;
|
|
18
|
+
/**
|
|
19
|
+
* MLflow assessment names reject `.` (and we avoid spaces/parens too), so map
|
|
20
|
+
* anything outside `[A-Za-z0-9_-]` to `_`. The judge check on the raw label
|
|
21
|
+
* (`judge.`-prefixed) is unaffected — it runs before sanitization.
|
|
22
|
+
*/
|
|
23
|
+
function sanitizeName(label) {
|
|
24
|
+
return label.replace(/[^A-Za-z0-9_-]/g, "_");
|
|
25
|
+
}
|
|
26
|
+
/**
|
|
27
|
+
* Build the Feedback assessments for an eval result: one per assertion (judge
|
|
28
|
+
* assertions tagged `LLM_JUDGE` with their numeric score + rationale, so they
|
|
29
|
+
* render as judge feedback in MLflow) plus an overall `appkit_eval` pass/fail.
|
|
30
|
+
* Returns [] when there's no trace to attach to or the eval was skipped.
|
|
31
|
+
*/
|
|
32
|
+
/** One Feedback assessment for a single assertion (judge assertions tagged `LLM_JUDGE`). */
|
|
33
|
+
function assertionAssessment(a, traceId, name, evalId) {
|
|
34
|
+
return {
|
|
35
|
+
trace_id: traceId,
|
|
36
|
+
assessment_name: name,
|
|
37
|
+
source: a.label.startsWith("judge.") ? {
|
|
38
|
+
source_type: "LLM_JUDGE",
|
|
39
|
+
source_id: "appkit-judge"
|
|
40
|
+
} : {
|
|
41
|
+
source_type: "CODE",
|
|
42
|
+
source_id: "appkit-eval"
|
|
43
|
+
},
|
|
44
|
+
feedback: { value: a.score ?? a.pass },
|
|
45
|
+
rationale: a.detail,
|
|
46
|
+
metadata: {
|
|
47
|
+
eval_id: evalId,
|
|
48
|
+
severity: a.severity
|
|
49
|
+
}
|
|
50
|
+
};
|
|
51
|
+
}
|
|
52
|
+
/** The overall `appkit_eval` pass/fail assessment for an eval result. */
|
|
53
|
+
function overallAssessment(result, traceId) {
|
|
54
|
+
return {
|
|
55
|
+
trace_id: traceId,
|
|
56
|
+
assessment_name: "appkit_eval",
|
|
57
|
+
source: {
|
|
58
|
+
source_type: "CODE",
|
|
59
|
+
source_id: "appkit-eval"
|
|
60
|
+
},
|
|
61
|
+
feedback: { value: result.passed },
|
|
62
|
+
rationale: result.error ? "eval errored" : result.passed ? "all gates passed" : "one or more gates failed",
|
|
63
|
+
metadata: { eval_id: result.id }
|
|
64
|
+
};
|
|
65
|
+
}
|
|
66
|
+
function buildAssessments(result) {
|
|
67
|
+
if (!result.traceId || result.skipped) return [];
|
|
68
|
+
const traceId = result.traceId;
|
|
69
|
+
const out = [];
|
|
70
|
+
const used = /* @__PURE__ */ new Map();
|
|
71
|
+
for (const a of result.assertions) {
|
|
72
|
+
const base = sanitizeName(a.label);
|
|
73
|
+
const seen = used.get(base) ?? 0;
|
|
74
|
+
used.set(base, seen + 1);
|
|
75
|
+
const name = seen === 0 ? base : `${base}_${seen + 1}`;
|
|
76
|
+
out.push(assertionAssessment(a, traceId, name, result.id));
|
|
77
|
+
}
|
|
78
|
+
out.push(overallAssessment(result, traceId));
|
|
79
|
+
return out;
|
|
80
|
+
}
|
|
81
|
+
/**
|
|
82
|
+
* A Unity Catalog V4 trace id: `trace:/<catalog.schema[.prefix]>/<otel_hex>`.
|
|
83
|
+
* Databricks addresses these with the location and the bare hex as *separate*
|
|
84
|
+
* path segments (mirrors MLflow's `parse_trace_id_v4`).
|
|
85
|
+
*/
|
|
86
|
+
const V4_TRACE_ID = /^trace:\/([^/]+)\/([^/]+)$/;
|
|
87
|
+
/**
|
|
88
|
+
* Build the REST request to create one assessment, dispatching on the trace-id
|
|
89
|
+
* form:
|
|
90
|
+
* - **V4 UC** (`trace:/<location>/<hex>`) → `POST /api/4.0/mlflow/traces/
|
|
91
|
+
* {location}/{hex}/assessments`, body is the bare assessment (the Databricks
|
|
92
|
+
* RPC maps `location_id` from the path). This is the path UC-backed
|
|
93
|
+
* experiments require — the V3 endpoint 400s on a V4 id.
|
|
94
|
+
* - **V3** (`tr-...`, classic experiments) → `POST /api/3.0/mlflow/traces/
|
|
95
|
+
* {trace_id}/assessments`, body wraps the assessment in `{ assessment }`.
|
|
96
|
+
*/
|
|
97
|
+
function assessmentRequest(assessment, sqlWarehouseId) {
|
|
98
|
+
const v4 = V4_TRACE_ID.exec(assessment.trace_id);
|
|
99
|
+
if (v4) {
|
|
100
|
+
const [, location, id] = v4;
|
|
101
|
+
const query = sqlWarehouseId ? `?sql_warehouse_id=${encodeURIComponent(sqlWarehouseId)}` : "";
|
|
102
|
+
return {
|
|
103
|
+
path: `/api/4.0/mlflow/traces/${encodeURIComponent(location)}/${id}/assessments${query}`,
|
|
104
|
+
body: assessment
|
|
105
|
+
};
|
|
106
|
+
}
|
|
107
|
+
return {
|
|
108
|
+
path: `/api/3.0/mlflow/traces/${encodeURIComponent(assessment.trace_id)}/assessments`,
|
|
109
|
+
body: { assessment }
|
|
110
|
+
};
|
|
111
|
+
}
|
|
112
|
+
const sleep = (ms) => new Promise((resolve) => setTimeout(resolve, ms));
|
|
113
|
+
/**
|
|
114
|
+
* Write one assessment, retrying on a 404 (trace not yet ingested) with linear
|
|
115
|
+
* backoff. Returns the final {@link PostResult}; non-404 failures return on the
|
|
116
|
+
* first attempt.
|
|
117
|
+
*/
|
|
118
|
+
async function writeAssessment(client, assessment, sqlWarehouseId) {
|
|
119
|
+
const { path, body } = assessmentRequest(assessment, sqlWarehouseId);
|
|
120
|
+
let res = await client.postResult(path, body);
|
|
121
|
+
for (let attempt = 1; attempt <= ASSESSMENT_WRITE_RETRIES && !res.ok && res.status === TRACE_NOT_FOUND; attempt++) {
|
|
122
|
+
await sleep(ASSESSMENT_RETRY_BASE_MS * attempt);
|
|
123
|
+
res = await client.postResult(path, body);
|
|
124
|
+
}
|
|
125
|
+
return res;
|
|
126
|
+
}
|
|
127
|
+
/**
|
|
128
|
+
* Write one pass/fail assessment per eval result to the Databricks MLflow REST
|
|
129
|
+
* API. Never throws — failures are collected so the run still reports.
|
|
130
|
+
*/
|
|
131
|
+
async function reportToMlflow(client, results, sqlWarehouseId) {
|
|
132
|
+
const outcome = {
|
|
133
|
+
written: 0,
|
|
134
|
+
skipped: 0,
|
|
135
|
+
failures: []
|
|
136
|
+
};
|
|
137
|
+
const assessments = [];
|
|
138
|
+
for (const result of results) {
|
|
139
|
+
const built = buildAssessments(result);
|
|
140
|
+
if (built.length === 0) {
|
|
141
|
+
outcome.skipped++;
|
|
142
|
+
continue;
|
|
143
|
+
}
|
|
144
|
+
assessments.push(...built);
|
|
145
|
+
}
|
|
146
|
+
const posts = await mapPool(assessments, ASSESSMENT_WRITE_CONCURRENCY, (a) => writeAssessment(client, a, sqlWarehouseId));
|
|
147
|
+
for (let i = 0; i < posts.length; i++) {
|
|
148
|
+
const res = posts[i];
|
|
149
|
+
if (res.ok) outcome.written++;
|
|
150
|
+
else outcome.failures.push({
|
|
151
|
+
traceId: assessments[i].trace_id,
|
|
152
|
+
status: res.status,
|
|
153
|
+
error: res.error
|
|
154
|
+
});
|
|
155
|
+
}
|
|
156
|
+
return outcome;
|
|
157
|
+
}
|
|
158
|
+
|
|
159
|
+
//#endregion
|
|
160
|
+
export { buildAssessments, reportToMlflow };
|
|
161
|
+
//# sourceMappingURL=mlflow-report.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"mlflow-report.js","names":[],"sources":["../../src/evals/mlflow-report.ts"],"sourcesContent":["import type { MlflowClient, PostResult } from \"../connectors/mlflow\";\nimport { mapPool } from \"./pool\";\nimport type { AssertionResult, EvalResult } from \"./types\";\n\n/**\n * Max assessment writes in flight. Independent per-trace REST POSTs to the\n * Databricks MLflow API; bounded to stay well under its rate limits.\n */\nconst ASSESSMENT_WRITE_CONCURRENCY = 8;\n\n/**\n * Retry budget for a 404 on an assessment write. The app exports each turn's\n * trace asynchronously, so a write issued right after the run can beat the\n * trace into the store (404 \"trace not found\"); linear backoff rides out that\n * ingestion lag. Only 404s retry — a 4xx/5xx won't resolve by waiting.\n */\nconst ASSESSMENT_WRITE_RETRIES = 5;\nconst ASSESSMENT_RETRY_BASE_MS = 500;\nconst TRACE_NOT_FOUND = 404;\n\n/** A Feedback assessment in the MLflow REST proto-JSON shape. */\nexport interface Assessment {\n trace_id: string;\n assessment_name: string;\n source: { source_type: \"CODE\" | \"HUMAN\" | \"LLM_JUDGE\"; source_id: string };\n feedback: { value: unknown };\n rationale?: string;\n metadata?: Record<string, string>;\n}\n\nexport interface ReportOutcome {\n written: number;\n skipped: number;\n failures: Array<{ traceId: string; status?: number; error?: string }>;\n}\n\n/**\n * MLflow assessment names reject `.` (and we avoid spaces/parens too), so map\n * anything outside `[A-Za-z0-9_-]` to `_`. The judge check on the raw label\n * (`judge.`-prefixed) is unaffected — it runs before sanitization.\n */\nfunction sanitizeName(label: string): string {\n return label.replace(/[^A-Za-z0-9_-]/g, \"_\");\n}\n\n/**\n * Build the Feedback assessments for an eval result: one per assertion (judge\n * assertions tagged `LLM_JUDGE` with their numeric score + rationale, so they\n * render as judge feedback in MLflow) plus an overall `appkit_eval` pass/fail.\n * Returns [] when there's no trace to attach to or the eval was skipped.\n */\n/** One Feedback assessment for a single assertion (judge assertions tagged `LLM_JUDGE`). */\nfunction assertionAssessment(\n a: AssertionResult,\n traceId: string,\n name: string,\n evalId: string,\n): Assessment {\n const isJudge = a.label.startsWith(\"judge.\");\n return {\n trace_id: traceId,\n assessment_name: name,\n source: isJudge\n ? { source_type: \"LLM_JUDGE\", source_id: \"appkit-judge\" }\n : { source_type: \"CODE\", source_id: \"appkit-eval\" },\n // Judges report a numeric score; deterministic assertions a boolean.\n feedback: { value: a.score ?? a.pass },\n rationale: a.detail,\n metadata: { eval_id: evalId, severity: a.severity },\n };\n}\n\n/** The overall `appkit_eval` pass/fail assessment for an eval result. */\nfunction overallAssessment(result: EvalResult, traceId: string): Assessment {\n return {\n trace_id: traceId,\n assessment_name: \"appkit_eval\",\n source: { source_type: \"CODE\", source_id: \"appkit-eval\" },\n feedback: { value: result.passed },\n // Persist only a generic marker, never `result.error` itself: this rationale\n // is POSTed to MLflow and readable by anyone with access to the experiment,\n // a broader audience than the runner. The full error stays on the operator\n // console (printed via the reporter's per-eval detail line).\n rationale: result.error\n ? \"eval errored\"\n : result.passed\n ? \"all gates passed\"\n : \"one or more gates failed\",\n metadata: { eval_id: result.id },\n };\n}\n\nexport function buildAssessments(result: EvalResult): Assessment[] {\n if (!result.traceId || result.skipped) return [];\n const traceId = result.traceId;\n const out: Assessment[] = [];\n const used = new Map<string, number>();\n\n for (const a of result.assertions) {\n const base = sanitizeName(a.label);\n const seen = used.get(base) ?? 0;\n used.set(base, seen + 1);\n const name = seen === 0 ? base : `${base}_${seen + 1}`;\n out.push(assertionAssessment(a, traceId, name, result.id));\n }\n\n out.push(overallAssessment(result, traceId));\n return out;\n}\n\n/**\n * A Unity Catalog V4 trace id: `trace:/<catalog.schema[.prefix]>/<otel_hex>`.\n * Databricks addresses these with the location and the bare hex as *separate*\n * path segments (mirrors MLflow's `parse_trace_id_v4`).\n */\nconst V4_TRACE_ID = /^trace:\\/([^/]+)\\/([^/]+)$/;\n\n/**\n * Build the REST request to create one assessment, dispatching on the trace-id\n * form:\n * - **V4 UC** (`trace:/<location>/<hex>`) → `POST /api/4.0/mlflow/traces/\n * {location}/{hex}/assessments`, body is the bare assessment (the Databricks\n * RPC maps `location_id` from the path). This is the path UC-backed\n * experiments require — the V3 endpoint 400s on a V4 id.\n * - **V3** (`tr-...`, classic experiments) → `POST /api/3.0/mlflow/traces/\n * {trace_id}/assessments`, body wraps the assessment in `{ assessment }`.\n */\nfunction assessmentRequest(\n assessment: Assessment,\n sqlWarehouseId?: string,\n): {\n path: string;\n body: unknown;\n} {\n const v4 = V4_TRACE_ID.exec(assessment.trace_id);\n if (v4) {\n const [, location, id] = v4;\n // UC trace assessments are backed by a SQL warehouse; Databricks requires\n // its id as a query param (mlflow's `_append_sql_warehouse_id_param`).\n const query = sqlWarehouseId\n ? `?sql_warehouse_id=${encodeURIComponent(sqlWarehouseId)}`\n : \"\";\n return {\n path: `/api/4.0/mlflow/traces/${encodeURIComponent(location)}/${id}/assessments${query}`,\n body: assessment,\n };\n }\n return {\n path: `/api/3.0/mlflow/traces/${encodeURIComponent(assessment.trace_id)}/assessments`,\n body: { assessment },\n };\n}\n\nconst sleep = (ms: number): Promise<void> =>\n new Promise((resolve) => setTimeout(resolve, ms));\n\n/**\n * Write one assessment, retrying on a 404 (trace not yet ingested) with linear\n * backoff. Returns the final {@link PostResult}; non-404 failures return on the\n * first attempt.\n */\nasync function writeAssessment(\n client: MlflowClient,\n assessment: Assessment,\n sqlWarehouseId?: string,\n): Promise<PostResult> {\n const { path, body } = assessmentRequest(assessment, sqlWarehouseId);\n let res = await client.postResult(path, body);\n for (\n let attempt = 1;\n attempt <= ASSESSMENT_WRITE_RETRIES &&\n !res.ok &&\n res.status === TRACE_NOT_FOUND;\n attempt++\n ) {\n await sleep(ASSESSMENT_RETRY_BASE_MS * attempt);\n res = await client.postResult(path, body);\n }\n return res;\n}\n\n/**\n * Write one pass/fail assessment per eval result to the Databricks MLflow REST\n * API. Never throws — failures are collected so the run still reports.\n */\nexport async function reportToMlflow(\n client: MlflowClient,\n results: EvalResult[],\n sqlWarehouseId?: string,\n): Promise<ReportOutcome> {\n const outcome: ReportOutcome = { written: 0, skipped: 0, failures: [] };\n\n // Build every assessment first (pure, no I/O). `skipped` counts results with\n // no trace to attach to; `written`/`failures` are counted per assessment.\n const assessments: Assessment[] = [];\n for (const result of results) {\n const built = buildAssessments(result);\n if (built.length === 0) {\n outcome.skipped++;\n continue;\n }\n assessments.push(...built);\n }\n\n // The writes are independent per-trace REST calls, so run them through a\n // bounded pool instead of strictly serial. postResult never throws.\n const posts = await mapPool(assessments, ASSESSMENT_WRITE_CONCURRENCY, (a) =>\n writeAssessment(client, a, sqlWarehouseId),\n );\n for (let i = 0; i < posts.length; i++) {\n const res = posts[i];\n if (res.ok) {\n outcome.written++;\n } else {\n outcome.failures.push({\n traceId: assessments[i].trace_id,\n status: res.status,\n error: res.error,\n });\n }\n }\n return outcome;\n}\n"],"mappings":";;;;;;;AAQA,MAAM,+BAA+B;;;;;;;AAQrC,MAAM,2BAA2B;AACjC,MAAM,2BAA2B;AACjC,MAAM,kBAAkB;;;;;;AAuBxB,SAAS,aAAa,OAAuB;AAC3C,QAAO,MAAM,QAAQ,mBAAmB,IAAI;;;;;;;;;AAU9C,SAAS,oBACP,GACA,SACA,MACA,QACY;AAEZ,QAAO;EACL,UAAU;EACV,iBAAiB;EACjB,QAJc,EAAE,MAAM,WAAW,SAAS,GAKtC;GAAE,aAAa;GAAa,WAAW;GAAgB,GACvD;GAAE,aAAa;GAAQ,WAAW;GAAe;EAErD,UAAU,EAAE,OAAO,EAAE,SAAS,EAAE,MAAM;EACtC,WAAW,EAAE;EACb,UAAU;GAAE,SAAS;GAAQ,UAAU,EAAE;GAAU;EACpD;;;AAIH,SAAS,kBAAkB,QAAoB,SAA6B;AAC1E,QAAO;EACL,UAAU;EACV,iBAAiB;EACjB,QAAQ;GAAE,aAAa;GAAQ,WAAW;GAAe;EACzD,UAAU,EAAE,OAAO,OAAO,QAAQ;EAKlC,WAAW,OAAO,QACd,iBACA,OAAO,SACL,qBACA;EACN,UAAU,EAAE,SAAS,OAAO,IAAI;EACjC;;AAGH,SAAgB,iBAAiB,QAAkC;AACjE,KAAI,CAAC,OAAO,WAAW,OAAO,QAAS,QAAO,EAAE;CAChD,MAAM,UAAU,OAAO;CACvB,MAAM,MAAoB,EAAE;CAC5B,MAAM,uBAAO,IAAI,KAAqB;AAEtC,MAAK,MAAM,KAAK,OAAO,YAAY;EACjC,MAAM,OAAO,aAAa,EAAE,MAAM;EAClC,MAAM,OAAO,KAAK,IAAI,KAAK,IAAI;AAC/B,OAAK,IAAI,MAAM,OAAO,EAAE;EACxB,MAAM,OAAO,SAAS,IAAI,OAAO,GAAG,KAAK,GAAG,OAAO;AACnD,MAAI,KAAK,oBAAoB,GAAG,SAAS,MAAM,OAAO,GAAG,CAAC;;AAG5D,KAAI,KAAK,kBAAkB,QAAQ,QAAQ,CAAC;AAC5C,QAAO;;;;;;;AAQT,MAAM,cAAc;;;;;;;;;;;AAYpB,SAAS,kBACP,YACA,gBAIA;CACA,MAAM,KAAK,YAAY,KAAK,WAAW,SAAS;AAChD,KAAI,IAAI;EACN,MAAM,GAAG,UAAU,MAAM;EAGzB,MAAM,QAAQ,iBACV,qBAAqB,mBAAmB,eAAe,KACvD;AACJ,SAAO;GACL,MAAM,0BAA0B,mBAAmB,SAAS,CAAC,GAAG,GAAG,cAAc;GACjF,MAAM;GACP;;AAEH,QAAO;EACL,MAAM,0BAA0B,mBAAmB,WAAW,SAAS,CAAC;EACxE,MAAM,EAAE,YAAY;EACrB;;AAGH,MAAM,SAAS,OACb,IAAI,SAAS,YAAY,WAAW,SAAS,GAAG,CAAC;;;;;;AAOnD,eAAe,gBACb,QACA,YACA,gBACqB;CACrB,MAAM,EAAE,MAAM,SAAS,kBAAkB,YAAY,eAAe;CACpE,IAAI,MAAM,MAAM,OAAO,WAAW,MAAM,KAAK;AAC7C,MACE,IAAI,UAAU,GACd,WAAW,4BACX,CAAC,IAAI,MACL,IAAI,WAAW,iBACf,WACA;AACA,QAAM,MAAM,2BAA2B,QAAQ;AAC/C,QAAM,MAAM,OAAO,WAAW,MAAM,KAAK;;AAE3C,QAAO;;;;;;AAOT,eAAsB,eACpB,QACA,SACA,gBACwB;CACxB,MAAM,UAAyB;EAAE,SAAS;EAAG,SAAS;EAAG,UAAU,EAAE;EAAE;CAIvE,MAAM,cAA4B,EAAE;AACpC,MAAK,MAAM,UAAU,SAAS;EAC5B,MAAM,QAAQ,iBAAiB,OAAO;AACtC,MAAI,MAAM,WAAW,GAAG;AACtB,WAAQ;AACR;;AAEF,cAAY,KAAK,GAAG,MAAM;;CAK5B,MAAM,QAAQ,MAAM,QAAQ,aAAa,+BAA+B,MACtE,gBAAgB,QAAQ,GAAG,eAAe,CAC3C;AACD,MAAK,IAAI,IAAI,GAAG,IAAI,MAAM,QAAQ,KAAK;EACrC,MAAM,MAAM,MAAM;AAClB,MAAI,IAAI,GACN,SAAQ;MAER,SAAQ,SAAS,KAAK;GACpB,SAAS,YAAY,GAAG;GACxB,QAAQ,IAAI;GACZ,OAAO,IAAI;GACZ,CAAC;;AAGN,QAAO"}
|
|
@@ -0,0 +1,13 @@
|
|
|
1
|
+
import "../connectors/mlflow/index.js";
|
|
2
|
+
|
|
3
|
+
//#region src/evals/mlflow-run.d.ts
|
|
4
|
+
interface FinishOutcome {
|
|
5
|
+
finished: boolean;
|
|
6
|
+
/** Metric logging is best-effort; set when it failed (the run is still finished). */
|
|
7
|
+
metricsError?: string;
|
|
8
|
+
/** Set when the FINISHED update itself failed (run may be left RUNNING). */
|
|
9
|
+
finishError?: string;
|
|
10
|
+
}
|
|
11
|
+
//#endregion
|
|
12
|
+
export { FinishOutcome };
|
|
13
|
+
//# sourceMappingURL=mlflow-run.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"mlflow-run.d.ts","names":[],"sources":["../../src/evals/mlflow-run.ts"],"mappings":";;;UA6EiB,aAAA;EACf,QAAA;;EAEA,YAAA;;EAEA,WAAA;AAAA"}
|
|
@@ -0,0 +1,101 @@
|
|
|
1
|
+
//#region src/evals/mlflow-run.ts
|
|
2
|
+
/** Run tag value that makes a run appear under the experiment's "Evaluation runs". */
|
|
3
|
+
const GENAI_EVALUATE_RUN_TYPE = "genai_evaluate";
|
|
4
|
+
/**
|
|
5
|
+
* Source tags on the eval run. Traces linked via `mlflow.sourceRun` surface the
|
|
6
|
+
* run's source in the traces-table "Source" column (mirrors how Python's
|
|
7
|
+
* `evaluate()` shows the script name), so tagging the run is what populates it.
|
|
8
|
+
*/
|
|
9
|
+
const EVAL_SOURCE_NAME = "appkit agent eval";
|
|
10
|
+
/**
|
|
11
|
+
* Create an MLflow run tagged as a GenAI evaluation, so it shows under the
|
|
12
|
+
* experiment's "Evaluation runs". Returns the run id; link traces to it via the
|
|
13
|
+
* `mlflow.sourceRun` trace metadata and log results before finishing.
|
|
14
|
+
*/
|
|
15
|
+
async function createEvalRun(client, options) {
|
|
16
|
+
const created = await client.post("/api/2.0/mlflow/runs/create", {
|
|
17
|
+
experiment_id: options.experimentId,
|
|
18
|
+
start_time: options.startTime,
|
|
19
|
+
...options.runName ? { run_name: options.runName } : {},
|
|
20
|
+
tags: [
|
|
21
|
+
{
|
|
22
|
+
key: "mlflow.runType",
|
|
23
|
+
value: GENAI_EVALUATE_RUN_TYPE
|
|
24
|
+
},
|
|
25
|
+
{
|
|
26
|
+
key: "mlflow.source.name",
|
|
27
|
+
value: EVAL_SOURCE_NAME
|
|
28
|
+
},
|
|
29
|
+
{
|
|
30
|
+
key: "mlflow.source.type",
|
|
31
|
+
value: "LOCAL"
|
|
32
|
+
}
|
|
33
|
+
]
|
|
34
|
+
});
|
|
35
|
+
const runId = created.run?.info?.run_id ?? created.run?.info?.run_uuid;
|
|
36
|
+
if (!runId) throw new Error("runs/create returned no run id");
|
|
37
|
+
return runId;
|
|
38
|
+
}
|
|
39
|
+
/** Aggregate per-eval results into MLflow run metrics. */
|
|
40
|
+
function aggregateMetrics(results, timestamp) {
|
|
41
|
+
const scored = results.filter((r) => !r.skipped);
|
|
42
|
+
const passed = scored.filter((r) => r.passed).length;
|
|
43
|
+
return [
|
|
44
|
+
{
|
|
45
|
+
key: "eval/total",
|
|
46
|
+
value: results.length,
|
|
47
|
+
timestamp,
|
|
48
|
+
step: 0
|
|
49
|
+
},
|
|
50
|
+
{
|
|
51
|
+
key: "eval/scored",
|
|
52
|
+
value: scored.length,
|
|
53
|
+
timestamp,
|
|
54
|
+
step: 0
|
|
55
|
+
},
|
|
56
|
+
{
|
|
57
|
+
key: "eval/passed",
|
|
58
|
+
value: passed,
|
|
59
|
+
timestamp,
|
|
60
|
+
step: 0
|
|
61
|
+
},
|
|
62
|
+
{
|
|
63
|
+
key: "eval/pass_rate",
|
|
64
|
+
value: scored.length ? passed / scored.length : 0,
|
|
65
|
+
timestamp,
|
|
66
|
+
step: 0
|
|
67
|
+
}
|
|
68
|
+
];
|
|
69
|
+
}
|
|
70
|
+
/**
|
|
71
|
+
* Log aggregate metrics (best-effort) and mark the eval run FINISHED. Never
|
|
72
|
+
* throws — a metric-logging failure must not prevent the run from being closed,
|
|
73
|
+
* or it would be left stuck in RUNNING forever.
|
|
74
|
+
*/
|
|
75
|
+
async function finishEvalRun(client, options) {
|
|
76
|
+
const outcome = { finished: false };
|
|
77
|
+
const metrics = aggregateMetrics(options.results, options.endTime);
|
|
78
|
+
if (metrics.length) try {
|
|
79
|
+
await client.post("/api/2.0/mlflow/runs/log-batch", {
|
|
80
|
+
run_id: options.runId,
|
|
81
|
+
metrics
|
|
82
|
+
});
|
|
83
|
+
} catch (err) {
|
|
84
|
+
outcome.metricsError = err instanceof Error ? err.message : String(err);
|
|
85
|
+
}
|
|
86
|
+
try {
|
|
87
|
+
await client.post("/api/2.0/mlflow/runs/update", {
|
|
88
|
+
run_id: options.runId,
|
|
89
|
+
status: "FINISHED",
|
|
90
|
+
end_time: options.endTime
|
|
91
|
+
});
|
|
92
|
+
outcome.finished = true;
|
|
93
|
+
} catch (err) {
|
|
94
|
+
outcome.finishError = err instanceof Error ? err.message : String(err);
|
|
95
|
+
}
|
|
96
|
+
return outcome;
|
|
97
|
+
}
|
|
98
|
+
|
|
99
|
+
//#endregion
|
|
100
|
+
export { createEvalRun, finishEvalRun };
|
|
101
|
+
//# sourceMappingURL=mlflow-run.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"mlflow-run.js","names":[],"sources":["../../src/evals/mlflow-run.ts"],"sourcesContent":["import type { MlflowClient } from \"../connectors/mlflow\";\nimport type { EvalResult } from \"./types\";\n\n/** Run tag value that makes a run appear under the experiment's \"Evaluation runs\". */\nconst GENAI_EVALUATE_RUN_TYPE = \"genai_evaluate\";\n\n/**\n * Source tags on the eval run. Traces linked via `mlflow.sourceRun` surface the\n * run's source in the traces-table \"Source\" column (mirrors how Python's\n * `evaluate()` shows the script name), so tagging the run is what populates it.\n */\nconst EVAL_SOURCE_NAME = \"appkit agent eval\";\n\ninterface CreateRunResponse {\n run: { info: { run_id?: string; run_uuid?: string } };\n}\n\ninterface MlflowMetric {\n key: string;\n value: number;\n timestamp: number;\n step: number;\n}\n\n/**\n * Create an MLflow run tagged as a GenAI evaluation, so it shows under the\n * experiment's \"Evaluation runs\". Returns the run id; link traces to it via the\n * `mlflow.sourceRun` trace metadata and log results before finishing.\n */\nexport async function createEvalRun(\n client: MlflowClient,\n options: {\n experimentId: string;\n runName?: string;\n startTime: number;\n },\n): Promise<string> {\n const created = await client.post<CreateRunResponse>(\n \"/api/2.0/mlflow/runs/create\",\n {\n experiment_id: options.experimentId,\n start_time: options.startTime,\n ...(options.runName ? { run_name: options.runName } : {}),\n tags: [\n { key: \"mlflow.runType\", value: GENAI_EVALUATE_RUN_TYPE },\n { key: \"mlflow.source.name\", value: EVAL_SOURCE_NAME },\n { key: \"mlflow.source.type\", value: \"LOCAL\" },\n ],\n },\n );\n const runId = created.run?.info?.run_id ?? created.run?.info?.run_uuid;\n if (!runId) {\n throw new Error(\"runs/create returned no run id\");\n }\n return runId;\n}\n\n/** Aggregate per-eval results into MLflow run metrics. */\nexport function aggregateMetrics(\n results: EvalResult[],\n timestamp: number,\n): MlflowMetric[] {\n const scored = results.filter((r) => !r.skipped);\n const passed = scored.filter((r) => r.passed).length;\n return [\n { key: \"eval/total\", value: results.length, timestamp, step: 0 },\n { key: \"eval/scored\", value: scored.length, timestamp, step: 0 },\n { key: \"eval/passed\", value: passed, timestamp, step: 0 },\n {\n key: \"eval/pass_rate\",\n value: scored.length ? passed / scored.length : 0,\n timestamp,\n step: 0,\n },\n ];\n}\n\nexport interface FinishOutcome {\n finished: boolean;\n /** Metric logging is best-effort; set when it failed (the run is still finished). */\n metricsError?: string;\n /** Set when the FINISHED update itself failed (run may be left RUNNING). */\n finishError?: string;\n}\n\n/**\n * Log aggregate metrics (best-effort) and mark the eval run FINISHED. Never\n * throws — a metric-logging failure must not prevent the run from being closed,\n * or it would be left stuck in RUNNING forever.\n */\nexport async function finishEvalRun(\n client: MlflowClient,\n options: {\n runId: string;\n results: EvalResult[];\n endTime: number;\n },\n): Promise<FinishOutcome> {\n const outcome: FinishOutcome = { finished: false };\n\n const metrics = aggregateMetrics(options.results, options.endTime);\n if (metrics.length) {\n try {\n await client.post(\"/api/2.0/mlflow/runs/log-batch\", {\n run_id: options.runId,\n metrics,\n });\n } catch (err) {\n outcome.metricsError = err instanceof Error ? err.message : String(err);\n }\n }\n\n try {\n await client.post(\"/api/2.0/mlflow/runs/update\", {\n run_id: options.runId,\n status: \"FINISHED\",\n end_time: options.endTime,\n });\n outcome.finished = true;\n } catch (err) {\n outcome.finishError = err instanceof Error ? err.message : String(err);\n }\n\n return outcome;\n}\n"],"mappings":";;AAIA,MAAM,0BAA0B;;;;;;AAOhC,MAAM,mBAAmB;;;;;;AAkBzB,eAAsB,cACpB,QACA,SAKiB;CACjB,MAAM,UAAU,MAAM,OAAO,KAC3B,+BACA;EACE,eAAe,QAAQ;EACvB,YAAY,QAAQ;EACpB,GAAI,QAAQ,UAAU,EAAE,UAAU,QAAQ,SAAS,GAAG,EAAE;EACxD,MAAM;GACJ;IAAE,KAAK;IAAkB,OAAO;IAAyB;GACzD;IAAE,KAAK;IAAsB,OAAO;IAAkB;GACtD;IAAE,KAAK;IAAsB,OAAO;IAAS;GAC9C;EACF,CACF;CACD,MAAM,QAAQ,QAAQ,KAAK,MAAM,UAAU,QAAQ,KAAK,MAAM;AAC9D,KAAI,CAAC,MACH,OAAM,IAAI,MAAM,iCAAiC;AAEnD,QAAO;;;AAIT,SAAgB,iBACd,SACA,WACgB;CAChB,MAAM,SAAS,QAAQ,QAAQ,MAAM,CAAC,EAAE,QAAQ;CAChD,MAAM,SAAS,OAAO,QAAQ,MAAM,EAAE,OAAO,CAAC;AAC9C,QAAO;EACL;GAAE,KAAK;GAAc,OAAO,QAAQ;GAAQ;GAAW,MAAM;GAAG;EAChE;GAAE,KAAK;GAAe,OAAO,OAAO;GAAQ;GAAW,MAAM;GAAG;EAChE;GAAE,KAAK;GAAe,OAAO;GAAQ;GAAW,MAAM;GAAG;EACzD;GACE,KAAK;GACL,OAAO,OAAO,SAAS,SAAS,OAAO,SAAS;GAChD;GACA,MAAM;GACP;EACF;;;;;;;AAgBH,eAAsB,cACpB,QACA,SAKwB;CACxB,MAAM,UAAyB,EAAE,UAAU,OAAO;CAElD,MAAM,UAAU,iBAAiB,QAAQ,SAAS,QAAQ,QAAQ;AAClE,KAAI,QAAQ,OACV,KAAI;AACF,QAAM,OAAO,KAAK,kCAAkC;GAClD,QAAQ,QAAQ;GAChB;GACD,CAAC;UACK,KAAK;AACZ,UAAQ,eAAe,eAAe,QAAQ,IAAI,UAAU,OAAO,IAAI;;AAI3E,KAAI;AACF,QAAM,OAAO,KAAK,+BAA+B;GAC/C,QAAQ,QAAQ;GAChB,QAAQ;GACR,UAAU,QAAQ;GACnB,CAAC;AACF,UAAQ,WAAW;UACZ,KAAK;AACZ,UAAQ,cAAc,eAAe,QAAQ,IAAI,UAAU,OAAO,IAAI;;AAGxE,QAAO"}
|
|
@@ -0,0 +1,24 @@
|
|
|
1
|
+
//#region src/evals/pool.ts
|
|
2
|
+
/**
|
|
3
|
+
* Run `fn` over `items` with at most `concurrency` calls in flight, preserving
|
|
4
|
+
* input order in the result array. `concurrency` is clamped to `[1, length]`.
|
|
5
|
+
* `fn` must not reject — a rejection abandons the other in-flight items.
|
|
6
|
+
*/
|
|
7
|
+
async function mapPool(items, concurrency, fn) {
|
|
8
|
+
const results = new Array(items.length);
|
|
9
|
+
const n = Number.isFinite(concurrency) ? concurrency : 1;
|
|
10
|
+
const limit = Math.max(1, Math.min(n, items.length));
|
|
11
|
+
let cursor = 0;
|
|
12
|
+
const worker = async () => {
|
|
13
|
+
while (cursor < items.length) {
|
|
14
|
+
const index = cursor++;
|
|
15
|
+
results[index] = await fn(items[index], index);
|
|
16
|
+
}
|
|
17
|
+
};
|
|
18
|
+
await Promise.all(Array.from({ length: limit }, () => worker()));
|
|
19
|
+
return results;
|
|
20
|
+
}
|
|
21
|
+
|
|
22
|
+
//#endregion
|
|
23
|
+
export { mapPool };
|
|
24
|
+
//# sourceMappingURL=pool.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"pool.js","names":[],"sources":["../../src/evals/pool.ts"],"sourcesContent":["/**\n * Run `fn` over `items` with at most `concurrency` calls in flight, preserving\n * input order in the result array. `concurrency` is clamped to `[1, length]`.\n * `fn` must not reject — a rejection abandons the other in-flight items.\n */\nexport async function mapPool<T, R>(\n items: T[],\n concurrency: number,\n fn: (item: T, index: number) => Promise<R>,\n): Promise<R[]> {\n const results = new Array<R>(items.length);\n // Coerce a non-finite concurrency (e.g. NaN from a bad CLI `--concurrency`)\n // to 1: otherwise Math.min(NaN, len) is NaN, Array.from({length: NaN}) is [],\n // and zero workers spawn — silently skipping every item and leaving `undefined`\n // holes in the result.\n const n = Number.isFinite(concurrency) ? concurrency : 1;\n const limit = Math.max(1, Math.min(n, items.length));\n let cursor = 0;\n // Each worker pulls the next index off the shared cursor until exhausted.\n // `cursor++` is atomic between awaits (single-threaded), so no index is\n // handed to two workers.\n const worker = async (): Promise<void> => {\n while (cursor < items.length) {\n const index = cursor++;\n results[index] = await fn(items[index], index);\n }\n };\n await Promise.all(Array.from({ length: limit }, () => worker()));\n return results;\n}\n"],"mappings":";;;;;;AAKA,eAAsB,QACpB,OACA,aACA,IACc;CACd,MAAM,UAAU,IAAI,MAAS,MAAM,OAAO;CAK1C,MAAM,IAAI,OAAO,SAAS,YAAY,GAAG,cAAc;CACvD,MAAM,QAAQ,KAAK,IAAI,GAAG,KAAK,IAAI,GAAG,MAAM,OAAO,CAAC;CACpD,IAAI,SAAS;CAIb,MAAM,SAAS,YAA2B;AACxC,SAAO,SAAS,MAAM,QAAQ;GAC5B,MAAM,QAAQ;AACd,WAAQ,SAAS,MAAM,GAAG,MAAM,QAAQ,MAAM;;;AAGlD,OAAM,QAAQ,IAAI,MAAM,KAAK,EAAE,QAAQ,OAAO,QAAQ,QAAQ,CAAC,CAAC;AAChE,QAAO"}
|
|
@@ -0,0 +1,25 @@
|
|
|
1
|
+
import { EvalResult } from "./types.js";
|
|
2
|
+
|
|
3
|
+
//#region src/evals/report.d.ts
|
|
4
|
+
interface EvalSummary {
|
|
5
|
+
total: number;
|
|
6
|
+
passed: number;
|
|
7
|
+
failed: number;
|
|
8
|
+
skipped: number;
|
|
9
|
+
/** True when no eval failed (skips don't count as failures). */
|
|
10
|
+
allPassed: boolean;
|
|
11
|
+
}
|
|
12
|
+
declare function summarize(results: EvalResult[]): EvalSummary;
|
|
13
|
+
/** Status glyph for a single eval result. */
|
|
14
|
+
declare function evalGlyph(result: EvalResult): string;
|
|
15
|
+
/** The one-line header for a single eval result (no failure detail). */
|
|
16
|
+
declare function formatEvalHeadline(result: EvalResult): string;
|
|
17
|
+
/** Indented detail lines for a failing eval (error + failing assertions). */
|
|
18
|
+
declare function formatEvalDetail(result: EvalResult): string[];
|
|
19
|
+
/** The final PASS/FAIL summary line. */
|
|
20
|
+
declare function formatSummaryLine(results: EvalResult[]): string;
|
|
21
|
+
/** Render all results as a human-readable console report (non-streaming). */
|
|
22
|
+
declare function formatEvalResults(results: EvalResult[]): string;
|
|
23
|
+
//#endregion
|
|
24
|
+
export { EvalSummary, evalGlyph, formatEvalDetail, formatEvalHeadline, formatEvalResults, formatSummaryLine, summarize };
|
|
25
|
+
//# sourceMappingURL=report.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"report.d.ts","names":[],"sources":["../../src/evals/report.ts"],"mappings":";;;UAEiB,WAAA;EACf,KAAA;EACA,MAAA;EACA,MAAA;EACA,OAAA;EAJ0B;EAM1B,SAAA;AAAA;AAAA,iBAGc,SAAA,CAAU,OAAA,EAAS,UAAA,KAAe,WAAA;;iBAmBlC,SAAA,CAAU,MAAA,EAAQ,UAAA;;iBAMlB,kBAAA,CAAmB,MAAA,EAAQ,UAAA;AAzB3C;AAAA,iBAqCgB,gBAAA,CAAiB,MAAA,EAAQ,UAAA;;iBAYzB,iBAAA,CAAkB,OAAA,EAAS,UAAA;;iBAM3B,iBAAA,CAAkB,OAAA,EAAS,UAAA"}
|
|
@@ -0,0 +1,57 @@
|
|
|
1
|
+
//#region src/evals/report.ts
|
|
2
|
+
function summarize(results) {
|
|
3
|
+
let passed = 0;
|
|
4
|
+
let failed = 0;
|
|
5
|
+
let skipped = 0;
|
|
6
|
+
for (const r of results) if (r.skipped) skipped++;
|
|
7
|
+
else if (r.passed) passed++;
|
|
8
|
+
else failed++;
|
|
9
|
+
return {
|
|
10
|
+
total: results.length,
|
|
11
|
+
passed,
|
|
12
|
+
failed,
|
|
13
|
+
skipped,
|
|
14
|
+
allPassed: failed === 0
|
|
15
|
+
};
|
|
16
|
+
}
|
|
17
|
+
/** Status glyph for a single eval result. */
|
|
18
|
+
function evalGlyph(result) {
|
|
19
|
+
if (result.skipped) return "−";
|
|
20
|
+
return result.passed ? "✓" : "✗";
|
|
21
|
+
}
|
|
22
|
+
/** The one-line header for a single eval result (no failure detail). */
|
|
23
|
+
function formatEvalHeadline(result) {
|
|
24
|
+
if (result.skipped) return `− ${result.id} (skipped${result.skipped.reason ? `: ${result.skipped.reason}` : ""})`;
|
|
25
|
+
return `${evalGlyph(result)} ${result.id}${result.description ? ` — ${result.description}` : ""}`;
|
|
26
|
+
}
|
|
27
|
+
/** Indented detail lines for a failing eval (error + failing assertions). */
|
|
28
|
+
function formatEvalDetail(result) {
|
|
29
|
+
const lines = [];
|
|
30
|
+
if (result.error) lines.push(` error: ${result.error}`);
|
|
31
|
+
for (const a of result.assertions) {
|
|
32
|
+
if (a.pass) continue;
|
|
33
|
+
const tag = a.severity === "soft" ? "soft" : "gate";
|
|
34
|
+
lines.push(` ✗ [${tag}] ${a.label}${a.detail ? ` — ${a.detail}` : ""}`);
|
|
35
|
+
}
|
|
36
|
+
return lines;
|
|
37
|
+
}
|
|
38
|
+
/** The final PASS/FAIL summary line. */
|
|
39
|
+
function formatSummaryLine(results) {
|
|
40
|
+
const s = summarize(results);
|
|
41
|
+
return `${s.allPassed ? "PASS" : "FAIL"} — ${s.passed} passed, ${s.failed} failed, ${s.skipped} skipped (${s.total} total)`;
|
|
42
|
+
}
|
|
43
|
+
/** Render all results as a human-readable console report (non-streaming). */
|
|
44
|
+
function formatEvalResults(results) {
|
|
45
|
+
const lines = [];
|
|
46
|
+
for (const r of results) {
|
|
47
|
+
lines.push(formatEvalHeadline(r));
|
|
48
|
+
lines.push(...formatEvalDetail(r));
|
|
49
|
+
}
|
|
50
|
+
lines.push("");
|
|
51
|
+
lines.push(formatSummaryLine(results));
|
|
52
|
+
return lines.join("\n");
|
|
53
|
+
}
|
|
54
|
+
|
|
55
|
+
//#endregion
|
|
56
|
+
export { evalGlyph, formatEvalDetail, formatEvalHeadline, formatEvalResults, formatSummaryLine, summarize };
|
|
57
|
+
//# sourceMappingURL=report.js.map
|