@gleanwork/mcp-server-tester 1.0.1-beta.0 → 1.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/cli/index.js +1 -1
- package/dist/fixtures/mcp.js +1 -1
- package/dist/fixtures/mcp.js.map +1 -1
- package/dist/index-BcUWzQCx.d.cts +4282 -0
- package/dist/index-BcUWzQCx.d.ts +4282 -0
- package/dist/index.cjs +677 -26
- package/dist/index.cjs.map +1 -1
- package/dist/index.d.cts +551 -4365
- package/dist/index.d.ts +551 -4365
- package/dist/index.js +667 -29
- package/dist/index.js.map +1 -1
- package/dist/reporters/mcpReporter.cjs +294 -1
- package/dist/reporters/mcpReporter.cjs.map +1 -1
- package/dist/reporters/mcpReporter.d.cts +426 -3
- package/dist/reporters/mcpReporter.d.ts +426 -3
- package/dist/reporters/mcpReporter.js +294 -1
- package/dist/reporters/mcpReporter.js.map +1 -1
- package/dist/reporters/ui-dist/app.js +1 -1
- package/dist/types/index.cjs +19 -0
- package/dist/types/index.cjs.map +1 -0
- package/dist/types/index.d.cts +9 -0
- package/dist/types/index.d.ts +9 -0
- package/dist/types/index.js +17 -0
- package/dist/types/index.js.map +1 -0
- package/package.json +8 -1
package/dist/index.cjs
CHANGED
|
@@ -3126,6 +3126,20 @@ var init_dist3 = __esm({
|
|
|
3126
3126
|
};
|
|
3127
3127
|
}
|
|
3128
3128
|
});
|
|
3129
|
+
|
|
3130
|
+
// src/assertions/validators/types.ts
|
|
3131
|
+
var SnapshotSanitizers = {
|
|
3132
|
+
/** Replaces Unix timestamps (seconds and milliseconds) with a stable placeholder */
|
|
3133
|
+
TIMESTAMP: "timestamp",
|
|
3134
|
+
/** Replaces UUID v1-v5 strings with a stable placeholder */
|
|
3135
|
+
UUID: "uuid",
|
|
3136
|
+
/** Replaces ISO 8601 date/datetime strings with a stable placeholder */
|
|
3137
|
+
ISO_DATE: "iso-date",
|
|
3138
|
+
/** Replaces MongoDB ObjectId strings with a stable placeholder */
|
|
3139
|
+
OBJECT_ID: "objectId",
|
|
3140
|
+
/** Replaces JWT tokens with a stable placeholder */
|
|
3141
|
+
JWT: "jwt"
|
|
3142
|
+
};
|
|
3129
3143
|
var MCPHostCapabilitiesSchema = zod.z.object({
|
|
3130
3144
|
sampling: zod.z.record(zod.z.string(), zod.z.unknown()).optional(),
|
|
3131
3145
|
roots: zod.z.object({
|
|
@@ -4411,7 +4425,7 @@ function escapeHtml(text) {
|
|
|
4411
4425
|
|
|
4412
4426
|
// package.json
|
|
4413
4427
|
var package_default = {
|
|
4414
|
-
version: "1.
|
|
4428
|
+
version: "1.1.0"};
|
|
4415
4429
|
|
|
4416
4430
|
// src/mcp/clientFactory.ts
|
|
4417
4431
|
function getRetryAfterDelayMs(err) {
|
|
@@ -5935,20 +5949,6 @@ async function validateJudge(response, config) {
|
|
|
5935
5949
|
}
|
|
5936
5950
|
}
|
|
5937
5951
|
|
|
5938
|
-
// src/assertions/validators/types.ts
|
|
5939
|
-
var SnapshotSanitizers = {
|
|
5940
|
-
/** Replaces Unix timestamps (seconds and milliseconds) with a stable placeholder */
|
|
5941
|
-
TIMESTAMP: "timestamp",
|
|
5942
|
-
/** Replaces UUID v1-v5 strings with a stable placeholder */
|
|
5943
|
-
UUID: "uuid",
|
|
5944
|
-
/** Replaces ISO 8601 date/datetime strings with a stable placeholder */
|
|
5945
|
-
ISO_DATE: "iso-date",
|
|
5946
|
-
/** Replaces MongoDB ObjectId strings with a stable placeholder */
|
|
5947
|
-
OBJECT_ID: "objectId",
|
|
5948
|
-
/** Replaces JWT tokens with a stable placeholder */
|
|
5949
|
-
JWT: "jwt"
|
|
5950
|
-
};
|
|
5951
|
-
|
|
5952
5952
|
// src/mcp/fixtures/mcpFixture.ts
|
|
5953
5953
|
var DEFAULT_CALL_TIMEOUT_MS = 3e4;
|
|
5954
5954
|
function withCallTimeout(promise, ms, opName) {
|
|
@@ -7307,6 +7307,216 @@ function buildBaselinePassMap(baseline) {
|
|
|
7307
7307
|
}
|
|
7308
7308
|
return map;
|
|
7309
7309
|
}
|
|
7310
|
+
var KIND_DIRS = {
|
|
7311
|
+
"eval-runner-result": "eval-runs",
|
|
7312
|
+
"reporter-run": "reporter-runs",
|
|
7313
|
+
"eval-run-comparison": "comparisons/eval-runs",
|
|
7314
|
+
"server-comparison": "comparisons/servers"
|
|
7315
|
+
};
|
|
7316
|
+
function createEvalResultStore(config) {
|
|
7317
|
+
if (config.provider === "file") {
|
|
7318
|
+
return new FileEvalResultStore(config);
|
|
7319
|
+
}
|
|
7320
|
+
return new GCSEvalResultStore(config);
|
|
7321
|
+
}
|
|
7322
|
+
function resolveEvalResultStore(store) {
|
|
7323
|
+
return isEvalResultStore(store) ? store : createEvalResultStore(store);
|
|
7324
|
+
}
|
|
7325
|
+
function isEvalResultStore(value) {
|
|
7326
|
+
return typeof value === "object" && value !== null && "saveArtifact" in value && "loadArtifact" in value && "loadLatestArtifact" in value && "listArtifacts" in value;
|
|
7327
|
+
}
|
|
7328
|
+
function createStoredEvalArtifact(options) {
|
|
7329
|
+
const createdAt = options.createdAt ?? (/* @__PURE__ */ new Date()).toISOString();
|
|
7330
|
+
return {
|
|
7331
|
+
schemaVersion: 1,
|
|
7332
|
+
kind: options.kind,
|
|
7333
|
+
id: options.id ?? createDefaultArtifactId(createdAt),
|
|
7334
|
+
createdAt,
|
|
7335
|
+
metadata: {
|
|
7336
|
+
...defaultEnvironmentMetadata(),
|
|
7337
|
+
...options.metadata ?? {}
|
|
7338
|
+
},
|
|
7339
|
+
data: options.data
|
|
7340
|
+
};
|
|
7341
|
+
}
|
|
7342
|
+
function createDefaultArtifactId(timestamp = (/* @__PURE__ */ new Date()).toISOString()) {
|
|
7343
|
+
const safeTimestamp = timestamp.replace(/[:.]/g, "-");
|
|
7344
|
+
const runNumber = process.env.GITHUB_RUN_NUMBER;
|
|
7345
|
+
const sha = process.env.GITHUB_SHA?.slice(0, 12);
|
|
7346
|
+
const suffix = runNumber ?? sha;
|
|
7347
|
+
return suffix ? `${safeTimestamp}-${suffix}` : safeTimestamp;
|
|
7348
|
+
}
|
|
7349
|
+
function defaultEnvironmentMetadata() {
|
|
7350
|
+
return {
|
|
7351
|
+
...process.env.GITHUB_SHA !== void 0 && {
|
|
7352
|
+
gitHash: process.env.GITHUB_SHA
|
|
7353
|
+
},
|
|
7354
|
+
...process.env.GITHUB_REF_NAME !== void 0 && {
|
|
7355
|
+
branch: process.env.GITHUB_REF_NAME
|
|
7356
|
+
},
|
|
7357
|
+
...process.env.GITHUB_RUN_NUMBER !== void 0 && {
|
|
7358
|
+
runNumber: process.env.GITHUB_RUN_NUMBER
|
|
7359
|
+
},
|
|
7360
|
+
...process.env.GITHUB_EVENT_NAME !== void 0 && {
|
|
7361
|
+
trigger: process.env.GITHUB_EVENT_NAME
|
|
7362
|
+
}
|
|
7363
|
+
};
|
|
7364
|
+
}
|
|
7365
|
+
var FileEvalResultStore = class {
|
|
7366
|
+
dir;
|
|
7367
|
+
constructor(config) {
|
|
7368
|
+
this.dir = config.dir;
|
|
7369
|
+
}
|
|
7370
|
+
async saveArtifact(artifact) {
|
|
7371
|
+
const artifactDir = path2.join(this.dir, KIND_DIRS[artifact.kind]);
|
|
7372
|
+
await fs$1.mkdir(artifactDir, { recursive: true });
|
|
7373
|
+
const serialized = JSON.stringify(artifact, null, 2);
|
|
7374
|
+
await fs$1.writeFile(
|
|
7375
|
+
path2.join(artifactDir, `${artifact.id}.json`),
|
|
7376
|
+
serialized,
|
|
7377
|
+
"utf8"
|
|
7378
|
+
);
|
|
7379
|
+
await fs$1.writeFile(path2.join(artifactDir, "latest.json"), serialized, "utf8");
|
|
7380
|
+
}
|
|
7381
|
+
async loadArtifact(kind, id) {
|
|
7382
|
+
const raw = await fs$1.readFile(
|
|
7383
|
+
path2.join(this.dir, KIND_DIRS[kind], `${id}.json`),
|
|
7384
|
+
"utf8"
|
|
7385
|
+
);
|
|
7386
|
+
return JSON.parse(raw);
|
|
7387
|
+
}
|
|
7388
|
+
async loadLatestArtifact(kind) {
|
|
7389
|
+
try {
|
|
7390
|
+
const raw = await fs$1.readFile(
|
|
7391
|
+
path2.join(this.dir, KIND_DIRS[kind], "latest.json"),
|
|
7392
|
+
"utf8"
|
|
7393
|
+
);
|
|
7394
|
+
return JSON.parse(raw);
|
|
7395
|
+
} catch (error) {
|
|
7396
|
+
if (isMissingFileError(error)) {
|
|
7397
|
+
return null;
|
|
7398
|
+
}
|
|
7399
|
+
throw error;
|
|
7400
|
+
}
|
|
7401
|
+
}
|
|
7402
|
+
async listArtifacts(kind, options = {}) {
|
|
7403
|
+
let files;
|
|
7404
|
+
try {
|
|
7405
|
+
files = await fs$1.readdir(path2.join(this.dir, KIND_DIRS[kind]));
|
|
7406
|
+
} catch (error) {
|
|
7407
|
+
if (isMissingFileError(error)) {
|
|
7408
|
+
return [];
|
|
7409
|
+
}
|
|
7410
|
+
throw error;
|
|
7411
|
+
}
|
|
7412
|
+
const summaries = await Promise.all(
|
|
7413
|
+
files.filter((f) => f.endsWith(".json") && f !== "latest.json").map(async (file) => {
|
|
7414
|
+
const raw = await fs$1.readFile(
|
|
7415
|
+
path2.join(this.dir, KIND_DIRS[kind], file),
|
|
7416
|
+
"utf8"
|
|
7417
|
+
);
|
|
7418
|
+
return toSummary(JSON.parse(raw));
|
|
7419
|
+
})
|
|
7420
|
+
);
|
|
7421
|
+
return summaries.sort((a, b) => b.createdAt.localeCompare(a.createdAt)).slice(0, options.limit);
|
|
7422
|
+
}
|
|
7423
|
+
};
|
|
7424
|
+
var GCSEvalResultStore = class {
|
|
7425
|
+
bucketName;
|
|
7426
|
+
prefix;
|
|
7427
|
+
storage;
|
|
7428
|
+
constructor(config) {
|
|
7429
|
+
this.bucketName = config.bucket;
|
|
7430
|
+
this.prefix = trimSlashes(config.prefix ?? "");
|
|
7431
|
+
}
|
|
7432
|
+
async saveArtifact(artifact) {
|
|
7433
|
+
const bucket = await this.getBucket();
|
|
7434
|
+
const serialized = JSON.stringify(artifact, null, 2);
|
|
7435
|
+
await bucket.file(
|
|
7436
|
+
this.objectPath(
|
|
7437
|
+
artifact.kind,
|
|
7438
|
+
`${encodeURIComponent(artifact.id)}.json`
|
|
7439
|
+
)
|
|
7440
|
+
).save(serialized, {
|
|
7441
|
+
contentType: "application/json",
|
|
7442
|
+
resumable: false,
|
|
7443
|
+
validation: false
|
|
7444
|
+
});
|
|
7445
|
+
await bucket.file(this.objectPath(artifact.kind, "latest.json")).save(serialized, {
|
|
7446
|
+
contentType: "application/json",
|
|
7447
|
+
resumable: false,
|
|
7448
|
+
validation: false
|
|
7449
|
+
});
|
|
7450
|
+
}
|
|
7451
|
+
async loadArtifact(kind, id) {
|
|
7452
|
+
const bucket = await this.getBucket();
|
|
7453
|
+
const file = bucket.file(
|
|
7454
|
+
this.objectPath(kind, `${encodeURIComponent(id)}.json`)
|
|
7455
|
+
);
|
|
7456
|
+
const [buffer] = await file.download();
|
|
7457
|
+
return JSON.parse(buffer.toString("utf8"));
|
|
7458
|
+
}
|
|
7459
|
+
async loadLatestArtifact(kind) {
|
|
7460
|
+
const bucket = await this.getBucket();
|
|
7461
|
+
const file = bucket.file(this.objectPath(kind, "latest.json"));
|
|
7462
|
+
const [exists] = await file.exists();
|
|
7463
|
+
if (!exists) return null;
|
|
7464
|
+
const [buffer] = await file.download();
|
|
7465
|
+
return JSON.parse(buffer.toString("utf8"));
|
|
7466
|
+
}
|
|
7467
|
+
async listArtifacts(kind, options = {}) {
|
|
7468
|
+
const bucket = await this.getBucket();
|
|
7469
|
+
const [files] = await bucket.getFiles({
|
|
7470
|
+
prefix: this.objectPath(kind, "")
|
|
7471
|
+
});
|
|
7472
|
+
const summaries = await Promise.all(
|
|
7473
|
+
files.filter(
|
|
7474
|
+
(f) => f.name.endsWith(".json") && !f.name.endsWith("/latest.json")
|
|
7475
|
+
).map(async (file) => {
|
|
7476
|
+
const [buffer] = await file.download();
|
|
7477
|
+
return toSummary(
|
|
7478
|
+
JSON.parse(buffer.toString("utf8"))
|
|
7479
|
+
);
|
|
7480
|
+
})
|
|
7481
|
+
);
|
|
7482
|
+
return summaries.sort((a, b) => b.createdAt.localeCompare(a.createdAt)).slice(0, options.limit);
|
|
7483
|
+
}
|
|
7484
|
+
async getBucket() {
|
|
7485
|
+
if (!this.storage) {
|
|
7486
|
+
const moduleName = "@google-cloud/storage";
|
|
7487
|
+
let Storage;
|
|
7488
|
+
try {
|
|
7489
|
+
const mod = await import(moduleName);
|
|
7490
|
+
Storage = mod.Storage;
|
|
7491
|
+
} catch (error) {
|
|
7492
|
+
throw new Error(
|
|
7493
|
+
`GCS result storage requires the optional \`@google-cloud/storage\` package. Install it and authenticate with Application Default Credentials via GOOGLE_APPLICATION_CREDENTIALS.
|
|
7494
|
+
Original error: ${error instanceof Error ? error.message : String(error)}`
|
|
7495
|
+
);
|
|
7496
|
+
}
|
|
7497
|
+
this.storage = new Storage();
|
|
7498
|
+
}
|
|
7499
|
+
return this.storage.bucket(this.bucketName);
|
|
7500
|
+
}
|
|
7501
|
+
objectPath(kind, filename) {
|
|
7502
|
+
const parts = [this.prefix, KIND_DIRS[kind], filename].filter(Boolean);
|
|
7503
|
+
return parts.join("/");
|
|
7504
|
+
}
|
|
7505
|
+
};
|
|
7506
|
+
function toSummary(artifact) {
|
|
7507
|
+
return {
|
|
7508
|
+
kind: artifact.kind,
|
|
7509
|
+
id: artifact.id,
|
|
7510
|
+
createdAt: artifact.createdAt,
|
|
7511
|
+
metadata: artifact.metadata
|
|
7512
|
+
};
|
|
7513
|
+
}
|
|
7514
|
+
function trimSlashes(value) {
|
|
7515
|
+
return value.replace(/^\/+|\/+$/g, "");
|
|
7516
|
+
}
|
|
7517
|
+
function isMissingFileError(error) {
|
|
7518
|
+
return typeof error === "object" && error !== null && "code" in error && error.code === "ENOENT";
|
|
7519
|
+
}
|
|
7310
7520
|
var execFileAsync = util.promisify(child_process.execFile);
|
|
7311
7521
|
async function execFileNoThrow(file, args) {
|
|
7312
7522
|
try {
|
|
@@ -7353,6 +7563,41 @@ function sumUsage(a, b) {
|
|
|
7353
7563
|
}
|
|
7354
7564
|
|
|
7355
7565
|
// src/evals/evalRunner.ts
|
|
7566
|
+
function createToolOverrideMCP(mcp, variant) {
|
|
7567
|
+
return {
|
|
7568
|
+
...mcp,
|
|
7569
|
+
async listTools() {
|
|
7570
|
+
const tools = await mcp.listTools();
|
|
7571
|
+
const knownToolNames = new Set(tools.map((tool2) => tool2.name));
|
|
7572
|
+
const unknownToolNames = Object.keys(variant.tools).filter(
|
|
7573
|
+
(name15) => !knownToolNames.has(name15)
|
|
7574
|
+
);
|
|
7575
|
+
if (unknownToolNames.length > 0) {
|
|
7576
|
+
throw new Error(
|
|
7577
|
+
`[mcp-server-tester] toolOverrides variant "${variant.id}" references unknown tool(s): ` + unknownToolNames.join(", ")
|
|
7578
|
+
);
|
|
7579
|
+
}
|
|
7580
|
+
return tools.map((tool2) => {
|
|
7581
|
+
const override = variant.tools[tool2.name];
|
|
7582
|
+
if (!override) {
|
|
7583
|
+
return tool2;
|
|
7584
|
+
}
|
|
7585
|
+
return {
|
|
7586
|
+
...tool2,
|
|
7587
|
+
...override.description !== void 0 && {
|
|
7588
|
+
description: override.description
|
|
7589
|
+
},
|
|
7590
|
+
...override.inputSchema !== void 0 && {
|
|
7591
|
+
inputSchema: override.inputSchema
|
|
7592
|
+
}
|
|
7593
|
+
};
|
|
7594
|
+
});
|
|
7595
|
+
},
|
|
7596
|
+
async callTool(name15, args) {
|
|
7597
|
+
return mcp.callTool(name15, args);
|
|
7598
|
+
}
|
|
7599
|
+
};
|
|
7600
|
+
}
|
|
7356
7601
|
async function executeToolCall(evalCase, mcp) {
|
|
7357
7602
|
const mode = evalCase.mode || "direct";
|
|
7358
7603
|
try {
|
|
@@ -7537,9 +7782,12 @@ async function runExpectBlockValidations(expectBlock, response, config) {
|
|
|
7537
7782
|
}
|
|
7538
7783
|
return { expectations: results, toolPrecision, toolRecall };
|
|
7539
7784
|
}
|
|
7540
|
-
function buildRequest(evalCase) {
|
|
7785
|
+
function buildRequest(evalCase, toolOverrideVariantId) {
|
|
7541
7786
|
const request = {};
|
|
7542
7787
|
if (evalCase.description) request.description = evalCase.description;
|
|
7788
|
+
if (toolOverrideVariantId !== void 0) {
|
|
7789
|
+
request.toolOverrideVariantId = toolOverrideVariantId;
|
|
7790
|
+
}
|
|
7543
7791
|
if (evalCase.mode === "mcp_host") {
|
|
7544
7792
|
if (evalCase.scenario) request.scenario = evalCase.scenario;
|
|
7545
7793
|
if (evalCase.mcpHostConfig) {
|
|
@@ -7604,7 +7852,7 @@ async function runSingleIteration(evalCase, context, options) {
|
|
|
7604
7852
|
toolName: evalCase.scenario != null ? "mcp_host" : evalCase.toolName ?? "unknown",
|
|
7605
7853
|
source: "eval",
|
|
7606
7854
|
pass: didCasePass(error, expectationResults),
|
|
7607
|
-
request: buildRequest(evalCase),
|
|
7855
|
+
request: buildRequest(evalCase, options.toolOverrideVariantId),
|
|
7608
7856
|
response,
|
|
7609
7857
|
error,
|
|
7610
7858
|
expectations: expectationResults,
|
|
@@ -7683,7 +7931,8 @@ async function runEvalCase(evalCase, context, options = {}) {
|
|
|
7683
7931
|
authType: context.mcp.authType,
|
|
7684
7932
|
project: context.mcp.project,
|
|
7685
7933
|
durationMs: 0,
|
|
7686
|
-
tags: evalCase.tags
|
|
7934
|
+
tags: evalCase.tags,
|
|
7935
|
+
request: buildRequest(evalCase, options.toolOverrideVariantId)
|
|
7687
7936
|
};
|
|
7688
7937
|
const totalHostUsage = iterationResults.reduce(
|
|
7689
7938
|
(acc, r) => sumUsage(acc, r.hostUsage),
|
|
@@ -7742,11 +7991,15 @@ async function runEvalDataset(options, context) {
|
|
|
7742
7991
|
filterTags,
|
|
7743
7992
|
saveResultsTo,
|
|
7744
7993
|
omitResponsesFromBaseline = true,
|
|
7994
|
+
redactStoredResponses,
|
|
7995
|
+
resultStore,
|
|
7745
7996
|
baselineResultsFrom,
|
|
7997
|
+
toolOverrides,
|
|
7746
7998
|
mcpHostModel,
|
|
7747
7999
|
judgeModel
|
|
7748
8000
|
} = options;
|
|
7749
8001
|
const startTime = Date.now();
|
|
8002
|
+
const effectiveContext = toolOverrides ? { ...context, mcp: createToolOverrideMCP(context.mcp, toolOverrides) } : context;
|
|
7750
8003
|
const allSchemas = {
|
|
7751
8004
|
...dataset.schemas,
|
|
7752
8005
|
...schemas
|
|
@@ -7778,9 +8031,10 @@ async function runEvalDataset(options, context) {
|
|
|
7778
8031
|
}
|
|
7779
8032
|
}
|
|
7780
8033
|
const effectiveCase = withIterations.judgeReps === void 0 && defaultJudgeReps !== void 0 ? { ...withIterations, judgeReps: defaultJudgeReps } : withIterations;
|
|
7781
|
-
const result2 = await runEvalCase(effectiveCase,
|
|
8034
|
+
const result2 = await runEvalCase(effectiveCase, effectiveContext, {
|
|
7782
8035
|
datasetName: dataset.name,
|
|
7783
|
-
schemas: allSchemas
|
|
8036
|
+
schemas: allSchemas,
|
|
8037
|
+
toolOverrideVariantId: toolOverrides?.id
|
|
7784
8038
|
});
|
|
7785
8039
|
if (onCaseComplete) {
|
|
7786
8040
|
await onCaseComplete(result2);
|
|
@@ -7805,6 +8059,9 @@ async function runEvalDataset(options, context) {
|
|
|
7805
8059
|
gitHash,
|
|
7806
8060
|
timestamp: (/* @__PURE__ */ new Date()).toISOString(),
|
|
7807
8061
|
packageVersion: package_default.version,
|
|
8062
|
+
...toolOverrides !== void 0 && {
|
|
8063
|
+
toolOverrideVariantId: toolOverrides.id
|
|
8064
|
+
},
|
|
7808
8065
|
...mcpHostModel !== void 0 && { mcpHostModel },
|
|
7809
8066
|
...judgeModel !== void 0 && { judgeModel }
|
|
7810
8067
|
};
|
|
@@ -7823,7 +8080,7 @@ async function runEvalDataset(options, context) {
|
|
|
7823
8080
|
};
|
|
7824
8081
|
if (baselineResultsFrom) {
|
|
7825
8082
|
try {
|
|
7826
|
-
const baseline = await loadBaseline(baselineResultsFrom);
|
|
8083
|
+
const baseline = typeof baselineResultsFrom === "string" ? await loadBaseline(baselineResultsFrom) : await loadStoredBaseline(baselineResultsFrom, resultStore);
|
|
7827
8084
|
const baselinePassRate = baseline.total > 0 ? baseline.passed / baseline.total : 0;
|
|
7828
8085
|
const baselineMap = buildBaselinePassMap(baseline);
|
|
7829
8086
|
const currentCaseIds = result.caseResults.map((cr) => cr.id);
|
|
@@ -7851,7 +8108,7 @@ async function runEvalDataset(options, context) {
|
|
|
7851
8108
|
result.deltaPassRate = result.total > 0 ? result.passed / result.total - baselinePassRate : 0;
|
|
7852
8109
|
} catch (err) {
|
|
7853
8110
|
console.warn(
|
|
7854
|
-
`[mcp-server-tester] Could not load baseline from ${baselineResultsFrom}: ${err instanceof Error ? err.message : String(err)}`
|
|
8111
|
+
`[mcp-server-tester] Could not load baseline from ${formatBaselineRef(baselineResultsFrom)}: ${err instanceof Error ? err.message : String(err)}`
|
|
7855
8112
|
);
|
|
7856
8113
|
}
|
|
7857
8114
|
}
|
|
@@ -7866,9 +8123,26 @@ async function runEvalDataset(options, context) {
|
|
|
7866
8123
|
result.datasetToolF1 = avgPrec + avgRecall > 0 ? 2 * avgPrec * avgRecall / (avgPrec + avgRecall) : 0;
|
|
7867
8124
|
}
|
|
7868
8125
|
if (saveResultsTo) {
|
|
7869
|
-
|
|
7870
|
-
|
|
7871
|
-
|
|
8126
|
+
if (typeof saveResultsTo === "string") {
|
|
8127
|
+
await saveBaseline(result, saveResultsTo, {
|
|
8128
|
+
omitResponses: omitResponsesFromBaseline
|
|
8129
|
+
});
|
|
8130
|
+
} else {
|
|
8131
|
+
await saveStoredEvalResult(result, saveResultsTo, {
|
|
8132
|
+
resultStore,
|
|
8133
|
+
omitResponses: redactStoredResponses ?? true,
|
|
8134
|
+
metadata: {
|
|
8135
|
+
datasetName: dataset.name,
|
|
8136
|
+
...toolOverrides?.id !== void 0 && {
|
|
8137
|
+
toolOverrideVariantId: toolOverrides.id
|
|
8138
|
+
},
|
|
8139
|
+
...mcpHostModel !== void 0 && { mcpHostModel },
|
|
8140
|
+
...judgeModel !== void 0 && { judgeModel },
|
|
8141
|
+
...gitHash !== void 0 && { gitHash },
|
|
8142
|
+
packageVersion: package_default.version
|
|
8143
|
+
}
|
|
8144
|
+
});
|
|
8145
|
+
}
|
|
7872
8146
|
}
|
|
7873
8147
|
if (context.testInfo) {
|
|
7874
8148
|
await context.testInfo.attach("mcp-test-results", {
|
|
@@ -7882,6 +8156,50 @@ async function runEvalDataset(options, context) {
|
|
|
7882
8156
|
}
|
|
7883
8157
|
return result;
|
|
7884
8158
|
}
|
|
8159
|
+
async function loadStoredBaseline(baselineResultsFrom, resultStore) {
|
|
8160
|
+
if (!resultStore) {
|
|
8161
|
+
throw new Error("resultStore is required for store-backed baselines");
|
|
8162
|
+
}
|
|
8163
|
+
const store = resolveEvalResultStore(resultStore);
|
|
8164
|
+
const artifact = baselineResultsFrom.ref === "latest" ? await store.loadLatestArtifact("eval-runner-result") : await store.loadArtifact(
|
|
8165
|
+
"eval-runner-result",
|
|
8166
|
+
baselineResultsFrom.ref.id
|
|
8167
|
+
);
|
|
8168
|
+
if (!artifact) {
|
|
8169
|
+
throw new Error("No latest eval run artifact found");
|
|
8170
|
+
}
|
|
8171
|
+
return artifact.data;
|
|
8172
|
+
}
|
|
8173
|
+
async function saveStoredEvalResult(result, saveResultsTo, options) {
|
|
8174
|
+
if (!options.resultStore) {
|
|
8175
|
+
throw new Error("resultStore is required for store-backed saves");
|
|
8176
|
+
}
|
|
8177
|
+
const store = resolveEvalResultStore(options.resultStore);
|
|
8178
|
+
const data = options.omitResponses ? omitResponsesFromResult(result) : result;
|
|
8179
|
+
const id = saveResultsTo.ref && saveResultsTo.ref !== "latest" ? saveResultsTo.ref.id : void 0;
|
|
8180
|
+
await store.saveArtifact(
|
|
8181
|
+
createStoredEvalArtifact({
|
|
8182
|
+
kind: "eval-runner-result",
|
|
8183
|
+
id,
|
|
8184
|
+
data,
|
|
8185
|
+
metadata: options.metadata
|
|
8186
|
+
})
|
|
8187
|
+
);
|
|
8188
|
+
}
|
|
8189
|
+
function omitResponsesFromResult(result) {
|
|
8190
|
+
return {
|
|
8191
|
+
...result,
|
|
8192
|
+
caseResults: result.caseResults.map(
|
|
8193
|
+
({ response: _response, ...rest }) => rest
|
|
8194
|
+
)
|
|
8195
|
+
};
|
|
8196
|
+
}
|
|
8197
|
+
function formatBaselineRef(baselineResultsFrom) {
|
|
8198
|
+
if (typeof baselineResultsFrom === "string") {
|
|
8199
|
+
return baselineResultsFrom;
|
|
8200
|
+
}
|
|
8201
|
+
return baselineResultsFrom.ref === "latest" ? "resultStore latest" : `resultStore ${baselineResultsFrom.ref.id}`;
|
|
8202
|
+
}
|
|
7885
8203
|
|
|
7886
8204
|
// src/evals/serverComparison.ts
|
|
7887
8205
|
async function runServerComparison(options, contextA, contextB) {
|
|
@@ -7923,7 +8241,7 @@ async function runServerComparison(options, contextA, contextB) {
|
|
|
7923
8241
|
}
|
|
7924
8242
|
const total = cases.length;
|
|
7925
8243
|
const decidedCases = aWins + bWins + ties;
|
|
7926
|
-
|
|
8244
|
+
const comparison = {
|
|
7927
8245
|
dataset: options.dataset.name,
|
|
7928
8246
|
total,
|
|
7929
8247
|
aWins,
|
|
@@ -7940,6 +8258,326 @@ async function runServerComparison(options, contextA, contextB) {
|
|
|
7940
8258
|
serverBResult: resultB,
|
|
7941
8259
|
durationMs: Date.now() - startTime
|
|
7942
8260
|
};
|
|
8261
|
+
if (options.comparisonStore) {
|
|
8262
|
+
await saveServerComparison({
|
|
8263
|
+
store: options.comparisonStore,
|
|
8264
|
+
comparison,
|
|
8265
|
+
id: options.comparisonId,
|
|
8266
|
+
metadata: {
|
|
8267
|
+
datasetName: options.dataset.name,
|
|
8268
|
+
...options.comparisonMetadata ?? {}
|
|
8269
|
+
},
|
|
8270
|
+
redactStoredResponses: options.redactStoredResponses
|
|
8271
|
+
});
|
|
8272
|
+
}
|
|
8273
|
+
return comparison;
|
|
8274
|
+
}
|
|
8275
|
+
async function saveServerComparison(options) {
|
|
8276
|
+
const store = resolveEvalResultStore(options.store);
|
|
8277
|
+
const data = options.redactStoredResponses ? redactResponses(options.comparison) : options.comparison;
|
|
8278
|
+
const artifact = createStoredEvalArtifact({
|
|
8279
|
+
kind: "server-comparison",
|
|
8280
|
+
id: options.id,
|
|
8281
|
+
data,
|
|
8282
|
+
metadata: {
|
|
8283
|
+
datasetName: options.comparison.dataset,
|
|
8284
|
+
...options.metadata ?? {}
|
|
8285
|
+
}
|
|
8286
|
+
});
|
|
8287
|
+
await store.saveArtifact(artifact);
|
|
8288
|
+
return artifact;
|
|
8289
|
+
}
|
|
8290
|
+
function redactResponses(value) {
|
|
8291
|
+
return JSON.parse(
|
|
8292
|
+
JSON.stringify(
|
|
8293
|
+
value,
|
|
8294
|
+
(key, currentValue) => key === "response" ? void 0 : currentValue
|
|
8295
|
+
)
|
|
8296
|
+
);
|
|
8297
|
+
}
|
|
8298
|
+
|
|
8299
|
+
// src/evals/evalRunComparison.ts
|
|
8300
|
+
function compareEvalRuns(options) {
|
|
8301
|
+
const { baseline, candidate, labels } = options;
|
|
8302
|
+
const candidateMap = new Map(
|
|
8303
|
+
candidate.caseResults.map((result) => [result.id, result])
|
|
8304
|
+
);
|
|
8305
|
+
const cases = [];
|
|
8306
|
+
const seenIds = /* @__PURE__ */ new Set();
|
|
8307
|
+
for (const baselineCase of baseline.caseResults) {
|
|
8308
|
+
seenIds.add(baselineCase.id);
|
|
8309
|
+
const candidateCase = candidateMap.get(baselineCase.id);
|
|
8310
|
+
if (!candidateCase) {
|
|
8311
|
+
cases.push({
|
|
8312
|
+
id: baselineCase.id,
|
|
8313
|
+
outcome: "MISSING_FROM_CANDIDATE",
|
|
8314
|
+
baseline: baselineCase
|
|
8315
|
+
});
|
|
8316
|
+
continue;
|
|
8317
|
+
}
|
|
8318
|
+
cases.push({
|
|
8319
|
+
id: baselineCase.id,
|
|
8320
|
+
outcome: compareCaseOutcome(baselineCase.pass, candidateCase.pass),
|
|
8321
|
+
baseline: baselineCase,
|
|
8322
|
+
candidate: candidateCase
|
|
8323
|
+
});
|
|
8324
|
+
}
|
|
8325
|
+
for (const candidateCase of candidate.caseResults) {
|
|
8326
|
+
if (seenIds.has(candidateCase.id)) {
|
|
8327
|
+
continue;
|
|
8328
|
+
}
|
|
8329
|
+
cases.push({
|
|
8330
|
+
id: candidateCase.id,
|
|
8331
|
+
outcome: "MISSING_FROM_BASELINE",
|
|
8332
|
+
candidate: candidateCase
|
|
8333
|
+
});
|
|
8334
|
+
}
|
|
8335
|
+
const baselinePassRate = passRate(baseline);
|
|
8336
|
+
const candidatePassRate = passRate(candidate);
|
|
8337
|
+
return {
|
|
8338
|
+
baselineLabel: labels?.baseline ?? "baseline",
|
|
8339
|
+
candidateLabel: labels?.candidate ?? candidate.metadata?.toolOverrideVariantId ?? "candidate",
|
|
8340
|
+
baselinePassRate,
|
|
8341
|
+
candidatePassRate,
|
|
8342
|
+
deltaPassRate: candidatePassRate - baselinePassRate,
|
|
8343
|
+
...metricDelta(
|
|
8344
|
+
"ToolPrecision",
|
|
8345
|
+
baseline.datasetToolPrecision,
|
|
8346
|
+
candidate.datasetToolPrecision
|
|
8347
|
+
),
|
|
8348
|
+
...metricDelta(
|
|
8349
|
+
"ToolRecall",
|
|
8350
|
+
baseline.datasetToolRecall,
|
|
8351
|
+
candidate.datasetToolRecall
|
|
8352
|
+
),
|
|
8353
|
+
...metricDelta("ToolF1", baseline.datasetToolF1, candidate.datasetToolF1),
|
|
8354
|
+
cases,
|
|
8355
|
+
improvedCases: cases.filter((c) => c.outcome === "IMPROVED"),
|
|
8356
|
+
regressedCases: cases.filter((c) => c.outcome === "REGRESSED"),
|
|
8357
|
+
unchangedPasses: cases.filter((c) => c.outcome === "UNCHANGED_PASS"),
|
|
8358
|
+
unchangedFailures: cases.filter((c) => c.outcome === "UNCHANGED_FAIL"),
|
|
8359
|
+
missingFromBaseline: cases.filter(
|
|
8360
|
+
(c) => c.outcome === "MISSING_FROM_BASELINE"
|
|
8361
|
+
),
|
|
8362
|
+
missingFromCandidate: cases.filter(
|
|
8363
|
+
(c) => c.outcome === "MISSING_FROM_CANDIDATE"
|
|
8364
|
+
)
|
|
8365
|
+
};
|
|
8366
|
+
}
|
|
8367
|
+
async function loadStoredEvalRunnerResult(storeLike, ref) {
|
|
8368
|
+
const store = resolveEvalResultStore(storeLike);
|
|
8369
|
+
const artifact = ref === "latest" ? await store.loadLatestArtifact("eval-runner-result") : await store.loadArtifact(
|
|
8370
|
+
"eval-runner-result",
|
|
8371
|
+
ref.id
|
|
8372
|
+
);
|
|
8373
|
+
if (!artifact) {
|
|
8374
|
+
throw new Error("No latest eval run artifact found");
|
|
8375
|
+
}
|
|
8376
|
+
return artifact;
|
|
8377
|
+
}
|
|
8378
|
+
async function saveEvalRunComparison(options) {
|
|
8379
|
+
const store = resolveEvalResultStore(options.store);
|
|
8380
|
+
const data = options.redactStoredResponses ? redactResponses2(options.comparison) : options.comparison;
|
|
8381
|
+
const artifact = createStoredEvalArtifact({
|
|
8382
|
+
kind: "eval-run-comparison",
|
|
8383
|
+
id: options.id,
|
|
8384
|
+
data,
|
|
8385
|
+
metadata: {
|
|
8386
|
+
labels: {
|
|
8387
|
+
baseline: options.comparison.baselineLabel,
|
|
8388
|
+
candidate: options.comparison.candidateLabel
|
|
8389
|
+
},
|
|
8390
|
+
...options.metadata ?? {}
|
|
8391
|
+
}
|
|
8392
|
+
});
|
|
8393
|
+
await store.saveArtifact(artifact);
|
|
8394
|
+
return artifact;
|
|
8395
|
+
}
|
|
8396
|
+
function compareCaseOutcome(baselinePass, candidatePass) {
|
|
8397
|
+
if (!baselinePass && candidatePass) return "IMPROVED";
|
|
8398
|
+
if (baselinePass && !candidatePass) return "REGRESSED";
|
|
8399
|
+
return baselinePass ? "UNCHANGED_PASS" : "UNCHANGED_FAIL";
|
|
8400
|
+
}
|
|
8401
|
+
function passRate(result) {
|
|
8402
|
+
return result.total > 0 ? result.passed / result.total : 0;
|
|
8403
|
+
}
|
|
8404
|
+
function metricDelta(name15, baselineValue, candidateValue) {
|
|
8405
|
+
const result = {};
|
|
8406
|
+
if (baselineValue !== void 0) {
|
|
8407
|
+
result[`baseline${name15}`] = baselineValue;
|
|
8408
|
+
}
|
|
8409
|
+
if (candidateValue !== void 0) {
|
|
8410
|
+
result[`candidate${name15}`] = candidateValue;
|
|
8411
|
+
}
|
|
8412
|
+
if (baselineValue !== void 0 && candidateValue !== void 0) {
|
|
8413
|
+
result[`delta${name15}`] = candidateValue - baselineValue;
|
|
8414
|
+
}
|
|
8415
|
+
return result;
|
|
8416
|
+
}
|
|
8417
|
+
function redactResponses2(value) {
|
|
8418
|
+
return JSON.parse(
|
|
8419
|
+
JSON.stringify(
|
|
8420
|
+
value,
|
|
8421
|
+
(key, currentValue) => key === "response" ? void 0 : currentValue
|
|
8422
|
+
)
|
|
8423
|
+
);
|
|
8424
|
+
}
|
|
8425
|
+
|
|
8426
|
+
// src/evals/variantExperiment.ts
|
|
8427
|
+
async function runVariantExperiment(options, context) {
|
|
8428
|
+
const metric = options.metric ?? "passRate";
|
|
8429
|
+
const maxRounds = options.maxRounds ?? 1;
|
|
8430
|
+
const minImprovement = options.minImprovement ?? 0;
|
|
8431
|
+
const allowRegressions = options.allowRegressions ?? false;
|
|
8432
|
+
const baseline = await runEvalDataset(
|
|
8433
|
+
buildRunOptions(options, void 0),
|
|
8434
|
+
context
|
|
8435
|
+
);
|
|
8436
|
+
const baselineValue = readMetric(baseline, metric);
|
|
8437
|
+
if (baselineValue === void 0) {
|
|
8438
|
+
throw new Error(
|
|
8439
|
+
`Metric '${metric}' is unavailable: the dataset produced no tool precision/recall data. Add mcp_host cases with toolsTriggered expectations, or use metric 'passRate'.`
|
|
8440
|
+
);
|
|
8441
|
+
}
|
|
8442
|
+
const rounds = [];
|
|
8443
|
+
let bestSoFar;
|
|
8444
|
+
let bestAttempted;
|
|
8445
|
+
let reason = "max-rounds";
|
|
8446
|
+
for (let round = 0; round < maxRounds; round++) {
|
|
8447
|
+
const variants = await gatherVariants(options, {
|
|
8448
|
+
round,
|
|
8449
|
+
baseline,
|
|
8450
|
+
metric,
|
|
8451
|
+
history: rounds,
|
|
8452
|
+
bestSoFar
|
|
8453
|
+
});
|
|
8454
|
+
if (variants.length === 0) {
|
|
8455
|
+
reason = round === 0 ? "no-variants" : "no-improvement";
|
|
8456
|
+
break;
|
|
8457
|
+
}
|
|
8458
|
+
const candidates = [];
|
|
8459
|
+
for (const variant of variants) {
|
|
8460
|
+
const candidate = await scoreVariant(
|
|
8461
|
+
options,
|
|
8462
|
+
context,
|
|
8463
|
+
baseline,
|
|
8464
|
+
baselineValue,
|
|
8465
|
+
metric,
|
|
8466
|
+
allowRegressions,
|
|
8467
|
+
variant
|
|
8468
|
+
);
|
|
8469
|
+
candidates.push(candidate);
|
|
8470
|
+
bestAttempted = pickBetter(bestAttempted, candidate, true);
|
|
8471
|
+
}
|
|
8472
|
+
const roundBest = candidates.reduce(
|
|
8473
|
+
(best, candidate) => pickBetter(best, candidate, false),
|
|
8474
|
+
void 0
|
|
8475
|
+
);
|
|
8476
|
+
rounds.push({ round, candidates, best: roundBest });
|
|
8477
|
+
if (roundBest) {
|
|
8478
|
+
const improvement = roundBest.metricValue - (bestSoFar?.metricValue ?? baselineValue);
|
|
8479
|
+
bestSoFar = pickBetter(bestSoFar, roundBest, false);
|
|
8480
|
+
if (improvement < minImprovement) {
|
|
8481
|
+
reason = "no-improvement";
|
|
8482
|
+
break;
|
|
8483
|
+
}
|
|
8484
|
+
}
|
|
8485
|
+
}
|
|
8486
|
+
const winner = bestSoFar;
|
|
8487
|
+
const proposalSource = winner ?? bestAttempted;
|
|
8488
|
+
const proposal = proposalSource ? buildProposal(metric, baselineValue, proposalSource, winner !== void 0) : void 0;
|
|
8489
|
+
return {
|
|
8490
|
+
metric,
|
|
8491
|
+
baseline,
|
|
8492
|
+
rounds,
|
|
8493
|
+
winner,
|
|
8494
|
+
proposal,
|
|
8495
|
+
converged: true,
|
|
8496
|
+
reason
|
|
8497
|
+
};
|
|
8498
|
+
}
|
|
8499
|
+
async function gatherVariants(options, context) {
|
|
8500
|
+
if (context.round === 0 && options.variants && options.variants.length > 0) {
|
|
8501
|
+
return options.variants;
|
|
8502
|
+
}
|
|
8503
|
+
if (options.proposeVariants) {
|
|
8504
|
+
return options.proposeVariants(context);
|
|
8505
|
+
}
|
|
8506
|
+
return [];
|
|
8507
|
+
}
|
|
8508
|
+
async function scoreVariant(options, context, baseline, baselineValue, metric, allowRegressions, variant) {
|
|
8509
|
+
const result = await runEvalDataset(
|
|
8510
|
+
buildRunOptions(options, variant),
|
|
8511
|
+
context
|
|
8512
|
+
);
|
|
8513
|
+
const comparison = compareEvalRuns({
|
|
8514
|
+
baseline,
|
|
8515
|
+
candidate: result,
|
|
8516
|
+
labels: { candidate: variant.id }
|
|
8517
|
+
});
|
|
8518
|
+
const metricValue = readMetric(result, metric) ?? baselineValue;
|
|
8519
|
+
const disqualified = !allowRegressions && comparison.regressedCases.length > 0;
|
|
8520
|
+
return {
|
|
8521
|
+
variant,
|
|
8522
|
+
result,
|
|
8523
|
+
comparison,
|
|
8524
|
+
metricValue,
|
|
8525
|
+
metricDelta: metricValue - baselineValue,
|
|
8526
|
+
disqualified
|
|
8527
|
+
};
|
|
8528
|
+
}
|
|
8529
|
+
function pickBetter(incumbent, challenger, includeDisqualified) {
|
|
8530
|
+
if (!includeDisqualified && challenger.disqualified) {
|
|
8531
|
+
return incumbent;
|
|
8532
|
+
}
|
|
8533
|
+
if (!incumbent) {
|
|
8534
|
+
return challenger;
|
|
8535
|
+
}
|
|
8536
|
+
return challenger.metricValue > incumbent.metricValue ? challenger : incumbent;
|
|
8537
|
+
}
|
|
8538
|
+
function buildProposal(metric, baselineValue, source, isWinner) {
|
|
8539
|
+
let recommendation;
|
|
8540
|
+
if (isWinner) {
|
|
8541
|
+
recommendation = source.metricDelta > 0 ? "apply" : "inconclusive";
|
|
8542
|
+
} else {
|
|
8543
|
+
recommendation = source.disqualified ? "reject" : "inconclusive";
|
|
8544
|
+
}
|
|
8545
|
+
return {
|
|
8546
|
+
variantId: source.variant.id,
|
|
8547
|
+
metric,
|
|
8548
|
+
baselineValue,
|
|
8549
|
+
candidateValue: source.metricValue,
|
|
8550
|
+
delta: source.metricDelta,
|
|
8551
|
+
toolChanges: source.variant.tools,
|
|
8552
|
+
improvedCaseIds: source.comparison.improvedCases.map((c) => c.id),
|
|
8553
|
+
regressedCaseIds: source.comparison.regressedCases.map((c) => c.id),
|
|
8554
|
+
recommendation
|
|
8555
|
+
};
|
|
8556
|
+
}
|
|
8557
|
+
function readMetric(result, metric) {
|
|
8558
|
+
switch (metric) {
|
|
8559
|
+
case "passRate":
|
|
8560
|
+
return result.total > 0 ? result.passed / result.total : 0;
|
|
8561
|
+
case "toolF1":
|
|
8562
|
+
return result.datasetToolF1;
|
|
8563
|
+
case "toolPrecision":
|
|
8564
|
+
return result.datasetToolPrecision;
|
|
8565
|
+
case "toolRecall":
|
|
8566
|
+
return result.datasetToolRecall;
|
|
8567
|
+
}
|
|
8568
|
+
}
|
|
8569
|
+
function buildRunOptions(options, toolOverrides) {
|
|
8570
|
+
return {
|
|
8571
|
+
dataset: options.dataset,
|
|
8572
|
+
toolOverrides,
|
|
8573
|
+
defaultLlmIterations: options.defaultLlmIterations,
|
|
8574
|
+
defaultJudgeReps: options.defaultJudgeReps,
|
|
8575
|
+
concurrency: options.concurrency,
|
|
8576
|
+
filterTags: options.filterTags,
|
|
8577
|
+
schemas: options.schemas,
|
|
8578
|
+
mcpHostModel: options.mcpHostModel,
|
|
8579
|
+
judgeModel: options.judgeModel
|
|
8580
|
+
};
|
|
7943
8581
|
}
|
|
7944
8582
|
|
|
7945
8583
|
// src/spec/conformanceChecks.ts
|
|
@@ -8119,15 +8757,22 @@ exports.DiscoveryError = DiscoveryError;
|
|
|
8119
8757
|
exports.ENV_VAR_NAMES = ENV_VAR_NAMES;
|
|
8120
8758
|
exports.EvalCaseSchema = EvalCaseSchema;
|
|
8121
8759
|
exports.EvalDatasetSchema = EvalDatasetSchema;
|
|
8760
|
+
exports.FileEvalResultStore = FileEvalResultStore;
|
|
8761
|
+
exports.GCSEvalResultStore = GCSEvalResultStore;
|
|
8122
8762
|
exports.MCPConfigSchema = MCPConfigSchema;
|
|
8123
8763
|
exports.MCP_PROTOCOL_VERSION = MCP_PROTOCOL_VERSION;
|
|
8124
8764
|
exports.SnapshotSanitizers = SnapshotSanitizers;
|
|
8125
8765
|
exports.clearJudgeRegistry = clearJudgeRegistry;
|
|
8126
8766
|
exports.closeMCPClient = closeMCPClient;
|
|
8767
|
+
exports.compareEvalRuns = compareEvalRuns;
|
|
8768
|
+
exports.createDefaultArtifactId = createDefaultArtifactId;
|
|
8769
|
+
exports.createEvalResultStore = createEvalResultStore;
|
|
8127
8770
|
exports.createJudge = createJudge;
|
|
8128
8771
|
exports.createMCPClientForConfig = createMCPClientForConfig;
|
|
8129
8772
|
exports.createMCPFixture = createMCPFixture;
|
|
8773
|
+
exports.createStoredEvalArtifact = createStoredEvalArtifact;
|
|
8130
8774
|
exports.createTokenAuthHeaders = createTokenAuthHeaders;
|
|
8775
|
+
exports.defaultEnvironmentMetadata = defaultEnvironmentMetadata;
|
|
8131
8776
|
exports.discoverAuthorizationServer = discoverAuthorizationServer;
|
|
8132
8777
|
exports.discoverProtectedResource = discoverProtectedResource;
|
|
8133
8778
|
exports.expect = expect;
|
|
@@ -8138,6 +8783,7 @@ exports.getResponseSizeBytes = getResponseSizeBytes;
|
|
|
8138
8783
|
exports.hasValidTokens = hasValidTokens;
|
|
8139
8784
|
exports.injectTokens = injectTokens;
|
|
8140
8785
|
exports.isBuiltInRubric = isBuiltInRubric;
|
|
8786
|
+
exports.isEvalResultStore = isEvalResultStore;
|
|
8141
8787
|
exports.isHttpConfig = isHttpConfig;
|
|
8142
8788
|
exports.isProviderAvailable = isProviderAvailable;
|
|
8143
8789
|
exports.isStdioConfig = isStdioConfig;
|
|
@@ -8146,6 +8792,7 @@ exports.isTokenExpiringSoon = isTokenExpiringSoon;
|
|
|
8146
8792
|
exports.loadBaseline = loadBaseline;
|
|
8147
8793
|
exports.loadEvalDataset = loadEvalDataset;
|
|
8148
8794
|
exports.loadEvalDatasetFromObject = loadEvalDatasetFromObject;
|
|
8795
|
+
exports.loadStoredEvalRunnerResult = loadStoredEvalRunnerResult;
|
|
8149
8796
|
exports.loadTokens = loadTokens;
|
|
8150
8797
|
exports.loadTokensFromEnv = loadTokensFromEnv;
|
|
8151
8798
|
exports.mcpAuthTest = test2;
|
|
@@ -8156,12 +8803,16 @@ exports.performOAuthSetup = performOAuthSetup;
|
|
|
8156
8803
|
exports.performOAuthSetupIfNeeded = performOAuthSetupIfNeeded;
|
|
8157
8804
|
exports.refreshAccessToken = refreshAccessToken;
|
|
8158
8805
|
exports.registerJudge = registerJudge;
|
|
8806
|
+
exports.resolveEvalResultStore = resolveEvalResultStore;
|
|
8159
8807
|
exports.resolveRubric = resolveRubric;
|
|
8160
8808
|
exports.runConformanceChecks = runConformanceChecks;
|
|
8161
8809
|
exports.runEvalCase = runEvalCase;
|
|
8162
8810
|
exports.runEvalDataset = runEvalDataset;
|
|
8163
8811
|
exports.runServerComparison = runServerComparison;
|
|
8812
|
+
exports.runVariantExperiment = runVariantExperiment;
|
|
8164
8813
|
exports.saveBaseline = saveBaseline;
|
|
8814
|
+
exports.saveEvalRunComparison = saveEvalRunComparison;
|
|
8815
|
+
exports.saveServerComparison = saveServerComparison;
|
|
8165
8816
|
exports.simulateMCPHost = simulateMCPHost;
|
|
8166
8817
|
exports.test = test;
|
|
8167
8818
|
exports.validateAccessToken = validateAccessToken;
|