@dalmia/calibrate-mcp 0.0.60 → 0.0.62
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +1 -1
- package/bin/mcp-server.js +65 -23
- package/bin/mcp-server.js.map +14 -14
- package/esm/landing-page.js +1 -1
- package/esm/lib/config.d.ts +2 -2
- package/esm/lib/config.js +2 -2
- package/esm/mcp-server/mcp-server.js +1 -1
- package/esm/mcp-server/server.js +1 -1
- package/esm/models/agenttestrunlistitem.js +1 -1
- package/esm/models/agenttestrunlistitem.js.map +1 -1
- package/esm/models/getagenttestrunsagenttestsagentagentuuidrunsgetop.js +1 -1
- package/esm/models/getagenttestrunsagenttestsagentagentuuidrunsgetop.js.map +1 -1
- package/esm/models/modelresult.js +1 -1
- package/esm/models/modelresult.js.map +1 -1
- package/esm/models/modelrunsummary.d.ts +1 -0
- package/esm/models/modelrunsummary.d.ts.map +1 -1
- package/esm/models/modelrunsummary.js +2 -1
- package/esm/models/modelrunsummary.js.map +1 -1
- package/esm/models/testrunstatusresponse.js +1 -1
- package/esm/models/testrunstatusresponse.js.map +1 -1
- package/package.json +1 -1
- package/src/landing-page.ts +1 -1
- package/src/lib/config.ts +2 -2
- package/src/mcp-server/mcp-server.ts +1 -1
- package/src/mcp-server/server.ts +1 -1
- package/src/models/agenttestrunlistitem.ts +1 -1
- package/src/models/getagenttestrunsagenttestsagentagentuuidrunsgetop.ts +1 -1
- package/src/models/modelresult.ts +1 -1
- package/src/models/modelrunsummary.ts +5 -1
- package/src/models/testrunstatusresponse.ts +1 -1
package/README.md
CHANGED
|
@@ -61,4 +61,4 @@ Point Cursor at the local build by using `node` with the path to `bin/mcp-server
|
|
|
61
61
|
## Resources
|
|
62
62
|
|
|
63
63
|
- `npx @dalmia/calibrate-mcp start --help` — all flags and transports
|
|
64
|
-
- [Calibrate docs](https://calibrate.artpark.ai
|
|
64
|
+
- [Calibrate docs](https://docs.calibrate.artpark.ai) — API reference
|
package/bin/mcp-server.js
CHANGED
|
@@ -42191,6 +42191,22 @@ var require_utils4 = __commonJS(function(exports, module) {
|
|
|
42191
42191
|
}
|
|
42192
42192
|
return output;
|
|
42193
42193
|
}
|
|
42194
|
+
function recomposeZonedIPv6Host(component) {
|
|
42195
|
+
const zone = component.ipv6Zone;
|
|
42196
|
+
if (zone === undefined)
|
|
42197
|
+
return;
|
|
42198
|
+
const host = component.host;
|
|
42199
|
+
const separator = host.indexOf("%");
|
|
42200
|
+
if (separator === -1)
|
|
42201
|
+
return;
|
|
42202
|
+
const address = host.slice(0, separator);
|
|
42203
|
+
const escaped = address + "%25" + zone;
|
|
42204
|
+
const derived = normalizeIPv6("[" + escaped + "]");
|
|
42205
|
+
if (derived.isIPV6 === true && derived.host === host) {
|
|
42206
|
+
return "[" + escaped + "]";
|
|
42207
|
+
}
|
|
42208
|
+
return;
|
|
42209
|
+
}
|
|
42194
42210
|
function recomposeAuthority(component) {
|
|
42195
42211
|
const uriTokens = [];
|
|
42196
42212
|
if (component.userinfo !== undefined) {
|
|
@@ -42200,15 +42216,20 @@ var require_utils4 = __commonJS(function(exports, module) {
|
|
|
42200
42216
|
if (component.host !== undefined) {
|
|
42201
42217
|
let host = component.host;
|
|
42202
42218
|
if (!isIPv4(host)) {
|
|
42203
|
-
|
|
42204
|
-
if (
|
|
42205
|
-
host =
|
|
42206
|
-
ipV6res = normalizeIPv6(host);
|
|
42207
|
-
}
|
|
42208
|
-
if (ipV6res.isIPV6 === true || ipV6res.isIPVFuture === true) {
|
|
42209
|
-
host = `[${ipV6res.escapedHost}]`;
|
|
42219
|
+
const zonedHost = recomposeZonedIPv6Host(component);
|
|
42220
|
+
if (zonedHost !== undefined) {
|
|
42221
|
+
host = zonedHost;
|
|
42210
42222
|
} else {
|
|
42211
|
-
|
|
42223
|
+
let ipV6res = normalizeIPv6(host);
|
|
42224
|
+
if (ipV6res.isIPV6 !== true && ipV6res.isIPVFuture !== true) {
|
|
42225
|
+
host = normalizePercentEncoding(host, true);
|
|
42226
|
+
ipV6res = normalizeIPv6(host);
|
|
42227
|
+
}
|
|
42228
|
+
if (ipV6res.isIPV6 === true || ipV6res.isIPVFuture === true) {
|
|
42229
|
+
host = `[${ipV6res.escapedHost}]`;
|
|
42230
|
+
} else {
|
|
42231
|
+
host = reescapeHostDelimiters(host, false);
|
|
42232
|
+
}
|
|
42212
42233
|
}
|
|
42213
42234
|
}
|
|
42214
42235
|
uriTokens.push(host);
|
|
@@ -42421,6 +42442,7 @@ var require_schemes = __commonJS(function(exports, module) {
|
|
|
42421
42442
|
var HEX_PAIR = /^[\da-f]{2}$/iu;
|
|
42422
42443
|
var MAILTO_DOMAIN_LITERAL = /^\[[\x21-\x5A\x5E-\x7E]*\]$/u;
|
|
42423
42444
|
var MAILTO_DOMAIN_ERROR = "URI mailto has an invalid recipient domain.";
|
|
42445
|
+
var MAILTO_AUTHORITY_ERROR = "URI mailto must not have an authority component.";
|
|
42424
42446
|
var HAS_SURROGATE = /[\uD800-\uDFFF]/u;
|
|
42425
42447
|
function decodeHex(str) {
|
|
42426
42448
|
if (typeof str !== "string" || str.indexOf("%") === -1) {
|
|
@@ -42560,6 +42582,12 @@ var require_schemes = __commonJS(function(exports, module) {
|
|
|
42560
42582
|
mailtoComponent.headers = headers;
|
|
42561
42583
|
}
|
|
42562
42584
|
mailtoComponent.query = undefined;
|
|
42585
|
+
if (to.length > 0 && (mailtoComponent.userinfo !== undefined || mailtoComponent.host !== undefined || mailtoComponent.port !== undefined)) {
|
|
42586
|
+
mailtoComponent.userinfo = undefined;
|
|
42587
|
+
mailtoComponent.host = undefined;
|
|
42588
|
+
mailtoComponent.port = undefined;
|
|
42589
|
+
mailtoComponent.error = mailtoComponent.error || MAILTO_AUTHORITY_ERROR;
|
|
42590
|
+
}
|
|
42563
42591
|
for (let i = 0;i < to.length; i++) {
|
|
42564
42592
|
const rawAddr = to[i];
|
|
42565
42593
|
const atIdx = rawAddr.lastIndexOf("@");
|
|
@@ -42602,6 +42630,9 @@ var require_schemes = __commonJS(function(exports, module) {
|
|
|
42602
42630
|
headers.body = mailtoComponent.body;
|
|
42603
42631
|
mailtoComponent.headers = headers;
|
|
42604
42632
|
if (to.length) {
|
|
42633
|
+
mailtoComponent.userinfo = undefined;
|
|
42634
|
+
mailtoComponent.host = undefined;
|
|
42635
|
+
mailtoComponent.port = undefined;
|
|
42605
42636
|
for (let i = 0;i < to.length; i++) {
|
|
42606
42637
|
const addr = String(to[i]);
|
|
42607
42638
|
const atIdx = addr.lastIndexOf("@");
|
|
@@ -42715,6 +42746,12 @@ var require_fast_uri = __commonJS(function(exports, module) {
|
|
|
42715
42746
|
schemelessOptions.skipEscape = true;
|
|
42716
42747
|
return serialize(resolved, schemelessOptions);
|
|
42717
42748
|
}
|
|
42749
|
+
function copyHost(target, source) {
|
|
42750
|
+
target.host = source.host;
|
|
42751
|
+
if (source.ipv6Zone !== undefined) {
|
|
42752
|
+
target.ipv6Zone = source.ipv6Zone;
|
|
42753
|
+
}
|
|
42754
|
+
}
|
|
42718
42755
|
function resolveComponent(base, relative, options, skipNormalization) {
|
|
42719
42756
|
const target = {};
|
|
42720
42757
|
if (!skipNormalization) {
|
|
@@ -42725,14 +42762,14 @@ var require_fast_uri = __commonJS(function(exports, module) {
|
|
|
42725
42762
|
if (!options.tolerant && relative.scheme) {
|
|
42726
42763
|
target.scheme = relative.scheme;
|
|
42727
42764
|
target.userinfo = relative.userinfo;
|
|
42728
|
-
target
|
|
42765
|
+
copyHost(target, relative);
|
|
42729
42766
|
target.port = relative.port;
|
|
42730
42767
|
target.path = removeDotSegments(relative.path || "");
|
|
42731
42768
|
target.query = relative.query;
|
|
42732
42769
|
} else {
|
|
42733
42770
|
if (relative.userinfo !== undefined || relative.host !== undefined || relative.port !== undefined) {
|
|
42734
42771
|
target.userinfo = relative.userinfo;
|
|
42735
|
-
target
|
|
42772
|
+
copyHost(target, relative);
|
|
42736
42773
|
target.port = relative.port;
|
|
42737
42774
|
target.path = removeDotSegments(relative.path || "");
|
|
42738
42775
|
target.query = relative.query;
|
|
@@ -42760,7 +42797,7 @@ var require_fast_uri = __commonJS(function(exports, module) {
|
|
|
42760
42797
|
target.query = relative.query;
|
|
42761
42798
|
}
|
|
42762
42799
|
target.userinfo = base.userinfo;
|
|
42763
|
-
target
|
|
42800
|
+
copyHost(target, base);
|
|
42764
42801
|
target.port = base.port;
|
|
42765
42802
|
}
|
|
42766
42803
|
target.scheme = base.scheme;
|
|
@@ -42776,6 +42813,7 @@ var require_fast_uri = __commonJS(function(exports, module) {
|
|
|
42776
42813
|
function serialize(cmpts, opts) {
|
|
42777
42814
|
const component = {
|
|
42778
42815
|
host: cmpts.host,
|
|
42816
|
+
ipv6Zone: cmpts.ipv6Zone,
|
|
42779
42817
|
scheme: cmpts.scheme,
|
|
42780
42818
|
userinfo: cmpts.userinfo,
|
|
42781
42819
|
port: cmpts.port,
|
|
@@ -42972,6 +43010,9 @@ var require_fast_uri = __commonJS(function(exports, module) {
|
|
|
42972
43010
|
isIP = ipv6result.isIPV6 || ipv6result.isIPVFuture === true;
|
|
42973
43011
|
malformedIPLiteral = hasIPLiteralBracket && (!bracketedIPLiteral || ipv6result.error === true);
|
|
42974
43012
|
parsed.host = isIP ? ipv6result.host : ipv6result.host.toLowerCase();
|
|
43013
|
+
if (isIP && ipv6result.isIPV6 === true && parsed.host.indexOf("%") !== -1) {
|
|
43014
|
+
parsed.ipv6Zone = parsed.host.slice(parsed.host.indexOf("%") + 1);
|
|
43015
|
+
}
|
|
42975
43016
|
if (malformedIPLiteral) {
|
|
42976
43017
|
parsed.error = parsed.error || "URI host is malformed.";
|
|
42977
43018
|
malformedAuthorityOrPort = true;
|
|
@@ -52761,9 +52802,9 @@ var init_config = __esm(() => {
|
|
|
52761
52802
|
SDK_METADATA = {
|
|
52762
52803
|
language: "typescript",
|
|
52763
52804
|
openapiDocVersion: "0.1.0",
|
|
52764
|
-
sdkVersion: "0.0.
|
|
52805
|
+
sdkVersion: "0.0.62",
|
|
52765
52806
|
genVersion: "2.915.1",
|
|
52766
|
-
userAgent: "speakeasy-sdk/mcp-typescript 0.0.
|
|
52807
|
+
userAgent: "speakeasy-sdk/mcp-typescript 0.0.62 2.915.1 0.1.0 @dalmia/calibrate-mcp"
|
|
52767
52808
|
};
|
|
52768
52809
|
});
|
|
52769
52810
|
|
|
@@ -55982,7 +56023,7 @@ var init_modelresult = __esm(() => {
|
|
|
55982
56023
|
ModelResult$zodSchema = object({
|
|
55983
56024
|
cost: record(string2(), any()).nullable().optional().describe("Aggregated cost as `{mean, min, max, count}` (USD)"),
|
|
55984
56025
|
evaluator_summary: array(record(string2(), any())).nullable().optional().describe("Aggregate summary for each evaluator for this model"),
|
|
55985
|
-
failed: int().nullable().optional().describe("Number of test cases that
|
|
56026
|
+
failed: int().nullable().optional().describe("Number of test cases that did not pass, which includes the ones that produced no answer"),
|
|
55986
56027
|
latency_ms: record(string2(), any()).nullable().optional().describe("Aggregated latency in milliseconds, as `{p50, p95, p99, count}`"),
|
|
55987
56028
|
message: string2().describe("Status or result message for this model"),
|
|
55988
56029
|
model: string2().describe("Model name these results are for"),
|
|
@@ -56212,7 +56253,7 @@ var init_testrunstatusresponse = __esm(() => {
|
|
|
56212
56253
|
error: string2().nullable().optional().describe("Why the run could not be carried out, when it failed before producing any result"),
|
|
56213
56254
|
evaluator_summary: array(record(string2(), any())).nullable().optional().describe("Totals for each evaluator over the whole run, matching the shape a benchmark reports for each model. Only evaluators that returned a verdict appear"),
|
|
56214
56255
|
evaluators: array(TestRunEvaluator$zodSchema).nullable().optional().describe("The evaluators used in this run. Each verdict in `judge_results` links to one of these by `evaluator_uuid`"),
|
|
56215
|
-
failed: int().nullable().optional().describe("Number of test cases that
|
|
56256
|
+
failed: int().nullable().optional().describe("Number of test cases that did not pass, which includes the ones that produced no answer"),
|
|
56216
56257
|
is_public: boolean2().default(false).describe("Whether the run is shared publicly"),
|
|
56217
56258
|
latency_ms: record(string2(), any()).nullable().optional().describe("Aggregated response latency in milliseconds, as `{p50, p95, p99, count}`"),
|
|
56218
56259
|
name: string2().describe("Name of the run. A run nobody has renamed shows its number instead, such as `Run 1` for a test run or `Benchmark 1` for a benchmark"),
|
|
@@ -56874,12 +56915,13 @@ var ModelRunSummary$zodSchema;
|
|
|
56874
56915
|
var init_modelrunsummary = __esm(() => {
|
|
56875
56916
|
init_zod();
|
|
56876
56917
|
ModelRunSummary$zodSchema = object({
|
|
56877
|
-
failed: int().nullable().optional().describe("Number of test cases that
|
|
56918
|
+
failed: int().nullable().optional().describe("Number of test cases that did not pass for this model, which includes the ones that produced no answer"),
|
|
56878
56919
|
message: string2().default("").describe("Status or result message for this model"),
|
|
56879
56920
|
model: string2().describe("Model name these results are for"),
|
|
56880
56921
|
passed: int().nullable().optional().describe("Number of test cases that passed for this model"),
|
|
56881
56922
|
success: boolean2().nullable().optional().describe("Whether this model's run succeeded"),
|
|
56882
|
-
total_tests: int().nullable().optional().describe("Total test cases for this model")
|
|
56923
|
+
total_tests: int().nullable().optional().describe("Total test cases for this model"),
|
|
56924
|
+
unanswered_tests: int().nullable().optional().describe("Number of this model's test cases that produced no answer, already counted in `failed`")
|
|
56883
56925
|
}).describe("Flat summary for one model in a benchmark run-LIST item. The full results\nfor each case of a model live on the benchmark detail endpoint\n(`GET /agent-tests/benchmark/{task_id}`), not here.");
|
|
56884
56926
|
});
|
|
56885
56927
|
|
|
@@ -56926,7 +56968,7 @@ var init_agenttestrunlistitem = __esm(() => {
|
|
|
56926
56968
|
created_at: string2().describe("When the run was created (ISO 8601 UTC)"),
|
|
56927
56969
|
error: boolean2().default(false).describe("True if the run failed"),
|
|
56928
56970
|
evaluators: array(RunListEvaluator$zodSchema).optional().describe("The evaluators that judged this run, deduplicated and in display order. A `Tool call` entry is appended when any test in the run was a tool-call test. That entry has no `uuid`, because it is not an evaluator in the library. Empty when the run had no evaluators"),
|
|
56929
|
-
failed: int().nullable().optional().describe("Number of test cases that
|
|
56971
|
+
failed: int().nullable().optional().describe("Number of test cases that did not pass, which includes the ones that produced no answer"),
|
|
56930
56972
|
is_public: boolean2().default(false).describe("Whether the run is shared publicly"),
|
|
56931
56973
|
latency_ms: record(string2(), any()).nullable().optional().describe("Aggregated latency in milliseconds, as `{p50, p95, p99, count}`"),
|
|
56932
56974
|
model_results: array(ModelRunSummary$zodSchema).nullable().optional().describe("Flat summary for each model in a benchmark run (fetch the benchmark detail for full results)"),
|
|
@@ -56972,7 +57014,7 @@ var init_getagenttestrunsagenttestsagentagentuuidrunsgetop = __esm(() => {
|
|
|
56972
57014
|
GetAgentTestRunsAgentTestsAgentAgentUuidRunsGetRequest$zodSchema = object({
|
|
56973
57015
|
agent_uuid: string2().describe("Agent whose test runs to list"),
|
|
56974
57016
|
around: string2().describe("ID of a run to jump to, returning the page that contains it instead of the page at `offset`").nullable().optional(),
|
|
56975
|
-
has_failures: boolean2().describe("Filter by whether
|
|
57017
|
+
has_failures: boolean2().describe("Filter by whether a test in the run did not pass. `true` returns only runs with a failing test, `false` only runs that got through every test and passed them all. A run that broke, was stopped, or gave up part way with no failing test is in neither: filter by `status` for those. Omit for both").nullable().optional(),
|
|
56976
57018
|
limit: int().describe("Maximum number of items to return. Omit for no limit (all items)").nullable().optional(),
|
|
56977
57019
|
offset: int().default(0).describe("Number of items to skip before returning results"),
|
|
56978
57020
|
status: TaskStatus$zodSchema.nullable().optional().describe("Filter by run status. Omit for all statuses"),
|
|
@@ -61564,7 +61606,7 @@ hits its trace limit.
|
|
|
61564
61606
|
function createMCPServer(deps) {
|
|
61565
61607
|
const server = new McpServer({
|
|
61566
61608
|
name: "CalibrateMcp",
|
|
61567
|
-
version: "0.0.
|
|
61609
|
+
version: "0.0.62"
|
|
61568
61610
|
});
|
|
61569
61611
|
const getClient = deps.getSDK || (() => new CalibrateMcpCore({
|
|
61570
61612
|
security: deps.security,
|
|
@@ -62853,7 +62895,7 @@ http_headers = { "api-key-auth" = "YOUR_API_KEY_AUTH" }`;
|
|
|
62853
62895
|
<h1>Instructions</h1>
|
|
62854
62896
|
<p>One-click installation for Claude Desktop users</p>
|
|
62855
62897
|
<div class="instruction-item">
|
|
62856
|
-
<a href="https://github.com/dalmia/calibrate-mcp/releases/download/v0.0.
|
|
62898
|
+
<a href="https://github.com/dalmia/calibrate-mcp/releases/download/v0.0.62/mcp-server.mcpb" download="mcp-server.mcpb" class="action-button header-action" style="display: inline-flex; margin-bottom: 16px;">
|
|
62857
62899
|
\uD83D\uDCE5 Download MCP Bundle
|
|
62858
62900
|
</a>
|
|
62859
62901
|
</div>
|
|
@@ -65737,7 +65779,7 @@ var routes = buildRouteMap({
|
|
|
65737
65779
|
var app = buildApplication(routes, {
|
|
65738
65780
|
name: "mcp",
|
|
65739
65781
|
versionInfo: {
|
|
65740
|
-
currentVersion: "0.0.
|
|
65782
|
+
currentVersion: "0.0.62"
|
|
65741
65783
|
}
|
|
65742
65784
|
});
|
|
65743
65785
|
run(app, process3.argv.slice(2), buildContext(process3));
|
|
@@ -65745,5 +65787,5 @@ export {
|
|
|
65745
65787
|
app
|
|
65746
65788
|
};
|
|
65747
65789
|
|
|
65748
|
-
//# debugId=
|
|
65790
|
+
//# debugId=C281A178ECB0D2FA64756E2164756E21
|
|
65749
65791
|
//# sourceMappingURL=mcp-server.js.map
|