@dalmia/calibrate-mcp 0.0.61 → 0.0.63
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +1 -1
- package/bin/mcp-server.js +15 -14
- package/bin/mcp-server.js.map +13 -13
- package/esm/landing-page.js +1 -1
- package/esm/lib/config.d.ts +2 -2
- package/esm/lib/config.js +2 -2
- package/esm/mcp-server/mcp-server.js +1 -1
- package/esm/mcp-server/server.js +1 -1
- package/esm/models/agentcreate.js +1 -1
- package/esm/models/agentcreate.js.map +1 -1
- package/esm/models/agenttestrunlistitem.js +1 -1
- package/esm/models/agenttestrunlistitem.js.map +1 -1
- package/esm/models/agentupdate.js +1 -1
- package/esm/models/agentupdate.js.map +1 -1
- package/esm/models/getagenttestrunsagenttestsagentagentuuidrunsgetop.js +1 -1
- package/esm/models/getagenttestrunsagenttestsagentagentuuidrunsgetop.js.map +1 -1
- package/esm/models/modelresult.js +1 -1
- package/esm/models/modelresult.js.map +1 -1
- package/esm/models/modelrunsummary.d.ts +1 -0
- package/esm/models/modelrunsummary.d.ts.map +1 -1
- package/esm/models/modelrunsummary.js +2 -1
- package/esm/models/modelrunsummary.js.map +1 -1
- package/esm/models/testrunstatusresponse.js +1 -1
- package/esm/models/testrunstatusresponse.js.map +1 -1
- package/package.json +1 -1
- package/src/landing-page.ts +1 -1
- package/src/lib/config.ts +2 -2
- package/src/mcp-server/mcp-server.ts +1 -1
- package/src/mcp-server/server.ts +1 -1
- package/src/models/agentcreate.ts +1 -1
- package/src/models/agenttestrunlistitem.ts +1 -1
- package/src/models/agentupdate.ts +1 -1
- package/src/models/getagenttestrunsagenttestsagentagentuuidrunsgetop.ts +1 -1
- package/src/models/modelresult.ts +1 -1
- package/src/models/modelrunsummary.ts +5 -1
- package/src/models/testrunstatusresponse.ts +1 -1
package/README.md
CHANGED
|
@@ -61,4 +61,4 @@ Point Cursor at the local build by using `node` with the path to `bin/mcp-server
|
|
|
61
61
|
## Resources
|
|
62
62
|
|
|
63
63
|
- `npx @dalmia/calibrate-mcp start --help` — all flags and transports
|
|
64
|
-
- [Calibrate docs](https://calibrate.artpark.ai
|
|
64
|
+
- [Calibrate docs](https://docs.calibrate.artpark.ai) — API reference
|
package/bin/mcp-server.js
CHANGED
|
@@ -52802,9 +52802,9 @@ var init_config = __esm(() => {
|
|
|
52802
52802
|
SDK_METADATA = {
|
|
52803
52803
|
language: "typescript",
|
|
52804
52804
|
openapiDocVersion: "0.1.0",
|
|
52805
|
-
sdkVersion: "0.0.
|
|
52805
|
+
sdkVersion: "0.0.63",
|
|
52806
52806
|
genVersion: "2.915.1",
|
|
52807
|
-
userAgent: "speakeasy-sdk/mcp-typescript 0.0.
|
|
52807
|
+
userAgent: "speakeasy-sdk/mcp-typescript 0.0.63 2.915.1 0.1.0 @dalmia/calibrate-mcp"
|
|
52808
52808
|
};
|
|
52809
52809
|
});
|
|
52810
52810
|
|
|
@@ -54263,7 +54263,7 @@ var init_agentcreate = __esm(() => {
|
|
|
54263
54263
|
"general"
|
|
54264
54264
|
]).describe('What the agent expects in the request body:\n\n- `conversation`: a normal back-and-forth agent, answers within an ongoing conversation. Receives `{"messages": [...]}`\n- `general`: a one-shot agent, takes a single plain input and produces a single plain output, no conversation. Receives `{"input": "..."}`');
|
|
54265
54265
|
AgentCreate$zodSchema = object({
|
|
54266
|
-
config: record(string2(), any()).nullable().optional().describe('Agent behavioral config. The keys depend on `type`.\n\n**`type=agent`**, built inside Calibrate:\n- `system_prompt`: the agent\'s instructions\n- `llm.model`: `provider/model`, e.g. `openai/gpt-4.1` or `google/gemini-2.5-flash`\n- `stt.provider`: `deepgram`, `openai`, `cartesia`, `elevenlabs`, `google`, `sarvam`, or `smallest`\n- `tts.provider`: `cartesia`, `openai`, `google`, `elevenlabs`, `sarvam`, or `smallest`\n- `settings.agent_speaks_first`, `settings.max_assistant_turns`\n- `system_tools.end_call`: let the agent end the call\n- `data_extraction_fields`: `[{name, type, description, required}]`\n\n```json\n{\n "system_prompt": "You are a helpful support agent.",\n "llm": {"model": "openai/gpt-4.1"},\n "stt": {"provider": "deepgram"},\n "tts": {"provider": "elevenlabs"},\n "settings": {"agent_speaks_first": true, "max_assistant_turns": 50}\n}\n```\n\n**`type=connection`**, your own HTTP endpoint:\n- `agent_url`: public HTTP(S) endpoint your agent is called at\n- `agent_headers`: headers sent on each request, e.g. auth\n- `benchmark_provider`: `openrouter` by default. Other values: `openai`, `google`, `anthropic`, `meta-llama`, `mistralai`, `deepseek`, `x-ai`, `cohere`, `qwen`, or `ai21`\n\n```json\n{\n "agent_url": "https://api.example.com/agent",\n "agent_headers": {"Authorization": "Bearer <token>"},\n "benchmark_provider": "openrouter"\n}\n```\n\nEvery request Calibrate makes to your endpoint carries the header\n`X-Calibrate-Eval: 1`. Read it to tell a test run from a real user, for example\nto tag the trace you send back or to skip sending one.\n\nFor `type=agent`, omitted keys inherit managed defaults. Omit `config` entirely to use all defaults. For `type=connection`, `config` is stored as-is and must contain `agent_url`'),
|
|
54266
|
+
config: record(string2(), any()).nullable().optional().describe('Agent behavioral config. The keys depend on `type`.\n\n**`type=agent`**, built inside Calibrate:\n- `system_prompt`: the agent\'s instructions\n- `llm.model`: `provider/model`, e.g. `openai/gpt-4.1` or `google/gemini-2.5-flash`\n- `stt.provider`: `deepgram`, `openai`, `cartesia`, `elevenlabs`, `google`, `sarvam`, or `smallest`\n- `tts.provider`: `cartesia`, `openai`, `google`, `elevenlabs`, `sarvam`, or `smallest`\n- `settings.agent_speaks_first`, `settings.max_assistant_turns`\n- `system_tools.end_call`: let the agent end the call\n- `data_extraction_fields`: `[{name, type, description, required}]`\n\n```json\n{\n "system_prompt": "You are a helpful support agent.",\n "llm": {"model": "openai/gpt-4.1"},\n "stt": {"provider": "deepgram"},\n "tts": {"provider": "elevenlabs"},\n "settings": {"agent_speaks_first": true, "max_assistant_turns": 50}\n}\n```\n\n**`type=connection`**, your own HTTP endpoint:\n- `agent_url`: public HTTP(S) endpoint your agent is called at\n- `agent_headers`: headers sent on each request, e.g. auth\n- `benchmark_provider`: `openrouter` by default. Other values: `openai`, `google`, `anthropic`, `meta-llama`, `mistralai`, `deepseek`, `x-ai`, `cohere`, `qwen`, or `ai21`\n\n```json\n{\n "agent_url": "https://api.example.com/agent",\n "agent_headers": {"Authorization": "Bearer <token>"},\n "benchmark_provider": "openrouter"\n}\n```\n\n**Either type**:\n- `traces.scoring.enabled`: whether the traces you send for this agent are scored by its linked evaluators. On unless you set it to `false`. Turning it back on needs at least one linked evaluator that can score traces\n\n```json\n{"traces": {"scoring": {"enabled": false}}}\n```\n\nEvery request Calibrate makes to your endpoint carries the header\n`X-Calibrate-Eval: 1`. Read it to tell a test run from a real user, for example\nto tag the trace you send back or to skip sending one.\n\nFor `type=agent`, omitted keys inherit managed defaults. Omit `config` entirely to use all defaults. For `type=connection`, `config` is stored as-is and must contain `agent_url`'),
|
|
54267
54267
|
interaction_type: AgentCreateInteractionType$zodSchema.default("conversation").describe('What the agent expects in the request body:\n\n- `conversation`: a normal back-and-forth agent, answers within an ongoing conversation. Receives `{"messages": [...]}`\n- `general`: a one-shot agent, takes a single plain input and produces a single plain output, no conversation. Receives `{"input": "..."}`'),
|
|
54268
54268
|
name: string2().describe("Agent name, unique within the workspace"),
|
|
54269
54269
|
type: AgentCreateType$zodSchema.default("agent").describe("- `agent`: built inside Calibrate\n- `connection`: your existing agent connected to Calibrate")
|
|
@@ -55334,7 +55334,7 @@ var AgentUpdate$zodSchema;
|
|
|
55334
55334
|
var init_agentupdate = __esm(() => {
|
|
55335
55335
|
init_zod();
|
|
55336
55336
|
AgentUpdate$zodSchema = object({
|
|
55337
|
-
config: record(string2(), any()).nullable().optional().describe('Agent behavioral config. The keys depend on `type`.\n\n**`type=agent`**, built inside Calibrate:\n- `system_prompt`: the agent\'s instructions\n- `llm.model`: `provider/model`, e.g. `openai/gpt-4.1` or `google/gemini-2.5-flash`\n- `stt.provider`: `deepgram`, `openai`, `cartesia`, `elevenlabs`, `google`, `sarvam`, or `smallest`\n- `tts.provider`: `cartesia`, `openai`, `google`, `elevenlabs`, `sarvam`, or `smallest`\n- `settings.agent_speaks_first`, `settings.max_assistant_turns`\n- `system_tools.end_call`: let the agent end the call\n- `data_extraction_fields`: `[{name, type, description, required}]`\n\n```json\n{\n "system_prompt": "You are a helpful support agent.",\n "llm": {"model": "openai/gpt-4.1"},\n "stt": {"provider": "deepgram"},\n "tts": {"provider": "elevenlabs"},\n "settings": {"agent_speaks_first": true, "max_assistant_turns": 50}\n}\n```\n\n**`type=connection`**, your own HTTP endpoint:\n- `agent_url`: public HTTP(S) endpoint your agent is called at\n- `agent_headers`: headers sent on each request, e.g. auth\n- `benchmark_provider`: `openrouter` by default. Other values: `openai`, `google`, `anthropic`, `meta-llama`, `mistralai`, `deepseek`, `x-ai`, `cohere`, `qwen`, or `ai21`\n\n```json\n{\n "agent_url": "https://api.example.com/agent",\n "agent_headers": {"Authorization": "Bearer <token>"},\n "benchmark_provider": "openrouter"\n}\n```\n\nEvery request Calibrate makes to your endpoint carries the header\n`X-Calibrate-Eval: 1`. Read it to tell a test run from a real user, for example\nto tag the trace you send back or to skip sending one.\n\nReplaces the stored config. Omit to leave unchanged\n\nFor `type=connection`, changing `agent_url` or `agent_headers` resets the connection and benchmark verification flags'),
|
|
55337
|
+
config: record(string2(), any()).nullable().optional().describe('Agent behavioral config. The keys depend on `type`.\n\n**`type=agent`**, built inside Calibrate:\n- `system_prompt`: the agent\'s instructions\n- `llm.model`: `provider/model`, e.g. `openai/gpt-4.1` or `google/gemini-2.5-flash`\n- `stt.provider`: `deepgram`, `openai`, `cartesia`, `elevenlabs`, `google`, `sarvam`, or `smallest`\n- `tts.provider`: `cartesia`, `openai`, `google`, `elevenlabs`, `sarvam`, or `smallest`\n- `settings.agent_speaks_first`, `settings.max_assistant_turns`\n- `system_tools.end_call`: let the agent end the call\n- `data_extraction_fields`: `[{name, type, description, required}]`\n\n```json\n{\n "system_prompt": "You are a helpful support agent.",\n "llm": {"model": "openai/gpt-4.1"},\n "stt": {"provider": "deepgram"},\n "tts": {"provider": "elevenlabs"},\n "settings": {"agent_speaks_first": true, "max_assistant_turns": 50}\n}\n```\n\n**`type=connection`**, your own HTTP endpoint:\n- `agent_url`: public HTTP(S) endpoint your agent is called at\n- `agent_headers`: headers sent on each request, e.g. auth\n- `benchmark_provider`: `openrouter` by default. Other values: `openai`, `google`, `anthropic`, `meta-llama`, `mistralai`, `deepseek`, `x-ai`, `cohere`, `qwen`, or `ai21`\n\n```json\n{\n "agent_url": "https://api.example.com/agent",\n "agent_headers": {"Authorization": "Bearer <token>"},\n "benchmark_provider": "openrouter"\n}\n```\n\n**Either type**:\n- `traces.scoring.enabled`: whether the traces you send for this agent are scored by its linked evaluators. On unless you set it to `false`. Turning it back on needs at least one linked evaluator that can score traces\n\n```json\n{"traces": {"scoring": {"enabled": false}}}\n```\n\nEvery request Calibrate makes to your endpoint carries the header\n`X-Calibrate-Eval: 1`. Read it to tell a test run from a real user, for example\nto tag the trace you send back or to skip sending one.\n\nReplaces the stored config. Omit to leave unchanged\n\nFor `type=connection`, changing `agent_url` or `agent_headers` resets the connection and benchmark verification flags'),
|
|
55338
55338
|
name: string2().nullable().optional().describe("New agent name. Omit to leave the name unchanged")
|
|
55339
55339
|
});
|
|
55340
55340
|
});
|
|
@@ -56023,7 +56023,7 @@ var init_modelresult = __esm(() => {
|
|
|
56023
56023
|
ModelResult$zodSchema = object({
|
|
56024
56024
|
cost: record(string2(), any()).nullable().optional().describe("Aggregated cost as `{mean, min, max, count}` (USD)"),
|
|
56025
56025
|
evaluator_summary: array(record(string2(), any())).nullable().optional().describe("Aggregate summary for each evaluator for this model"),
|
|
56026
|
-
failed: int().nullable().optional().describe("Number of test cases that
|
|
56026
|
+
failed: int().nullable().optional().describe("Number of test cases that did not pass, which includes the ones that produced no answer"),
|
|
56027
56027
|
latency_ms: record(string2(), any()).nullable().optional().describe("Aggregated latency in milliseconds, as `{p50, p95, p99, count}`"),
|
|
56028
56028
|
message: string2().describe("Status or result message for this model"),
|
|
56029
56029
|
model: string2().describe("Model name these results are for"),
|
|
@@ -56253,7 +56253,7 @@ var init_testrunstatusresponse = __esm(() => {
|
|
|
56253
56253
|
error: string2().nullable().optional().describe("Why the run could not be carried out, when it failed before producing any result"),
|
|
56254
56254
|
evaluator_summary: array(record(string2(), any())).nullable().optional().describe("Totals for each evaluator over the whole run, matching the shape a benchmark reports for each model. Only evaluators that returned a verdict appear"),
|
|
56255
56255
|
evaluators: array(TestRunEvaluator$zodSchema).nullable().optional().describe("The evaluators used in this run. Each verdict in `judge_results` links to one of these by `evaluator_uuid`"),
|
|
56256
|
-
failed: int().nullable().optional().describe("Number of test cases that
|
|
56256
|
+
failed: int().nullable().optional().describe("Number of test cases that did not pass, which includes the ones that produced no answer"),
|
|
56257
56257
|
is_public: boolean2().default(false).describe("Whether the run is shared publicly"),
|
|
56258
56258
|
latency_ms: record(string2(), any()).nullable().optional().describe("Aggregated response latency in milliseconds, as `{p50, p95, p99, count}`"),
|
|
56259
56259
|
name: string2().describe("Name of the run. A run nobody has renamed shows its number instead, such as `Run 1` for a test run or `Benchmark 1` for a benchmark"),
|
|
@@ -56915,12 +56915,13 @@ var ModelRunSummary$zodSchema;
|
|
|
56915
56915
|
var init_modelrunsummary = __esm(() => {
|
|
56916
56916
|
init_zod();
|
|
56917
56917
|
ModelRunSummary$zodSchema = object({
|
|
56918
|
-
failed: int().nullable().optional().describe("Number of test cases that
|
|
56918
|
+
failed: int().nullable().optional().describe("Number of test cases that did not pass for this model, which includes the ones that produced no answer"),
|
|
56919
56919
|
message: string2().default("").describe("Status or result message for this model"),
|
|
56920
56920
|
model: string2().describe("Model name these results are for"),
|
|
56921
56921
|
passed: int().nullable().optional().describe("Number of test cases that passed for this model"),
|
|
56922
56922
|
success: boolean2().nullable().optional().describe("Whether this model's run succeeded"),
|
|
56923
|
-
total_tests: int().nullable().optional().describe("Total test cases for this model")
|
|
56923
|
+
total_tests: int().nullable().optional().describe("Total test cases for this model"),
|
|
56924
|
+
unanswered_tests: int().nullable().optional().describe("Number of this model's test cases that produced no answer, already counted in `failed`")
|
|
56924
56925
|
}).describe("Flat summary for one model in a benchmark run-LIST item. The full results\nfor each case of a model live on the benchmark detail endpoint\n(`GET /agent-tests/benchmark/{task_id}`), not here.");
|
|
56925
56926
|
});
|
|
56926
56927
|
|
|
@@ -56967,7 +56968,7 @@ var init_agenttestrunlistitem = __esm(() => {
|
|
|
56967
56968
|
created_at: string2().describe("When the run was created (ISO 8601 UTC)"),
|
|
56968
56969
|
error: boolean2().default(false).describe("True if the run failed"),
|
|
56969
56970
|
evaluators: array(RunListEvaluator$zodSchema).optional().describe("The evaluators that judged this run, deduplicated and in display order. A `Tool call` entry is appended when any test in the run was a tool-call test. That entry has no `uuid`, because it is not an evaluator in the library. Empty when the run had no evaluators"),
|
|
56970
|
-
failed: int().nullable().optional().describe("Number of test cases that
|
|
56971
|
+
failed: int().nullable().optional().describe("Number of test cases that did not pass, which includes the ones that produced no answer"),
|
|
56971
56972
|
is_public: boolean2().default(false).describe("Whether the run is shared publicly"),
|
|
56972
56973
|
latency_ms: record(string2(), any()).nullable().optional().describe("Aggregated latency in milliseconds, as `{p50, p95, p99, count}`"),
|
|
56973
56974
|
model_results: array(ModelRunSummary$zodSchema).nullable().optional().describe("Flat summary for each model in a benchmark run (fetch the benchmark detail for full results)"),
|
|
@@ -57013,7 +57014,7 @@ var init_getagenttestrunsagenttestsagentagentuuidrunsgetop = __esm(() => {
|
|
|
57013
57014
|
GetAgentTestRunsAgentTestsAgentAgentUuidRunsGetRequest$zodSchema = object({
|
|
57014
57015
|
agent_uuid: string2().describe("Agent whose test runs to list"),
|
|
57015
57016
|
around: string2().describe("ID of a run to jump to, returning the page that contains it instead of the page at `offset`").nullable().optional(),
|
|
57016
|
-
has_failures: boolean2().describe("Filter by whether
|
|
57017
|
+
has_failures: boolean2().describe("Filter by whether a test in the run did not pass. `true` returns only runs with a failing test, `false` only runs that got through every test and passed them all. A run that broke, was stopped, or gave up part way with no failing test is in neither: filter by `status` for those. Omit for both").nullable().optional(),
|
|
57017
57018
|
limit: int().describe("Maximum number of items to return. Omit for no limit (all items)").nullable().optional(),
|
|
57018
57019
|
offset: int().default(0).describe("Number of items to skip before returning results"),
|
|
57019
57020
|
status: TaskStatus$zodSchema.nullable().optional().describe("Filter by run status. Omit for all statuses"),
|
|
@@ -61605,7 +61606,7 @@ hits its trace limit.
|
|
|
61605
61606
|
function createMCPServer(deps) {
|
|
61606
61607
|
const server = new McpServer({
|
|
61607
61608
|
name: "CalibrateMcp",
|
|
61608
|
-
version: "0.0.
|
|
61609
|
+
version: "0.0.63"
|
|
61609
61610
|
});
|
|
61610
61611
|
const getClient = deps.getSDK || (() => new CalibrateMcpCore({
|
|
61611
61612
|
security: deps.security,
|
|
@@ -62894,7 +62895,7 @@ http_headers = { "api-key-auth" = "YOUR_API_KEY_AUTH" }`;
|
|
|
62894
62895
|
<h1>Instructions</h1>
|
|
62895
62896
|
<p>One-click installation for Claude Desktop users</p>
|
|
62896
62897
|
<div class="instruction-item">
|
|
62897
|
-
<a href="https://github.com/dalmia/calibrate-mcp/releases/download/v0.0.
|
|
62898
|
+
<a href="https://github.com/dalmia/calibrate-mcp/releases/download/v0.0.63/mcp-server.mcpb" download="mcp-server.mcpb" class="action-button header-action" style="display: inline-flex; margin-bottom: 16px;">
|
|
62898
62899
|
\uD83D\uDCE5 Download MCP Bundle
|
|
62899
62900
|
</a>
|
|
62900
62901
|
</div>
|
|
@@ -65778,7 +65779,7 @@ var routes = buildRouteMap({
|
|
|
65778
65779
|
var app = buildApplication(routes, {
|
|
65779
65780
|
name: "mcp",
|
|
65780
65781
|
versionInfo: {
|
|
65781
|
-
currentVersion: "0.0.
|
|
65782
|
+
currentVersion: "0.0.63"
|
|
65782
65783
|
}
|
|
65783
65784
|
});
|
|
65784
65785
|
run(app, process3.argv.slice(2), buildContext(process3));
|
|
@@ -65786,5 +65787,5 @@ export {
|
|
|
65786
65787
|
app
|
|
65787
65788
|
};
|
|
65788
65789
|
|
|
65789
|
-
//# debugId=
|
|
65790
|
+
//# debugId=DF7D04973BECCF3264756E2164756E21
|
|
65790
65791
|
//# sourceMappingURL=mcp-server.js.map
|