@dalmia/calibrate-mcp 0.0.61 → 0.0.63

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (36) hide show
  1. package/README.md +1 -1
  2. package/bin/mcp-server.js +15 -14
  3. package/bin/mcp-server.js.map +13 -13
  4. package/esm/landing-page.js +1 -1
  5. package/esm/lib/config.d.ts +2 -2
  6. package/esm/lib/config.js +2 -2
  7. package/esm/mcp-server/mcp-server.js +1 -1
  8. package/esm/mcp-server/server.js +1 -1
  9. package/esm/models/agentcreate.js +1 -1
  10. package/esm/models/agentcreate.js.map +1 -1
  11. package/esm/models/agenttestrunlistitem.js +1 -1
  12. package/esm/models/agenttestrunlistitem.js.map +1 -1
  13. package/esm/models/agentupdate.js +1 -1
  14. package/esm/models/agentupdate.js.map +1 -1
  15. package/esm/models/getagenttestrunsagenttestsagentagentuuidrunsgetop.js +1 -1
  16. package/esm/models/getagenttestrunsagenttestsagentagentuuidrunsgetop.js.map +1 -1
  17. package/esm/models/modelresult.js +1 -1
  18. package/esm/models/modelresult.js.map +1 -1
  19. package/esm/models/modelrunsummary.d.ts +1 -0
  20. package/esm/models/modelrunsummary.d.ts.map +1 -1
  21. package/esm/models/modelrunsummary.js +2 -1
  22. package/esm/models/modelrunsummary.js.map +1 -1
  23. package/esm/models/testrunstatusresponse.js +1 -1
  24. package/esm/models/testrunstatusresponse.js.map +1 -1
  25. package/package.json +1 -1
  26. package/src/landing-page.ts +1 -1
  27. package/src/lib/config.ts +2 -2
  28. package/src/mcp-server/mcp-server.ts +1 -1
  29. package/src/mcp-server/server.ts +1 -1
  30. package/src/models/agentcreate.ts +1 -1
  31. package/src/models/agenttestrunlistitem.ts +1 -1
  32. package/src/models/agentupdate.ts +1 -1
  33. package/src/models/getagenttestrunsagenttestsagentagentuuidrunsgetop.ts +1 -1
  34. package/src/models/modelresult.ts +1 -1
  35. package/src/models/modelrunsummary.ts +5 -1
  36. package/src/models/testrunstatusresponse.ts +1 -1
package/README.md CHANGED
@@ -61,4 +61,4 @@ Point Cursor at the local build by using `node` with the path to `bin/mcp-server
61
61
  ## Resources
62
62
 
63
63
  - `npx @dalmia/calibrate-mcp start --help` — all flags and transports
64
- - [Calibrate docs](https://calibrate.artpark.ai/docs) — API reference
64
+ - [Calibrate docs](https://docs.calibrate.artpark.ai) — API reference
package/bin/mcp-server.js CHANGED
@@ -52802,9 +52802,9 @@ var init_config = __esm(() => {
52802
52802
  SDK_METADATA = {
52803
52803
  language: "typescript",
52804
52804
  openapiDocVersion: "0.1.0",
52805
- sdkVersion: "0.0.61",
52805
+ sdkVersion: "0.0.63",
52806
52806
  genVersion: "2.915.1",
52807
- userAgent: "speakeasy-sdk/mcp-typescript 0.0.61 2.915.1 0.1.0 @dalmia/calibrate-mcp"
52807
+ userAgent: "speakeasy-sdk/mcp-typescript 0.0.63 2.915.1 0.1.0 @dalmia/calibrate-mcp"
52808
52808
  };
52809
52809
  });
52810
52810
 
@@ -54263,7 +54263,7 @@ var init_agentcreate = __esm(() => {
54263
54263
  "general"
54264
54264
  ]).describe('What the agent expects in the request body:\n\n- `conversation`: a normal back-and-forth agent, answers within an ongoing conversation. Receives `{"messages": [...]}`\n- `general`: a one-shot agent, takes a single plain input and produces a single plain output, no conversation. Receives `{"input": "..."}`');
54265
54265
  AgentCreate$zodSchema = object({
54266
- config: record(string2(), any()).nullable().optional().describe('Agent behavioral config. The keys depend on `type`.\n\n**`type=agent`**, built inside Calibrate:\n- `system_prompt`: the agent\'s instructions\n- `llm.model`: `provider/model`, e.g. `openai/gpt-4.1` or `google/gemini-2.5-flash`\n- `stt.provider`: `deepgram`, `openai`, `cartesia`, `elevenlabs`, `google`, `sarvam`, or `smallest`\n- `tts.provider`: `cartesia`, `openai`, `google`, `elevenlabs`, `sarvam`, or `smallest`\n- `settings.agent_speaks_first`, `settings.max_assistant_turns`\n- `system_tools.end_call`: let the agent end the call\n- `data_extraction_fields`: `[{name, type, description, required}]`\n\n```json\n{\n "system_prompt": "You are a helpful support agent.",\n "llm": {"model": "openai/gpt-4.1"},\n "stt": {"provider": "deepgram"},\n "tts": {"provider": "elevenlabs"},\n "settings": {"agent_speaks_first": true, "max_assistant_turns": 50}\n}\n```\n\n**`type=connection`**, your own HTTP endpoint:\n- `agent_url`: public HTTP(S) endpoint your agent is called at\n- `agent_headers`: headers sent on each request, e.g. auth\n- `benchmark_provider`: `openrouter` by default. Other values: `openai`, `google`, `anthropic`, `meta-llama`, `mistralai`, `deepseek`, `x-ai`, `cohere`, `qwen`, or `ai21`\n\n```json\n{\n "agent_url": "https://api.example.com/agent",\n "agent_headers": {"Authorization": "Bearer <token>"},\n "benchmark_provider": "openrouter"\n}\n```\n\nEvery request Calibrate makes to your endpoint carries the header\n`X-Calibrate-Eval: 1`. Read it to tell a test run from a real user, for example\nto tag the trace you send back or to skip sending one.\n\nFor `type=agent`, omitted keys inherit managed defaults. Omit `config` entirely to use all defaults. For `type=connection`, `config` is stored as-is and must contain `agent_url`'),
54266
+ config: record(string2(), any()).nullable().optional().describe('Agent behavioral config. The keys depend on `type`.\n\n**`type=agent`**, built inside Calibrate:\n- `system_prompt`: the agent\'s instructions\n- `llm.model`: `provider/model`, e.g. `openai/gpt-4.1` or `google/gemini-2.5-flash`\n- `stt.provider`: `deepgram`, `openai`, `cartesia`, `elevenlabs`, `google`, `sarvam`, or `smallest`\n- `tts.provider`: `cartesia`, `openai`, `google`, `elevenlabs`, `sarvam`, or `smallest`\n- `settings.agent_speaks_first`, `settings.max_assistant_turns`\n- `system_tools.end_call`: let the agent end the call\n- `data_extraction_fields`: `[{name, type, description, required}]`\n\n```json\n{\n "system_prompt": "You are a helpful support agent.",\n "llm": {"model": "openai/gpt-4.1"},\n "stt": {"provider": "deepgram"},\n "tts": {"provider": "elevenlabs"},\n "settings": {"agent_speaks_first": true, "max_assistant_turns": 50}\n}\n```\n\n**`type=connection`**, your own HTTP endpoint:\n- `agent_url`: public HTTP(S) endpoint your agent is called at\n- `agent_headers`: headers sent on each request, e.g. auth\n- `benchmark_provider`: `openrouter` by default. Other values: `openai`, `google`, `anthropic`, `meta-llama`, `mistralai`, `deepseek`, `x-ai`, `cohere`, `qwen`, or `ai21`\n\n```json\n{\n "agent_url": "https://api.example.com/agent",\n "agent_headers": {"Authorization": "Bearer <token>"},\n "benchmark_provider": "openrouter"\n}\n```\n\n**Either type**:\n- `traces.scoring.enabled`: whether the traces you send for this agent are scored by its linked evaluators. On unless you set it to `false`. Turning it back on needs at least one linked evaluator that can score traces\n\n```json\n{"traces": {"scoring": {"enabled": false}}}\n```\n\nEvery request Calibrate makes to your endpoint carries the header\n`X-Calibrate-Eval: 1`. Read it to tell a test run from a real user, for example\nto tag the trace you send back or to skip sending one.\n\nFor `type=agent`, omitted keys inherit managed defaults. Omit `config` entirely to use all defaults. For `type=connection`, `config` is stored as-is and must contain `agent_url`'),
54267
54267
  interaction_type: AgentCreateInteractionType$zodSchema.default("conversation").describe('What the agent expects in the request body:\n\n- `conversation`: a normal back-and-forth agent, answers within an ongoing conversation. Receives `{"messages": [...]}`\n- `general`: a one-shot agent, takes a single plain input and produces a single plain output, no conversation. Receives `{"input": "..."}`'),
54268
54268
  name: string2().describe("Agent name, unique within the workspace"),
54269
54269
  type: AgentCreateType$zodSchema.default("agent").describe("- `agent`: built inside Calibrate\n- `connection`: your existing agent connected to Calibrate")
@@ -55334,7 +55334,7 @@ var AgentUpdate$zodSchema;
55334
55334
  var init_agentupdate = __esm(() => {
55335
55335
  init_zod();
55336
55336
  AgentUpdate$zodSchema = object({
55337
- config: record(string2(), any()).nullable().optional().describe('Agent behavioral config. The keys depend on `type`.\n\n**`type=agent`**, built inside Calibrate:\n- `system_prompt`: the agent\'s instructions\n- `llm.model`: `provider/model`, e.g. `openai/gpt-4.1` or `google/gemini-2.5-flash`\n- `stt.provider`: `deepgram`, `openai`, `cartesia`, `elevenlabs`, `google`, `sarvam`, or `smallest`\n- `tts.provider`: `cartesia`, `openai`, `google`, `elevenlabs`, `sarvam`, or `smallest`\n- `settings.agent_speaks_first`, `settings.max_assistant_turns`\n- `system_tools.end_call`: let the agent end the call\n- `data_extraction_fields`: `[{name, type, description, required}]`\n\n```json\n{\n "system_prompt": "You are a helpful support agent.",\n "llm": {"model": "openai/gpt-4.1"},\n "stt": {"provider": "deepgram"},\n "tts": {"provider": "elevenlabs"},\n "settings": {"agent_speaks_first": true, "max_assistant_turns": 50}\n}\n```\n\n**`type=connection`**, your own HTTP endpoint:\n- `agent_url`: public HTTP(S) endpoint your agent is called at\n- `agent_headers`: headers sent on each request, e.g. auth\n- `benchmark_provider`: `openrouter` by default. Other values: `openai`, `google`, `anthropic`, `meta-llama`, `mistralai`, `deepseek`, `x-ai`, `cohere`, `qwen`, or `ai21`\n\n```json\n{\n "agent_url": "https://api.example.com/agent",\n "agent_headers": {"Authorization": "Bearer <token>"},\n "benchmark_provider": "openrouter"\n}\n```\n\nEvery request Calibrate makes to your endpoint carries the header\n`X-Calibrate-Eval: 1`. Read it to tell a test run from a real user, for example\nto tag the trace you send back or to skip sending one.\n\nReplaces the stored config. Omit to leave unchanged\n\nFor `type=connection`, changing `agent_url` or `agent_headers` resets the connection and benchmark verification flags'),
55337
+ config: record(string2(), any()).nullable().optional().describe('Agent behavioral config. The keys depend on `type`.\n\n**`type=agent`**, built inside Calibrate:\n- `system_prompt`: the agent\'s instructions\n- `llm.model`: `provider/model`, e.g. `openai/gpt-4.1` or `google/gemini-2.5-flash`\n- `stt.provider`: `deepgram`, `openai`, `cartesia`, `elevenlabs`, `google`, `sarvam`, or `smallest`\n- `tts.provider`: `cartesia`, `openai`, `google`, `elevenlabs`, `sarvam`, or `smallest`\n- `settings.agent_speaks_first`, `settings.max_assistant_turns`\n- `system_tools.end_call`: let the agent end the call\n- `data_extraction_fields`: `[{name, type, description, required}]`\n\n```json\n{\n "system_prompt": "You are a helpful support agent.",\n "llm": {"model": "openai/gpt-4.1"},\n "stt": {"provider": "deepgram"},\n "tts": {"provider": "elevenlabs"},\n "settings": {"agent_speaks_first": true, "max_assistant_turns": 50}\n}\n```\n\n**`type=connection`**, your own HTTP endpoint:\n- `agent_url`: public HTTP(S) endpoint your agent is called at\n- `agent_headers`: headers sent on each request, e.g. auth\n- `benchmark_provider`: `openrouter` by default. Other values: `openai`, `google`, `anthropic`, `meta-llama`, `mistralai`, `deepseek`, `x-ai`, `cohere`, `qwen`, or `ai21`\n\n```json\n{\n "agent_url": "https://api.example.com/agent",\n "agent_headers": {"Authorization": "Bearer <token>"},\n "benchmark_provider": "openrouter"\n}\n```\n\n**Either type**:\n- `traces.scoring.enabled`: whether the traces you send for this agent are scored by its linked evaluators. On unless you set it to `false`. Turning it back on needs at least one linked evaluator that can score traces\n\n```json\n{"traces": {"scoring": {"enabled": false}}}\n```\n\nEvery request Calibrate makes to your endpoint carries the header\n`X-Calibrate-Eval: 1`. Read it to tell a test run from a real user, for example\nto tag the trace you send back or to skip sending one.\n\nReplaces the stored config. Omit to leave unchanged\n\nFor `type=connection`, changing `agent_url` or `agent_headers` resets the connection and benchmark verification flags'),
55338
55338
  name: string2().nullable().optional().describe("New agent name. Omit to leave the name unchanged")
55339
55339
  });
55340
55340
  });
@@ -56023,7 +56023,7 @@ var init_modelresult = __esm(() => {
56023
56023
  ModelResult$zodSchema = object({
56024
56024
  cost: record(string2(), any()).nullable().optional().describe("Aggregated cost as `{mean, min, max, count}` (USD)"),
56025
56025
  evaluator_summary: array(record(string2(), any())).nullable().optional().describe("Aggregate summary for each evaluator for this model"),
56026
- failed: int().nullable().optional().describe("Number of test cases that failed"),
56026
+ failed: int().nullable().optional().describe("Number of test cases that did not pass, which includes the ones that produced no answer"),
56027
56027
  latency_ms: record(string2(), any()).nullable().optional().describe("Aggregated latency in milliseconds, as `{p50, p95, p99, count}`"),
56028
56028
  message: string2().describe("Status or result message for this model"),
56029
56029
  model: string2().describe("Model name these results are for"),
@@ -56253,7 +56253,7 @@ var init_testrunstatusresponse = __esm(() => {
56253
56253
  error: string2().nullable().optional().describe("Why the run could not be carried out, when it failed before producing any result"),
56254
56254
  evaluator_summary: array(record(string2(), any())).nullable().optional().describe("Totals for each evaluator over the whole run, matching the shape a benchmark reports for each model. Only evaluators that returned a verdict appear"),
56255
56255
  evaluators: array(TestRunEvaluator$zodSchema).nullable().optional().describe("The evaluators used in this run. Each verdict in `judge_results` links to one of these by `evaluator_uuid`"),
56256
- failed: int().nullable().optional().describe("Number of test cases that failed"),
56256
+ failed: int().nullable().optional().describe("Number of test cases that did not pass, which includes the ones that produced no answer"),
56257
56257
  is_public: boolean2().default(false).describe("Whether the run is shared publicly"),
56258
56258
  latency_ms: record(string2(), any()).nullable().optional().describe("Aggregated response latency in milliseconds, as `{p50, p95, p99, count}`"),
56259
56259
  name: string2().describe("Name of the run. A run nobody has renamed shows its number instead, such as `Run 1` for a test run or `Benchmark 1` for a benchmark"),
@@ -56915,12 +56915,13 @@ var ModelRunSummary$zodSchema;
56915
56915
  var init_modelrunsummary = __esm(() => {
56916
56916
  init_zod();
56917
56917
  ModelRunSummary$zodSchema = object({
56918
- failed: int().nullable().optional().describe("Number of test cases that failed for this model"),
56918
+ failed: int().nullable().optional().describe("Number of test cases that did not pass for this model, which includes the ones that produced no answer"),
56919
56919
  message: string2().default("").describe("Status or result message for this model"),
56920
56920
  model: string2().describe("Model name these results are for"),
56921
56921
  passed: int().nullable().optional().describe("Number of test cases that passed for this model"),
56922
56922
  success: boolean2().nullable().optional().describe("Whether this model's run succeeded"),
56923
- total_tests: int().nullable().optional().describe("Total test cases for this model")
56923
+ total_tests: int().nullable().optional().describe("Total test cases for this model"),
56924
+ unanswered_tests: int().nullable().optional().describe("Number of this model's test cases that produced no answer, already counted in `failed`")
56924
56925
  }).describe("Flat summary for one model in a benchmark run-LIST item. The full results\nfor each case of a model live on the benchmark detail endpoint\n(`GET /agent-tests/benchmark/{task_id}`), not here.");
56925
56926
  });
56926
56927
 
@@ -56967,7 +56968,7 @@ var init_agenttestrunlistitem = __esm(() => {
56967
56968
  created_at: string2().describe("When the run was created (ISO 8601 UTC)"),
56968
56969
  error: boolean2().default(false).describe("True if the run failed"),
56969
56970
  evaluators: array(RunListEvaluator$zodSchema).optional().describe("The evaluators that judged this run, deduplicated and in display order. A `Tool call` entry is appended when any test in the run was a tool-call test. That entry has no `uuid`, because it is not an evaluator in the library. Empty when the run had no evaluators"),
56970
- failed: int().nullable().optional().describe("Number of test cases that failed"),
56971
+ failed: int().nullable().optional().describe("Number of test cases that did not pass, which includes the ones that produced no answer"),
56971
56972
  is_public: boolean2().default(false).describe("Whether the run is shared publicly"),
56972
56973
  latency_ms: record(string2(), any()).nullable().optional().describe("Aggregated latency in milliseconds, as `{p50, p95, p99, count}`"),
56973
56974
  model_results: array(ModelRunSummary$zodSchema).nullable().optional().describe("Flat summary for each model in a benchmark run (fetch the benchmark detail for full results)"),
@@ -57013,7 +57014,7 @@ var init_getagenttestrunsagenttestsagentagentuuidrunsgetop = __esm(() => {
57013
57014
  GetAgentTestRunsAgentTestsAgentAgentUuidRunsGetRequest$zodSchema = object({
57014
57015
  agent_uuid: string2().describe("Agent whose test runs to list"),
57015
57016
  around: string2().describe("ID of a run to jump to, returning the page that contains it instead of the page at `offset`").nullable().optional(),
57016
- has_failures: boolean2().describe("Filter by whether the run has any failing test case or model. `true` returns only runs with failures (or errors), `false` only clean runs. Omit for both").nullable().optional(),
57017
+ has_failures: boolean2().describe("Filter by whether a test in the run did not pass. `true` returns only runs with a failing test, `false` only runs that got through every test and passed them all. A run that broke, was stopped, or gave up part way with no failing test is in neither: filter by `status` for those. Omit for both").nullable().optional(),
57017
57018
  limit: int().describe("Maximum number of items to return. Omit for no limit (all items)").nullable().optional(),
57018
57019
  offset: int().default(0).describe("Number of items to skip before returning results"),
57019
57020
  status: TaskStatus$zodSchema.nullable().optional().describe("Filter by run status. Omit for all statuses"),
@@ -61605,7 +61606,7 @@ hits its trace limit.
61605
61606
  function createMCPServer(deps) {
61606
61607
  const server = new McpServer({
61607
61608
  name: "CalibrateMcp",
61608
- version: "0.0.61"
61609
+ version: "0.0.63"
61609
61610
  });
61610
61611
  const getClient = deps.getSDK || (() => new CalibrateMcpCore({
61611
61612
  security: deps.security,
@@ -62894,7 +62895,7 @@ http_headers = { "api-key-auth" = "YOUR_API_KEY_AUTH" }`;
62894
62895
  <h1>Instructions</h1>
62895
62896
  <p>One-click installation for Claude Desktop users</p>
62896
62897
  <div class="instruction-item">
62897
- <a href="https://github.com/dalmia/calibrate-mcp/releases/download/v0.0.61/mcp-server.mcpb" download="mcp-server.mcpb" class="action-button header-action" style="display: inline-flex; margin-bottom: 16px;">
62898
+ <a href="https://github.com/dalmia/calibrate-mcp/releases/download/v0.0.63/mcp-server.mcpb" download="mcp-server.mcpb" class="action-button header-action" style="display: inline-flex; margin-bottom: 16px;">
62898
62899
  \uD83D\uDCE5 Download MCP Bundle
62899
62900
  </a>
62900
62901
  </div>
@@ -65778,7 +65779,7 @@ var routes = buildRouteMap({
65778
65779
  var app = buildApplication(routes, {
65779
65780
  name: "mcp",
65780
65781
  versionInfo: {
65781
- currentVersion: "0.0.61"
65782
+ currentVersion: "0.0.63"
65782
65783
  }
65783
65784
  });
65784
65785
  run(app, process3.argv.slice(2), buildContext(process3));
@@ -65786,5 +65787,5 @@ export {
65786
65787
  app
65787
65788
  };
65788
65789
 
65789
- //# debugId=92996C3E08F6D0B164756E2164756E21
65790
+ //# debugId=DF7D04973BECCF3264756E2164756E21
65790
65791
  //# sourceMappingURL=mcp-server.js.map