crosscheck-mcp 0.2.21 → 0.2.22

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1200,7 +1200,27 @@ var PROVIDER_CAPS = {
1200
1200
  // temperature, so the prefix is the fix.
1201
1201
  reasoning_prefixes: ["gpt-5", "gpt-6", "o1", "o3", "o4"]
1202
1202
  },
1203
- xai: { family: "openai_chat", system_role: "inline", supports_temperature: true },
1203
+ xai: {
1204
+ family: "openai_chat",
1205
+ system_role: "inline",
1206
+ // Kept `true`, unlike other reasoning models: grok-4 accepts temperature
1207
+ // (verified 200 at 0.4 and 1.0) and a panel wants the variance. Dropping
1208
+ // it would buy nothing and cost diversity.
1209
+ supports_temperature: true,
1210
+ // grok-4 IS reasoning-class — it emitted 1,432 reasoning tokens against 64
1211
+ // visible on one call. Saying so strips the "think step by step" preambles
1212
+ // it does not need, and lifts it off the non-reasoning ceiling (1500) onto
1213
+ // the reasoning-safe one (2048).
1214
+ //
1215
+ // NOT the reason grok answers short, which was the initial guess and was
1216
+ // wrong: at caps of 1500, 2048 and 6144 it returned 445, 404 and 461
1217
+ // visible tokens, finish_reason=stop every time. The cap was never
1218
+ // binding; grok is simply terse. The ceiling change removes a latent
1219
+ // constraint, it does not make grok say more.
1220
+ reasoning_prefixes: ["grok-4"],
1221
+ // Measured: xAI's cap bounds visible output only. See the field docs.
1222
+ reasoning_shares_output_budget: false
1223
+ },
1204
1224
  mistral: { family: "openai_chat", system_role: "inline", supports_temperature: true },
1205
1225
  groq: { family: "openai_chat", system_role: "inline", supports_temperature: true },
1206
1226
  deepseek: { family: "openai_chat", system_role: "inline", supports_temperature: true },
@@ -1239,6 +1259,10 @@ function supportsTemperature(provider, model) {
1239
1259
  if (caps.supports_temperature === "model") return !isReasoningModel(provider, model);
1240
1260
  return Boolean(caps.supports_temperature);
1241
1261
  }
1262
+ function reasoningSharesOutputBudget(provider) {
1263
+ const caps = PROVIDER_CAPS[provider.toLowerCase()];
1264
+ return caps?.reasoning_shares_output_budget ?? true;
1265
+ }
1242
1266
 
1243
1267
  // src/providers/types.ts
1244
1268
  init_cjs_shims();
@@ -1491,6 +1515,23 @@ async function acquireRateLimit(provider, deps) {
1491
1515
  var ANTHROPIC_API_URL = "https://api.anthropic.com/v1/messages";
1492
1516
  var ANTHROPIC_VERSION_HEADER = "2023-06-01";
1493
1517
  var ANTHROPIC_STRUCTURED_TOOL_NAME = "structured_output";
1518
+ function applyPromptCaching(body, env = process.env) {
1519
+ const flag = (env["CROSSCHECK_ANTHROPIC_PROMPT_CACHE"] ?? "").trim();
1520
+ if (flag === "0" || flag.toLowerCase() === "false") return;
1521
+ if (typeof body.system === "string" && body.system !== "") {
1522
+ body.system = [{ type: "text", text: body.system, cache_control: { type: "ephemeral" } }];
1523
+ }
1524
+ if (body.messages.length >= 2) {
1525
+ const idx = body.messages.length - 2;
1526
+ const m = body.messages[idx];
1527
+ if (typeof m.content === "string" && m.content !== "") {
1528
+ body.messages[idx] = {
1529
+ role: m.role,
1530
+ content: [{ type: "text", text: m.content, cache_control: { type: "ephemeral" } }]
1531
+ };
1532
+ }
1533
+ }
1534
+ }
1494
1535
  function buildAnthropicRequest(opts) {
1495
1536
  let system;
1496
1537
  const convo = [];
@@ -1515,6 +1556,7 @@ function buildAnthropicRequest(opts) {
1515
1556
  if (system !== void 0) {
1516
1557
  body.system = system;
1517
1558
  }
1559
+ applyPromptCaching(body);
1518
1560
  if (isReasoningModel("anthropic", opts.model)) {
1519
1561
  if (opts.model.toLowerCase().startsWith("claude-fable-5")) {
1520
1562
  body.thinking = { type: "adaptive" };
@@ -1574,6 +1616,7 @@ function parseAnthropicResponse(opts) {
1574
1616
  const prompt = Math.trunc(Number(u["input_tokens"] ?? 0)) || 0;
1575
1617
  const cached = Math.trunc(Number(u["cache_read_input_tokens"] ?? 0)) || 0;
1576
1618
  const completion = Math.trunc(Number(u["output_tokens"] ?? 0)) || 0;
1619
+ const cacheWrite = Math.trunc(Number(u["cache_creation_input_tokens"] ?? 0)) || 0;
1577
1620
  const usage = {
1578
1621
  provider: "anthropic",
1579
1622
  model: opts.model,
@@ -1581,7 +1624,7 @@ function parseAnthropicResponse(opts) {
1581
1624
  // production helper folds them in so prompt_tokens is the FULL
1582
1625
  // input volume; calculateCost then bills the cached subset at the
1583
1626
  // cached rate (cached <= prompt_tokens, since prompt = input+cached).
1584
- prompt_tokens: prompt + cached,
1627
+ prompt_tokens: prompt + cached + cacheWrite,
1585
1628
  completion_tokens: completion,
1586
1629
  cached_tokens: cached,
1587
1630
  total_tokens: 0,
@@ -1933,16 +1976,21 @@ function parseOpenAICompatibleResponse(opts) {
1933
1976
  const u = r["usage"] ?? {};
1934
1977
  const details = u["prompt_tokens_details"] ?? {};
1935
1978
  const cached = Math.trunc(Number(details["cached_tokens"] ?? 0)) || 0;
1979
+ const promptTokens = Math.trunc(Number(u["prompt_tokens"] ?? 0)) || 0;
1980
+ const reportedCompletion = Math.trunc(Number(u["completion_tokens"] ?? 0)) || 0;
1981
+ const reportedTotal = Math.trunc(Number(u["total_tokens"] ?? 0)) || 0;
1982
+ const derivedCompletion = reportedTotal > promptTokens ? reportedTotal - promptTokens : reportedCompletion;
1983
+ const completionTokens = Math.max(reportedCompletion, derivedCompletion);
1936
1984
  const usage = {
1937
1985
  provider: opts.provider,
1938
1986
  model: opts.model,
1939
- prompt_tokens: Math.trunc(Number(u["prompt_tokens"] ?? 0)) || 0,
1940
- completion_tokens: Math.trunc(Number(u["completion_tokens"] ?? 0)) || 0,
1987
+ prompt_tokens: promptTokens,
1988
+ completion_tokens: completionTokens,
1941
1989
  cached_tokens: cached,
1942
1990
  // Use the response's reported total_tokens directly. Python's
1943
1991
  // `Usage.to_dict()` falls back to prompt+completion only when
1944
1992
  // total_tokens is 0/missing; mirror that.
1945
- total_tokens: Math.trunc(Number(u["total_tokens"] ?? 0)) || 0,
1993
+ total_tokens: reportedTotal,
1946
1994
  cost_usd: 0,
1947
1995
  estimated: Object.keys(u).length === 0,
1948
1996
  purpose: opts.purpose
@@ -1984,7 +2032,11 @@ async function sendOpenAICompatible(args) {
1984
2032
  body.reasoning_effort = effort;
1985
2033
  }
1986
2034
  }
1987
- if (isReasoningModel(args.provider, args.model) && typeof body.max_completion_tokens === "number") {
2035
+ if (isReasoningModel(args.provider, args.model) && // Only where the cap is a SHARED budget. On a provider that bounds visible
2036
+ // output only (xAI), headroom cannot protect the answer from being crowded
2037
+ // out — nothing is crowding it — so it would merely authorise a
2038
+ // 25,000-token reply and the bill that comes with it.
2039
+ reasoningSharesOutputBudget(args.provider) && typeof body.max_completion_tokens === "number") {
1988
2040
  const raw = Number(process.env["CROSSCHECK_OPENAI_REASONING_HEADROOM_TOKENS"]);
1989
2041
  const headroom = Number.isFinite(raw) && raw >= 0 ? Math.trunc(raw) : 25e3;
1990
2042
  body.max_completion_tokens += headroom;
@@ -2300,7 +2352,7 @@ var import_zod = require("zod");
2300
2352
  // src/server-meta.ts
2301
2353
  init_cjs_shims();
2302
2354
  var SERVER_NAME = "crosscheck-agent";
2303
- var SERVER_VERSION = true ? "0.2.21" : "0.0.0-dev";
2355
+ var SERVER_VERSION = true ? "0.2.22" : "0.0.0-dev";
2304
2356
 
2305
2357
  // src/tools/audit.ts
2306
2358
  init_cjs_shims();
@@ -12792,7 +12844,7 @@ var CHECK_INTERVAL_SECONDS = 3 * 24 * 60 * 60;
12792
12844
  var DEFAULT_PACKAGE = "crosscheck-cli";
12793
12845
  var FETCH_TIMEOUT_MS = 3e3;
12794
12846
  function engineVersion() {
12795
- return true ? "0.2.21" : "0.0.0-dev";
12847
+ return true ? "0.2.22" : "0.0.0-dev";
12796
12848
  }
12797
12849
  function defaultUpdateCachePath() {
12798
12850
  const base = process.env["CROSSCHECK_DATA_DIR"] || import_node_path13.default.join(import_node_os2.default.homedir() || import_node_os2.default.tmpdir(), ".crosscheck");
@@ -14836,9 +14888,7 @@ function usageEventsFromEnvelope(out, toolName) {
14836
14888
  const timing = out["timing"];
14837
14889
  const timingByCall = timing && typeof timing === "object" ? timing["by_call"] : void 0;
14838
14890
  const allTiming = Array.isArray(timingByCall) ? timingByCall : [];
14839
- const latencies = allTiming.filter(
14840
- (t) => !t?.["error_kind"]
14841
- );
14891
+ const latencies = allTiming;
14842
14892
  const aligned = latencies.length === byCall.length;
14843
14893
  const pattern = toolToPattern(toolName);
14844
14894
  const events = [];
@@ -14848,13 +14898,18 @@ function usageEventsFromEnvelope(out, toolName) {
14848
14898
  if (!provider) continue;
14849
14899
  const model = String(u.model ?? "").slice(0, 128) || "unknown";
14850
14900
  const cost = Math.max(0, Number(u.cost_usd) || 0);
14851
- const latencyMs = aligned ? intNonNeg(latencies[i]?.["wall_ms"]) : 0;
14901
+ const timingRow = aligned ? latencies[i] : void 0;
14902
+ if (timingRow?.["error_kind"]) continue;
14903
+ const promptTokens = intNonNeg(u.prompt_tokens);
14904
+ const completionTokens = intNonNeg(u.completion_tokens);
14905
+ if (promptTokens === 0 && completionTokens === 0) continue;
14906
+ const latencyMs = intNonNeg(timingRow?.["wall_ms"]);
14852
14907
  events.push({
14853
14908
  provider,
14854
14909
  model,
14855
14910
  pattern,
14856
- promptTokens: intNonNeg(u.prompt_tokens),
14857
- completionTokens: intNonNeg(u.completion_tokens),
14911
+ promptTokens,
14912
+ completionTokens,
14858
14913
  costUsdEstimate: cost,
14859
14914
  latencyMs,
14860
14915
  status: "ok",