crosscheck-mcp 0.2.20 → 0.2.22

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1176,7 +1176,27 @@ var PROVIDER_CAPS = {
1176
1176
  // temperature, so the prefix is the fix.
1177
1177
  reasoning_prefixes: ["gpt-5", "gpt-6", "o1", "o3", "o4"]
1178
1178
  },
1179
- xai: { family: "openai_chat", system_role: "inline", supports_temperature: true },
1179
+ xai: {
1180
+ family: "openai_chat",
1181
+ system_role: "inline",
1182
+ // Kept `true`, unlike other reasoning models: grok-4 accepts temperature
1183
+ // (verified 200 at 0.4 and 1.0) and a panel wants the variance. Dropping
1184
+ // it would buy nothing and cost diversity.
1185
+ supports_temperature: true,
1186
+ // grok-4 IS reasoning-class — it emitted 1,432 reasoning tokens against 64
1187
+ // visible on one call. Saying so strips the "think step by step" preambles
1188
+ // it does not need, and lifts it off the non-reasoning ceiling (1500) onto
1189
+ // the reasoning-safe one (2048).
1190
+ //
1191
+ // NOT the reason grok answers short, which was the initial guess and was
1192
+ // wrong: at caps of 1500, 2048 and 6144 it returned 445, 404 and 461
1193
+ // visible tokens, finish_reason=stop every time. The cap was never
1194
+ // binding; grok is simply terse. The ceiling change removes a latent
1195
+ // constraint, it does not make grok say more.
1196
+ reasoning_prefixes: ["grok-4"],
1197
+ // Measured: xAI's cap bounds visible output only. See the field docs.
1198
+ reasoning_shares_output_budget: false
1199
+ },
1180
1200
  mistral: { family: "openai_chat", system_role: "inline", supports_temperature: true },
1181
1201
  groq: { family: "openai_chat", system_role: "inline", supports_temperature: true },
1182
1202
  deepseek: { family: "openai_chat", system_role: "inline", supports_temperature: true },
@@ -1215,6 +1235,10 @@ function supportsTemperature(provider, model) {
1215
1235
  if (caps.supports_temperature === "model") return !isReasoningModel(provider, model);
1216
1236
  return Boolean(caps.supports_temperature);
1217
1237
  }
1238
+ function reasoningSharesOutputBudget(provider) {
1239
+ const caps = PROVIDER_CAPS[provider.toLowerCase()];
1240
+ return caps?.reasoning_shares_output_budget ?? true;
1241
+ }
1218
1242
 
1219
1243
  // src/providers/types.ts
1220
1244
  init_esm_shims();
@@ -1230,12 +1254,20 @@ var ProviderError = class extends Error {
1230
1254
  * actually fix. Distinct from `transient`: this is never worth
1231
1255
  * retrying the SAME model, but IS worth trying a different one. */
1232
1256
  modelAccessFailure;
1257
+ /** True for "the key works, the account is out of credit / over a spend
1258
+ * limit." Rides alongside kind "auth" rather than replacing it, because
1259
+ * every retry and fallback decision downstream is already tuned to that
1260
+ * kind — but the two are completely different things to a human, and
1261
+ * reporting "API key rejected" for an unpaid invoice sends people to
1262
+ * regenerate a key that was never the problem. */
1263
+ billing;
1233
1264
  constructor(kind, message, opts) {
1234
1265
  super(message);
1235
1266
  this.kind = kind;
1236
1267
  this.status = opts?.status;
1237
1268
  this.transient = opts?.transient ?? defaultTransient(kind);
1238
1269
  this.modelAccessFailure = opts?.modelAccessFailure ?? false;
1270
+ this.billing = opts?.billing ?? false;
1239
1271
  if (opts?.retryAfterS !== void 0) {
1240
1272
  this.retryAfterS = opts.retryAfterS;
1241
1273
  }
@@ -1274,7 +1306,7 @@ function httpFailureToProviderError(provider, status, bodyText, retryAfterS, mod
1274
1306
  return new ProviderError(
1275
1307
  "auth",
1276
1308
  `${provider}: out of credits or billing isn't set up (HTTP ${status}). Your API key is reaching ${provider}, but the account has no usable balance \u2014 add credits / enable billing in your ${provider} console, then retry. Detail: ${detail}`,
1277
- { status }
1309
+ { status, billing: true }
1278
1310
  );
1279
1311
  }
1280
1312
  if (status === 401 || status === 403) {
@@ -1459,6 +1491,23 @@ async function acquireRateLimit(provider, deps) {
1459
1491
  var ANTHROPIC_API_URL = "https://api.anthropic.com/v1/messages";
1460
1492
  var ANTHROPIC_VERSION_HEADER = "2023-06-01";
1461
1493
  var ANTHROPIC_STRUCTURED_TOOL_NAME = "structured_output";
1494
+ function applyPromptCaching(body, env = process.env) {
1495
+ const flag = (env["CROSSCHECK_ANTHROPIC_PROMPT_CACHE"] ?? "").trim();
1496
+ if (flag === "0" || flag.toLowerCase() === "false") return;
1497
+ if (typeof body.system === "string" && body.system !== "") {
1498
+ body.system = [{ type: "text", text: body.system, cache_control: { type: "ephemeral" } }];
1499
+ }
1500
+ if (body.messages.length >= 2) {
1501
+ const idx = body.messages.length - 2;
1502
+ const m = body.messages[idx];
1503
+ if (typeof m.content === "string" && m.content !== "") {
1504
+ body.messages[idx] = {
1505
+ role: m.role,
1506
+ content: [{ type: "text", text: m.content, cache_control: { type: "ephemeral" } }]
1507
+ };
1508
+ }
1509
+ }
1510
+ }
1462
1511
  function buildAnthropicRequest(opts) {
1463
1512
  let system;
1464
1513
  const convo = [];
@@ -1483,6 +1532,7 @@ function buildAnthropicRequest(opts) {
1483
1532
  if (system !== void 0) {
1484
1533
  body.system = system;
1485
1534
  }
1535
+ applyPromptCaching(body);
1486
1536
  if (isReasoningModel("anthropic", opts.model)) {
1487
1537
  if (opts.model.toLowerCase().startsWith("claude-fable-5")) {
1488
1538
  body.thinking = { type: "adaptive" };
@@ -1542,6 +1592,7 @@ function parseAnthropicResponse(opts) {
1542
1592
  const prompt = Math.trunc(Number(u["input_tokens"] ?? 0)) || 0;
1543
1593
  const cached = Math.trunc(Number(u["cache_read_input_tokens"] ?? 0)) || 0;
1544
1594
  const completion = Math.trunc(Number(u["output_tokens"] ?? 0)) || 0;
1595
+ const cacheWrite = Math.trunc(Number(u["cache_creation_input_tokens"] ?? 0)) || 0;
1545
1596
  const usage = {
1546
1597
  provider: "anthropic",
1547
1598
  model: opts.model,
@@ -1549,7 +1600,7 @@ function parseAnthropicResponse(opts) {
1549
1600
  // production helper folds them in so prompt_tokens is the FULL
1550
1601
  // input volume; calculateCost then bills the cached subset at the
1551
1602
  // cached rate (cached <= prompt_tokens, since prompt = input+cached).
1552
- prompt_tokens: prompt + cached,
1603
+ prompt_tokens: prompt + cached + cacheWrite,
1553
1604
  completion_tokens: completion,
1554
1605
  cached_tokens: cached,
1555
1606
  total_tokens: 0,
@@ -1901,16 +1952,21 @@ function parseOpenAICompatibleResponse(opts) {
1901
1952
  const u = r["usage"] ?? {};
1902
1953
  const details = u["prompt_tokens_details"] ?? {};
1903
1954
  const cached = Math.trunc(Number(details["cached_tokens"] ?? 0)) || 0;
1955
+ const promptTokens = Math.trunc(Number(u["prompt_tokens"] ?? 0)) || 0;
1956
+ const reportedCompletion = Math.trunc(Number(u["completion_tokens"] ?? 0)) || 0;
1957
+ const reportedTotal = Math.trunc(Number(u["total_tokens"] ?? 0)) || 0;
1958
+ const derivedCompletion = reportedTotal > promptTokens ? reportedTotal - promptTokens : reportedCompletion;
1959
+ const completionTokens = Math.max(reportedCompletion, derivedCompletion);
1904
1960
  const usage = {
1905
1961
  provider: opts.provider,
1906
1962
  model: opts.model,
1907
- prompt_tokens: Math.trunc(Number(u["prompt_tokens"] ?? 0)) || 0,
1908
- completion_tokens: Math.trunc(Number(u["completion_tokens"] ?? 0)) || 0,
1963
+ prompt_tokens: promptTokens,
1964
+ completion_tokens: completionTokens,
1909
1965
  cached_tokens: cached,
1910
1966
  // Use the response's reported total_tokens directly. Python's
1911
1967
  // `Usage.to_dict()` falls back to prompt+completion only when
1912
1968
  // total_tokens is 0/missing; mirror that.
1913
- total_tokens: Math.trunc(Number(u["total_tokens"] ?? 0)) || 0,
1969
+ total_tokens: reportedTotal,
1914
1970
  cost_usd: 0,
1915
1971
  estimated: Object.keys(u).length === 0,
1916
1972
  purpose: opts.purpose
@@ -1952,7 +2008,11 @@ async function sendOpenAICompatible(args) {
1952
2008
  body.reasoning_effort = effort;
1953
2009
  }
1954
2010
  }
1955
- if (isReasoningModel(args.provider, args.model) && typeof body.max_completion_tokens === "number") {
2011
+ if (isReasoningModel(args.provider, args.model) && // Only where the cap is a SHARED budget. On a provider that bounds visible
2012
+ // output only (xAI), headroom cannot protect the answer from being crowded
2013
+ // out — nothing is crowding it — so it would merely authorise a
2014
+ // 25,000-token reply and the bill that comes with it.
2015
+ reasoningSharesOutputBudget(args.provider) && typeof body.max_completion_tokens === "number") {
1956
2016
  const raw = Number(process.env["CROSSCHECK_OPENAI_REASONING_HEADROOM_TOKENS"]);
1957
2017
  const headroom = Number.isFinite(raw) && raw >= 0 ? Math.trunc(raw) : 25e3;
1958
2018
  body.max_completion_tokens += headroom;
@@ -2271,7 +2331,7 @@ import { z } from "zod";
2271
2331
  // src/server-meta.ts
2272
2332
  init_esm_shims();
2273
2333
  var SERVER_NAME = "crosscheck-agent";
2274
- var SERVER_VERSION = true ? "0.2.20" : "0.0.0-dev";
2334
+ var SERVER_VERSION = true ? "0.2.22" : "0.0.0-dev";
2275
2335
 
2276
2336
  // src/tools/audit.ts
2277
2337
  init_esm_shims();
@@ -2784,7 +2844,9 @@ async function attachUsageBlock(result, answers, opts) {
2784
2844
  purpose: a.usage?.purpose ?? null,
2785
2845
  wall_ms: a.elapsed_ms ?? 0,
2786
2846
  cpu_ms: a.cpu_ms ?? 0,
2787
- cache_hit: Boolean(a.cache_hit)
2847
+ cache_hit: Boolean(a.cache_hit),
2848
+ ...a.error_kind ? { error_kind: a.error_kind } : {},
2849
+ ...a.error_billing ? { error_billing: true } : {}
2788
2850
  }))
2789
2851
  };
2790
2852
  result["timing"] = timing;
@@ -3093,11 +3155,13 @@ async function askOne(provider, messages, opts) {
3093
3155
  const cpuMs = Math.trunc((cpu.user + cpu.system) / 1e3);
3094
3156
  const kind = e instanceof ProviderError ? e.kind : "other";
3095
3157
  const msg = e instanceof Error ? e.message : String(e);
3158
+ const billing = e instanceof ProviderError && e.billing;
3096
3159
  return {
3097
3160
  provider: provider.name,
3098
3161
  model: provider.model,
3099
3162
  error: msg,
3100
3163
  error_kind: kind,
3164
+ ...billing ? { error_billing: true } : {},
3101
3165
  attempts: 0,
3102
3166
  usage: emptyUsage(provider.name, provider.model, opts.purpose),
3103
3167
  cache_hit: false,
@@ -12776,7 +12840,7 @@ var CHECK_INTERVAL_SECONDS = 3 * 24 * 60 * 60;
12776
12840
  var DEFAULT_PACKAGE = "crosscheck-cli";
12777
12841
  var FETCH_TIMEOUT_MS = 3e3;
12778
12842
  function engineVersion() {
12779
- return true ? "0.2.20" : "0.0.0-dev";
12843
+ return true ? "0.2.22" : "0.0.0-dev";
12780
12844
  }
12781
12845
  function defaultUpdateCachePath() {
12782
12846
  const base = process.env["CROSSCHECK_DATA_DIR"] || path10.join(os.homedir() || os.tmpdir(), ".crosscheck");
@@ -14791,6 +14855,23 @@ function getConfig2() {
14791
14855
  config = { url, seatId, orgId, seatKey };
14792
14856
  return config;
14793
14857
  }
14858
+ function errorClassFor(kind) {
14859
+ switch (kind) {
14860
+ case "auth":
14861
+ return "auth";
14862
+ case "rate_limit":
14863
+ return "rate_limit";
14864
+ case "timeout":
14865
+ return "timeout";
14866
+ case "server":
14867
+ return "server";
14868
+ // network/parse/client/other all mean "we don't know that the key is
14869
+ // bad", and guessing wrong here is what produces a dashboard that cries
14870
+ // wolf. Say unknown.
14871
+ default:
14872
+ return "unknown";
14873
+ }
14874
+ }
14794
14875
  function intNonNeg(v) {
14795
14876
  const n = Math.trunc(Number(v));
14796
14877
  return Number.isFinite(n) && n > 0 ? n : 0;
@@ -14798,12 +14879,12 @@ function intNonNeg(v) {
14798
14879
  function usageEventsFromEnvelope(out, toolName) {
14799
14880
  if (!out || typeof out !== "object" || Array.isArray(out)) return [];
14800
14881
  const usage = out["usage"];
14801
- if (!usage || typeof usage !== "object") return [];
14802
- const byCall = usage["by_call"];
14803
- if (!Array.isArray(byCall) || byCall.length === 0) return [];
14882
+ const rawByCall = usage && typeof usage === "object" ? usage["by_call"] : void 0;
14883
+ const byCall = Array.isArray(rawByCall) ? rawByCall : [];
14804
14884
  const timing = out["timing"];
14805
14885
  const timingByCall = timing && typeof timing === "object" ? timing["by_call"] : void 0;
14806
- const latencies = Array.isArray(timingByCall) ? timingByCall : [];
14886
+ const allTiming = Array.isArray(timingByCall) ? timingByCall : [];
14887
+ const latencies = allTiming;
14807
14888
  const aligned = latencies.length === byCall.length;
14808
14889
  const pattern = toolToPattern(toolName);
14809
14890
  const events = [];
@@ -14813,13 +14894,18 @@ function usageEventsFromEnvelope(out, toolName) {
14813
14894
  if (!provider) continue;
14814
14895
  const model = String(u.model ?? "").slice(0, 128) || "unknown";
14815
14896
  const cost = Math.max(0, Number(u.cost_usd) || 0);
14816
- const latencyMs = aligned ? intNonNeg(latencies[i]?.["wall_ms"]) : 0;
14897
+ const timingRow = aligned ? latencies[i] : void 0;
14898
+ if (timingRow?.["error_kind"]) continue;
14899
+ const promptTokens = intNonNeg(u.prompt_tokens);
14900
+ const completionTokens = intNonNeg(u.completion_tokens);
14901
+ if (promptTokens === 0 && completionTokens === 0) continue;
14902
+ const latencyMs = intNonNeg(timingRow?.["wall_ms"]);
14817
14903
  events.push({
14818
14904
  provider,
14819
14905
  model,
14820
14906
  pattern,
14821
- promptTokens: intNonNeg(u.prompt_tokens),
14822
- completionTokens: intNonNeg(u.completion_tokens),
14907
+ promptTokens,
14908
+ completionTokens,
14823
14909
  costUsdEstimate: cost,
14824
14910
  latencyMs,
14825
14911
  status: "ok",
@@ -14828,6 +14914,28 @@ function usageEventsFromEnvelope(out, toolName) {
14828
14914
  purpose: normalizePurpose(u.purpose)
14829
14915
  });
14830
14916
  }
14917
+ for (const t of allTiming) {
14918
+ const row = t;
14919
+ const kind = row["error_kind"];
14920
+ if (!kind) continue;
14921
+ const provider = providerToSchema(String(row["provider"] ?? ""));
14922
+ if (!provider) continue;
14923
+ events.push({
14924
+ provider,
14925
+ model: String(row["model"] ?? "").slice(0, 128) || "unknown",
14926
+ pattern,
14927
+ // A failure consumed no tokens and cost nothing. Recording it with a
14928
+ // nonzero cost would corrupt the per-feature ledger with money nobody
14929
+ // spent.
14930
+ promptTokens: 0,
14931
+ completionTokens: 0,
14932
+ costUsdEstimate: 0,
14933
+ latencyMs: intNonNeg(row["wall_ms"]),
14934
+ status: "error",
14935
+ errorClass: row["error_billing"] ? "auth" : errorClassFor(String(kind)),
14936
+ purpose: normalizePurpose(row["purpose"])
14937
+ });
14938
+ }
14831
14939
  return events;
14832
14940
  }
14833
14941
  var queue = [];
@@ -14854,6 +14962,10 @@ function enqueue(cfg2, input, runId, fingerprint) {
14854
14962
  costUsdEstimate: input.costUsdEstimate,
14855
14963
  latencyMs: input.latencyMs,
14856
14964
  status: input.status,
14965
+ // Only present on failures. The canonical body renders an absent
14966
+ // errorClass as "", so omitting and sending "" sign identically — but
14967
+ // omitting keeps the JSON honest about what we actually know.
14968
+ ...input.errorClass === void 0 ? {} : { errorClass: input.errorClass },
14857
14969
  runId,
14858
14970
  // Omit entirely (not "") when unknown, so the canonical body stays
14859
14971
  // consistent with what the server reconstructs from the JSON.