crosscheck-mcp 0.2.21 → 0.2.23

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1176,7 +1176,27 @@ var PROVIDER_CAPS = {
1176
1176
  // temperature, so the prefix is the fix.
1177
1177
  reasoning_prefixes: ["gpt-5", "gpt-6", "o1", "o3", "o4"]
1178
1178
  },
1179
- xai: { family: "openai_chat", system_role: "inline", supports_temperature: true },
1179
+ xai: {
1180
+ family: "openai_chat",
1181
+ system_role: "inline",
1182
+ // Kept `true`, unlike other reasoning models: grok-4 accepts temperature
1183
+ // (verified 200 at 0.4 and 1.0) and a panel wants the variance. Dropping
1184
+ // it would buy nothing and cost diversity.
1185
+ supports_temperature: true,
1186
+ // grok-4 IS reasoning-class — it emitted 1,432 reasoning tokens against 64
1187
+ // visible on one call. Saying so strips the "think step by step" preambles
1188
+ // it does not need, and lifts it off the non-reasoning ceiling (1500) onto
1189
+ // the reasoning-safe one (2048).
1190
+ //
1191
+ // NOT the reason grok answers short, which was the initial guess and was
1192
+ // wrong: at caps of 1500, 2048 and 6144 it returned 445, 404 and 461
1193
+ // visible tokens, finish_reason=stop every time. The cap was never
1194
+ // binding; grok is simply terse. The ceiling change removes a latent
1195
+ // constraint, it does not make grok say more.
1196
+ reasoning_prefixes: ["grok-4"],
1197
+ // Measured: xAI's cap bounds visible output only. See the field docs.
1198
+ reasoning_shares_output_budget: false
1199
+ },
1180
1200
  mistral: { family: "openai_chat", system_role: "inline", supports_temperature: true },
1181
1201
  groq: { family: "openai_chat", system_role: "inline", supports_temperature: true },
1182
1202
  deepseek: { family: "openai_chat", system_role: "inline", supports_temperature: true },
@@ -1215,6 +1235,10 @@ function supportsTemperature(provider, model) {
1215
1235
  if (caps.supports_temperature === "model") return !isReasoningModel(provider, model);
1216
1236
  return Boolean(caps.supports_temperature);
1217
1237
  }
1238
+ function reasoningSharesOutputBudget(provider) {
1239
+ const caps = PROVIDER_CAPS[provider.toLowerCase()];
1240
+ return caps?.reasoning_shares_output_budget ?? true;
1241
+ }
1218
1242
 
1219
1243
  // src/providers/types.ts
1220
1244
  init_esm_shims();
@@ -1467,6 +1491,23 @@ async function acquireRateLimit(provider, deps) {
1467
1491
  var ANTHROPIC_API_URL = "https://api.anthropic.com/v1/messages";
1468
1492
  var ANTHROPIC_VERSION_HEADER = "2023-06-01";
1469
1493
  var ANTHROPIC_STRUCTURED_TOOL_NAME = "structured_output";
1494
+ function applyPromptCaching(body, env = process.env) {
1495
+ const flag = (env["CROSSCHECK_ANTHROPIC_PROMPT_CACHE"] ?? "").trim();
1496
+ if (flag === "0" || flag.toLowerCase() === "false") return;
1497
+ if (typeof body.system === "string" && body.system !== "") {
1498
+ body.system = [{ type: "text", text: body.system, cache_control: { type: "ephemeral" } }];
1499
+ }
1500
+ if (body.messages.length >= 2) {
1501
+ const idx = body.messages.length - 2;
1502
+ const m = body.messages[idx];
1503
+ if (typeof m.content === "string" && m.content !== "") {
1504
+ body.messages[idx] = {
1505
+ role: m.role,
1506
+ content: [{ type: "text", text: m.content, cache_control: { type: "ephemeral" } }]
1507
+ };
1508
+ }
1509
+ }
1510
+ }
1470
1511
  function buildAnthropicRequest(opts) {
1471
1512
  let system;
1472
1513
  const convo = [];
@@ -1491,6 +1532,7 @@ function buildAnthropicRequest(opts) {
1491
1532
  if (system !== void 0) {
1492
1533
  body.system = system;
1493
1534
  }
1535
+ applyPromptCaching(body);
1494
1536
  if (isReasoningModel("anthropic", opts.model)) {
1495
1537
  if (opts.model.toLowerCase().startsWith("claude-fable-5")) {
1496
1538
  body.thinking = { type: "adaptive" };
@@ -1550,6 +1592,7 @@ function parseAnthropicResponse(opts) {
1550
1592
  const prompt = Math.trunc(Number(u["input_tokens"] ?? 0)) || 0;
1551
1593
  const cached = Math.trunc(Number(u["cache_read_input_tokens"] ?? 0)) || 0;
1552
1594
  const completion = Math.trunc(Number(u["output_tokens"] ?? 0)) || 0;
1595
+ const cacheWrite = Math.trunc(Number(u["cache_creation_input_tokens"] ?? 0)) || 0;
1553
1596
  const usage = {
1554
1597
  provider: "anthropic",
1555
1598
  model: opts.model,
@@ -1557,7 +1600,7 @@ function parseAnthropicResponse(opts) {
1557
1600
  // production helper folds them in so prompt_tokens is the FULL
1558
1601
  // input volume; calculateCost then bills the cached subset at the
1559
1602
  // cached rate (cached <= prompt_tokens, since prompt = input+cached).
1560
- prompt_tokens: prompt + cached,
1603
+ prompt_tokens: prompt + cached + cacheWrite,
1561
1604
  completion_tokens: completion,
1562
1605
  cached_tokens: cached,
1563
1606
  total_tokens: 0,
@@ -1909,16 +1952,21 @@ function parseOpenAICompatibleResponse(opts) {
1909
1952
  const u = r["usage"] ?? {};
1910
1953
  const details = u["prompt_tokens_details"] ?? {};
1911
1954
  const cached = Math.trunc(Number(details["cached_tokens"] ?? 0)) || 0;
1955
+ const promptTokens = Math.trunc(Number(u["prompt_tokens"] ?? 0)) || 0;
1956
+ const reportedCompletion = Math.trunc(Number(u["completion_tokens"] ?? 0)) || 0;
1957
+ const reportedTotal = Math.trunc(Number(u["total_tokens"] ?? 0)) || 0;
1958
+ const derivedCompletion = reportedTotal > promptTokens ? reportedTotal - promptTokens : reportedCompletion;
1959
+ const completionTokens = Math.max(reportedCompletion, derivedCompletion);
1912
1960
  const usage = {
1913
1961
  provider: opts.provider,
1914
1962
  model: opts.model,
1915
- prompt_tokens: Math.trunc(Number(u["prompt_tokens"] ?? 0)) || 0,
1916
- completion_tokens: Math.trunc(Number(u["completion_tokens"] ?? 0)) || 0,
1963
+ prompt_tokens: promptTokens,
1964
+ completion_tokens: completionTokens,
1917
1965
  cached_tokens: cached,
1918
1966
  // Use the response's reported total_tokens directly. Python's
1919
1967
  // `Usage.to_dict()` falls back to prompt+completion only when
1920
1968
  // total_tokens is 0/missing; mirror that.
1921
- total_tokens: Math.trunc(Number(u["total_tokens"] ?? 0)) || 0,
1969
+ total_tokens: reportedTotal,
1922
1970
  cost_usd: 0,
1923
1971
  estimated: Object.keys(u).length === 0,
1924
1972
  purpose: opts.purpose
@@ -1960,7 +2008,11 @@ async function sendOpenAICompatible(args) {
1960
2008
  body.reasoning_effort = effort;
1961
2009
  }
1962
2010
  }
1963
- if (isReasoningModel(args.provider, args.model) && typeof body.max_completion_tokens === "number") {
2011
+ if (isReasoningModel(args.provider, args.model) && // Only where the cap is a SHARED budget. On a provider that bounds visible
2012
+ // output only (xAI), headroom cannot protect the answer from being crowded
2013
+ // out — nothing is crowding it — so it would merely authorise a
2014
+ // 25,000-token reply and the bill that comes with it.
2015
+ reasoningSharesOutputBudget(args.provider) && typeof body.max_completion_tokens === "number") {
1964
2016
  const raw = Number(process.env["CROSSCHECK_OPENAI_REASONING_HEADROOM_TOKENS"]);
1965
2017
  const headroom = Number.isFinite(raw) && raw >= 0 ? Math.trunc(raw) : 25e3;
1966
2018
  body.max_completion_tokens += headroom;
@@ -2279,7 +2331,7 @@ import { z } from "zod";
2279
2331
  // src/server-meta.ts
2280
2332
  init_esm_shims();
2281
2333
  var SERVER_NAME = "crosscheck-agent";
2282
- var SERVER_VERSION = true ? "0.2.21" : "0.0.0-dev";
2334
+ var SERVER_VERSION = true ? "0.2.23" : "0.0.0-dev";
2283
2335
 
2284
2336
  // src/tools/audit.ts
2285
2337
  init_esm_shims();
@@ -10146,6 +10198,66 @@ function isObj6(v) {
10146
10198
 
10147
10199
  // src/tools/debate.ts
10148
10200
  init_esm_shims();
10201
+
10202
+ // src/core/debate-compaction.ts
10203
+ init_esm_shims();
10204
+ var DEFAULT_PRIOR_BUDGET_CHARS = 12e3;
10205
+ function renderTurn(e) {
10206
+ return `[${e.provider} \u2014 round ${e.round}]
10207
+ ${e.response ?? "(error)"}`;
10208
+ }
10209
+ function buildPriorTurns(transcript, opts = {}) {
10210
+ if (transcript.length === 0) {
10211
+ return { body: "", includedTurns: 0, omittedTurns: 0, chars: 0 };
10212
+ }
10213
+ const rendered = transcript.map(renderTurn);
10214
+ const full = rendered.join("\n\n");
10215
+ if (!opts.compress) {
10216
+ return { body: full, includedTurns: transcript.length, omittedTurns: 0, chars: full.length };
10217
+ }
10218
+ const budget = Math.max(0, opts.budgetChars ?? DEFAULT_PRIOR_BUDGET_CHARS);
10219
+ if (full.length <= budget) {
10220
+ return { body: full, includedTurns: transcript.length, omittedTurns: 0, chars: full.length };
10221
+ }
10222
+ const latestRound = Math.max(...transcript.map((e) => e.round));
10223
+ const keep = /* @__PURE__ */ new Set();
10224
+ let spent = 0;
10225
+ for (let i = transcript.length - 1; i >= 0; i -= 1) {
10226
+ const isLatest = transcript[i].round === latestRound;
10227
+ const cost = rendered[i].length + 2;
10228
+ if (isLatest) {
10229
+ keep.add(i);
10230
+ spent += cost;
10231
+ continue;
10232
+ }
10233
+ if (spent + cost > budget) continue;
10234
+ keep.add(i);
10235
+ spent += cost;
10236
+ }
10237
+ const parts = [];
10238
+ let omitted = 0;
10239
+ let pendingGap = 0;
10240
+ for (let i = 0; i < transcript.length; i += 1) {
10241
+ if (keep.has(i)) {
10242
+ if (pendingGap > 0) {
10243
+ parts.push(`[${pendingGap} earlier turn(s) omitted for length]`);
10244
+ pendingGap = 0;
10245
+ }
10246
+ parts.push(rendered[i]);
10247
+ } else {
10248
+ omitted += 1;
10249
+ pendingGap += 1;
10250
+ }
10251
+ }
10252
+ if (pendingGap > 0) parts.push(`[${pendingGap} earlier turn(s) omitted for length]`);
10253
+ const body = parts.join("\n\n");
10254
+ return { body, includedTurns: keep.size, omittedTurns: omitted, chars: body.length };
10255
+ }
10256
+ function buildModeratorTranscript(transcript) {
10257
+ return transcript.map(renderTurn).join("\n\n");
10258
+ }
10259
+
10260
+ // src/tools/debate.ts
10149
10261
  import { performance as performance10 } from "perf_hooks";
10150
10262
  var DEFERRED_OPTS4 = [
10151
10263
  // empty — every debate opt runs natively when its deps
@@ -10193,6 +10305,9 @@ async function runDebate(args, opts) {
10193
10305
  1,
10194
10306
  Math.trunc(Number(args["max_rounds"] ?? 3)) || 3
10195
10307
  );
10308
+ const compressRounds = args["compress_rounds"] === true;
10309
+ const priorBudgetChars = Math.trunc(Number(args["prior_budget_chars"] ?? DEFAULT_PRIOR_BUDGET_CHARS)) || DEFAULT_PRIOR_BUDGET_CHARS;
10310
+ let compactionMeta = null;
10196
10311
  const sessionId = typeof args["session_id"] === "string" ? args["session_id"] : null;
10197
10312
  const breakerEnv = await maybeBreakerEnvelope(
10198
10313
  opts.storage,
@@ -10303,7 +10418,7 @@ async function runDebate(args, opts) {
10303
10418
  const roundMessages = [
10304
10419
  {
10305
10420
  role: "system",
10306
- content: `You are debating peers from other model families. Round ${rnd}/${maxRounds}. Disagree where warranted, concede where right, and keep replies short and specific.`
10421
+ content: "You are debating peers from other model families. Disagree where warranted, concede where right, and keep replies short and specific."
10307
10422
  }
10308
10423
  ];
10309
10424
  if (context) {
@@ -10315,13 +10430,22 @@ ${context}` });
10315
10430
  TOPIC: ${topic}` : `TOPIC: ${topic}`;
10316
10431
  roundMessages.push({ role: "user", content: topicLine });
10317
10432
  if (transcript.length > 0) {
10318
- const prior = transcript.map(
10319
- (e) => `[${e.provider} \u2014 round ${e.round}]
10320
- ${e.response ?? "(error)"}`
10321
- ).join("\n\n");
10433
+ const prior = buildPriorTurns(transcript, {
10434
+ compress: compressRounds,
10435
+ budgetChars: priorBudgetChars
10436
+ });
10437
+ if (prior.omittedTurns > 0) compactionMeta = {
10438
+ omitted_turns: prior.omittedTurns,
10439
+ included_turns: prior.includedTurns,
10440
+ budget_chars: priorBudgetChars
10441
+ };
10322
10442
  roundMessages.push({ role: "user", content: `PRIOR TURNS:
10323
- ${prior}` });
10443
+ ${prior.body}` });
10324
10444
  }
10445
+ roundMessages.push({
10446
+ role: "user",
10447
+ content: `This is round ${rnd} of ${maxRounds}. Reply for this round.`
10448
+ });
10325
10449
  for (const p of selected) {
10326
10450
  const entry = await callPanelist(p, roundMessages);
10327
10451
  const withRound = { ...entry, round: rnd };
@@ -10369,10 +10493,7 @@ ${prior}` });
10369
10493
  let coReasoned = false;
10370
10494
  if (moderator) {
10371
10495
  const synthProvider = superRequested && opts.providers["anthropic"] ? retargetForSuper(opts.providers["anthropic"]) : upgraded ? retargetProvider(opts.providers["anthropic"], UPGRADE_MODEL) : moderator;
10372
- const condensed = transcript.map(
10373
- (e) => `[${e.provider} \u2014 round ${e.round}]
10374
- ${e.response ?? "(error)"}`
10375
- ).join("\n\n");
10496
+ const condensed = buildModeratorTranscript(transcript);
10376
10497
  const persona = buildPersonaInjection({
10377
10498
  toolName: "debate",
10378
10499
  prompt: topic,
@@ -10455,6 +10576,7 @@ ${condensed}`
10455
10576
  synthesis
10456
10577
  };
10457
10578
  if (personaMeta && personaMeta.used !== null) result["persona"] = personaMeta;
10579
+ if (compactionMeta) result["round_compaction"] = compactionMeta;
10458
10580
  if (upgraded) {
10459
10581
  result["reasoning_upgrade"] = {
10460
10582
  applied: true,
@@ -12788,7 +12910,7 @@ var CHECK_INTERVAL_SECONDS = 3 * 24 * 60 * 60;
12788
12910
  var DEFAULT_PACKAGE = "crosscheck-cli";
12789
12911
  var FETCH_TIMEOUT_MS = 3e3;
12790
12912
  function engineVersion() {
12791
- return true ? "0.2.21" : "0.0.0-dev";
12913
+ return true ? "0.2.23" : "0.0.0-dev";
12792
12914
  }
12793
12915
  function defaultUpdateCachePath() {
12794
12916
  const base = process.env["CROSSCHECK_DATA_DIR"] || path10.join(os.homedir() || os.tmpdir(), ".crosscheck");
@@ -14832,9 +14954,7 @@ function usageEventsFromEnvelope(out, toolName) {
14832
14954
  const timing = out["timing"];
14833
14955
  const timingByCall = timing && typeof timing === "object" ? timing["by_call"] : void 0;
14834
14956
  const allTiming = Array.isArray(timingByCall) ? timingByCall : [];
14835
- const latencies = allTiming.filter(
14836
- (t) => !t?.["error_kind"]
14837
- );
14957
+ const latencies = allTiming;
14838
14958
  const aligned = latencies.length === byCall.length;
14839
14959
  const pattern = toolToPattern(toolName);
14840
14960
  const events = [];
@@ -14844,13 +14964,18 @@ function usageEventsFromEnvelope(out, toolName) {
14844
14964
  if (!provider) continue;
14845
14965
  const model = String(u.model ?? "").slice(0, 128) || "unknown";
14846
14966
  const cost = Math.max(0, Number(u.cost_usd) || 0);
14847
- const latencyMs = aligned ? intNonNeg(latencies[i]?.["wall_ms"]) : 0;
14967
+ const timingRow = aligned ? latencies[i] : void 0;
14968
+ if (timingRow?.["error_kind"]) continue;
14969
+ const promptTokens = intNonNeg(u.prompt_tokens);
14970
+ const completionTokens = intNonNeg(u.completion_tokens);
14971
+ if (promptTokens === 0 && completionTokens === 0) continue;
14972
+ const latencyMs = intNonNeg(timingRow?.["wall_ms"]);
14848
14973
  events.push({
14849
14974
  provider,
14850
14975
  model,
14851
14976
  pattern,
14852
- promptTokens: intNonNeg(u.prompt_tokens),
14853
- completionTokens: intNonNeg(u.completion_tokens),
14977
+ promptTokens,
14978
+ completionTokens,
14854
14979
  costUsdEstimate: cost,
14855
14980
  latencyMs,
14856
14981
  status: "ok",