crosscheck-mcp 0.2.21 → 0.2.23

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -446,7 +446,27 @@ var PROVIDER_CAPS = {
446
446
  // temperature, so the prefix is the fix.
447
447
  reasoning_prefixes: ["gpt-5", "gpt-6", "o1", "o3", "o4"]
448
448
  },
449
- xai: { family: "openai_chat", system_role: "inline", supports_temperature: true },
449
+ xai: {
450
+ family: "openai_chat",
451
+ system_role: "inline",
452
+ // Kept `true`, unlike other reasoning models: grok-4 accepts temperature
453
+ // (verified 200 at 0.4 and 1.0) and a panel wants the variance. Dropping
454
+ // it would buy nothing and cost diversity.
455
+ supports_temperature: true,
456
+ // grok-4 IS reasoning-class — it emitted 1,432 reasoning tokens against 64
457
+ // visible on one call. Saying so strips the "think step by step" preambles
458
+ // it does not need, and lifts it off the non-reasoning ceiling (1500) onto
459
+ // the reasoning-safe one (2048).
460
+ //
461
+ // NOT the reason grok answers short, which was the initial guess and was
462
+ // wrong: at caps of 1500, 2048 and 6144 it returned 445, 404 and 461
463
+ // visible tokens, finish_reason=stop every time. The cap was never
464
+ // binding; grok is simply terse. The ceiling change removes a latent
465
+ // constraint, it does not make grok say more.
466
+ reasoning_prefixes: ["grok-4"],
467
+ // Measured: xAI's cap bounds visible output only. See the field docs.
468
+ reasoning_shares_output_budget: false
469
+ },
450
470
  mistral: { family: "openai_chat", system_role: "inline", supports_temperature: true },
451
471
  groq: { family: "openai_chat", system_role: "inline", supports_temperature: true },
452
472
  deepseek: { family: "openai_chat", system_role: "inline", supports_temperature: true },
@@ -485,6 +505,10 @@ function supportsTemperature(provider, model) {
485
505
  if (caps.supports_temperature === "model") return !isReasoningModel(provider, model);
486
506
  return Boolean(caps.supports_temperature);
487
507
  }
508
+ function reasoningSharesOutputBudget(provider) {
509
+ const caps = PROVIDER_CAPS[provider.toLowerCase()];
510
+ return caps?.reasoning_shares_output_budget ?? true;
511
+ }
488
512
 
489
513
  // src/providers/types.ts
490
514
  var ProviderError = class extends Error {
@@ -726,6 +750,23 @@ async function acquireRateLimit(provider, deps) {
726
750
  var ANTHROPIC_API_URL = "https://api.anthropic.com/v1/messages";
727
751
  var ANTHROPIC_VERSION_HEADER = "2023-06-01";
728
752
  var ANTHROPIC_STRUCTURED_TOOL_NAME = "structured_output";
753
+ function applyPromptCaching(body, env = process.env) {
754
+ const flag = (env["CROSSCHECK_ANTHROPIC_PROMPT_CACHE"] ?? "").trim();
755
+ if (flag === "0" || flag.toLowerCase() === "false") return;
756
+ if (typeof body.system === "string" && body.system !== "") {
757
+ body.system = [{ type: "text", text: body.system, cache_control: { type: "ephemeral" } }];
758
+ }
759
+ if (body.messages.length >= 2) {
760
+ const idx = body.messages.length - 2;
761
+ const m = body.messages[idx];
762
+ if (typeof m.content === "string" && m.content !== "") {
763
+ body.messages[idx] = {
764
+ role: m.role,
765
+ content: [{ type: "text", text: m.content, cache_control: { type: "ephemeral" } }]
766
+ };
767
+ }
768
+ }
769
+ }
729
770
  function buildAnthropicRequest(opts) {
730
771
  let system;
731
772
  const convo = [];
@@ -750,6 +791,7 @@ function buildAnthropicRequest(opts) {
750
791
  if (system !== void 0) {
751
792
  body.system = system;
752
793
  }
794
+ applyPromptCaching(body);
753
795
  if (isReasoningModel("anthropic", opts.model)) {
754
796
  if (opts.model.toLowerCase().startsWith("claude-fable-5")) {
755
797
  body.thinking = { type: "adaptive" };
@@ -809,6 +851,7 @@ function parseAnthropicResponse(opts) {
809
851
  const prompt = Math.trunc(Number(u["input_tokens"] ?? 0)) || 0;
810
852
  const cached = Math.trunc(Number(u["cache_read_input_tokens"] ?? 0)) || 0;
811
853
  const completion = Math.trunc(Number(u["output_tokens"] ?? 0)) || 0;
854
+ const cacheWrite = Math.trunc(Number(u["cache_creation_input_tokens"] ?? 0)) || 0;
812
855
  const usage = {
813
856
  provider: "anthropic",
814
857
  model: opts.model,
@@ -816,7 +859,7 @@ function parseAnthropicResponse(opts) {
816
859
  // production helper folds them in so prompt_tokens is the FULL
817
860
  // input volume; calculateCost then bills the cached subset at the
818
861
  // cached rate (cached <= prompt_tokens, since prompt = input+cached).
819
- prompt_tokens: prompt + cached,
862
+ prompt_tokens: prompt + cached + cacheWrite,
820
863
  completion_tokens: completion,
821
864
  cached_tokens: cached,
822
865
  total_tokens: 0,
@@ -1166,16 +1209,21 @@ function parseOpenAICompatibleResponse(opts) {
1166
1209
  const u = r["usage"] ?? {};
1167
1210
  const details = u["prompt_tokens_details"] ?? {};
1168
1211
  const cached = Math.trunc(Number(details["cached_tokens"] ?? 0)) || 0;
1212
+ const promptTokens = Math.trunc(Number(u["prompt_tokens"] ?? 0)) || 0;
1213
+ const reportedCompletion = Math.trunc(Number(u["completion_tokens"] ?? 0)) || 0;
1214
+ const reportedTotal = Math.trunc(Number(u["total_tokens"] ?? 0)) || 0;
1215
+ const derivedCompletion = reportedTotal > promptTokens ? reportedTotal - promptTokens : reportedCompletion;
1216
+ const completionTokens = Math.max(reportedCompletion, derivedCompletion);
1169
1217
  const usage = {
1170
1218
  provider: opts.provider,
1171
1219
  model: opts.model,
1172
- prompt_tokens: Math.trunc(Number(u["prompt_tokens"] ?? 0)) || 0,
1173
- completion_tokens: Math.trunc(Number(u["completion_tokens"] ?? 0)) || 0,
1220
+ prompt_tokens: promptTokens,
1221
+ completion_tokens: completionTokens,
1174
1222
  cached_tokens: cached,
1175
1223
  // Use the response's reported total_tokens directly. Python's
1176
1224
  // `Usage.to_dict()` falls back to prompt+completion only when
1177
1225
  // total_tokens is 0/missing; mirror that.
1178
- total_tokens: Math.trunc(Number(u["total_tokens"] ?? 0)) || 0,
1226
+ total_tokens: reportedTotal,
1179
1227
  cost_usd: 0,
1180
1228
  estimated: Object.keys(u).length === 0,
1181
1229
  purpose: opts.purpose
@@ -1217,7 +1265,11 @@ async function sendOpenAICompatible(args) {
1217
1265
  body.reasoning_effort = effort;
1218
1266
  }
1219
1267
  }
1220
- if (isReasoningModel(args.provider, args.model) && typeof body.max_completion_tokens === "number") {
1268
+ if (isReasoningModel(args.provider, args.model) && // Only where the cap is a SHARED budget. On a provider that bounds visible
1269
+ // output only (xAI), headroom cannot protect the answer from being crowded
1270
+ // out — nothing is crowding it — so it would merely authorise a
1271
+ // 25,000-token reply and the bill that comes with it.
1272
+ reasoningSharesOutputBudget(args.provider) && typeof body.max_completion_tokens === "number") {
1221
1273
  const raw = Number(process.env["CROSSCHECK_OPENAI_REASONING_HEADROOM_TOKENS"]);
1222
1274
  const headroom = Number.isFinite(raw) && raw >= 0 ? Math.trunc(raw) : 25e3;
1223
1275
  body.max_completion_tokens += headroom;
@@ -1392,7 +1444,7 @@ var import_zod = require("zod");
1392
1444
 
1393
1445
  // src/server-meta.ts
1394
1446
  var SERVER_NAME = "crosscheck-agent";
1395
- var SERVER_VERSION = true ? "0.2.21" : "0.0.0-dev";
1447
+ var SERVER_VERSION = true ? "0.2.23" : "0.0.0-dev";
1396
1448
 
1397
1449
  // src/tools/audit.ts
1398
1450
  var import_node_fs4 = require("fs");
@@ -9169,6 +9221,63 @@ function isObj6(v) {
9169
9221
  return typeof v === "object" && v !== null && !Array.isArray(v);
9170
9222
  }
9171
9223
 
9224
+ // src/core/debate-compaction.ts
9225
+ var DEFAULT_PRIOR_BUDGET_CHARS = 12e3;
9226
+ function renderTurn(e) {
9227
+ return `[${e.provider} \u2014 round ${e.round}]
9228
+ ${e.response ?? "(error)"}`;
9229
+ }
9230
+ function buildPriorTurns(transcript, opts = {}) {
9231
+ if (transcript.length === 0) {
9232
+ return { body: "", includedTurns: 0, omittedTurns: 0, chars: 0 };
9233
+ }
9234
+ const rendered = transcript.map(renderTurn);
9235
+ const full = rendered.join("\n\n");
9236
+ if (!opts.compress) {
9237
+ return { body: full, includedTurns: transcript.length, omittedTurns: 0, chars: full.length };
9238
+ }
9239
+ const budget = Math.max(0, opts.budgetChars ?? DEFAULT_PRIOR_BUDGET_CHARS);
9240
+ if (full.length <= budget) {
9241
+ return { body: full, includedTurns: transcript.length, omittedTurns: 0, chars: full.length };
9242
+ }
9243
+ const latestRound = Math.max(...transcript.map((e) => e.round));
9244
+ const keep = /* @__PURE__ */ new Set();
9245
+ let spent = 0;
9246
+ for (let i = transcript.length - 1; i >= 0; i -= 1) {
9247
+ const isLatest = transcript[i].round === latestRound;
9248
+ const cost = rendered[i].length + 2;
9249
+ if (isLatest) {
9250
+ keep.add(i);
9251
+ spent += cost;
9252
+ continue;
9253
+ }
9254
+ if (spent + cost > budget) continue;
9255
+ keep.add(i);
9256
+ spent += cost;
9257
+ }
9258
+ const parts = [];
9259
+ let omitted = 0;
9260
+ let pendingGap = 0;
9261
+ for (let i = 0; i < transcript.length; i += 1) {
9262
+ if (keep.has(i)) {
9263
+ if (pendingGap > 0) {
9264
+ parts.push(`[${pendingGap} earlier turn(s) omitted for length]`);
9265
+ pendingGap = 0;
9266
+ }
9267
+ parts.push(rendered[i]);
9268
+ } else {
9269
+ omitted += 1;
9270
+ pendingGap += 1;
9271
+ }
9272
+ }
9273
+ if (pendingGap > 0) parts.push(`[${pendingGap} earlier turn(s) omitted for length]`);
9274
+ const body = parts.join("\n\n");
9275
+ return { body, includedTurns: keep.size, omittedTurns: omitted, chars: body.length };
9276
+ }
9277
+ function buildModeratorTranscript(transcript) {
9278
+ return transcript.map(renderTurn).join("\n\n");
9279
+ }
9280
+
9172
9281
  // src/tools/debate.ts
9173
9282
  var import_node_perf_hooks9 = require("perf_hooks");
9174
9283
  var DEFERRED_OPTS4 = [
@@ -9217,6 +9326,9 @@ async function runDebate(args, opts) {
9217
9326
  1,
9218
9327
  Math.trunc(Number(args["max_rounds"] ?? 3)) || 3
9219
9328
  );
9329
+ const compressRounds = args["compress_rounds"] === true;
9330
+ const priorBudgetChars = Math.trunc(Number(args["prior_budget_chars"] ?? DEFAULT_PRIOR_BUDGET_CHARS)) || DEFAULT_PRIOR_BUDGET_CHARS;
9331
+ let compactionMeta = null;
9220
9332
  const sessionId = typeof args["session_id"] === "string" ? args["session_id"] : null;
9221
9333
  const breakerEnv = await maybeBreakerEnvelope(
9222
9334
  opts.storage,
@@ -9327,7 +9439,7 @@ async function runDebate(args, opts) {
9327
9439
  const roundMessages = [
9328
9440
  {
9329
9441
  role: "system",
9330
- content: `You are debating peers from other model families. Round ${rnd}/${maxRounds}. Disagree where warranted, concede where right, and keep replies short and specific.`
9442
+ content: "You are debating peers from other model families. Disagree where warranted, concede where right, and keep replies short and specific."
9331
9443
  }
9332
9444
  ];
9333
9445
  if (context) {
@@ -9339,13 +9451,22 @@ ${context}` });
9339
9451
  TOPIC: ${topic}` : `TOPIC: ${topic}`;
9340
9452
  roundMessages.push({ role: "user", content: topicLine });
9341
9453
  if (transcript.length > 0) {
9342
- const prior = transcript.map(
9343
- (e) => `[${e.provider} \u2014 round ${e.round}]
9344
- ${e.response ?? "(error)"}`
9345
- ).join("\n\n");
9454
+ const prior = buildPriorTurns(transcript, {
9455
+ compress: compressRounds,
9456
+ budgetChars: priorBudgetChars
9457
+ });
9458
+ if (prior.omittedTurns > 0) compactionMeta = {
9459
+ omitted_turns: prior.omittedTurns,
9460
+ included_turns: prior.includedTurns,
9461
+ budget_chars: priorBudgetChars
9462
+ };
9346
9463
  roundMessages.push({ role: "user", content: `PRIOR TURNS:
9347
- ${prior}` });
9464
+ ${prior.body}` });
9348
9465
  }
9466
+ roundMessages.push({
9467
+ role: "user",
9468
+ content: `This is round ${rnd} of ${maxRounds}. Reply for this round.`
9469
+ });
9349
9470
  for (const p of selected) {
9350
9471
  const entry = await callPanelist(p, roundMessages);
9351
9472
  const withRound = { ...entry, round: rnd };
@@ -9393,10 +9514,7 @@ ${prior}` });
9393
9514
  let coReasoned = false;
9394
9515
  if (moderator) {
9395
9516
  const synthProvider = superRequested && opts.providers["anthropic"] ? retargetForSuper(opts.providers["anthropic"]) : upgraded ? retargetProvider(opts.providers["anthropic"], UPGRADE_MODEL) : moderator;
9396
- const condensed = transcript.map(
9397
- (e) => `[${e.provider} \u2014 round ${e.round}]
9398
- ${e.response ?? "(error)"}`
9399
- ).join("\n\n");
9517
+ const condensed = buildModeratorTranscript(transcript);
9400
9518
  const persona = buildPersonaInjection({
9401
9519
  toolName: "debate",
9402
9520
  prompt: topic,
@@ -9479,6 +9597,7 @@ ${condensed}`
9479
9597
  synthesis
9480
9598
  };
9481
9599
  if (personaMeta && personaMeta.used !== null) result["persona"] = personaMeta;
9600
+ if (compactionMeta) result["round_compaction"] = compactionMeta;
9482
9601
  if (upgraded) {
9483
9602
  result["reasoning_upgrade"] = {
9484
9603
  applied: true,
@@ -11798,7 +11917,7 @@ var CHECK_INTERVAL_SECONDS = 3 * 24 * 60 * 60;
11798
11917
  var DEFAULT_PACKAGE = "crosscheck-cli";
11799
11918
  var FETCH_TIMEOUT_MS = 3e3;
11800
11919
  function engineVersion() {
11801
- return true ? "0.2.21" : "0.0.0-dev";
11920
+ return true ? "0.2.23" : "0.0.0-dev";
11802
11921
  }
11803
11922
  function defaultUpdateCachePath() {
11804
11923
  const base = process.env["CROSSCHECK_DATA_DIR"] || import_node_path12.default.join(import_node_os2.default.homedir() || import_node_os2.default.tmpdir(), ".crosscheck");