crosscheck-mcp 0.2.21 → 0.2.23

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -413,7 +413,27 @@ var PROVIDER_CAPS = {
413
413
  // temperature, so the prefix is the fix.
414
414
  reasoning_prefixes: ["gpt-5", "gpt-6", "o1", "o3", "o4"]
415
415
  },
416
- xai: { family: "openai_chat", system_role: "inline", supports_temperature: true },
416
+ xai: {
417
+ family: "openai_chat",
418
+ system_role: "inline",
419
+ // Kept `true`, unlike other reasoning models: grok-4 accepts temperature
420
+ // (verified 200 at 0.4 and 1.0) and a panel wants the variance. Dropping
421
+ // it would buy nothing and cost diversity.
422
+ supports_temperature: true,
423
+ // grok-4 IS reasoning-class — it emitted 1,432 reasoning tokens against 64
424
+ // visible on one call. Saying so strips the "think step by step" preambles
425
+ // it does not need, and lifts it off the non-reasoning ceiling (1500) onto
426
+ // the reasoning-safe one (2048).
427
+ //
428
+ // NOT the reason grok answers short, which was the initial guess and was
429
+ // wrong: at caps of 1500, 2048 and 6144 it returned 445, 404 and 461
430
+ // visible tokens, finish_reason=stop every time. The cap was never
431
+ // binding; grok is simply terse. The ceiling change removes a latent
432
+ // constraint, it does not make grok say more.
433
+ reasoning_prefixes: ["grok-4"],
434
+ // Measured: xAI's cap bounds visible output only. See the field docs.
435
+ reasoning_shares_output_budget: false
436
+ },
417
437
  mistral: { family: "openai_chat", system_role: "inline", supports_temperature: true },
418
438
  groq: { family: "openai_chat", system_role: "inline", supports_temperature: true },
419
439
  deepseek: { family: "openai_chat", system_role: "inline", supports_temperature: true },
@@ -452,6 +472,10 @@ function supportsTemperature(provider, model) {
452
472
  if (caps.supports_temperature === "model") return !isReasoningModel(provider, model);
453
473
  return Boolean(caps.supports_temperature);
454
474
  }
475
+ function reasoningSharesOutputBudget(provider) {
476
+ const caps = PROVIDER_CAPS[provider.toLowerCase()];
477
+ return caps?.reasoning_shares_output_budget ?? true;
478
+ }
455
479
 
456
480
  // src/providers/types.ts
457
481
  var ProviderError = class extends Error {
@@ -693,6 +717,23 @@ async function acquireRateLimit(provider, deps) {
693
717
  var ANTHROPIC_API_URL = "https://api.anthropic.com/v1/messages";
694
718
  var ANTHROPIC_VERSION_HEADER = "2023-06-01";
695
719
  var ANTHROPIC_STRUCTURED_TOOL_NAME = "structured_output";
720
+ function applyPromptCaching(body, env = process.env) {
721
+ const flag = (env["CROSSCHECK_ANTHROPIC_PROMPT_CACHE"] ?? "").trim();
722
+ if (flag === "0" || flag.toLowerCase() === "false") return;
723
+ if (typeof body.system === "string" && body.system !== "") {
724
+ body.system = [{ type: "text", text: body.system, cache_control: { type: "ephemeral" } }];
725
+ }
726
+ if (body.messages.length >= 2) {
727
+ const idx = body.messages.length - 2;
728
+ const m = body.messages[idx];
729
+ if (typeof m.content === "string" && m.content !== "") {
730
+ body.messages[idx] = {
731
+ role: m.role,
732
+ content: [{ type: "text", text: m.content, cache_control: { type: "ephemeral" } }]
733
+ };
734
+ }
735
+ }
736
+ }
696
737
  function buildAnthropicRequest(opts) {
697
738
  let system;
698
739
  const convo = [];
@@ -717,6 +758,7 @@ function buildAnthropicRequest(opts) {
717
758
  if (system !== void 0) {
718
759
  body.system = system;
719
760
  }
761
+ applyPromptCaching(body);
720
762
  if (isReasoningModel("anthropic", opts.model)) {
721
763
  if (opts.model.toLowerCase().startsWith("claude-fable-5")) {
722
764
  body.thinking = { type: "adaptive" };
@@ -776,6 +818,7 @@ function parseAnthropicResponse(opts) {
776
818
  const prompt = Math.trunc(Number(u["input_tokens"] ?? 0)) || 0;
777
819
  const cached = Math.trunc(Number(u["cache_read_input_tokens"] ?? 0)) || 0;
778
820
  const completion = Math.trunc(Number(u["output_tokens"] ?? 0)) || 0;
821
+ const cacheWrite = Math.trunc(Number(u["cache_creation_input_tokens"] ?? 0)) || 0;
779
822
  const usage = {
780
823
  provider: "anthropic",
781
824
  model: opts.model,
@@ -783,7 +826,7 @@ function parseAnthropicResponse(opts) {
783
826
  // production helper folds them in so prompt_tokens is the FULL
784
827
  // input volume; calculateCost then bills the cached subset at the
785
828
  // cached rate (cached <= prompt_tokens, since prompt = input+cached).
786
- prompt_tokens: prompt + cached,
829
+ prompt_tokens: prompt + cached + cacheWrite,
787
830
  completion_tokens: completion,
788
831
  cached_tokens: cached,
789
832
  total_tokens: 0,
@@ -1133,16 +1176,21 @@ function parseOpenAICompatibleResponse(opts) {
1133
1176
  const u = r["usage"] ?? {};
1134
1177
  const details = u["prompt_tokens_details"] ?? {};
1135
1178
  const cached = Math.trunc(Number(details["cached_tokens"] ?? 0)) || 0;
1179
+ const promptTokens = Math.trunc(Number(u["prompt_tokens"] ?? 0)) || 0;
1180
+ const reportedCompletion = Math.trunc(Number(u["completion_tokens"] ?? 0)) || 0;
1181
+ const reportedTotal = Math.trunc(Number(u["total_tokens"] ?? 0)) || 0;
1182
+ const derivedCompletion = reportedTotal > promptTokens ? reportedTotal - promptTokens : reportedCompletion;
1183
+ const completionTokens = Math.max(reportedCompletion, derivedCompletion);
1136
1184
  const usage = {
1137
1185
  provider: opts.provider,
1138
1186
  model: opts.model,
1139
- prompt_tokens: Math.trunc(Number(u["prompt_tokens"] ?? 0)) || 0,
1140
- completion_tokens: Math.trunc(Number(u["completion_tokens"] ?? 0)) || 0,
1187
+ prompt_tokens: promptTokens,
1188
+ completion_tokens: completionTokens,
1141
1189
  cached_tokens: cached,
1142
1190
  // Use the response's reported total_tokens directly. Python's
1143
1191
  // `Usage.to_dict()` falls back to prompt+completion only when
1144
1192
  // total_tokens is 0/missing; mirror that.
1145
- total_tokens: Math.trunc(Number(u["total_tokens"] ?? 0)) || 0,
1193
+ total_tokens: reportedTotal,
1146
1194
  cost_usd: 0,
1147
1195
  estimated: Object.keys(u).length === 0,
1148
1196
  purpose: opts.purpose
@@ -1184,7 +1232,11 @@ async function sendOpenAICompatible(args) {
1184
1232
  body.reasoning_effort = effort;
1185
1233
  }
1186
1234
  }
1187
- if (isReasoningModel(args.provider, args.model) && typeof body.max_completion_tokens === "number") {
1235
+ if (isReasoningModel(args.provider, args.model) && // Only where the cap is a SHARED budget. On a provider that bounds visible
1236
+ // output only (xAI), headroom cannot protect the answer from being crowded
1237
+ // out — nothing is crowding it — so it would merely authorise a
1238
+ // 25,000-token reply and the bill that comes with it.
1239
+ reasoningSharesOutputBudget(args.provider) && typeof body.max_completion_tokens === "number") {
1188
1240
  const raw = Number(process.env["CROSSCHECK_OPENAI_REASONING_HEADROOM_TOKENS"]);
1189
1241
  const headroom = Number.isFinite(raw) && raw >= 0 ? Math.trunc(raw) : 25e3;
1190
1242
  body.max_completion_tokens += headroom;
@@ -1359,7 +1411,7 @@ import { z } from "zod";
1359
1411
 
1360
1412
  // src/server-meta.ts
1361
1413
  var SERVER_NAME = "crosscheck-agent";
1362
- var SERVER_VERSION = true ? "0.2.21" : "0.0.0-dev";
1414
+ var SERVER_VERSION = true ? "0.2.23" : "0.0.0-dev";
1363
1415
 
1364
1416
  // src/tools/audit.ts
1365
1417
  import { readdirSync, readFileSync as readFileSync3, statSync } from "fs";
@@ -9153,6 +9205,63 @@ function isObj6(v) {
9153
9205
  return typeof v === "object" && v !== null && !Array.isArray(v);
9154
9206
  }
9155
9207
 
9208
+ // src/core/debate-compaction.ts
9209
+ var DEFAULT_PRIOR_BUDGET_CHARS = 12e3;
9210
+ function renderTurn(e) {
9211
+ return `[${e.provider} \u2014 round ${e.round}]
9212
+ ${e.response ?? "(error)"}`;
9213
+ }
9214
+ function buildPriorTurns(transcript, opts = {}) {
9215
+ if (transcript.length === 0) {
9216
+ return { body: "", includedTurns: 0, omittedTurns: 0, chars: 0 };
9217
+ }
9218
+ const rendered = transcript.map(renderTurn);
9219
+ const full = rendered.join("\n\n");
9220
+ if (!opts.compress) {
9221
+ return { body: full, includedTurns: transcript.length, omittedTurns: 0, chars: full.length };
9222
+ }
9223
+ const budget = Math.max(0, opts.budgetChars ?? DEFAULT_PRIOR_BUDGET_CHARS);
9224
+ if (full.length <= budget) {
9225
+ return { body: full, includedTurns: transcript.length, omittedTurns: 0, chars: full.length };
9226
+ }
9227
+ const latestRound = Math.max(...transcript.map((e) => e.round));
9228
+ const keep = /* @__PURE__ */ new Set();
9229
+ let spent = 0;
9230
+ for (let i = transcript.length - 1; i >= 0; i -= 1) {
9231
+ const isLatest = transcript[i].round === latestRound;
9232
+ const cost = rendered[i].length + 2;
9233
+ if (isLatest) {
9234
+ keep.add(i);
9235
+ spent += cost;
9236
+ continue;
9237
+ }
9238
+ if (spent + cost > budget) continue;
9239
+ keep.add(i);
9240
+ spent += cost;
9241
+ }
9242
+ const parts = [];
9243
+ let omitted = 0;
9244
+ let pendingGap = 0;
9245
+ for (let i = 0; i < transcript.length; i += 1) {
9246
+ if (keep.has(i)) {
9247
+ if (pendingGap > 0) {
9248
+ parts.push(`[${pendingGap} earlier turn(s) omitted for length]`);
9249
+ pendingGap = 0;
9250
+ }
9251
+ parts.push(rendered[i]);
9252
+ } else {
9253
+ omitted += 1;
9254
+ pendingGap += 1;
9255
+ }
9256
+ }
9257
+ if (pendingGap > 0) parts.push(`[${pendingGap} earlier turn(s) omitted for length]`);
9258
+ const body = parts.join("\n\n");
9259
+ return { body, includedTurns: keep.size, omittedTurns: omitted, chars: body.length };
9260
+ }
9261
+ function buildModeratorTranscript(transcript) {
9262
+ return transcript.map(renderTurn).join("\n\n");
9263
+ }
9264
+
9156
9265
  // src/tools/debate.ts
9157
9266
  import { performance as performance10 } from "perf_hooks";
9158
9267
  var DEFERRED_OPTS4 = [
@@ -9201,6 +9310,9 @@ async function runDebate(args, opts) {
9201
9310
  1,
9202
9311
  Math.trunc(Number(args["max_rounds"] ?? 3)) || 3
9203
9312
  );
9313
+ const compressRounds = args["compress_rounds"] === true;
9314
+ const priorBudgetChars = Math.trunc(Number(args["prior_budget_chars"] ?? DEFAULT_PRIOR_BUDGET_CHARS)) || DEFAULT_PRIOR_BUDGET_CHARS;
9315
+ let compactionMeta = null;
9204
9316
  const sessionId = typeof args["session_id"] === "string" ? args["session_id"] : null;
9205
9317
  const breakerEnv = await maybeBreakerEnvelope(
9206
9318
  opts.storage,
@@ -9311,7 +9423,7 @@ async function runDebate(args, opts) {
9311
9423
  const roundMessages = [
9312
9424
  {
9313
9425
  role: "system",
9314
- content: `You are debating peers from other model families. Round ${rnd}/${maxRounds}. Disagree where warranted, concede where right, and keep replies short and specific.`
9426
+ content: "You are debating peers from other model families. Disagree where warranted, concede where right, and keep replies short and specific."
9315
9427
  }
9316
9428
  ];
9317
9429
  if (context) {
@@ -9323,13 +9435,22 @@ ${context}` });
9323
9435
  TOPIC: ${topic}` : `TOPIC: ${topic}`;
9324
9436
  roundMessages.push({ role: "user", content: topicLine });
9325
9437
  if (transcript.length > 0) {
9326
- const prior = transcript.map(
9327
- (e) => `[${e.provider} \u2014 round ${e.round}]
9328
- ${e.response ?? "(error)"}`
9329
- ).join("\n\n");
9438
+ const prior = buildPriorTurns(transcript, {
9439
+ compress: compressRounds,
9440
+ budgetChars: priorBudgetChars
9441
+ });
9442
+ if (prior.omittedTurns > 0) compactionMeta = {
9443
+ omitted_turns: prior.omittedTurns,
9444
+ included_turns: prior.includedTurns,
9445
+ budget_chars: priorBudgetChars
9446
+ };
9330
9447
  roundMessages.push({ role: "user", content: `PRIOR TURNS:
9331
- ${prior}` });
9448
+ ${prior.body}` });
9332
9449
  }
9450
+ roundMessages.push({
9451
+ role: "user",
9452
+ content: `This is round ${rnd} of ${maxRounds}. Reply for this round.`
9453
+ });
9333
9454
  for (const p of selected) {
9334
9455
  const entry = await callPanelist(p, roundMessages);
9335
9456
  const withRound = { ...entry, round: rnd };
@@ -9377,10 +9498,7 @@ ${prior}` });
9377
9498
  let coReasoned = false;
9378
9499
  if (moderator) {
9379
9500
  const synthProvider = superRequested && opts.providers["anthropic"] ? retargetForSuper(opts.providers["anthropic"]) : upgraded ? retargetProvider(opts.providers["anthropic"], UPGRADE_MODEL) : moderator;
9380
- const condensed = transcript.map(
9381
- (e) => `[${e.provider} \u2014 round ${e.round}]
9382
- ${e.response ?? "(error)"}`
9383
- ).join("\n\n");
9501
+ const condensed = buildModeratorTranscript(transcript);
9384
9502
  const persona = buildPersonaInjection({
9385
9503
  toolName: "debate",
9386
9504
  prompt: topic,
@@ -9463,6 +9581,7 @@ ${condensed}`
9463
9581
  synthesis
9464
9582
  };
9465
9583
  if (personaMeta && personaMeta.used !== null) result["persona"] = personaMeta;
9584
+ if (compactionMeta) result["round_compaction"] = compactionMeta;
9466
9585
  if (upgraded) {
9467
9586
  result["reasoning_upgrade"] = {
9468
9587
  applied: true,
@@ -11782,7 +11901,7 @@ var CHECK_INTERVAL_SECONDS = 3 * 24 * 60 * 60;
11782
11901
  var DEFAULT_PACKAGE = "crosscheck-cli";
11783
11902
  var FETCH_TIMEOUT_MS = 3e3;
11784
11903
  function engineVersion() {
11785
- return true ? "0.2.21" : "0.0.0-dev";
11904
+ return true ? "0.2.23" : "0.0.0-dev";
11786
11905
  }
11787
11906
  function defaultUpdateCachePath() {
11788
11907
  const base = process.env["CROSSCHECK_DATA_DIR"] || path9.join(os.homedir() || os.tmpdir(), ".crosscheck");