crosscheck-mcp 0.2.21 → 0.2.23

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1200,7 +1200,27 @@ var PROVIDER_CAPS = {
1200
1200
  // temperature, so the prefix is the fix.
1201
1201
  reasoning_prefixes: ["gpt-5", "gpt-6", "o1", "o3", "o4"]
1202
1202
  },
1203
- xai: { family: "openai_chat", system_role: "inline", supports_temperature: true },
1203
+ xai: {
1204
+ family: "openai_chat",
1205
+ system_role: "inline",
1206
+ // Kept `true`, unlike other reasoning models: grok-4 accepts temperature
1207
+ // (verified 200 at 0.4 and 1.0) and a panel wants the variance. Dropping
1208
+ // it would buy nothing and cost diversity.
1209
+ supports_temperature: true,
1210
+ // grok-4 IS reasoning-class — it emitted 1,432 reasoning tokens against 64
1211
+ // visible on one call. Saying so strips the "think step by step" preambles
1212
+ // it does not need, and lifts it off the non-reasoning ceiling (1500) onto
1213
+ // the reasoning-safe one (2048).
1214
+ //
1215
+ // NOT the reason grok answers short, which was the initial guess and was
1216
+ // wrong: at caps of 1500, 2048 and 6144 it returned 445, 404 and 461
1217
+ // visible tokens, finish_reason=stop every time. The cap was never
1218
+ // binding; grok is simply terse. The ceiling change removes a latent
1219
+ // constraint, it does not make grok say more.
1220
+ reasoning_prefixes: ["grok-4"],
1221
+ // Measured: xAI's cap bounds visible output only. See the field docs.
1222
+ reasoning_shares_output_budget: false
1223
+ },
1204
1224
  mistral: { family: "openai_chat", system_role: "inline", supports_temperature: true },
1205
1225
  groq: { family: "openai_chat", system_role: "inline", supports_temperature: true },
1206
1226
  deepseek: { family: "openai_chat", system_role: "inline", supports_temperature: true },
@@ -1239,6 +1259,10 @@ function supportsTemperature(provider, model) {
1239
1259
  if (caps.supports_temperature === "model") return !isReasoningModel(provider, model);
1240
1260
  return Boolean(caps.supports_temperature);
1241
1261
  }
1262
+ function reasoningSharesOutputBudget(provider) {
1263
+ const caps = PROVIDER_CAPS[provider.toLowerCase()];
1264
+ return caps?.reasoning_shares_output_budget ?? true;
1265
+ }
1242
1266
 
1243
1267
  // src/providers/types.ts
1244
1268
  init_cjs_shims();
@@ -1491,6 +1515,23 @@ async function acquireRateLimit(provider, deps) {
1491
1515
  var ANTHROPIC_API_URL = "https://api.anthropic.com/v1/messages";
1492
1516
  var ANTHROPIC_VERSION_HEADER = "2023-06-01";
1493
1517
  var ANTHROPIC_STRUCTURED_TOOL_NAME = "structured_output";
1518
+ function applyPromptCaching(body, env = process.env) {
1519
+ const flag = (env["CROSSCHECK_ANTHROPIC_PROMPT_CACHE"] ?? "").trim();
1520
+ if (flag === "0" || flag.toLowerCase() === "false") return;
1521
+ if (typeof body.system === "string" && body.system !== "") {
1522
+ body.system = [{ type: "text", text: body.system, cache_control: { type: "ephemeral" } }];
1523
+ }
1524
+ if (body.messages.length >= 2) {
1525
+ const idx = body.messages.length - 2;
1526
+ const m = body.messages[idx];
1527
+ if (typeof m.content === "string" && m.content !== "") {
1528
+ body.messages[idx] = {
1529
+ role: m.role,
1530
+ content: [{ type: "text", text: m.content, cache_control: { type: "ephemeral" } }]
1531
+ };
1532
+ }
1533
+ }
1534
+ }
1494
1535
  function buildAnthropicRequest(opts) {
1495
1536
  let system;
1496
1537
  const convo = [];
@@ -1515,6 +1556,7 @@ function buildAnthropicRequest(opts) {
1515
1556
  if (system !== void 0) {
1516
1557
  body.system = system;
1517
1558
  }
1559
+ applyPromptCaching(body);
1518
1560
  if (isReasoningModel("anthropic", opts.model)) {
1519
1561
  if (opts.model.toLowerCase().startsWith("claude-fable-5")) {
1520
1562
  body.thinking = { type: "adaptive" };
@@ -1574,6 +1616,7 @@ function parseAnthropicResponse(opts) {
1574
1616
  const prompt = Math.trunc(Number(u["input_tokens"] ?? 0)) || 0;
1575
1617
  const cached = Math.trunc(Number(u["cache_read_input_tokens"] ?? 0)) || 0;
1576
1618
  const completion = Math.trunc(Number(u["output_tokens"] ?? 0)) || 0;
1619
+ const cacheWrite = Math.trunc(Number(u["cache_creation_input_tokens"] ?? 0)) || 0;
1577
1620
  const usage = {
1578
1621
  provider: "anthropic",
1579
1622
  model: opts.model,
@@ -1581,7 +1624,7 @@ function parseAnthropicResponse(opts) {
1581
1624
  // production helper folds them in so prompt_tokens is the FULL
1582
1625
  // input volume; calculateCost then bills the cached subset at the
1583
1626
  // cached rate (cached <= prompt_tokens, since prompt = input+cached).
1584
- prompt_tokens: prompt + cached,
1627
+ prompt_tokens: prompt + cached + cacheWrite,
1585
1628
  completion_tokens: completion,
1586
1629
  cached_tokens: cached,
1587
1630
  total_tokens: 0,
@@ -1933,16 +1976,21 @@ function parseOpenAICompatibleResponse(opts) {
1933
1976
  const u = r["usage"] ?? {};
1934
1977
  const details = u["prompt_tokens_details"] ?? {};
1935
1978
  const cached = Math.trunc(Number(details["cached_tokens"] ?? 0)) || 0;
1979
+ const promptTokens = Math.trunc(Number(u["prompt_tokens"] ?? 0)) || 0;
1980
+ const reportedCompletion = Math.trunc(Number(u["completion_tokens"] ?? 0)) || 0;
1981
+ const reportedTotal = Math.trunc(Number(u["total_tokens"] ?? 0)) || 0;
1982
+ const derivedCompletion = reportedTotal > promptTokens ? reportedTotal - promptTokens : reportedCompletion;
1983
+ const completionTokens = Math.max(reportedCompletion, derivedCompletion);
1936
1984
  const usage = {
1937
1985
  provider: opts.provider,
1938
1986
  model: opts.model,
1939
- prompt_tokens: Math.trunc(Number(u["prompt_tokens"] ?? 0)) || 0,
1940
- completion_tokens: Math.trunc(Number(u["completion_tokens"] ?? 0)) || 0,
1987
+ prompt_tokens: promptTokens,
1988
+ completion_tokens: completionTokens,
1941
1989
  cached_tokens: cached,
1942
1990
  // Use the response's reported total_tokens directly. Python's
1943
1991
  // `Usage.to_dict()` falls back to prompt+completion only when
1944
1992
  // total_tokens is 0/missing; mirror that.
1945
- total_tokens: Math.trunc(Number(u["total_tokens"] ?? 0)) || 0,
1993
+ total_tokens: reportedTotal,
1946
1994
  cost_usd: 0,
1947
1995
  estimated: Object.keys(u).length === 0,
1948
1996
  purpose: opts.purpose
@@ -1984,7 +2032,11 @@ async function sendOpenAICompatible(args) {
1984
2032
  body.reasoning_effort = effort;
1985
2033
  }
1986
2034
  }
1987
- if (isReasoningModel(args.provider, args.model) && typeof body.max_completion_tokens === "number") {
2035
+ if (isReasoningModel(args.provider, args.model) && // Only where the cap is a SHARED budget. On a provider that bounds visible
2036
+ // output only (xAI), headroom cannot protect the answer from being crowded
2037
+ // out — nothing is crowding it — so it would merely authorise a
2038
+ // 25,000-token reply and the bill that comes with it.
2039
+ reasoningSharesOutputBudget(args.provider) && typeof body.max_completion_tokens === "number") {
1988
2040
  const raw = Number(process.env["CROSSCHECK_OPENAI_REASONING_HEADROOM_TOKENS"]);
1989
2041
  const headroom = Number.isFinite(raw) && raw >= 0 ? Math.trunc(raw) : 25e3;
1990
2042
  body.max_completion_tokens += headroom;
@@ -2300,7 +2352,7 @@ var import_zod = require("zod");
2300
2352
  // src/server-meta.ts
2301
2353
  init_cjs_shims();
2302
2354
  var SERVER_NAME = "crosscheck-agent";
2303
- var SERVER_VERSION = true ? "0.2.21" : "0.0.0-dev";
2355
+ var SERVER_VERSION = true ? "0.2.23" : "0.0.0-dev";
2304
2356
 
2305
2357
  // src/tools/audit.ts
2306
2358
  init_cjs_shims();
@@ -10150,6 +10202,66 @@ function isObj6(v) {
10150
10202
 
10151
10203
  // src/tools/debate.ts
10152
10204
  init_cjs_shims();
10205
+
10206
+ // src/core/debate-compaction.ts
10207
+ init_cjs_shims();
10208
+ var DEFAULT_PRIOR_BUDGET_CHARS = 12e3;
10209
+ function renderTurn(e) {
10210
+ return `[${e.provider} \u2014 round ${e.round}]
10211
+ ${e.response ?? "(error)"}`;
10212
+ }
10213
+ function buildPriorTurns(transcript, opts = {}) {
10214
+ if (transcript.length === 0) {
10215
+ return { body: "", includedTurns: 0, omittedTurns: 0, chars: 0 };
10216
+ }
10217
+ const rendered = transcript.map(renderTurn);
10218
+ const full = rendered.join("\n\n");
10219
+ if (!opts.compress) {
10220
+ return { body: full, includedTurns: transcript.length, omittedTurns: 0, chars: full.length };
10221
+ }
10222
+ const budget = Math.max(0, opts.budgetChars ?? DEFAULT_PRIOR_BUDGET_CHARS);
10223
+ if (full.length <= budget) {
10224
+ return { body: full, includedTurns: transcript.length, omittedTurns: 0, chars: full.length };
10225
+ }
10226
+ const latestRound = Math.max(...transcript.map((e) => e.round));
10227
+ const keep = /* @__PURE__ */ new Set();
10228
+ let spent = 0;
10229
+ for (let i = transcript.length - 1; i >= 0; i -= 1) {
10230
+ const isLatest = transcript[i].round === latestRound;
10231
+ const cost = rendered[i].length + 2;
10232
+ if (isLatest) {
10233
+ keep.add(i);
10234
+ spent += cost;
10235
+ continue;
10236
+ }
10237
+ if (spent + cost > budget) continue;
10238
+ keep.add(i);
10239
+ spent += cost;
10240
+ }
10241
+ const parts = [];
10242
+ let omitted = 0;
10243
+ let pendingGap = 0;
10244
+ for (let i = 0; i < transcript.length; i += 1) {
10245
+ if (keep.has(i)) {
10246
+ if (pendingGap > 0) {
10247
+ parts.push(`[${pendingGap} earlier turn(s) omitted for length]`);
10248
+ pendingGap = 0;
10249
+ }
10250
+ parts.push(rendered[i]);
10251
+ } else {
10252
+ omitted += 1;
10253
+ pendingGap += 1;
10254
+ }
10255
+ }
10256
+ if (pendingGap > 0) parts.push(`[${pendingGap} earlier turn(s) omitted for length]`);
10257
+ const body = parts.join("\n\n");
10258
+ return { body, includedTurns: keep.size, omittedTurns: omitted, chars: body.length };
10259
+ }
10260
+ function buildModeratorTranscript(transcript) {
10261
+ return transcript.map(renderTurn).join("\n\n");
10262
+ }
10263
+
10264
+ // src/tools/debate.ts
10153
10265
  var import_node_perf_hooks9 = require("perf_hooks");
10154
10266
  var DEFERRED_OPTS4 = [
10155
10267
  // empty — every debate opt runs natively when its deps
@@ -10197,6 +10309,9 @@ async function runDebate(args, opts) {
10197
10309
  1,
10198
10310
  Math.trunc(Number(args["max_rounds"] ?? 3)) || 3
10199
10311
  );
10312
+ const compressRounds = args["compress_rounds"] === true;
10313
+ const priorBudgetChars = Math.trunc(Number(args["prior_budget_chars"] ?? DEFAULT_PRIOR_BUDGET_CHARS)) || DEFAULT_PRIOR_BUDGET_CHARS;
10314
+ let compactionMeta = null;
10200
10315
  const sessionId = typeof args["session_id"] === "string" ? args["session_id"] : null;
10201
10316
  const breakerEnv = await maybeBreakerEnvelope(
10202
10317
  opts.storage,
@@ -10307,7 +10422,7 @@ async function runDebate(args, opts) {
10307
10422
  const roundMessages = [
10308
10423
  {
10309
10424
  role: "system",
10310
- content: `You are debating peers from other model families. Round ${rnd}/${maxRounds}. Disagree where warranted, concede where right, and keep replies short and specific.`
10425
+ content: "You are debating peers from other model families. Disagree where warranted, concede where right, and keep replies short and specific."
10311
10426
  }
10312
10427
  ];
10313
10428
  if (context) {
@@ -10319,13 +10434,22 @@ ${context}` });
10319
10434
  TOPIC: ${topic}` : `TOPIC: ${topic}`;
10320
10435
  roundMessages.push({ role: "user", content: topicLine });
10321
10436
  if (transcript.length > 0) {
10322
- const prior = transcript.map(
10323
- (e) => `[${e.provider} \u2014 round ${e.round}]
10324
- ${e.response ?? "(error)"}`
10325
- ).join("\n\n");
10437
+ const prior = buildPriorTurns(transcript, {
10438
+ compress: compressRounds,
10439
+ budgetChars: priorBudgetChars
10440
+ });
10441
+ if (prior.omittedTurns > 0) compactionMeta = {
10442
+ omitted_turns: prior.omittedTurns,
10443
+ included_turns: prior.includedTurns,
10444
+ budget_chars: priorBudgetChars
10445
+ };
10326
10446
  roundMessages.push({ role: "user", content: `PRIOR TURNS:
10327
- ${prior}` });
10447
+ ${prior.body}` });
10328
10448
  }
10449
+ roundMessages.push({
10450
+ role: "user",
10451
+ content: `This is round ${rnd} of ${maxRounds}. Reply for this round.`
10452
+ });
10329
10453
  for (const p of selected) {
10330
10454
  const entry = await callPanelist(p, roundMessages);
10331
10455
  const withRound = { ...entry, round: rnd };
@@ -10373,10 +10497,7 @@ ${prior}` });
10373
10497
  let coReasoned = false;
10374
10498
  if (moderator) {
10375
10499
  const synthProvider = superRequested && opts.providers["anthropic"] ? retargetForSuper(opts.providers["anthropic"]) : upgraded ? retargetProvider(opts.providers["anthropic"], UPGRADE_MODEL) : moderator;
10376
- const condensed = transcript.map(
10377
- (e) => `[${e.provider} \u2014 round ${e.round}]
10378
- ${e.response ?? "(error)"}`
10379
- ).join("\n\n");
10500
+ const condensed = buildModeratorTranscript(transcript);
10380
10501
  const persona = buildPersonaInjection({
10381
10502
  toolName: "debate",
10382
10503
  prompt: topic,
@@ -10459,6 +10580,7 @@ ${condensed}`
10459
10580
  synthesis
10460
10581
  };
10461
10582
  if (personaMeta && personaMeta.used !== null) result["persona"] = personaMeta;
10583
+ if (compactionMeta) result["round_compaction"] = compactionMeta;
10462
10584
  if (upgraded) {
10463
10585
  result["reasoning_upgrade"] = {
10464
10586
  applied: true,
@@ -12792,7 +12914,7 @@ var CHECK_INTERVAL_SECONDS = 3 * 24 * 60 * 60;
12792
12914
  var DEFAULT_PACKAGE = "crosscheck-cli";
12793
12915
  var FETCH_TIMEOUT_MS = 3e3;
12794
12916
  function engineVersion() {
12795
- return true ? "0.2.21" : "0.0.0-dev";
12917
+ return true ? "0.2.23" : "0.0.0-dev";
12796
12918
  }
12797
12919
  function defaultUpdateCachePath() {
12798
12920
  const base = process.env["CROSSCHECK_DATA_DIR"] || import_node_path13.default.join(import_node_os2.default.homedir() || import_node_os2.default.tmpdir(), ".crosscheck");
@@ -14836,9 +14958,7 @@ function usageEventsFromEnvelope(out, toolName) {
14836
14958
  const timing = out["timing"];
14837
14959
  const timingByCall = timing && typeof timing === "object" ? timing["by_call"] : void 0;
14838
14960
  const allTiming = Array.isArray(timingByCall) ? timingByCall : [];
14839
- const latencies = allTiming.filter(
14840
- (t) => !t?.["error_kind"]
14841
- );
14961
+ const latencies = allTiming;
14842
14962
  const aligned = latencies.length === byCall.length;
14843
14963
  const pattern = toolToPattern(toolName);
14844
14964
  const events = [];
@@ -14848,13 +14968,18 @@ function usageEventsFromEnvelope(out, toolName) {
14848
14968
  if (!provider) continue;
14849
14969
  const model = String(u.model ?? "").slice(0, 128) || "unknown";
14850
14970
  const cost = Math.max(0, Number(u.cost_usd) || 0);
14851
- const latencyMs = aligned ? intNonNeg(latencies[i]?.["wall_ms"]) : 0;
14971
+ const timingRow = aligned ? latencies[i] : void 0;
14972
+ if (timingRow?.["error_kind"]) continue;
14973
+ const promptTokens = intNonNeg(u.prompt_tokens);
14974
+ const completionTokens = intNonNeg(u.completion_tokens);
14975
+ if (promptTokens === 0 && completionTokens === 0) continue;
14976
+ const latencyMs = intNonNeg(timingRow?.["wall_ms"]);
14852
14977
  events.push({
14853
14978
  provider,
14854
14979
  model,
14855
14980
  pattern,
14856
- promptTokens: intNonNeg(u.prompt_tokens),
14857
- completionTokens: intNonNeg(u.completion_tokens),
14981
+ promptTokens,
14982
+ completionTokens,
14858
14983
  costUsdEstimate: cost,
14859
14984
  latencyMs,
14860
14985
  status: "ok",