crosscheck-mcp 0.2.21 → 0.2.22
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/browser-ext.cjs +60 -8
- package/dist/browser-ext.cjs.map +1 -1
- package/dist/browser-ext.js +60 -8
- package/dist/browser-ext.js.map +1 -1
- package/dist/node-stdio.cjs +69 -14
- package/dist/node-stdio.cjs.map +1 -1
- package/dist/node-stdio.js +69 -14
- package/dist/node-stdio.js.map +1 -1
- package/package.json +1 -1
package/dist/node-stdio.js
CHANGED
|
@@ -1176,7 +1176,27 @@ var PROVIDER_CAPS = {
|
|
|
1176
1176
|
// temperature, so the prefix is the fix.
|
|
1177
1177
|
reasoning_prefixes: ["gpt-5", "gpt-6", "o1", "o3", "o4"]
|
|
1178
1178
|
},
|
|
1179
|
-
xai: {
|
|
1179
|
+
xai: {
|
|
1180
|
+
family: "openai_chat",
|
|
1181
|
+
system_role: "inline",
|
|
1182
|
+
// Kept `true`, unlike other reasoning models: grok-4 accepts temperature
|
|
1183
|
+
// (verified 200 at 0.4 and 1.0) and a panel wants the variance. Dropping
|
|
1184
|
+
// it would buy nothing and cost diversity.
|
|
1185
|
+
supports_temperature: true,
|
|
1186
|
+
// grok-4 IS reasoning-class — it emitted 1,432 reasoning tokens against 64
|
|
1187
|
+
// visible on one call. Saying so strips the "think step by step" preambles
|
|
1188
|
+
// it does not need, and lifts it off the non-reasoning ceiling (1500) onto
|
|
1189
|
+
// the reasoning-safe one (2048).
|
|
1190
|
+
//
|
|
1191
|
+
// NOT the reason grok answers short, which was the initial guess and was
|
|
1192
|
+
// wrong: at caps of 1500, 2048 and 6144 it returned 445, 404 and 461
|
|
1193
|
+
// visible tokens, finish_reason=stop every time. The cap was never
|
|
1194
|
+
// binding; grok is simply terse. The ceiling change removes a latent
|
|
1195
|
+
// constraint, it does not make grok say more.
|
|
1196
|
+
reasoning_prefixes: ["grok-4"],
|
|
1197
|
+
// Measured: xAI's cap bounds visible output only. See the field docs.
|
|
1198
|
+
reasoning_shares_output_budget: false
|
|
1199
|
+
},
|
|
1180
1200
|
mistral: { family: "openai_chat", system_role: "inline", supports_temperature: true },
|
|
1181
1201
|
groq: { family: "openai_chat", system_role: "inline", supports_temperature: true },
|
|
1182
1202
|
deepseek: { family: "openai_chat", system_role: "inline", supports_temperature: true },
|
|
@@ -1215,6 +1235,10 @@ function supportsTemperature(provider, model) {
|
|
|
1215
1235
|
if (caps.supports_temperature === "model") return !isReasoningModel(provider, model);
|
|
1216
1236
|
return Boolean(caps.supports_temperature);
|
|
1217
1237
|
}
|
|
1238
|
+
function reasoningSharesOutputBudget(provider) {
|
|
1239
|
+
const caps = PROVIDER_CAPS[provider.toLowerCase()];
|
|
1240
|
+
return caps?.reasoning_shares_output_budget ?? true;
|
|
1241
|
+
}
|
|
1218
1242
|
|
|
1219
1243
|
// src/providers/types.ts
|
|
1220
1244
|
init_esm_shims();
|
|
@@ -1467,6 +1491,23 @@ async function acquireRateLimit(provider, deps) {
|
|
|
1467
1491
|
var ANTHROPIC_API_URL = "https://api.anthropic.com/v1/messages";
|
|
1468
1492
|
var ANTHROPIC_VERSION_HEADER = "2023-06-01";
|
|
1469
1493
|
var ANTHROPIC_STRUCTURED_TOOL_NAME = "structured_output";
|
|
1494
|
+
function applyPromptCaching(body, env = process.env) {
|
|
1495
|
+
const flag = (env["CROSSCHECK_ANTHROPIC_PROMPT_CACHE"] ?? "").trim();
|
|
1496
|
+
if (flag === "0" || flag.toLowerCase() === "false") return;
|
|
1497
|
+
if (typeof body.system === "string" && body.system !== "") {
|
|
1498
|
+
body.system = [{ type: "text", text: body.system, cache_control: { type: "ephemeral" } }];
|
|
1499
|
+
}
|
|
1500
|
+
if (body.messages.length >= 2) {
|
|
1501
|
+
const idx = body.messages.length - 2;
|
|
1502
|
+
const m = body.messages[idx];
|
|
1503
|
+
if (typeof m.content === "string" && m.content !== "") {
|
|
1504
|
+
body.messages[idx] = {
|
|
1505
|
+
role: m.role,
|
|
1506
|
+
content: [{ type: "text", text: m.content, cache_control: { type: "ephemeral" } }]
|
|
1507
|
+
};
|
|
1508
|
+
}
|
|
1509
|
+
}
|
|
1510
|
+
}
|
|
1470
1511
|
function buildAnthropicRequest(opts) {
|
|
1471
1512
|
let system;
|
|
1472
1513
|
const convo = [];
|
|
@@ -1491,6 +1532,7 @@ function buildAnthropicRequest(opts) {
|
|
|
1491
1532
|
if (system !== void 0) {
|
|
1492
1533
|
body.system = system;
|
|
1493
1534
|
}
|
|
1535
|
+
applyPromptCaching(body);
|
|
1494
1536
|
if (isReasoningModel("anthropic", opts.model)) {
|
|
1495
1537
|
if (opts.model.toLowerCase().startsWith("claude-fable-5")) {
|
|
1496
1538
|
body.thinking = { type: "adaptive" };
|
|
@@ -1550,6 +1592,7 @@ function parseAnthropicResponse(opts) {
|
|
|
1550
1592
|
const prompt = Math.trunc(Number(u["input_tokens"] ?? 0)) || 0;
|
|
1551
1593
|
const cached = Math.trunc(Number(u["cache_read_input_tokens"] ?? 0)) || 0;
|
|
1552
1594
|
const completion = Math.trunc(Number(u["output_tokens"] ?? 0)) || 0;
|
|
1595
|
+
const cacheWrite = Math.trunc(Number(u["cache_creation_input_tokens"] ?? 0)) || 0;
|
|
1553
1596
|
const usage = {
|
|
1554
1597
|
provider: "anthropic",
|
|
1555
1598
|
model: opts.model,
|
|
@@ -1557,7 +1600,7 @@ function parseAnthropicResponse(opts) {
|
|
|
1557
1600
|
// production helper folds them in so prompt_tokens is the FULL
|
|
1558
1601
|
// input volume; calculateCost then bills the cached subset at the
|
|
1559
1602
|
// cached rate (cached <= prompt_tokens, since prompt = input+cached).
|
|
1560
|
-
prompt_tokens: prompt + cached,
|
|
1603
|
+
prompt_tokens: prompt + cached + cacheWrite,
|
|
1561
1604
|
completion_tokens: completion,
|
|
1562
1605
|
cached_tokens: cached,
|
|
1563
1606
|
total_tokens: 0,
|
|
@@ -1909,16 +1952,21 @@ function parseOpenAICompatibleResponse(opts) {
|
|
|
1909
1952
|
const u = r["usage"] ?? {};
|
|
1910
1953
|
const details = u["prompt_tokens_details"] ?? {};
|
|
1911
1954
|
const cached = Math.trunc(Number(details["cached_tokens"] ?? 0)) || 0;
|
|
1955
|
+
const promptTokens = Math.trunc(Number(u["prompt_tokens"] ?? 0)) || 0;
|
|
1956
|
+
const reportedCompletion = Math.trunc(Number(u["completion_tokens"] ?? 0)) || 0;
|
|
1957
|
+
const reportedTotal = Math.trunc(Number(u["total_tokens"] ?? 0)) || 0;
|
|
1958
|
+
const derivedCompletion = reportedTotal > promptTokens ? reportedTotal - promptTokens : reportedCompletion;
|
|
1959
|
+
const completionTokens = Math.max(reportedCompletion, derivedCompletion);
|
|
1912
1960
|
const usage = {
|
|
1913
1961
|
provider: opts.provider,
|
|
1914
1962
|
model: opts.model,
|
|
1915
|
-
prompt_tokens:
|
|
1916
|
-
completion_tokens:
|
|
1963
|
+
prompt_tokens: promptTokens,
|
|
1964
|
+
completion_tokens: completionTokens,
|
|
1917
1965
|
cached_tokens: cached,
|
|
1918
1966
|
// Use the response's reported total_tokens directly. Python's
|
|
1919
1967
|
// `Usage.to_dict()` falls back to prompt+completion only when
|
|
1920
1968
|
// total_tokens is 0/missing; mirror that.
|
|
1921
|
-
total_tokens:
|
|
1969
|
+
total_tokens: reportedTotal,
|
|
1922
1970
|
cost_usd: 0,
|
|
1923
1971
|
estimated: Object.keys(u).length === 0,
|
|
1924
1972
|
purpose: opts.purpose
|
|
@@ -1960,7 +2008,11 @@ async function sendOpenAICompatible(args) {
|
|
|
1960
2008
|
body.reasoning_effort = effort;
|
|
1961
2009
|
}
|
|
1962
2010
|
}
|
|
1963
|
-
if (isReasoningModel(args.provider, args.model) &&
|
|
2011
|
+
if (isReasoningModel(args.provider, args.model) && // Only where the cap is a SHARED budget. On a provider that bounds visible
|
|
2012
|
+
// output only (xAI), headroom cannot protect the answer from being crowded
|
|
2013
|
+
// out — nothing is crowding it — so it would merely authorise a
|
|
2014
|
+
// 25,000-token reply and the bill that comes with it.
|
|
2015
|
+
reasoningSharesOutputBudget(args.provider) && typeof body.max_completion_tokens === "number") {
|
|
1964
2016
|
const raw = Number(process.env["CROSSCHECK_OPENAI_REASONING_HEADROOM_TOKENS"]);
|
|
1965
2017
|
const headroom = Number.isFinite(raw) && raw >= 0 ? Math.trunc(raw) : 25e3;
|
|
1966
2018
|
body.max_completion_tokens += headroom;
|
|
@@ -2279,7 +2331,7 @@ import { z } from "zod";
|
|
|
2279
2331
|
// src/server-meta.ts
|
|
2280
2332
|
init_esm_shims();
|
|
2281
2333
|
var SERVER_NAME = "crosscheck-agent";
|
|
2282
|
-
var SERVER_VERSION = true ? "0.2.
|
|
2334
|
+
var SERVER_VERSION = true ? "0.2.22" : "0.0.0-dev";
|
|
2283
2335
|
|
|
2284
2336
|
// src/tools/audit.ts
|
|
2285
2337
|
init_esm_shims();
|
|
@@ -12788,7 +12840,7 @@ var CHECK_INTERVAL_SECONDS = 3 * 24 * 60 * 60;
|
|
|
12788
12840
|
var DEFAULT_PACKAGE = "crosscheck-cli";
|
|
12789
12841
|
var FETCH_TIMEOUT_MS = 3e3;
|
|
12790
12842
|
function engineVersion() {
|
|
12791
|
-
return true ? "0.2.
|
|
12843
|
+
return true ? "0.2.22" : "0.0.0-dev";
|
|
12792
12844
|
}
|
|
12793
12845
|
function defaultUpdateCachePath() {
|
|
12794
12846
|
const base = process.env["CROSSCHECK_DATA_DIR"] || path10.join(os.homedir() || os.tmpdir(), ".crosscheck");
|
|
@@ -14832,9 +14884,7 @@ function usageEventsFromEnvelope(out, toolName) {
|
|
|
14832
14884
|
const timing = out["timing"];
|
|
14833
14885
|
const timingByCall = timing && typeof timing === "object" ? timing["by_call"] : void 0;
|
|
14834
14886
|
const allTiming = Array.isArray(timingByCall) ? timingByCall : [];
|
|
14835
|
-
const latencies = allTiming
|
|
14836
|
-
(t) => !t?.["error_kind"]
|
|
14837
|
-
);
|
|
14887
|
+
const latencies = allTiming;
|
|
14838
14888
|
const aligned = latencies.length === byCall.length;
|
|
14839
14889
|
const pattern = toolToPattern(toolName);
|
|
14840
14890
|
const events = [];
|
|
@@ -14844,13 +14894,18 @@ function usageEventsFromEnvelope(out, toolName) {
|
|
|
14844
14894
|
if (!provider) continue;
|
|
14845
14895
|
const model = String(u.model ?? "").slice(0, 128) || "unknown";
|
|
14846
14896
|
const cost = Math.max(0, Number(u.cost_usd) || 0);
|
|
14847
|
-
const
|
|
14897
|
+
const timingRow = aligned ? latencies[i] : void 0;
|
|
14898
|
+
if (timingRow?.["error_kind"]) continue;
|
|
14899
|
+
const promptTokens = intNonNeg(u.prompt_tokens);
|
|
14900
|
+
const completionTokens = intNonNeg(u.completion_tokens);
|
|
14901
|
+
if (promptTokens === 0 && completionTokens === 0) continue;
|
|
14902
|
+
const latencyMs = intNonNeg(timingRow?.["wall_ms"]);
|
|
14848
14903
|
events.push({
|
|
14849
14904
|
provider,
|
|
14850
14905
|
model,
|
|
14851
14906
|
pattern,
|
|
14852
|
-
promptTokens
|
|
14853
|
-
completionTokens
|
|
14907
|
+
promptTokens,
|
|
14908
|
+
completionTokens,
|
|
14854
14909
|
costUsdEstimate: cost,
|
|
14855
14910
|
latencyMs,
|
|
14856
14911
|
status: "ok",
|