mslxdff 0.1.133 → 0.1.135

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "mslxdff",
3
- "version": "0.1.133",
3
+ "version": "0.1.135",
4
4
  "description": "测试项目,请勿使用。",
5
5
  "type": "module",
6
6
  "bin": {
@@ -0,0 +1,88 @@
1
+ // `-stats` 模型用量报表:近 N 小时每模型 token 消耗 + 首字/总耗时/速度。
2
+ // 数据来自 src/usage/(逐请求 JSONL),不是 state.json 的终生 EMA —— 见 .scratch/stats-report/SPEC.md
3
+ import { usageReport } from "../../usage/report.js";
4
+ import { usageEnabled, usageKeepDays } from "../../usage/record.js";
5
+
6
+ const FLAGS = ["-stats", "--stats"];
7
+
8
+ export function isStatsFlag(args) {
9
+ return FLAGS.some((f) => args.includes(f));
10
+ }
11
+
12
+ export function parseStatsArgs(args) {
13
+ const hoursIdx = args.findIndex((a) => a === "--hours" || a === "-hours");
14
+ const rawHours = hoursIdx >= 0 ? Number(args[hoursIdx + 1]) : NaN;
15
+ const hours = Number.isFinite(rawHours) && rawHours > 0 ? Math.min(168, Math.floor(rawHours)) : 24;
16
+ const modelIdx = args.findIndex((a) => a === "--model" || a === "-model");
17
+ const rawModel = modelIdx >= 0 && args[modelIdx + 1] && !String(args[modelIdx + 1]).startsWith("-") ? args[modelIdx + 1] : null;
18
+ return { hours, model: rawModel, json: args.includes("--json") };
19
+ }
20
+
21
+ function fmtTok(n) {
22
+ const v = Number(n) || 0;
23
+ if (v < 1000) return String(v);
24
+ if (v < 1_000_000) return `${(v / 1000).toFixed(1)}k`;
25
+ return `${(v / 1_000_000).toFixed(2)}M`;
26
+ }
27
+
28
+ function fmtMs(v) {
29
+ if (v == null || !Number.isFinite(v)) return "—";
30
+ return v < 1000 ? `${v}ms` : `${(v / 1000).toFixed(1)}s`;
31
+ }
32
+
33
+ function fmtTps(v) {
34
+ return v == null || !Number.isFinite(v) ? "—" : `${v} tok/s`;
35
+ }
36
+
37
+ // 中文字符按 2 列宽算,否则表格错位
38
+ function width(s) {
39
+ let w = 0;
40
+ for (const ch of String(s)) w += /[\u1100-\u115F\u2E80-\uA4CF\uAC00-\uD7A3\uF900-\uFAFF\uFE30-\uFE4F\uFF00-\uFF60\uFFE0-\uFFE6]/.test(ch) ? 2 : 1;
41
+ return w;
42
+ }
43
+
44
+ function padW(s, w) {
45
+ const t = String(s);
46
+ return t + " ".repeat(Math.max(0, w - width(t)));
47
+ }
48
+
49
+ function rowText(id, r) {
50
+ return ` ${padW(id, 30)} ${padW(r.requests, 6)} ${padW(fmtTok(r.promptTokens), 8)} ${padW(fmtTok(r.completionTokens), 10)} ${padW(fmtTok(r.totalTokens), 8)} ${padW(fmtMs(r.avgTtfbMs), 7)} ${padW(fmtMs(r.avgTotalMs), 8)} ${fmtTps(r.avgTps)}`;
51
+ }
52
+
53
+ export function renderStats(report, { hours = 24, model = null } = {}) {
54
+ const lines = [];
55
+ const { models, totals } = report;
56
+ if (!models.length) {
57
+ lines.push(`暂无用量记录(近 ${hours}h${model ? ` · 模型 ${model}` : ""})— 经 8989 网关发一次请求后出现`);
58
+ lines.push("提示:mslxdff -status 看当前体检 · mslxdff -log 20 看最近事件");
59
+ lines.push("说明:-chat 直连 mimo/big-pickle 不经网关,不计入本表");
60
+ return lines.join("\n");
61
+ }
62
+ lines.push(`模型用量(近 ${hours}h · 成功请求 ${totals.requests} 次 · ${models.length} 个模型)`);
63
+ lines.push(` ${padW("模型", 30)} ${padW("请求", 6)} ${padW("prompt", 8)} ${padW("输出", 10)} ${padW("合计", 8)} ${padW("首字", 7)} ${padW("总耗时", 8)} 速度`);
64
+ for (const m of models) lines.push(rowText(m.id, m));
65
+ lines.push(` ${"-".repeat(76)}`);
66
+ lines.push(rowText("合计", totals));
67
+ if (totals.reasoningTokens > 0) lines.push(` 其中思考 tokens:${fmtTok(totals.reasoningTokens)}`);
68
+ lines.push("");
69
+ lines.push("说明:速度 = 输出 tokens ÷ 生成耗时(总耗时−首字),按窗口加权;只统计成功请求。");
70
+ lines.push(` -chat 直连 mimo/big-pickle 不经 8989 网关,不计入。数据保留 ${usageKeepDays()} 天,mslxdff -stats --hours 1|--json|--model <id> 可调。`);
71
+ return lines.join("\n");
72
+ }
73
+
74
+ export async function handleStats(args) {
75
+ if (!isStatsFlag(args)) return false;
76
+ const { hours, model, json } = parseStatsArgs(args);
77
+ if (!usageEnabled()) {
78
+ console.log("用量记录已关闭(MSLXDFF_USAGE_LOG=0)—— 去掉该 env 后重启 daemon 即可恢复采集。");
79
+ return true;
80
+ }
81
+ const report = await usageReport({ hours, model, now: Date.now() });
82
+ if (json) {
83
+ console.log(JSON.stringify(report, null, 2));
84
+ return true;
85
+ }
86
+ console.log(renderStats(report, { hours, model }));
87
+ return true;
88
+ }
package/src/cli/help.js CHANGED
@@ -5,6 +5,8 @@ Usage:
5
5
  mslxdff start as a background daemon and exit (status + help if one is already running)
6
6
  mslxdff -d start as a background daemon
7
7
  mslxdff -status show current status (daemon/health/port/config, upstream providers, models + metrics/体检表, autostart/plugins, groups/peers, recent calls with ttfb/tps, last error)
8
+ mslxdff -stats [--hours N] [--json] [--model <id>] per-model token usage + speed over the last N hours (default 24; success requests only, from the local usage JSONL)
9
+
8
10
  mslxdff -log [N] show last N events (default 10, e.g. -log 100)
9
11
  mslxdff -models interactive picker: ↑/↓ select a model, Enter sets it as the default (non-TTY: plain list)
10
12
  mslxdff -model list list the free models this proxy serves (cached)
package/src/cli/index.js CHANGED
@@ -23,6 +23,8 @@ export async function run(args = process.argv.slice(2)) {
23
23
  if (await handlePlugins(args)) return;
24
24
  if (await handleChat(args)) return;
25
25
  if (await handleStatus(args, VERSION)) return;
26
+ const { handleStats } = await import("./commands/stats.js");
27
+ if (await handleStats(args)) return;
26
28
 
27
29
  const { handleModel } = await import("./commands/model.js");
28
30
  if (await handleModel(args)) return;
package/src/cli/status.js CHANGED
@@ -243,25 +243,20 @@ export async function printStatus(VERSION) {
243
243
  }
244
244
 
245
245
  const { recentCalls, lastError } = await import("../logs.js");
246
- console.log("\nrecent calls: (gateway 持久化,最近5条,含首字/tok/s)");
246
+ console.log("\nrecent calls: (gateway 持久化,最近5条;token/速度看 mslxdff -stats)");
247
247
  const calls = recentCalls(5);
248
248
  if (calls.length) {
249
- let sumDur = 0, sumTps = 0, tpsN = 0;
249
+ let sumDur = 0;
250
250
  for (const c of calls) {
251
251
  if (Number.isFinite(c.durationMs)) sumDur += c.durationMs;
252
252
  else if (Number.isFinite(c.totalMs)) sumDur += c.totalMs;
253
- if (Number.isFinite(c.tps)) { sumTps += c.tps; tpsN++; } else if (Number.isFinite(c.charsPerSec)) { sumTps += c.charsPerSec; tpsN++; }
254
253
  }
255
254
  const avgDur = calls.length ? Math.round(sumDur / calls.length) : null;
256
- const avgTps = tpsN ? Math.round(sumTps / tpsN) : null;
257
- console.log(` avg ${avgDur ? avgDur + "ms" : "—"}${avgTps ? ` · ${avgTps} tok/s` : ""} — mslxdff -log 20 查看详情`);
255
+ console.log(` avg ${avgDur ? avgDur + "ms" : "—"} — mslxdff -log 20 查看详情`);
258
256
  for (const c of calls) {
259
257
  const dur = c.totalMs ?? c.durationMs;
260
- const ttfb = c.ttfbMs != null ? ` 首字${c.ttfbMs}ms` : "";
261
- const tps = c.tps != null ? ` ${c.tps}tok/s` : (c.charsPerSec ? ` ${c.charsPerSec}ch/s` : "");
262
- const tok = c.usage?.completion_tokens != null ? ` tok${c.usage.completion_tokens}` : (c.chars ? ` ch${c.chars}` : "");
263
258
  const tm = fmtTs(c.ts);
264
- console.log(` ${tm} ${(c.model || "-").padEnd(28)} ${String(c.status || "-").padEnd(4)} ${dur ? dur + "ms" : ""}${ttfb}${tps}${tok}${c.auto ? " auto" : ""}`);
259
+ console.log(` ${tm} ${(c.model || "-").padEnd(28)} ${String(c.status || "-").padEnd(4)} ${dur ? dur + "ms" : ""}${c.stream ? " stream" : ""}${c.auto ? " auto" : ""}`);
265
260
  }
266
261
  } else {
267
262
  console.log(" (none yet — 发一次请求后出现,mslxdff -chat hi)");
@@ -0,0 +1,131 @@
1
+ // zen 免费层"agent 形状"门禁(2026-09-18 上线,bisect 实测):
2
+ // 403 FreeTierError ≤ 请求必须同时满足:① stream:true ② tools 含 bash/edit/glob/grep/read 五个核心名。
3
+ // 另 UA 版本需 ≥ opencode/1.18.0(低版本返回 426 UpgradeRequired,见 opencode-identity.js)。
4
+ // 官方客户端自带这五个工具天然通过;-chat 直连(非流式、工具名不同)与裸 API 客户端需补形状。
5
+ // 逃生阀:MSLXDFF_FREE_LANE=0 关闭(注入与强制流式都跳过),上游若撤门禁可回退。
6
+
7
+ export const CORE_AGENT_TOOL_NAMES = ["bash", "edit", "glob", "grep", "read"];
8
+
9
+ function laneDisabled(env = process.env) {
10
+ const raw = env.MSLXDFF_FREE_LANE;
11
+ if (raw === undefined || raw === null || raw === "") return false;
12
+ const s = String(raw).trim().toLowerCase();
13
+ return s === "0" || s === "false" || s === "off" || s === "no";
14
+ }
15
+
16
+ const chatTool = (name) => ({
17
+ type: "function",
18
+ function: { name, description: `The ${name} tool.`, parameters: { type: "object", properties: {} } },
19
+ });
20
+
21
+ const responsesTool = (name) => ({
22
+ type: "function",
23
+ name,
24
+ description: `The ${name} tool.`,
25
+ parameters: { type: "object", properties: {} },
26
+ });
27
+
28
+ function toolName(tool, responses) {
29
+ return responses ? tool?.name : tool?.function?.name;
30
+ }
31
+
32
+ /**
33
+ * 给免费层请求补 agent 形状(幂等,原地改 body)。
34
+ * @returns {{injected: string[], forcedStream: boolean, disabled: boolean}}
35
+ */
36
+ export function ensureFreeLaneShape(body, { responses = false, env = process.env } = {}) {
37
+ if (!body || typeof body !== "object") return { injected: [], forcedStream: false, disabled: true };
38
+ if (laneDisabled(env)) return { injected: [], forcedStream: false, disabled: true };
39
+ const forcedStream = body.stream !== true;
40
+ body.stream = true;
41
+ const mk = responses ? responsesTool : chatTool;
42
+ const list = Array.isArray(body.tools) ? body.tools : [];
43
+ const have = new Set(list.map((t) => toolName(t, responses)).filter(Boolean));
44
+ const injected = CORE_AGENT_TOOL_NAMES.filter((n) => !have.has(n));
45
+ if (injected.length) body.tools = [...list, ...injected.map(mk)];
46
+ return { injected, forcedStream, disabled: false };
47
+ }
48
+
49
+ function mergeToolCall(target, idx, delta) {
50
+ const cur = target[idx] || (target[idx] = { index: idx, id: undefined, type: "function", function: { name: undefined, arguments: "" } });
51
+ if (delta?.id) cur.id = delta.id;
52
+ if (delta?.type) cur.type = delta.type;
53
+ if (delta?.function?.name) cur.function.name = delta.function.name;
54
+ if (delta?.function?.arguments) cur.function.arguments += delta.function.arguments;
55
+ }
56
+
57
+ function finishSseJson(acc) {
58
+ const message = { role: "assistant", content: acc.content };
59
+ if (acc.reasoning) message.reasoning_content = acc.reasoning;
60
+ if (acc.toolCalls.length) {
61
+ message.tool_calls = acc.toolCalls
62
+ .filter(Boolean)
63
+ .map((c, i) => ({ ...c, index: undefined, id: c.id || `call_${i}`, function: { name: c.function.name || "", arguments: c.function.arguments || "{}" } }));
64
+ if (!message.content) message.content = "";
65
+ }
66
+ const json = {
67
+ id: acc.id || "chatcmpl-aggregated",
68
+ object: "chat.completion",
69
+ created: Math.floor(Date.now() / 1000),
70
+ model: acc.model,
71
+ choices: [{ index: 0, message, finish_reason: acc.finishReason || (acc.toolCalls.length ? "tool_calls" : "stop"), logprobs: null }],
72
+ };
73
+ if (acc.usage) json.usage = acc.usage;
74
+ return json;
75
+ }
76
+
77
+ /**
78
+ * 把上游 SSE 流聚合回非流式 chat completion JSON。
79
+ * 仅消费 body,不改状态;解析失败时抛错(调用方自行回退)。
80
+ */
81
+ export async function aggregateChatSse(res) {
82
+ const reader = res.body?.getReader?.();
83
+ if (!reader) return res;
84
+ const acc = { id: null, model: null, content: "", reasoning: "", toolCalls: [], finishReason: null, usage: null, error: null };
85
+ const decoder = new TextDecoder();
86
+ let buf = "";
87
+ const eat = (frame) => {
88
+ for (const line of frame.split("\n")) {
89
+ const s = line.trim();
90
+ if (!s.startsWith("data:")) continue;
91
+ const payload = s.slice(5).trim();
92
+ if (!payload || payload === "[DONE]") continue;
93
+ let j;
94
+ try { j = JSON.parse(payload); } catch { continue; }
95
+ if (j?.error) { acc.error = acc.error || j.error; continue; }
96
+ if (j?.id) acc.id = j.id;
97
+ if (j?.model) acc.model = j.model;
98
+ if (j?.usage) acc.usage = j.usage;
99
+ const ch = j?.choices?.[0];
100
+ if (!ch) continue;
101
+ const d = ch.delta || {};
102
+ if (typeof d.content === "string") acc.content += d.content;
103
+ if (typeof d.reasoning_content === "string") acc.reasoning += d.reasoning_content;
104
+ else if (typeof d.reasoning === "string") acc.reasoning += d.reasoning;
105
+ if (Array.isArray(d.tool_calls)) for (const tc of d.tool_calls) mergeToolCall(acc.toolCalls, tc.index ?? 0, tc);
106
+ if (ch.finish_reason) acc.finishReason = ch.finish_reason;
107
+ if (ch.message && typeof ch.message.content === "string" && !acc.content) {
108
+ acc.content = ch.message.content;
109
+ if (Array.isArray(ch.message.tool_calls)) for (const tc of ch.message.tool_calls) mergeToolCall(acc.toolCalls, tc.index ?? 0, tc);
110
+ }
111
+ }
112
+ };
113
+ try {
114
+ for (;;) {
115
+ const { done, value } = await reader.read();
116
+ if (done) break;
117
+ buf += decoder.decode(value, { stream: true });
118
+ const frames = buf.split("\n\n");
119
+ buf = frames.pop() || "";
120
+ for (const f of frames) eat(f);
121
+ }
122
+ if (buf.trim()) eat(buf);
123
+ } finally {
124
+ try { reader.releaseLock?.(); } catch {}
125
+ }
126
+ if (acc.error && !acc.content && !acc.toolCalls.length) {
127
+ const body = JSON.stringify({ error: acc.error });
128
+ return new Response(body, { status: 502, headers: { "content-type": "application/json" } });
129
+ }
130
+ return new Response(JSON.stringify(finishSseJson(acc)), { status: res.status, headers: { "content-type": "application/json" } });
131
+ }
@@ -3,9 +3,10 @@ import crypto from "node:crypto";
3
3
  // opencode 客户端身份规格单一来源(源码 packages/schema/src/identifier.ts +
4
4
  // packages/opencode/src/session/llm/request.ts):id = <prefix>_ + 12 位 hex
5
5
  // (timestamp*4096+同毫秒计数,截 48bit) + 14 位 base62,共 26 字符;UA 必须 "opencode/<semver>"。
6
- // zen 免费层 2026-09-17 起按该规格放行(缺版本 UA 或 id 形状不符 → 403 FreeTierError)。
6
+ // zen 免费层 2026-09-17 起按该规格放行(缺版本 UA 或 id 形状不符 → 403 FreeTierError);
7
+ // 2026-09-18 起还要求 UA 版本 ≥ 1.18.0(低版本 426 UpgradeRequired),默认跟 npm latest。
7
8
  const ID_CHARS = "0123456789ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyz";
8
- const DEFAULT_OPENCODE_UA = "opencode/1.17.20";
9
+ const DEFAULT_OPENCODE_UA = "opencode/1.18.31";
9
10
  let idLastTs = 0;
10
11
  let idCounter = 0;
11
12
 
@@ -2,6 +2,7 @@ import { runHook } from "../../plugins.js";
2
2
  import { recordModelStats } from "../../state.js";
3
3
  import { normalizeFullId } from "../../providers/model-id.js";
4
4
  import { computeMetrics } from "../../metrics.js";
5
+ import { recordChatUsage } from "../../usage/record.js"; // 窗口报表唯一写入点(canonical 名单记防双计)— 见 .agents/notes/implemented/feature/2026-09-19-usage-report-jsonl.md
5
6
 
6
7
  // 唯一/最后候选没有 failover 去向:首块闸门退化为纯"防连接泄漏",放宽避免误杀慢模型
7
8
  // (参考 opencode:zen 通道不设超时;openai responses 硬编码 300s headerTimeout)
@@ -134,6 +135,15 @@ export function createRelayPipeline({
134
135
  }
135
136
  _evt("slow-model", { reqId, model: actual, elapsedMs: out.totalMs ?? (Date.now() - curStartedAt), threshold: C.STALL_TIMEOUT_MS, interrupted: true, detail: out.detail ?? null });
136
137
  try { _logCall(actual, 200); } catch {}
138
+ // interrupted 的 200 也是真实消耗(最贵的长生成)——照常落 usage 标 interrupted:1,口径与 5c 一致 — 见 .agents/notes/implemented/feature/2026-09-19-usage-report-jsonl.md
139
+ if (out.status === 200) {
140
+ try {
141
+ const u = out.detail?.usage || null;
142
+ const t1 = Number.isFinite(out.totalMs) && out.totalMs > 0 ? out.totalMs : (Date.now() - curStartedAt);
143
+ const t0 = Number.isFinite(out.ttfMs) && out.ttfMs > 0 ? out.ttfMs : null;
144
+ recordChatUsage({ model: normalizeFullId(actual), via, usage: u, interrupted: 1, ttfbMs: t0, totalMs: t1, tps: null }).catch(() => {});
145
+ } catch {}
146
+ }
137
147
  _evt("result", { reqId, model: actual, status: out.status, via, timing: upRes?._t ?? null, ttfMs: out.ttfMs, totalMs: out.totalMs, interrupted: true, detail: out.detail ?? null, fallback, requested, actual });
138
148
  _evt("client-response", { requested, actual, via, fallback, status: out.status, reqId, interrupted: true });
139
149
  if (plugins?.length) runHook(plugins, "request:completed", { reqId, requested, useAuto, hops, stream: Boolean(body?.stream), durationMs: Date.now() - curStartedAt, via, status: out.status, actual, interrupted: true, fallback }).catch(() => {});
@@ -183,6 +193,9 @@ export function createRelayPipeline({
183
193
  const fullId = normalizeFullId(actual);
184
194
  recordModelStats(fullId, { ttfbMs: ttfb, totalMs: total, tps, completionTokens: compTok });
185
195
  if (fullId !== actual) recordModelStats(actual, { ttfbMs: ttfb, totalMs: total, tps, completionTokens: compTok });
196
+ // 窗口报表:逐请求落 usage(行形状由 usage/record.js 拥有,含 prompt/total ——
197
+ // state 的 modelStats 只存 completion 的 EMA)。只按 canonical 名记一次,避免双计。
198
+ recordChatUsage({ model: fullId, via, usage, ttfbMs: ttfb, totalMs: total, tps: m.tps }).catch(() => {});
186
199
  } catch {}
187
200
  }
188
201
 
@@ -5,6 +5,7 @@
5
5
  import { createUpstreamClient, createOpencodeHeaderBuilder } from "../upstream.js";
6
6
  import { isResponsesModel } from "../upstream-responses.js";
7
7
  import { isFreeModel } from "../models.js";
8
+ import { ensureFreeLaneShape } from "../free-lane.js";
8
9
  import { createSdkChat } from "./sdk/chat.js";
9
10
  import { createSdkResponses } from "./sdk/responses.js";
10
11
  import { dispatcherFetch } from "./sdk/attempt.js";
@@ -32,9 +33,11 @@ export function createUpstreamEngine(opts = {}) {
32
33
  let logged = false;
33
34
 
34
35
  async function chat(body) {
35
- if (sdkDown) return legacy.chat(body);
36
- // doGenerate 聚合未实现:非流式(含 responses 非流式)委派 legacy,避免把 JSON 客户端 SSE 化
37
- if (body?.stream === false) return legacy.chat(body);
36
+ // zen 免费层 agent 形状门禁(2026-09-18):SDK 流式通道不经 legacy.chat,必须在这里补形状。
37
+ // 非流式先委派 legacy(它在自己内部注入并把 SSE 聚合回 JSON,避免这里先改 stream 导致误判)。
38
+ const wantsStream = body?.stream !== false;
39
+ if (sdkDown || !wantsStream) return legacy.chat(body);
40
+ if (isFreeModel(body?.model)) ensureFreeLaneShape(body);
38
41
  const useResponses = isResponsesModel(body?.model);
39
42
  try {
40
43
  const res = useResponses ? await responses.chat(body) : await sdk.chat(body);
package/src/upstream.js CHANGED
@@ -7,6 +7,7 @@ import { isFreeModel } from "./models.js";
7
7
  import { fmtShanghaiYMDHMS } from "./time.js";
8
8
  import { createTransport } from "./transport/index.js";
9
9
  import { isResponsesModel, chatToResponsesBody, toChatResponse, reshapeResponsesSse } from "./upstream-responses.js";
10
+ import { ensureFreeLaneShape, aggregateChatSse } from "./free-lane.js";
10
11
  import { digestIdTail, genId, opencodeClientIdentity, opencodeUa } from "./opencode-identity.js";
11
12
 
12
13
  export { opencodeClientIdentity, opencodeUa };
@@ -136,6 +137,27 @@ export function createUpstreamClient({
136
137
  const sessionId = opts?.sessionId || sessionFromMessages(body?.messages) || null;
137
138
  const url = isResp ? `${baseUrl}/zen/v1/responses` : `${baseUrl}/zen/v1/chat/completions`;
138
139
  const reqBody = isResp ? chatToResponsesBody(body) : body;
140
+ const clientWantsStream = body?.stream !== false;
141
+ // zen 免费层 agent 形状门禁(2026-09-18):必须流式 + 含核心五工具名,否则 403 FreeTierError
142
+ const freeLane = authToken === "public" && isFreeModel(body?.model);
143
+ if (freeLane) ensureFreeLaneShape(reqBody, { responses: isResp });
144
+ const laneDebug = freeLane && envInt("MSLXDFF_FREE_LANE_DEBUG", 0) === 1;
145
+ if (laneDebug) {
146
+ try { console.error(`[free-lane] send model=${body?.model} url=${url} stream=${reqBody.stream} tools=${(reqBody.tools || []).length} ua=${opencodeUa()} want=${clientWantsStream}`); } catch {}
147
+ }
148
+ // 非流式调用者实际拿到的是被强制流式的上游 → 聚合回 chat completion JSON
149
+ async function agentJson(resp) {
150
+ if (!(freeLane && !clientWantsStream && resp.ok)) return resp;
151
+ const ct = resp.headers?.get?.("content-type") || "";
152
+ if (!ct.includes("text/event-stream")) return resp;
153
+ try {
154
+ const out = await aggregateChatSse(resp);
155
+ out._t = resp._t;
156
+ return out;
157
+ } catch {
158
+ return resp;
159
+ }
160
+ }
139
161
  const t0 = performance.now();
140
162
 
141
163
  // 首发请求(transport 已处理 network/429 等重试)
@@ -146,7 +168,7 @@ export function createUpstreamClient({
146
168
  method: "POST",
147
169
  headers: buildHeaders(reqBody, { sessionId }),
148
170
  body: reqBody,
149
- stream: body?.stream !== false,
171
+ stream: reqBody.stream !== false,
150
172
  timeoutMs: connectTimeoutMs,
151
173
  retry,
152
174
  });
@@ -154,6 +176,9 @@ export function createUpstreamClient({
154
176
  e._t = e._t || { attempts: [], waitMs: 0, totalMs: Math.round(performance.now() - t0) };
155
177
  throw e;
156
178
  }
179
+ if (laneDebug) {
180
+ try { console.error(`[free-lane] resp model=${body?.model} status=${res.status} ct=${res.headers?.get?.("content-type") || ""}`); } catch {}
181
+ }
157
182
 
158
183
  // 匿名兜底:仅非 anonFirst 时,public 429 + free 模型才走 hermes 空头重试
159
184
  if (!anonFirst && res.status === 429 && isFreeModel(body?.model) && shouldTryAnonFree()) {
@@ -168,7 +193,7 @@ export function createUpstreamClient({
168
193
  method: "POST",
169
194
  headers: buildHeaders(reqBody, { anonymous: true, sessionId }),
170
195
  body: reqBody,
171
- stream: body?.stream !== false,
196
+ stream: reqBody.stream !== false,
172
197
  timeoutMs: connectTimeoutMs,
173
198
  retry,
174
199
  });
@@ -180,7 +205,7 @@ export function createUpstreamClient({
180
205
  let outAnon = anonRes;
181
206
  if (isResp && anonRes.ok) {
182
207
  const ctAnon = anonRes.headers.get("content-type") || "";
183
- const isStreamAnon = body?.stream !== false && ctAnon.includes("text/event-stream");
208
+ const isStreamAnon = (clientWantsStream || freeLane) && ctAnon.includes("text/event-stream");
184
209
  if (isStreamAnon) {
185
210
  outAnon = reshapeResponsesSse(anonRes, body.model);
186
211
  } else {
@@ -201,7 +226,7 @@ export function createUpstreamClient({
201
226
  if (consecutiveHits >= 2) {
202
227
  try { appendFile(freeAnonLogFile(), ` -> 连续额外额度 ${consecutiveHits} 次\n`).catch(() => {}); } catch {}
203
228
  }
204
- return outAnon;
229
+ return await agentJson(outAnon);
205
230
  }
206
231
  }
207
232
  if (anonRes) {
@@ -215,11 +240,11 @@ export function createUpstreamClient({
215
240
  // responses 模型成功态转 chat(复用 upstream-responses)
216
241
  if (isResp && res.ok) {
217
242
  const ct = res.headers.get("content-type") || "";
218
- const isStream = body?.stream !== false && ct.includes("text/event-stream");
243
+ const isStream = (clientWantsStream || freeLane) && ct.includes("text/event-stream");
219
244
  if (isStream) {
220
245
  const transformed = reshapeResponsesSse(res, body.model);
221
246
  transformed._t = { ...(res._t || {}), totalMs: Math.round(performance.now() - t0) };
222
- return transformed;
247
+ return await agentJson(transformed);
223
248
  }
224
249
  try {
225
250
  const txt = await res.text();
@@ -238,7 +263,7 @@ export function createUpstreamClient({
238
263
  }
239
264
  }
240
265
  res._t = { ...(res._t || {}), totalMs: res._t?.totalMs ?? Math.round(performance.now() - t0) };
241
- return res;
266
+ return await agentJson(res);
242
267
  }
243
268
 
244
269
  async function preheat() {
@@ -0,0 +1,99 @@
1
+ // 逐请求 usage 落盘:按日切片的 JSONL,供 -stats 做时间窗口聚合。
2
+ // Note: 与 logs.js 的 calls/errors/events 不同,这里按日分片 + 保留期删旧文件,
3
+ // 不走 1MB 环形截断 —— 环形会把 24h 窗口的数据裁到末 100 行。见 .scratch/stats-report/SPEC.md
4
+ import { appendFile, mkdir, readdir, unlink } from "node:fs/promises";
5
+ import { join } from "node:path";
6
+ import { logDir } from "../logs.js";
7
+
8
+ const DEFAULT_KEEP_DAYS = 2;
9
+ const DAY_MS = 86_400_000;
10
+
11
+ export function usageEnabled() {
12
+ return process.env.MSLXDFF_USAGE_LOG !== "0";
13
+ }
14
+
15
+ export function usageKeepDays() {
16
+ const n = Number(process.env.MSLXDFF_USAGE_KEEP_DAYS);
17
+ return Number.isInteger(n) && n > 0 ? n : DEFAULT_KEEP_DAYS;
18
+ }
19
+
20
+ export function usageDir({ dir } = {}) {
21
+ return join(dir || logDir(), "usage");
22
+ }
23
+
24
+ export function ymd(date) {
25
+ const y = date.getFullYear();
26
+ const m = String(date.getMonth() + 1).padStart(2, "0");
27
+ const d = String(date.getDate()).padStart(2, "0");
28
+ return `${y}-${m}-${d}`;
29
+ }
30
+
31
+ export function usageFileFor(date, { dir } = {}) {
32
+ return join(usageDir({ dir }), `${ymd(date)}.jsonl`);
33
+ }
34
+
35
+ // 删掉早于保留期的日文件。文件名是 YYYY-MM-DD,可直接字典序比较。
36
+ export async function pruneUsage({ dir, keepDays, now = new Date() } = {}) {
37
+ const keep = Number.isInteger(keepDays) && keepDays > 0 ? keepDays : usageKeepDays();
38
+ const cutoff = ymd(new Date(now.getTime() - keep * DAY_MS));
39
+ const removed = [];
40
+ let files = [];
41
+ try {
42
+ files = await readdir(usageDir({ dir }));
43
+ } catch {
44
+ return removed; // 目录还不存在 = 无事可做
45
+ }
46
+ for (const f of files) {
47
+ if (!f.endsWith(".jsonl")) continue;
48
+ const day = f.slice(0, -".jsonl".length);
49
+ if (!/^\d{4}-\d{2}-\d{2}$/.test(day) || day >= cutoff) continue;
50
+ try {
51
+ await unlink(join(usageDir({ dir }), f));
52
+ removed.push(f);
53
+ } catch {}
54
+ }
55
+ return removed;
56
+ }
57
+
58
+ let lastPrunedDay = null;
59
+
60
+ // 追加一行 usage。异步写,调用方可 fire-and-forget(不阻塞流式响应)。
61
+ // 清理是惰性的且每天最多触发一次,避免每个请求都去 readdir。
62
+ export async function recordUsage(entry, { dir, now = new Date() } = {}) {
63
+ if (!usageEnabled() || !entry || typeof entry !== "object") return null;
64
+ const row = { ts: now.getTime(), ...entry };
65
+ try {
66
+ await mkdir(usageDir({ dir }), { recursive: true });
67
+ await appendFile(usageFileFor(now, { dir }), JSON.stringify(row) + "\n");
68
+ } catch {
69
+ return null;
70
+ }
71
+ const today = ymd(now);
72
+ if (lastPrunedDay !== today) {
73
+ lastPrunedDay = today;
74
+ pruneUsage({ dir, now }).catch(() => {});
75
+ }
76
+ return row;
77
+ }
78
+
79
+ // 从 relay 结果组装一行 usage —— 行形状由本模块拥有,调用方只交原始字段。
80
+ // usage 即 metrics.js 的 extractUsageFromJson 输出(prompt/completion/total/reasoning)。
81
+ export function recordChatUsage({ model, via, usage, ttfbMs, totalMs, tps, interrupted } = {}) {
82
+ return recordUsage({
83
+ model,
84
+ via,
85
+ interrupted,
86
+ prompt_tokens: usage?.prompt_tokens ?? 0,
87
+ completion_tokens: usage?.completion_tokens ?? 0,
88
+ total_tokens: usage?.total_tokens ?? 0,
89
+ reasoning_tokens: usage?.reasoning_tokens ?? 0,
90
+ ttfbMs: Number.isFinite(ttfbMs) ? ttfbMs : null,
91
+ totalMs: Number.isFinite(totalMs) ? totalMs : null,
92
+ tps: Number.isFinite(tps) ? tps : null,
93
+ });
94
+ }
95
+
96
+ // 测试用:重置惰性清理标记,避免跨用例串味(置于文件末,业务函数在其上)
97
+ export function _resetPruneMarker() {
98
+ lastPrunedDay = null;
99
+ }
@@ -0,0 +1,152 @@
1
+ // 窗口用量聚合:JSONL 行数组 + 时间窗口 → 每模型 token/速度报表。
2
+ // 纯函数(aggregateUsage)与薄 IO(readUsageRows / usageReport)分开,前者可离线单测。
3
+ // Note: 速度用加权口径 Σcompletion / ΣcompletionMs,不用算术平均 —— 短回答会把算术均值拉飞。
4
+ import { readFile, readdir } from "node:fs/promises";
5
+ import { join } from "node:path";
6
+ import { usageDir } from "./record.js";
7
+
8
+ const HOUR_MS = 3_600_000;
9
+
10
+ function tok(v) {
11
+ const n = Number(v);
12
+ return Number.isFinite(n) && n > 0 ? n : 0;
13
+ }
14
+
15
+ function ms(v) {
16
+ const n = Number(v);
17
+ return Number.isFinite(n) && n >= 0 ? n : null;
18
+ }
19
+
20
+ function blank(id) {
21
+ return {
22
+ id,
23
+ requests: 0,
24
+ promptTokens: 0,
25
+ completionTokens: 0,
26
+ totalTokens: 0,
27
+ reasoningTokens: 0,
28
+ ttfbSumMs: 0,
29
+ ttfbN: 0,
30
+ totalSumMs: 0,
31
+ totalN: 0,
32
+ completionMsSum: 0,
33
+ tpsTokSum: 0,
34
+ };
35
+ }
36
+
37
+ // 把累计量收敛成对外字段:平均首字/平均总耗时/加权速度。
38
+ function finalize(a) {
39
+ return {
40
+ id: a.id,
41
+ requests: a.requests,
42
+ promptTokens: a.promptTokens,
43
+ completionTokens: a.completionTokens,
44
+ totalTokens: a.totalTokens,
45
+ reasoningTokens: a.reasoningTokens,
46
+ avgTtfbMs: a.ttfbN ? Math.round(a.ttfbSumMs / a.ttfbN) : null,
47
+ avgTotalMs: a.totalN ? Math.round(a.totalSumMs / a.totalN) : null,
48
+ avgTps: a.completionMsSum > 0 ? Number((a.tpsTokSum / (a.completionMsSum / 1000)).toFixed(1)) : null,
49
+ };
50
+ }
51
+
52
+ function fold(acc, r) {
53
+ acc.requests++;
54
+ const prompt = tok(r.prompt_tokens);
55
+ const comp = tok(r.completion_tokens);
56
+ acc.promptTokens += prompt;
57
+ acc.completionTokens += comp;
58
+ const total = tok(r.total_tokens);
59
+ acc.totalTokens += total || prompt + comp;
60
+ acc.reasoningTokens += tok(r.reasoning_tokens);
61
+
62
+ const ttfb = ms(r.ttfbMs ?? r.ttfb_ms);
63
+ if (ttfb != null) {
64
+ acc.ttfbSumMs += ttfb;
65
+ acc.ttfbN++;
66
+ }
67
+ const totalMs = ms(r.totalMs ?? r.total_ms);
68
+ if (totalMs != null && totalMs > 0) {
69
+ acc.totalSumMs += totalMs;
70
+ acc.totalN++;
71
+ }
72
+ // 生成阶段耗时 = 总耗时 - 首字(首字缺失按 0 计)
73
+ if (totalMs != null && totalMs > 0 && comp > 0) {
74
+ const compMs = Math.max(0, totalMs - (ttfb ?? 0));
75
+ if (compMs > 0) {
76
+ acc.completionMsSum += compMs;
77
+ acc.tpsTokSum += comp;
78
+ }
79
+ }
80
+ return acc;
81
+ }
82
+
83
+ // 纯函数:只做过滤 + 归并,不碰磁盘。
84
+ export function aggregateUsage(rows, { hours = 24, now = Date.now(), model = null } = {}) {
85
+ const windowHours = Number.isFinite(Number(hours)) && Number(hours) > 0 ? Number(hours) : 24;
86
+ const untilN = Number(now);
87
+ const until = Number.isFinite(untilN) ? untilN : Date.now();
88
+ const since = until - windowHours * HOUR_MS;
89
+ const byModel = new Map();
90
+ const sum = blank(null);
91
+ let scanned = 0;
92
+
93
+ if (Array.isArray(rows)) {
94
+ for (const r of rows) {
95
+ if (!r || typeof r !== "object") continue;
96
+ const ts = Number(r.ts);
97
+ if (!Number.isFinite(ts) || ts < since || ts > until) continue;
98
+ const id = String(r.model || "").trim();
99
+ if (!id) continue;
100
+ // model 过滤同时接受 canonical 全称与裸 id(如 --model big-pickle 匹配 opencode/big-pickle 行)
101
+ if (model && id !== model && !id.endsWith("/" + model)) continue;
102
+ scanned++;
103
+ let acc = byModel.get(id);
104
+ if (!acc) { acc = blank(id); byModel.set(id, acc); }
105
+ fold(acc, r);
106
+ fold(sum, r);
107
+ }
108
+ }
109
+ const models = [...byModel.values()].map(finalize);
110
+ models.sort((a, b) => (b.totalTokens - a.totalTokens) || (b.requests - a.requests) || a.id.localeCompare(b.id));
111
+ return {
112
+ windowHours,
113
+ since,
114
+ until,
115
+ rows: scanned,
116
+ models,
117
+ totals: finalize(sum),
118
+ };
119
+ }
120
+
121
+ // 读保留期内的日文件并只留窗口内的行(保留期默认 2 天,文件数很少,全读即可)。
122
+ export async function readUsageRows({ dir, hours = 24, now = Date.now() } = {}) {
123
+ const since = Number(now) - (Number(hours) > 0 ? Number(hours) : 24) * HOUR_MS;
124
+ let files = [];
125
+ try {
126
+ files = (await readdir(usageDir({ dir }))).filter((f) => f.endsWith(".jsonl")).sort();
127
+ } catch {
128
+ return [];
129
+ }
130
+ const rows = [];
131
+ for (const f of files) {
132
+ let text = "";
133
+ try {
134
+ text = await readFile(join(usageDir({ dir }), f), "utf8");
135
+ } catch {
136
+ continue;
137
+ }
138
+ for (const line of text.split("\n")) {
139
+ if (!line) continue;
140
+ try {
141
+ const r = JSON.parse(line);
142
+ if (Number(r?.ts) >= since) rows.push(r);
143
+ } catch {}
144
+ }
145
+ }
146
+ return rows;
147
+ }
148
+
149
+ export async function usageReport({ dir, hours = 24, now = Date.now(), model = null } = {}) {
150
+ const rows = await readUsageRows({ dir, hours, now });
151
+ return aggregateUsage(rows, { hours, now, model });
152
+ }