mslxdff 0.1.159 → 0.1.161
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/docs/cli_help_mini.md +3 -1
- package/package.json +1 -1
- package/src/autostart.js +3 -0
- package/src/chat-pipeline/auto-race.js +1 -1
- package/src/chat-pipeline/empty-turn.js +119 -0
- package/src/chat-pipeline/index.js +4 -1
- package/src/chat-pipeline/serial-trial.js +244 -210
- package/src/cli/commands/daemon.js +4 -4
- package/src/cli/commands/model/list-providers.js +1 -1
- package/src/cli/commands/provider/globalqwenwork-login.js +126 -0
- package/src/cli/commands/provider/index.js +2 -0
- package/src/cli/commands/provider/models.js +4 -1
- package/src/cli/commands/stats.js +20 -5
- package/src/cli/commands/system.js +2 -2
- package/src/cli/format.js +6 -0
- package/src/cli/policy.js +2 -2
- package/src/cli/provider-row.js +2 -2
- package/src/daemon.js +7 -2
- package/src/logs.js +14 -1
- package/src/model-trace.js +5 -2
- package/src/providers/AGENTS.md +46 -0
- package/src/providers/classify.js +1 -1
- package/src/providers/globalqwenwork/account-store.js +141 -0
- package/src/providers/globalqwenwork/constants.js +73 -0
- package/src/providers/globalqwenwork/cosy.js +123 -0
- package/src/providers/globalqwenwork/crypto.js +220 -0
- package/src/providers/globalqwenwork/http.js +22 -0
- package/src/providers/globalqwenwork/index.js +331 -0
- package/src/providers/globalqwenwork/payload.js +145 -0
- package/src/providers/globalqwenwork/rsa.js +56 -0
- package/src/providers/globalqwenwork/sse.js +270 -0
- package/src/providers/globalqwenwork/stream.js +132 -0
- package/src/providers/globalqwenwork/upstream.js +122 -0
- package/src/providers/globalqwenwork.js +1 -0
- package/src/providers/registry.js +8 -0
- package/src/providers/zcode/chat.js +11 -10
- package/src/providers/zcode/const.js +1 -1
- package/src/providers/zcode/context-shape.js +162 -0
- package/src/providers/zcode/headers.js +38 -0
- package/src/providers/zcode/sse.js +19 -2
- package/src/routes/AGENTS.md +37 -0
- package/src/routes/chat/broadband-handler.js +3 -2
- package/src/routes/chat/exhausted-handler.js +23 -8
- package/src/routes/chat/hedge-handler.js +3 -0
- package/src/routes/chat/local-handler.js +5 -1
- package/src/routes/chat/peer-handler.js +2 -0
- package/src/routes/chat/relay-pipeline.js +40 -14
- package/src/routes/chat/via-route-handler.js +2 -0
- package/src/routes/helpers.js +20 -4
- package/src/routes/stream-hold.js +138 -0
- package/src/routes/stream-scan.js +278 -0
- package/src/routes/stream.js +106 -158
- package/src/runtime/lifecycle-forensics.js +116 -0
- package/src/runtime/lifecycle-log.js +23 -0
- package/src/runtime/provider-gate.js +4 -3
- package/src/talk-log.js +226 -0
- package/src/timeline.js +5 -2
- package/src/usage/record.js +8 -1
- package/src/usage/report.js +39 -2
package/src/talk-log.js
ADDED
|
@@ -0,0 +1,226 @@
|
|
|
1
|
+
// 环形对话日志(talk log):按「供应商-模型-talk.log」分文件,只保留最近 1 小时的完整对话。
|
|
2
|
+
// 记三件事:我问了什么 / 模型怎么想(reasoning)/ 模型怎么答(content),供人工调研模型的回答方式。
|
|
3
|
+
// 与 model-trace 相反:那边刻意不落正文(「加日志」≠「倒数据」),这里专门落正文 —— 所以必须脱敏 + 封顶。
|
|
4
|
+
// 默认开;MSLXDFF_TALK_LOG=0 关闭,窗口 MSLXDFF_TALK_LOG_WINDOW_MIN(分,默认 60),单文件 MSLXDFF_TALK_LOG_MAX_KB(默认 5MB)。
|
|
5
|
+
import { appendFileSync, chmodSync, closeSync, existsSync, mkdirSync, openSync, readFileSync, readdirSync, readSync, renameSync, rmSync, statSync, writeFileSync } from "node:fs";
|
|
6
|
+
import { dirname, join } from "node:path";
|
|
7
|
+
import { logDir } from "./logs.js";
|
|
8
|
+
import { fmtShanghaiYMDHMS } from "./time.js";
|
|
9
|
+
|
|
10
|
+
const SUBDIR = "talk";
|
|
11
|
+
const DEFAULT_WINDOW_MIN = 60;
|
|
12
|
+
const DEFAULT_MAX_KB = 5 * 1024; // 5MB per talk-log file (ring buffer)
|
|
13
|
+
const MARK = "#talk-entry";
|
|
14
|
+
// 正常路径只 append;整文件重写(环形淘汰)只在「最老条目出窗」或「超字节上限」时发生。
|
|
15
|
+
// oldestCache 记住每个文件当前最老的 ts,避免每次写入都回读全文件。
|
|
16
|
+
const oldestCache = new Map();
|
|
17
|
+
|
|
18
|
+
export function talkLogEnabled() { return String(process.env.MSLXDFF_TALK_LOG ?? "1") !== "0"; }
|
|
19
|
+
function windowMs() { const n = Number(process.env.MSLXDFF_TALK_LOG_WINDOW_MIN); return Number.isFinite(n) && n > 0 ? n * 60000 : DEFAULT_WINDOW_MIN * 60000; }
|
|
20
|
+
function maxBytes() { const n = Number(process.env.MSLXDFF_TALK_LOG_MAX_KB); return Number.isFinite(n) && n > 0 ? n * 1024 : DEFAULT_MAX_KB * 1024; }
|
|
21
|
+
/** 仅测试用:进程重启等价物(环形窗口的内存基线清空)。 */
|
|
22
|
+
export function resetTalkLogCache() { oldestCache.clear(); }
|
|
23
|
+
|
|
24
|
+
/** 供应商-模型-talk.log:qoder/qfmodel → qoder-qfmodel-talk.log;裸 id(免费池)直接用一个名字段。 */
|
|
25
|
+
export function talkLogName(model) {
|
|
26
|
+
const s = String(model || "unknown").trim().toLowerCase();
|
|
27
|
+
const segs = s.includes("/") ? s.split("/").filter((x) => x && x !== "unknown") : [s];
|
|
28
|
+
const safe = segs.join("-").replace(/[^a-z0-9._-]+/g, "-").replace(/^[-.]+|[-.]+$/g, "").slice(0, 180);
|
|
29
|
+
return `${safe || "unknown"}-talk.log`;
|
|
30
|
+
}
|
|
31
|
+
|
|
32
|
+
export function talkDir() { return join(logDir(), SUBDIR); }
|
|
33
|
+
export function talkLogFile(model) { return join(talkDir(), talkLogName(model)); }
|
|
34
|
+
|
|
35
|
+
// 正文里的凭据一律先抹再落盘(保留换行,不像 model-trace 那样压成一行 —— 这里要读的是回答本身)。
|
|
36
|
+
const MASKS = [
|
|
37
|
+
[/([?&](?:token|refresh[_-]?token|access[_-]?token|api[_-]?key|apikey|key|secret|password|cookie)=)[^\s&]+/gi, "$1[已脱敏]"],
|
|
38
|
+
[/(authorization|bearer|proxy-authorization|cookie|x-api-key)\s*[:=]\s*[^\s,;]+/gi, "$1=[已脱敏]"],
|
|
39
|
+
[/("(?:apiKey|api_key|accessToken|refreshToken|secret|password|client_secret)"\s*:\s*)"[^"]*"/gi, '$1"[已脱敏]"'],
|
|
40
|
+
[/\beyJ[\w-]{6,}\.[\w-]{6,}\.[\w-]{6,}\b/g, "[已脱敏-JWT]"],
|
|
41
|
+
[/\b(?:sk|pk|ghp|gho|ghu|ghs|ghr|xox[baprs])-[\w-]{10,}\b/g, "[已脱敏-key]"],
|
|
42
|
+
// 裸 k=v(提问里常出现 token=…、password=…):只掐足够长的值,避免把「token 数量」这类正常技术讨论也糊掉
|
|
43
|
+
[/\b(token|accessToken|refreshToken|api[_-]?key|apiKey|secret|password|cookie|client_secret|authorization)\s*[:=]\s*[\w./+-]{12,}/gi, "$1=[已脱敏]"],
|
|
44
|
+
];
|
|
45
|
+
|
|
46
|
+
/** 脱敏 + 按字节封顶截断(中文 3 字节/字,逐字累加才不会超预算)。 */
|
|
47
|
+
export function maskText(v, cap = 100000) {
|
|
48
|
+
let s = String(v ?? "");
|
|
49
|
+
for (const [re, repl] of MASKS) s = s.replace(re, repl);
|
|
50
|
+
if (Buffer.byteLength(s, "utf8") <= cap) return s;
|
|
51
|
+
const tail = Buffer.from("\n…[已截断]", "utf8");
|
|
52
|
+
const budget = Math.max(0, cap - tail.length);
|
|
53
|
+
let bytes = 0, i = 0;
|
|
54
|
+
for (const ch of s) {
|
|
55
|
+
const n = Buffer.byteLength(ch, "utf8");
|
|
56
|
+
if (bytes + n > budget) break;
|
|
57
|
+
bytes += n;
|
|
58
|
+
i += ch.length;
|
|
59
|
+
}
|
|
60
|
+
return s.slice(0, i) + "\n…[已截断]";
|
|
61
|
+
}
|
|
62
|
+
|
|
63
|
+
function textOf(content) {
|
|
64
|
+
if (typeof content === "string") return content;
|
|
65
|
+
if (Array.isArray(content)) return content.map((p) => p?.text || (p?.image_url ? `[图片 ${String(p.image_url.url).slice(0, 60)}]` : "")).filter(Boolean).join("\n");
|
|
66
|
+
return "";
|
|
67
|
+
}
|
|
68
|
+
|
|
69
|
+
/** 只取「我发给大模型的问题」= 最后一条 user 消息(agent 回路每轮都带全量历史,全落会把环形窗口占满)。 */
|
|
70
|
+
export function lastUserQuestion(body) {
|
|
71
|
+
const msgs = Array.isArray(body?.messages) ? body.messages : null;
|
|
72
|
+
if (msgs) { for (let i = msgs.length - 1; i >= 0; i--) if (msgs[i]?.role === "user") return textOf(msgs[i].content); }
|
|
73
|
+
const inp = body?.input; // Responses 端点形状兜底
|
|
74
|
+
if (typeof inp === "string") return inp;
|
|
75
|
+
if (Array.isArray(inp)) { for (let i = inp.length - 1; i >= 0; i--) if (inp[i]?.role === "user") return textOf(inp[i].content); }
|
|
76
|
+
return "";
|
|
77
|
+
}
|
|
78
|
+
|
|
79
|
+
const CUT_IN = "----8<----", CUT_OUT = "---->8----";
|
|
80
|
+
|
|
81
|
+
function pushBlock(out, label, text, cap) {
|
|
82
|
+
const m = maskText(text, cap);
|
|
83
|
+
if (!m.trim()) return;
|
|
84
|
+
out.push(`[${label} · ${m.length} 字]`, CUT_IN, m, CUT_OUT, "");
|
|
85
|
+
}
|
|
86
|
+
|
|
87
|
+
/** 一条对话 = 一个可读块:头部一行元信息(ts= 是环形淘汰的唯一依据),正文三段。 */
|
|
88
|
+
export function formatTalkEntry({ reqId, model, via, hops, status, stream, elapsedMs, question, thinking, answer, tools, usage, upstream, account, pick, ts } = {}) {
|
|
89
|
+
const now = Number.isFinite(ts) && ts > 0 ? ts : Date.now();
|
|
90
|
+
const head = [
|
|
91
|
+
`${MARK} ts=${now}`, `time=${fmtShanghaiYMDHMS(new Date(now))}`, `req=${reqId || "-"}`, `model=${model || "-"}`,
|
|
92
|
+
via ? `via=${via}` : "", hops != null ? `hops=${hops}` : "", `status=${status ?? "-"}`,
|
|
93
|
+
stream ? "stream=1" : "stream=0", Number.isFinite(elapsedMs) ? `elapsed=${Math.round(elapsedMs)}ms` : "",
|
|
94
|
+
usage ? `usage(prompt=${usage.prompt_tokens ?? "-"},completion=${usage.completion_tokens ?? "-"})` : "",
|
|
95
|
+
upstream ? `upstream=${upstream}` : "", account ? `account=${account}` : "", pick ? `pick=${pick}` : "",
|
|
96
|
+
].filter(Boolean).join(" ");
|
|
97
|
+
const out = [head, ""];
|
|
98
|
+
pushBlock(out, "我问", question, 20000);
|
|
99
|
+
pushBlock(out, "思考", thinking, 100000);
|
|
100
|
+
pushBlock(out, "回答", answer, 200000);
|
|
101
|
+
const t = tools == null || tools === "" ? "" : String(tools);
|
|
102
|
+
pushBlock(out, "工具调用", t, 20000);
|
|
103
|
+
out.push("[END]", "");
|
|
104
|
+
return out.join("\n") + "\n";
|
|
105
|
+
}
|
|
106
|
+
|
|
107
|
+
/** 只读文件头 8KB 取最老一条的 ts(进程重启后重建内存基线,不必回读全文件)。 */
|
|
108
|
+
function firstTs(file) {
|
|
109
|
+
let fd = null;
|
|
110
|
+
try {
|
|
111
|
+
fd = openSync(file, "r");
|
|
112
|
+
const buf = Buffer.alloc(8192);
|
|
113
|
+
const n = readSync(fd, buf, 0, 8192, 0);
|
|
114
|
+
const m = buf.subarray(0, n).toString("utf8").match(/^#talk-entry ts=(\d+)/m);
|
|
115
|
+
return m ? Number(m[1]) : null;
|
|
116
|
+
} catch { return null; }
|
|
117
|
+
finally { if (fd != null) { try { closeSync(fd); } catch {} } }
|
|
118
|
+
}
|
|
119
|
+
|
|
120
|
+
/** 追加一条(正常路径只 append;到窗口或超预算才整写淘汰)。 */
|
|
121
|
+
export function appendTalkEntry({ file, block, atMs } = {}) {
|
|
122
|
+
try {
|
|
123
|
+
if (!file || !block) return 0;
|
|
124
|
+
mkdirSync(dirname(file), { recursive: true });
|
|
125
|
+
const isNew = !existsSync(file);
|
|
126
|
+
appendFileSync(file, block);
|
|
127
|
+
// mode 只在新建时生效;存量文件补一次 chmod(POSIX;Windows 无意义)
|
|
128
|
+
if (isNew) { try { chmodSync(file, 0o600); } catch {} }
|
|
129
|
+
const now = Number.isFinite(atMs) && atMs > 0 ? atMs : Date.now();
|
|
130
|
+
let base = oldestCache.get(file);
|
|
131
|
+
if (base == null) { base = firstTs(file); if (base == null) base = now; }
|
|
132
|
+
const size = (() => { try { return statSync(file).size; } catch { return 0; } })();
|
|
133
|
+
if (now - base > windowMs() || size > maxBytes()) trimTalkLog(file, now);
|
|
134
|
+
else oldestCache.set(file, Math.min(base, now));
|
|
135
|
+
return 1;
|
|
136
|
+
} catch { return 0; }
|
|
137
|
+
}
|
|
138
|
+
|
|
139
|
+
/** 按 `#talk-entry` 头行切块(内容行理论上可能撞前缀,故要求严格 ts= 形状才认头)。 */
|
|
140
|
+
export function splitEntries(txt) {
|
|
141
|
+
const entries = [];
|
|
142
|
+
const lines = String(txt ?? "").split("\n");
|
|
143
|
+
let start = -1;
|
|
144
|
+
const close = (end) => {
|
|
145
|
+
if (start < 0) return;
|
|
146
|
+
const chunk = lines.slice(start, end);
|
|
147
|
+
const m = chunk[0].match(/^#talk-entry ts=(\d+)/);
|
|
148
|
+
entries.push({ ts: m ? Number(m[1]) : 0, text: chunk.join("\n") + "\n" });
|
|
149
|
+
};
|
|
150
|
+
for (let i = 0; i < lines.length; i++) {
|
|
151
|
+
if (/^#talk-entry ts=\d+/.test(lines[i])) { close(i); start = i; }
|
|
152
|
+
}
|
|
153
|
+
close(lines.length);
|
|
154
|
+
return entries;
|
|
155
|
+
}
|
|
156
|
+
|
|
157
|
+
/** 环形淘汰:先按时间窗口丢,再按字节预算从最老的丢(至少留最新 1 条);tmp+rename 原子替换。 */
|
|
158
|
+
export function trimTalkLog(file, nowTs = Date.now()) {
|
|
159
|
+
try {
|
|
160
|
+
if (!existsSync(file)) return 0;
|
|
161
|
+
const limit = nowTs - windowMs();
|
|
162
|
+
const cap = maxBytes();
|
|
163
|
+
const all = splitEntries(readFileSync(file, "utf8"));
|
|
164
|
+
// 多进程(restart 交接窗口)可能交错追加 → 按 ts 重排,保证「环形」语义与阅读顺序一致
|
|
165
|
+
let kept = all.filter((e) => e.ts > 0 && e.ts >= limit).sort((a, b) => a.ts - b.ts);
|
|
166
|
+
if (kept.length === 0) { rmSync(file, { force: true }); oldestCache.delete(file); return all.length; }
|
|
167
|
+
let bytes = Buffer.byteLength(kept.map((e) => e.text).join(""), "utf8");
|
|
168
|
+
while (bytes > cap && kept.length > 1) bytes -= Buffer.byteLength(kept.shift().text, "utf8");
|
|
169
|
+
const tmp = `${file}.${process.pid}.trim`;
|
|
170
|
+
writeFileSync(tmp, kept.map((e) => e.text).join(""), { mode: 0o600 });
|
|
171
|
+
renameSync(tmp, file);
|
|
172
|
+
oldestCache.set(file, kept[0].ts);
|
|
173
|
+
return all.length - kept.length;
|
|
174
|
+
} catch { return 0; }
|
|
175
|
+
}
|
|
176
|
+
|
|
177
|
+
/**
|
|
178
|
+
* relay 落盘的唯一入口:把 relay() 累积的 detail.talk(思考/正文/工具分片)拼成一条对话。
|
|
179
|
+
* 空轮(零正文零思考零工具)不占环形窗口 —— 那是 relay-pipeline 已经另行报错的场景。
|
|
180
|
+
*/
|
|
181
|
+
export function recordRelayTalk({ reqId, model, via, hops, body, out, echo = {} } = {}) {
|
|
182
|
+
if (!talkLogEnabled()) return false;
|
|
183
|
+
try {
|
|
184
|
+
const talk = out?.detail?.talk;
|
|
185
|
+
const answer = (talk?.content || []).join("");
|
|
186
|
+
const thinking = (talk?.reasoning || []).join("");
|
|
187
|
+
const tools = talk?.tools?.length ? JSON.stringify(talk.tools) : "";
|
|
188
|
+
const status = out?.status;
|
|
189
|
+
if (!answer && !thinking && !tools && Number(status) !== 200) return false;
|
|
190
|
+
if (!answer && !thinking && !tools) return false;
|
|
191
|
+
const ts = Date.now();
|
|
192
|
+
const block = formatTalkEntry({
|
|
193
|
+
reqId, model, via, hops, status, stream: Boolean(body?.stream),
|
|
194
|
+
elapsedMs: out?.totalMs, question: lastUserQuestion(body || {}),
|
|
195
|
+
thinking, answer, tools, usage: out?.detail?.usage, ...echo, ts,
|
|
196
|
+
});
|
|
197
|
+
appendTalkEntry({ file: talkLogFile(model), block, atMs: ts });
|
|
198
|
+
return true;
|
|
199
|
+
} catch { return false; }
|
|
200
|
+
}
|
|
201
|
+
|
|
202
|
+
/** 读最近 N 条(倒序,最新在前)—— 给 CLI/排障用。 */
|
|
203
|
+
export function recentBlocks({ model, n = 10, file } = {}) {
|
|
204
|
+
try {
|
|
205
|
+
const target = file || talkLogFile(model);
|
|
206
|
+
if (!existsSync(target)) return [];
|
|
207
|
+
return splitEntries(readFileSync(target, "utf8")).map((e) => e.text).reverse().slice(0, n);
|
|
208
|
+
} catch { return []; }
|
|
209
|
+
}
|
|
210
|
+
|
|
211
|
+
/** 最近活跃的文件清单(哪些供应商×模型在这一小时里说过话)。 */
|
|
212
|
+
export function listRecentFiles({ limit = 20 } = {}) {
|
|
213
|
+
try {
|
|
214
|
+
const dir = talkDir();
|
|
215
|
+
if (!existsSync(dir)) return [];
|
|
216
|
+
return readdirSync(dir, { withFileTypes: true })
|
|
217
|
+
.filter((d) => d.isFile() && d.name.endsWith("-talk.log"))
|
|
218
|
+
.map((d) => {
|
|
219
|
+
let mtime = 0;
|
|
220
|
+
try { mtime = statSync(join(dir, d.name)).mtimeMs; } catch {}
|
|
221
|
+
return { name: d.name, mtime, time: fmtShanghaiYMDHMS(new Date(mtime)) };
|
|
222
|
+
})
|
|
223
|
+
.sort((a, b) => b.mtime - a.mtime)
|
|
224
|
+
.slice(0, limit);
|
|
225
|
+
} catch { return []; }
|
|
226
|
+
}
|
package/src/timeline.js
CHANGED
|
@@ -12,12 +12,15 @@ function outcome(status, reason) {
|
|
|
12
12
|
return `${status}${reason ? ` ${short(reason)}` : ""}`;
|
|
13
13
|
}
|
|
14
14
|
|
|
15
|
-
export function formatTimeline({ reqId, model, direct = [], peers = [], retries = 0, result = null, totalMs = 0 } = {}) {
|
|
15
|
+
export function formatTimeline({ reqId, model, direct = [], peers = [], retries = 0, retryResult = null, result = null, totalMs = 0 } = {}) {
|
|
16
16
|
const d = direct.length ? direct[direct.length - 1] : null;
|
|
17
17
|
const ps = peers.map((p) => `[peer=${hostPort(p.peer)} ${p.ok ? "win" : "fail"} ${p.latencyMs != null ? `${p.latencyMs}ms` : "timing-"} ${short(p.message || p.status || "", 70)}]`);
|
|
18
18
|
const r = result || {};
|
|
19
19
|
const detail = r.detail || {};
|
|
20
20
|
const res = r.status == null ? "-" : `${r.status}${detail.sawFinishReason ? ` ${detail.sawFinishReason}` : ""}${detail.toolCalls ? ` tools=${detail.toolCalls}` : ""}${detail.chars != null ? ` chars=${detail.chars}` : ""}${detail.interrupted ? " interrupted=1" : ""}${detail.timedOut ? " timedOut=1" : ""}`;
|
|
21
21
|
const directText = d ? outcome(d.status, d.reason) : "-";
|
|
22
|
-
|
|
22
|
+
// retry= 次数(保持既有 token 可 grep);win=1 重试后救回,lost=1 重试用尽仍零正文。
|
|
23
|
+
const _rc = retryResult === "ok" ? " [retry_win=1]" : retryResult === "lost" ? " [retry_lost=1]" : "";
|
|
24
|
+
const retryTag = (retries || retryResult) ? ` [retry=${retries || 0}]${_rc}` : "";
|
|
25
|
+
return `[req=${reqId || "-"}] [model=${model || "-"}] [direct=${directText}]${ps.length ? ` ${ps.join(" ")}` : ""}${retryTag} [result=${res}] [total=${Math.max(0, Math.round(Number(totalMs) || 0))}ms]`;
|
|
23
26
|
}
|
package/src/usage/record.js
CHANGED
|
@@ -78,7 +78,8 @@ export async function recordUsage(entry, { dir, now = new Date() } = {}) {
|
|
|
78
78
|
|
|
79
79
|
// 从 relay 结果组装一行 usage —— 行形状由本模块拥有,调用方只交原始字段。
|
|
80
80
|
// usage 即 metrics.js 的 extractUsageFromJson 输出(prompt/completion/total/reasoning)。
|
|
81
|
-
export function recordChatUsage({ model, via, usage, ttfbMs, totalMs, tps, interrupted } = {}) {
|
|
81
|
+
export function recordChatUsage({ model, via, usage, ttfbMs, totalMs, tps, interrupted, reasoningChars, stream } = {}) {
|
|
82
|
+
const rc = Number(reasoningChars);
|
|
82
83
|
return recordUsage({
|
|
83
84
|
model,
|
|
84
85
|
via,
|
|
@@ -87,6 +88,12 @@ export function recordChatUsage({ model, via, usage, ttfbMs, totalMs, tps, inter
|
|
|
87
88
|
completion_tokens: usage?.completion_tokens ?? 0,
|
|
88
89
|
total_tokens: usage?.total_tokens ?? 0,
|
|
89
90
|
reasoning_tokens: usage?.reasoning_tokens ?? 0,
|
|
91
|
+
// —— 以下为纯增字段:只存原始观测,chars/4 的估算留给聚合层(口径可演进,不必回填历史行)——
|
|
92
|
+
reasoning_chars: Number.isFinite(rc) && rc > 0 ? Math.trunc(rc) : 0,
|
|
93
|
+
// 上游是否明确上报过思考 tokens(metrics.js 只在数值有限时才写 reasoning_tokens)
|
|
94
|
+
reasoning_reported: Number.isFinite(usage?.reasoning_tokens) ? 1 : 0,
|
|
95
|
+
// 调用方没表态时整字段不落盘:「未知」≠「非流式」,覆盖度分母宁低不假高
|
|
96
|
+
...(stream === undefined ? {} : { stream: stream ? 1 : 0 }),
|
|
90
97
|
ttfbMs: Number.isFinite(ttfbMs) ? ttfbMs : null,
|
|
91
98
|
totalMs: Number.isFinite(totalMs) ? totalMs : null,
|
|
92
99
|
tps: Number.isFinite(tps) ? tps : null,
|
package/src/usage/report.js
CHANGED
|
@@ -16,6 +16,24 @@ function ms(v) {
|
|
|
16
16
|
const n = Number(v);
|
|
17
17
|
return Number.isFinite(n) && n >= 0 ? n : null;
|
|
18
18
|
}
|
|
19
|
+
// 思考 tokens 的四态判定(行级):上报优先 → 有思考字符就估算 → 明确报 0 才算 0 → 否则未知。
|
|
20
|
+
// 「上游报 0 却确有思考内容」按观测走估算:qoder 就是这形状,报 0 是上游没填,不是模型没想。
|
|
21
|
+
export function resolveReasoning(row) {
|
|
22
|
+
const reportedVal = Number(row?.reasoning_tokens);
|
|
23
|
+
// 兼容判据:修复前的旧行没有 reasoning_reported,但带着真实上报值 —— 不能因缺字段退化成未知
|
|
24
|
+
const hasReported = row?.reasoning_reported === 1 || (Number.isFinite(reportedVal) && reportedVal > 0);
|
|
25
|
+
const chars = Number(row?.reasoning_chars);
|
|
26
|
+
const charCount = Number.isFinite(chars) && chars > 0 ? Math.trunc(chars) : 0;
|
|
27
|
+
if (hasReported && Number.isFinite(reportedVal) && reportedVal > 0) return { value: reportedVal, source: "reported" };
|
|
28
|
+
if (charCount > 0) {
|
|
29
|
+
const cap = Number(row?.completion_tokens);
|
|
30
|
+
const est = Math.ceil(charCount / 4); // 与 src/providers/cline/usage.js 的既有估算口径一致
|
|
31
|
+
return { value: Number.isFinite(cap) && cap > 0 ? Math.min(est, cap) : est, source: "estimated" };
|
|
32
|
+
}
|
|
33
|
+
if (hasReported) return { value: 0, source: "reported" }; // 该轮真的没思考
|
|
34
|
+
return { value: 0, source: "none" }; // 无从判断 → 表格渲染 —,绝不渲染成 0
|
|
35
|
+
}
|
|
36
|
+
|
|
19
37
|
|
|
20
38
|
function blank(id) {
|
|
21
39
|
return {
|
|
@@ -31,6 +49,12 @@ function blank(id) {
|
|
|
31
49
|
totalN: 0,
|
|
32
50
|
completionMsSum: 0,
|
|
33
51
|
tpsTokSum: 0,
|
|
52
|
+
reasoningReported: 0,
|
|
53
|
+
reasoningEstimated: 0,
|
|
54
|
+
reasoningReportedRows: 0,
|
|
55
|
+
reasoningEstimatedRows: 0,
|
|
56
|
+
reasoningChars: 0,
|
|
57
|
+
streamRequests: 0,
|
|
34
58
|
};
|
|
35
59
|
}
|
|
36
60
|
|
|
@@ -42,9 +66,16 @@ function finalize(a) {
|
|
|
42
66
|
promptTokens: a.promptTokens,
|
|
43
67
|
completionTokens: a.completionTokens,
|
|
44
68
|
totalTokens: a.totalTokens,
|
|
45
|
-
reasoningTokens: a.
|
|
69
|
+
reasoningTokens: a.reasoningReported + a.reasoningEstimated,
|
|
46
70
|
avgTtfbMs: a.ttfbN ? Math.round(a.ttfbSumMs / a.ttfbN) : null,
|
|
47
71
|
avgTotalMs: a.totalN ? Math.round(a.totalSumMs / a.totalN) : null,
|
|
72
|
+
// 思考来源:上报与估算同时存在时是 mixed——合计值不冒充单一精确数(spec 锁死)
|
|
73
|
+
reasoningSource: a.reasoningReportedRows > 0 && a.reasoningEstimatedRows > 0 ? "mixed"
|
|
74
|
+
: a.reasoningReportedRows > 0 ? "reported"
|
|
75
|
+
: a.reasoningEstimatedRows > 0 ? "estimated" : "none",
|
|
76
|
+
reasoningChars: a.reasoningChars,
|
|
77
|
+
ttfSamples: a.ttfbN,
|
|
78
|
+
streamRequests: a.streamRequests,
|
|
48
79
|
avgTps: a.completionMsSum > 0 ? Number((a.tpsTokSum / (a.completionMsSum / 1000)).toFixed(1)) : null,
|
|
49
80
|
};
|
|
50
81
|
}
|
|
@@ -57,7 +88,13 @@ function fold(acc, r) {
|
|
|
57
88
|
acc.completionTokens += comp;
|
|
58
89
|
const total = tok(r.total_tokens);
|
|
59
90
|
acc.totalTokens += total || prompt + comp;
|
|
60
|
-
|
|
91
|
+
// 思考:四态分层累计(上报/估算各算各的,未知不贡献数值),原始字符数照实累加
|
|
92
|
+
const rs = resolveReasoning(r);
|
|
93
|
+
if (rs.source === "reported") { acc.reasoningReported += rs.value; acc.reasoningReportedRows++; }
|
|
94
|
+
else if (rs.source === "estimated") { acc.reasoningEstimated += rs.value; acc.reasoningEstimatedRows++; }
|
|
95
|
+
acc.reasoningChars += tok(r.reasoning_chars);
|
|
96
|
+
// 首字样本分母:非流式行既无首字也不该摊进分母;缺 stream 的旧行按未知保守计入
|
|
97
|
+
if (r.stream !== 0) acc.streamRequests++;
|
|
61
98
|
|
|
62
99
|
const ttfb = ms(r.ttfbMs ?? r.ttfb_ms);
|
|
63
100
|
if (ttfb != null) {
|