dsh-workbuddy-xdpool 1.3.0 → 1.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +44 -0
- package/lib/bin.js +153 -2
- package/lib/index.d.ts +44 -0
- package/lib/index.js +602 -23
- package/package.json +1 -1
package/CHANGELOG.md
CHANGED
|
@@ -4,6 +4,50 @@
|
|
|
4
4
|
|
|
5
5
|
版本号遵循 [语义化版本](https://semver.org/lang/zh-CN/)。
|
|
6
6
|
|
|
7
|
+
## 1.4.0 (2026-09-24)
|
|
8
|
+
|
|
9
|
+
### 长对话不再撞死在上下文上限上
|
|
10
|
+
|
|
11
|
+
以前聊到一定长度,就会弹出一个 `400 context_length_exceeded` 然后被送回主页,只能重开会话。
|
|
12
|
+
现在会自动收缩上下文继续跑,不再打断你。
|
|
13
|
+
|
|
14
|
+
这里其实有两层保险,各自管不同的情况:
|
|
15
|
+
|
|
16
|
+
**第一层是 Harness 自带的。** DSH 本来就装了 `compaction-basic`,它会在发请求之前先算一遍 token
|
|
17
|
+
压力,超了就摘要、重试。问题是它认错误的方式是「看这段文字像不像上下文溢出」,而插件原来报的是
|
|
18
|
+
一句友好英文 —— `the conversation exceeds <模型名>'s context window` —— 恰好是它唯一认不出来的
|
|
19
|
+
那种写法。于是它以为这是个普通的坏请求,跳过了恢复,直接把 400 抛给你。
|
|
20
|
+
|
|
21
|
+
(顺带一个坑:上游 WorkBuddy 自己那句 `input length too long`/业务码 `11115`,同样不在它的
|
|
22
|
+
识别清单里。所以「把上游原话原样透传」这条路也走不通,必须主动改写成它认得的措辞。)
|
|
23
|
+
|
|
24
|
+
现在溢出时会输出它认得的写法,Harness 的自动摘要就接上了。
|
|
25
|
+
|
|
26
|
+
**第二层是插件自己的。** 就算 Harness 那层没兜住(比如窗口值报得不准,或者你一口气粘了个巨大
|
|
27
|
+
文件),插件会自己压缩一遍再重试一次:先丢最旧的回合,保留 system 和最近几轮;还放不下就把被丢
|
|
28
|
+
掉的回合交给模型做个摘要塞回去;再不行才硬截断。摘要这一步哪怕失败也只会降级成直接截断,绝不
|
|
29
|
+
会反过来让你的这一轮请求失败。
|
|
30
|
+
|
|
31
|
+
### 「API 密钥无效」不再把所有账号卡死在第一个上
|
|
32
|
+
|
|
33
|
+
账号池的本意是:某个号不行就换下一个。但错误分类里有个漏洞 —— HTTP 401 只有在响应体里恰好
|
|
34
|
+
出现两条特定标记之一时,才会被认成「登录态失效」(可刷新、可轮换)。上游换个说法,比如
|
|
35
|
+
「api密钥无效」、`invalid api key`、`unauthorized`,就掉进了兜底的 `client` 类。
|
|
36
|
+
|
|
37
|
+
而 `client` 在轮换循环里是**终态**:直接 `break`,不再试后面的号。结果就是一个坏掉的号把整个池
|
|
38
|
+
钉死在第一个账号上,2/3/4 号全都白装着。
|
|
39
|
+
|
|
40
|
+
现在 401 和 403 无条件算「登录态失效」—— 刷新 token 然后换下一个号。另外把常见的中英文失效措辞
|
|
41
|
+
也补进了标记清单,这样即便状态码不是 401(比如被包在 200 信封里)也还能认出来。
|
|
42
|
+
|
|
43
|
+
### 其它
|
|
44
|
+
|
|
45
|
+
- 顺手修了个压缩逻辑里的真 bug:它用同一个变量同时表示「留下来的」和「被丢掉的」,结果明明已经
|
|
46
|
+
砍到只剩几条,却报告「没有变化」,导致摘要功能整个跳过、根本没执行。
|
|
47
|
+
- 新增两组测试:`context-overflow-contract.test.ts` 逐字复刻了 Harness 的错误匹配规则(它将来改
|
|
48
|
+
措辞时这个测试会变红,提醒你同步改);`upstream-error-classification.test.ts` 钉住了上面那个
|
|
49
|
+
401 分类回归。
|
|
50
|
+
|
|
7
51
|
## 1.3.0 (2026-09-23)
|
|
8
52
|
|
|
9
53
|
上一版把八类任务链接了进去,这一版修的是**「接进去了但你看不见」**的问题——功能在跑,界面上却没显示,等于白做。
|
package/lib/bin.js
CHANGED
|
@@ -29,6 +29,8 @@ const MODELS_CATALOG_PATH = "/v2/enterprises/personal/models";
|
|
|
29
29
|
const GLOBAL_CONFIG_PATH = "/v3/config";
|
|
30
30
|
const JSON_TIMEOUT_MS = 3e4;
|
|
31
31
|
const ERROR_BODY_LIMIT = 4096;
|
|
32
|
+
/** Cap on the reassembled compaction reply, guarding against a runaway stream. */
|
|
33
|
+
const COMPLETION_TEXT_LIMIT = 65536;
|
|
32
34
|
/** Insufficient-credit markers, ASCII lowercase plus the original Chinese. */
|
|
33
35
|
const HARD_CREDIT_MARKERS = [
|
|
34
36
|
"insufficient credit",
|
|
@@ -47,8 +49,31 @@ const HARD_CREDIT_MARKERS = [
|
|
|
47
49
|
"额度用尽",
|
|
48
50
|
"没有积分"
|
|
49
51
|
];
|
|
50
|
-
/** Session-invalidation markers that mean "
|
|
51
|
-
|
|
52
|
+
/** Session-invalidation markers that mean "this credential is dead; use another".
|
|
53
|
+
* Kept alongside the HTTP-status rule in `classifyUpstreamError`: the status is
|
|
54
|
+
* enough for a direct 401/403, but some failures arrive wrapped in a 200
|
|
55
|
+
* envelope or a 4xx the gateway words differently. Adding the English and
|
|
56
|
+
* Chinese phrasings the upstream actually uses keeps those recoverable too —
|
|
57
|
+
* an unmatched one fell through to `client`, which is terminal in the shim
|
|
58
|
+
* and pinned the pool to the first account (the "API 密钥无效" bug). */
|
|
59
|
+
const SESSION_DEAD_MARKERS = [
|
|
60
|
+
"Offline user session not found",
|
|
61
|
+
"12153",
|
|
62
|
+
"api key is invalid",
|
|
63
|
+
"invalid api key",
|
|
64
|
+
"invalid_api_key",
|
|
65
|
+
"api密钥无效",
|
|
66
|
+
"密钥无效",
|
|
67
|
+
"无效的密钥",
|
|
68
|
+
"unauthorized",
|
|
69
|
+
"token expired",
|
|
70
|
+
"token is invalid",
|
|
71
|
+
"login expired",
|
|
72
|
+
"please login",
|
|
73
|
+
"未登录",
|
|
74
|
+
"登录已失效",
|
|
75
|
+
"重新登录"
|
|
76
|
+
];
|
|
52
77
|
/**
|
|
53
78
|
* Markers for "already checked in today".
|
|
54
79
|
*
|
|
@@ -221,12 +246,82 @@ function envelopeError(status, envelope) {
|
|
|
221
246
|
return /* @__PURE__ */ new Error(`workbuddy upstream ${kind} (http ${status}): ${envelope.msg.slice(0, 160)}`);
|
|
222
247
|
}
|
|
223
248
|
/**
|
|
249
|
+
* Read an OpenAI-style SSE chat stream and concatenate the assistant text.
|
|
250
|
+
*
|
|
251
|
+
* The upstream always streams (`stream: true` is forced on every chat body),
|
|
252
|
+
* so a non-streaming internal call has to reassemble the deltas itself. Only
|
|
253
|
+
* `choices[0].delta.content` is collected; reasoning deltas are dropped
|
|
254
|
+
* because a compaction summary needs the final answer, not the scratchpad.
|
|
255
|
+
*/
|
|
256
|
+
async function readCompletionText(body) {
|
|
257
|
+
const decoder = new TextDecoder();
|
|
258
|
+
const reader = body.getReader();
|
|
259
|
+
let buffer = "";
|
|
260
|
+
let text = "";
|
|
261
|
+
try {
|
|
262
|
+
for (;;) {
|
|
263
|
+
const { done, value } = await reader.read();
|
|
264
|
+
if (done) break;
|
|
265
|
+
buffer += decoder.decode(value, { stream: true });
|
|
266
|
+
let split = buffer.indexOf("\n\n");
|
|
267
|
+
while (split !== -1) {
|
|
268
|
+
const frame = buffer.slice(0, split);
|
|
269
|
+
buffer = buffer.slice(split + 2);
|
|
270
|
+
text += contentOfFrame(frame);
|
|
271
|
+
if (text.length > COMPLETION_TEXT_LIMIT) return text.slice(0, COMPLETION_TEXT_LIMIT);
|
|
272
|
+
split = buffer.indexOf("\n\n");
|
|
273
|
+
}
|
|
274
|
+
}
|
|
275
|
+
if (buffer.trim() !== "") text += contentOfFrame(buffer);
|
|
276
|
+
} finally {
|
|
277
|
+
reader.releaseLock?.();
|
|
278
|
+
}
|
|
279
|
+
return text;
|
|
280
|
+
}
|
|
281
|
+
/** Pull `choices[0].delta.content` (or a non-streaming `message.content`) out of one SSE frame. */
|
|
282
|
+
function contentOfFrame(frame) {
|
|
283
|
+
let out = "";
|
|
284
|
+
for (const rawLine of frame.split(/\r?\n/u)) {
|
|
285
|
+
const line = rawLine.trim();
|
|
286
|
+
if (!line.startsWith("data:")) continue;
|
|
287
|
+
const payload = line.slice(5).trim();
|
|
288
|
+
if (payload === "" || payload === "[DONE]") continue;
|
|
289
|
+
let parsed;
|
|
290
|
+
try {
|
|
291
|
+
parsed = JSON.parse(payload);
|
|
292
|
+
} catch {
|
|
293
|
+
continue;
|
|
294
|
+
}
|
|
295
|
+
if (typeof parsed !== "object" || parsed === null) continue;
|
|
296
|
+
const choices = parsed["choices"];
|
|
297
|
+
if (!Array.isArray(choices) || choices.length === 0) continue;
|
|
298
|
+
const choice = choices[0];
|
|
299
|
+
const delta = choice["delta"];
|
|
300
|
+
if (typeof delta === "object" && delta !== null) {
|
|
301
|
+
const content = delta["content"];
|
|
302
|
+
if (typeof content === "string") out += content;
|
|
303
|
+
}
|
|
304
|
+
const message = choice["message"];
|
|
305
|
+
if (typeof message === "object" && message !== null) {
|
|
306
|
+
const content = message["content"];
|
|
307
|
+
if (typeof content === "string") out += content;
|
|
308
|
+
}
|
|
309
|
+
const data = parsed["data"];
|
|
310
|
+
if (typeof data === "object" && data !== null) {
|
|
311
|
+
const inner = data["content"];
|
|
312
|
+
if (typeof inner === "string") out += inner;
|
|
313
|
+
}
|
|
314
|
+
}
|
|
315
|
+
return out;
|
|
316
|
+
}
|
|
317
|
+
/**
|
|
224
318
|
* Classify an upstream failure from its HTTP status and body excerpt.
|
|
225
319
|
* Body markers win over status, because the upstream reuses 400/200 for
|
|
226
320
|
* several distinct conditions.
|
|
227
321
|
*/
|
|
228
322
|
function classifyUpstreamError(status, body) {
|
|
229
323
|
if (status === 402) return "hard_credit";
|
|
324
|
+
if (status === 401 || status === 403) return "session_dead";
|
|
230
325
|
const lower = body.toLowerCase();
|
|
231
326
|
for (const marker of HARD_CREDIT_MARKERS) if (lower.includes(marker.toLowerCase()) || body.includes(marker)) return "hard_credit";
|
|
232
327
|
for (const marker of SESSION_DEAD_MARKERS) if (body.includes(marker)) return "session_dead";
|
|
@@ -589,6 +684,51 @@ var WorkBuddyUpstreamClient = class {
|
|
|
589
684
|
}
|
|
590
685
|
return JSON.stringify(obj);
|
|
591
686
|
}
|
|
687
|
+
/**
|
|
688
|
+
* Parse a raw OpenAI chat body without normalising it.
|
|
689
|
+
*
|
|
690
|
+
* The compactor needs the message array as objects, while `chatStream` only
|
|
691
|
+
* accepts the serialised string form.
|
|
692
|
+
*/
|
|
693
|
+
parseChatBody(raw) {
|
|
694
|
+
let body;
|
|
695
|
+
try {
|
|
696
|
+
body = JSON.parse(raw);
|
|
697
|
+
} catch {
|
|
698
|
+
return;
|
|
699
|
+
}
|
|
700
|
+
if (typeof body !== "object" || body === null || Array.isArray(body)) return void 0;
|
|
701
|
+
return body;
|
|
702
|
+
}
|
|
703
|
+
/** Re-serialise `base` with a rewritten `messages` array, still normalised. */
|
|
704
|
+
buildChatBody(base, messages) {
|
|
705
|
+
return this.prepareChatBody(JSON.stringify({
|
|
706
|
+
...base,
|
|
707
|
+
messages
|
|
708
|
+
}));
|
|
709
|
+
}
|
|
710
|
+
/**
|
|
711
|
+
* Run one NON-streaming completion and return the assistant text.
|
|
712
|
+
*
|
|
713
|
+
* Used only for internal compaction (summarising dropped turns). The chat
|
|
714
|
+
* endpoint itself always streams, so this reassembles the SSE frames into a
|
|
715
|
+
* single string. Throws on any failure: the compactor then falls back to
|
|
716
|
+
* plain truncation rather than failing the user's turn.
|
|
717
|
+
*/
|
|
718
|
+
async completeChat(credential, prepared, signal) {
|
|
719
|
+
const response = await this.fetchImpl(`${chatBase(credential)}/v2/chat/completions`, {
|
|
720
|
+
method: "POST",
|
|
721
|
+
headers: chatHeaders(credential),
|
|
722
|
+
body: prepared,
|
|
723
|
+
...signal === void 0 ? {} : { signal }
|
|
724
|
+
});
|
|
725
|
+
if (!response.ok) {
|
|
726
|
+
const text = (await response.text().catch(() => "")).slice(0, ERROR_BODY_LIMIT);
|
|
727
|
+
throw new Error(`compaction upstream http ${response.status}: ${text}`);
|
|
728
|
+
}
|
|
729
|
+
if (response.body === null) throw new Error("compaction upstream returned no body");
|
|
730
|
+
return await readCompletionText(response.body);
|
|
731
|
+
}
|
|
592
732
|
/** Forward one chat completion. Never throws for upstream failures. */
|
|
593
733
|
async chatStream(credential, prepared, signal) {
|
|
594
734
|
let response;
|
|
@@ -3480,6 +3620,17 @@ var WorkBuddyScheduler = class {
|
|
|
3480
3620
|
this.logger.info?.(`automation travel ${account.label}: nothing to do (state=${travel.state})`);
|
|
3481
3621
|
}
|
|
3482
3622
|
};
|
|
3623
|
+
[
|
|
3624
|
+
"You are compacting an ongoing conversation so it can continue without the original history.",
|
|
3625
|
+
"Summarise the transcript below into a dense briefing for the next assistant turn.",
|
|
3626
|
+
"Preserve, in this order of priority:",
|
|
3627
|
+
"1. explicit user requirements, constraints and corrections;",
|
|
3628
|
+
"2. decisions already made, and the reasoning behind them;",
|
|
3629
|
+
"3. concrete facts: file paths, identifiers, commands, numbers, error messages;",
|
|
3630
|
+
"4. unfinished work and the current blocker.",
|
|
3631
|
+
"Drop pleasantries, repetition and superseded attempts.",
|
|
3632
|
+
"Write the briefing only — no preamble, no markdown fence."
|
|
3633
|
+
].join("\n");
|
|
3483
3634
|
//#endregion
|
|
3484
3635
|
//#region src/status.ts
|
|
3485
3636
|
/** Format the per-model credit multipliers. */
|
package/lib/index.d.ts
CHANGED
|
@@ -164,6 +164,32 @@ export declare function expertActualUseEvent(expert: MarketExpert, conversationI
|
|
|
164
164
|
*/
|
|
165
165
|
export declare function expertChatEvents(expert: MarketExpert, conversationId: string, requestId: string): Record<string, unknown>[];
|
|
166
166
|
//#endregion
|
|
167
|
+
//#region src/context-budget.d.ts
|
|
168
|
+
/**
|
|
169
|
+
* Context-window budgeting for the WorkBuddy shim.
|
|
170
|
+
*
|
|
171
|
+
* The upstream answers `context_length_exceeded` (business code 11115) when a
|
|
172
|
+
* request overruns the model's window. Rather than bouncing that back to the
|
|
173
|
+
* user as a dead turn, the shim compacts the conversation on the fly:
|
|
174
|
+
*
|
|
175
|
+
* 1. estimate the prompt cost locally (cheap, no round trip);
|
|
176
|
+
* 2. drop the oldest turns while keeping `system` + the newest exchange;
|
|
177
|
+
* 3. if that still overruns, ask the model itself to summarise the middle of
|
|
178
|
+
* the conversation and splice that summary back in as a system message.
|
|
179
|
+
*
|
|
180
|
+
* Everything here is pure and synchronous-free except `summarizeMessages`,
|
|
181
|
+
* which the caller drives through an injected chat function so this module
|
|
182
|
+
* stays testable without a network.
|
|
183
|
+
*
|
|
184
|
+
* @module dsh-workbuddy-xdpool/context-budget
|
|
185
|
+
*/
|
|
186
|
+
/** One OpenAI chat message, narrowed to the fields we must preserve. */
|
|
187
|
+
interface ChatMessage {
|
|
188
|
+
role: string;
|
|
189
|
+
content: unknown;
|
|
190
|
+
[key: string]: unknown;
|
|
191
|
+
}
|
|
192
|
+
//#endregion
|
|
167
193
|
//#region src/upstream.d.ts
|
|
168
194
|
/** Upstream failure classes the shim maps onto distinct HTTP answers. */
|
|
169
195
|
type UpstreamErrorKind = 'hard_credit' | 'soft_rate' | 'session_dead' | 'not_found' | 'server' | 'client';
|
|
@@ -388,6 +414,24 @@ export declare class WorkBuddyUpstreamClient {
|
|
|
388
414
|
* business code 11128), and flatten `tool_choice` into its string form.
|
|
389
415
|
*/
|
|
390
416
|
prepareChatBody(raw: string): string;
|
|
417
|
+
/**
|
|
418
|
+
* Parse a raw OpenAI chat body without normalising it.
|
|
419
|
+
*
|
|
420
|
+
* The compactor needs the message array as objects, while `chatStream` only
|
|
421
|
+
* accepts the serialised string form.
|
|
422
|
+
*/
|
|
423
|
+
parseChatBody(raw: string): Record<string, unknown> | undefined;
|
|
424
|
+
/** Re-serialise `base` with a rewritten `messages` array, still normalised. */
|
|
425
|
+
buildChatBody(base: Record<string, unknown>, messages: readonly ChatMessage[]): string;
|
|
426
|
+
/**
|
|
427
|
+
* Run one NON-streaming completion and return the assistant text.
|
|
428
|
+
*
|
|
429
|
+
* Used only for internal compaction (summarising dropped turns). The chat
|
|
430
|
+
* endpoint itself always streams, so this reassembles the SSE frames into a
|
|
431
|
+
* single string. Throws on any failure: the compactor then falls back to
|
|
432
|
+
* plain truncation rather than failing the user's turn.
|
|
433
|
+
*/
|
|
434
|
+
completeChat(credential: WorkBuddyCredential, prepared: string, signal?: AbortSignal): Promise<string>;
|
|
391
435
|
/** Forward one chat completion. Never throws for upstream failures. */
|
|
392
436
|
chatStream(credential: WorkBuddyCredential, prepared: string, signal?: AbortSignal): Promise<ChatStreamResult>;
|
|
393
437
|
/** POST the token-refresh endpoint; the caller merges the outcome. */
|
package/lib/index.js
CHANGED
|
@@ -31,6 +31,8 @@ const MODELS_CATALOG_PATH = "/v2/enterprises/personal/models";
|
|
|
31
31
|
const GLOBAL_CONFIG_PATH = "/v3/config";
|
|
32
32
|
const JSON_TIMEOUT_MS = 3e4;
|
|
33
33
|
const ERROR_BODY_LIMIT = 4096;
|
|
34
|
+
/** Cap on the reassembled compaction reply, guarding against a runaway stream. */
|
|
35
|
+
const COMPLETION_TEXT_LIMIT = 65536;
|
|
34
36
|
/** Insufficient-credit markers, ASCII lowercase plus the original Chinese. */
|
|
35
37
|
const HARD_CREDIT_MARKERS = [
|
|
36
38
|
"insufficient credit",
|
|
@@ -49,8 +51,31 @@ const HARD_CREDIT_MARKERS = [
|
|
|
49
51
|
"额度用尽",
|
|
50
52
|
"没有积分"
|
|
51
53
|
];
|
|
52
|
-
/** Session-invalidation markers that mean "
|
|
53
|
-
|
|
54
|
+
/** Session-invalidation markers that mean "this credential is dead; use another".
|
|
55
|
+
* Kept alongside the HTTP-status rule in `classifyUpstreamError`: the status is
|
|
56
|
+
* enough for a direct 401/403, but some failures arrive wrapped in a 200
|
|
57
|
+
* envelope or a 4xx the gateway words differently. Adding the English and
|
|
58
|
+
* Chinese phrasings the upstream actually uses keeps those recoverable too —
|
|
59
|
+
* an unmatched one fell through to `client`, which is terminal in the shim
|
|
60
|
+
* and pinned the pool to the first account (the "API 密钥无效" bug). */
|
|
61
|
+
const SESSION_DEAD_MARKERS = [
|
|
62
|
+
"Offline user session not found",
|
|
63
|
+
"12153",
|
|
64
|
+
"api key is invalid",
|
|
65
|
+
"invalid api key",
|
|
66
|
+
"invalid_api_key",
|
|
67
|
+
"api密钥无效",
|
|
68
|
+
"密钥无效",
|
|
69
|
+
"无效的密钥",
|
|
70
|
+
"unauthorized",
|
|
71
|
+
"token expired",
|
|
72
|
+
"token is invalid",
|
|
73
|
+
"login expired",
|
|
74
|
+
"please login",
|
|
75
|
+
"未登录",
|
|
76
|
+
"登录已失效",
|
|
77
|
+
"重新登录"
|
|
78
|
+
];
|
|
54
79
|
/**
|
|
55
80
|
* Markers for "already checked in today".
|
|
56
81
|
*
|
|
@@ -223,12 +248,82 @@ function envelopeError(status, envelope) {
|
|
|
223
248
|
return /* @__PURE__ */ new Error(`workbuddy upstream ${kind} (http ${status}): ${envelope.msg.slice(0, 160)}`);
|
|
224
249
|
}
|
|
225
250
|
/**
|
|
251
|
+
* Read an OpenAI-style SSE chat stream and concatenate the assistant text.
|
|
252
|
+
*
|
|
253
|
+
* The upstream always streams (`stream: true` is forced on every chat body),
|
|
254
|
+
* so a non-streaming internal call has to reassemble the deltas itself. Only
|
|
255
|
+
* `choices[0].delta.content` is collected; reasoning deltas are dropped
|
|
256
|
+
* because a compaction summary needs the final answer, not the scratchpad.
|
|
257
|
+
*/
|
|
258
|
+
async function readCompletionText(body) {
|
|
259
|
+
const decoder = new TextDecoder();
|
|
260
|
+
const reader = body.getReader();
|
|
261
|
+
let buffer = "";
|
|
262
|
+
let text = "";
|
|
263
|
+
try {
|
|
264
|
+
for (;;) {
|
|
265
|
+
const { done, value } = await reader.read();
|
|
266
|
+
if (done) break;
|
|
267
|
+
buffer += decoder.decode(value, { stream: true });
|
|
268
|
+
let split = buffer.indexOf("\n\n");
|
|
269
|
+
while (split !== -1) {
|
|
270
|
+
const frame = buffer.slice(0, split);
|
|
271
|
+
buffer = buffer.slice(split + 2);
|
|
272
|
+
text += contentOfFrame(frame);
|
|
273
|
+
if (text.length > COMPLETION_TEXT_LIMIT) return text.slice(0, COMPLETION_TEXT_LIMIT);
|
|
274
|
+
split = buffer.indexOf("\n\n");
|
|
275
|
+
}
|
|
276
|
+
}
|
|
277
|
+
if (buffer.trim() !== "") text += contentOfFrame(buffer);
|
|
278
|
+
} finally {
|
|
279
|
+
reader.releaseLock?.();
|
|
280
|
+
}
|
|
281
|
+
return text;
|
|
282
|
+
}
|
|
283
|
+
/** Pull `choices[0].delta.content` (or a non-streaming `message.content`) out of one SSE frame. */
|
|
284
|
+
function contentOfFrame(frame) {
|
|
285
|
+
let out = "";
|
|
286
|
+
for (const rawLine of frame.split(/\r?\n/u)) {
|
|
287
|
+
const line = rawLine.trim();
|
|
288
|
+
if (!line.startsWith("data:")) continue;
|
|
289
|
+
const payload = line.slice(5).trim();
|
|
290
|
+
if (payload === "" || payload === "[DONE]") continue;
|
|
291
|
+
let parsed;
|
|
292
|
+
try {
|
|
293
|
+
parsed = JSON.parse(payload);
|
|
294
|
+
} catch {
|
|
295
|
+
continue;
|
|
296
|
+
}
|
|
297
|
+
if (typeof parsed !== "object" || parsed === null) continue;
|
|
298
|
+
const choices = parsed["choices"];
|
|
299
|
+
if (!Array.isArray(choices) || choices.length === 0) continue;
|
|
300
|
+
const choice = choices[0];
|
|
301
|
+
const delta = choice["delta"];
|
|
302
|
+
if (typeof delta === "object" && delta !== null) {
|
|
303
|
+
const content = delta["content"];
|
|
304
|
+
if (typeof content === "string") out += content;
|
|
305
|
+
}
|
|
306
|
+
const message = choice["message"];
|
|
307
|
+
if (typeof message === "object" && message !== null) {
|
|
308
|
+
const content = message["content"];
|
|
309
|
+
if (typeof content === "string") out += content;
|
|
310
|
+
}
|
|
311
|
+
const data = parsed["data"];
|
|
312
|
+
if (typeof data === "object" && data !== null) {
|
|
313
|
+
const inner = data["content"];
|
|
314
|
+
if (typeof inner === "string") out += inner;
|
|
315
|
+
}
|
|
316
|
+
}
|
|
317
|
+
return out;
|
|
318
|
+
}
|
|
319
|
+
/**
|
|
226
320
|
* Classify an upstream failure from its HTTP status and body excerpt.
|
|
227
321
|
* Body markers win over status, because the upstream reuses 400/200 for
|
|
228
322
|
* several distinct conditions.
|
|
229
323
|
*/
|
|
230
324
|
function classifyUpstreamError(status, body) {
|
|
231
325
|
if (status === 402) return "hard_credit";
|
|
326
|
+
if (status === 401 || status === 403) return "session_dead";
|
|
232
327
|
const lower = body.toLowerCase();
|
|
233
328
|
for (const marker of HARD_CREDIT_MARKERS) if (lower.includes(marker.toLowerCase()) || body.includes(marker)) return "hard_credit";
|
|
234
329
|
for (const marker of SESSION_DEAD_MARKERS) if (body.includes(marker)) return "session_dead";
|
|
@@ -696,6 +791,51 @@ var WorkBuddyUpstreamClient = class {
|
|
|
696
791
|
}
|
|
697
792
|
return JSON.stringify(obj);
|
|
698
793
|
}
|
|
794
|
+
/**
|
|
795
|
+
* Parse a raw OpenAI chat body without normalising it.
|
|
796
|
+
*
|
|
797
|
+
* The compactor needs the message array as objects, while `chatStream` only
|
|
798
|
+
* accepts the serialised string form.
|
|
799
|
+
*/
|
|
800
|
+
parseChatBody(raw) {
|
|
801
|
+
let body;
|
|
802
|
+
try {
|
|
803
|
+
body = JSON.parse(raw);
|
|
804
|
+
} catch {
|
|
805
|
+
return;
|
|
806
|
+
}
|
|
807
|
+
if (typeof body !== "object" || body === null || Array.isArray(body)) return void 0;
|
|
808
|
+
return body;
|
|
809
|
+
}
|
|
810
|
+
/** Re-serialise `base` with a rewritten `messages` array, still normalised. */
|
|
811
|
+
buildChatBody(base, messages) {
|
|
812
|
+
return this.prepareChatBody(JSON.stringify({
|
|
813
|
+
...base,
|
|
814
|
+
messages
|
|
815
|
+
}));
|
|
816
|
+
}
|
|
817
|
+
/**
|
|
818
|
+
* Run one NON-streaming completion and return the assistant text.
|
|
819
|
+
*
|
|
820
|
+
* Used only for internal compaction (summarising dropped turns). The chat
|
|
821
|
+
* endpoint itself always streams, so this reassembles the SSE frames into a
|
|
822
|
+
* single string. Throws on any failure: the compactor then falls back to
|
|
823
|
+
* plain truncation rather than failing the user's turn.
|
|
824
|
+
*/
|
|
825
|
+
async completeChat(credential, prepared, signal) {
|
|
826
|
+
const response = await this.fetchImpl(`${chatBase(credential)}/v2/chat/completions`, {
|
|
827
|
+
method: "POST",
|
|
828
|
+
headers: chatHeaders(credential),
|
|
829
|
+
body: prepared,
|
|
830
|
+
...signal === void 0 ? {} : { signal }
|
|
831
|
+
});
|
|
832
|
+
if (!response.ok) {
|
|
833
|
+
const text = (await response.text().catch(() => "")).slice(0, ERROR_BODY_LIMIT);
|
|
834
|
+
throw new Error(`compaction upstream http ${response.status}: ${text}`);
|
|
835
|
+
}
|
|
836
|
+
if (response.body === null) throw new Error("compaction upstream returned no body");
|
|
837
|
+
return await readCompletionText(response.body);
|
|
838
|
+
}
|
|
699
839
|
/** Forward one chat completion. Never throws for upstream failures. */
|
|
700
840
|
async chatStream(credential, prepared, signal) {
|
|
701
841
|
let response;
|
|
@@ -3798,6 +3938,286 @@ var WorkBuddyScheduler = class {
|
|
|
3798
3938
|
}
|
|
3799
3939
|
};
|
|
3800
3940
|
//#endregion
|
|
3941
|
+
//#region src/context-budget.ts
|
|
3942
|
+
/** Rough character-per-token ratio. CJK is ~1 token/char, latin ~1/4. */
|
|
3943
|
+
const CHARS_PER_TOKEN_LATIN = 4;
|
|
3944
|
+
const CHARS_PER_TOKEN_CJK = 1;
|
|
3945
|
+
/** Fixed per-message overhead the chat template adds (role markers etc.). */
|
|
3946
|
+
const PER_MESSAGE_TOKEN_OVERHEAD = 4;
|
|
3947
|
+
/** Every image/tool part costs at least this much once decoded. */
|
|
3948
|
+
const PER_PART_TOKEN_FLOOR = 16;
|
|
3949
|
+
/**
|
|
3950
|
+
* Estimate the token cost of one message's `content`.
|
|
3951
|
+
*
|
|
3952
|
+
* Deliberately conservative (over-estimates) so we compact slightly early
|
|
3953
|
+
* rather than discovering the overrun upstream.
|
|
3954
|
+
*/
|
|
3955
|
+
function estimateContentTokens(content) {
|
|
3956
|
+
if (content === null || content === void 0) return 0;
|
|
3957
|
+
if (typeof content === "string") return estimateTextTokens(content);
|
|
3958
|
+
if (typeof content === "number" || typeof content === "boolean") return PER_PART_TOKEN_FLOOR;
|
|
3959
|
+
if (Array.isArray(content)) {
|
|
3960
|
+
let total = 0;
|
|
3961
|
+
for (const part of content) total += estimateContentTokens(part);
|
|
3962
|
+
return total;
|
|
3963
|
+
}
|
|
3964
|
+
if (typeof content === "object") {
|
|
3965
|
+
const record = content;
|
|
3966
|
+
let total = PER_PART_TOKEN_FLOOR;
|
|
3967
|
+
for (const key of [
|
|
3968
|
+
"text",
|
|
3969
|
+
"image_url",
|
|
3970
|
+
"input",
|
|
3971
|
+
"content"
|
|
3972
|
+
]) if (key in record) total += estimateContentTokens(record[key]);
|
|
3973
|
+
if (total === PER_PART_TOKEN_FLOOR) total += estimateTextTokens(safeStringify(record));
|
|
3974
|
+
return total;
|
|
3975
|
+
}
|
|
3976
|
+
return 0;
|
|
3977
|
+
}
|
|
3978
|
+
/** Estimate tokens for a plain string, accounting for CJK density. */
|
|
3979
|
+
function estimateTextTokens(text) {
|
|
3980
|
+
if (text === "") return 0;
|
|
3981
|
+
let cjk = 0;
|
|
3982
|
+
for (const char of text) if (isCjk(char.codePointAt(0) ?? 0)) cjk += 1;
|
|
3983
|
+
const latin = text.length - cjk;
|
|
3984
|
+
return Math.ceil(cjk / CHARS_PER_TOKEN_CJK + latin / CHARS_PER_TOKEN_LATIN);
|
|
3985
|
+
}
|
|
3986
|
+
function isCjk(code) {
|
|
3987
|
+
return code >= 12288 && code <= 12351 || code >= 12352 && code <= 12543 || code >= 13312 && code <= 19903 || code >= 19968 && code <= 40959 || code >= 63744 && code <= 64255 || code >= 65280 && code <= 65519 || code >= 131072 && code <= 191471;
|
|
3988
|
+
}
|
|
3989
|
+
function safeStringify(value) {
|
|
3990
|
+
try {
|
|
3991
|
+
return JSON.stringify(value) ?? "";
|
|
3992
|
+
} catch {
|
|
3993
|
+
return String(value);
|
|
3994
|
+
}
|
|
3995
|
+
}
|
|
3996
|
+
/** Estimate the prompt cost of a whole message array. */
|
|
3997
|
+
function estimateMessagesTokens(messages) {
|
|
3998
|
+
let total = 0;
|
|
3999
|
+
for (const message of messages) {
|
|
4000
|
+
total += PER_MESSAGE_TOKEN_OVERHEAD;
|
|
4001
|
+
total += estimateContentTokens(message.content);
|
|
4002
|
+
if (message["tool_calls"] !== void 0) total += estimateContentTokens(message["tool_calls"]);
|
|
4003
|
+
if (message["name"] !== void 0) total += estimateTextTokens(String(message["name"]));
|
|
4004
|
+
}
|
|
4005
|
+
return total;
|
|
4006
|
+
}
|
|
4007
|
+
/** True when `role` carries instructions that must survive compaction. */
|
|
4008
|
+
function isPinnedRole(role) {
|
|
4009
|
+
return role === "system" || role === "developer";
|
|
4010
|
+
}
|
|
4011
|
+
/**
|
|
4012
|
+
* Drop the oldest non-pinned messages until the estimate fits `budget`.
|
|
4013
|
+
*
|
|
4014
|
+
* Pinned (system/developer) messages and the newest `keepRecent` messages are
|
|
4015
|
+
* never dropped here — if those alone overrun the budget, the caller must fall
|
|
4016
|
+
* back to summarisation or give up.
|
|
4017
|
+
*/
|
|
4018
|
+
function compactMessages(messages, options) {
|
|
4019
|
+
const keepRecent = Math.max(1, options.keepRecent ?? 4);
|
|
4020
|
+
const estimate = estimateMessagesTokens(messages);
|
|
4021
|
+
if (estimate <= options.budget) return {
|
|
4022
|
+
messages: [...messages],
|
|
4023
|
+
dropped: [],
|
|
4024
|
+
tokens: estimate,
|
|
4025
|
+
changed: false
|
|
4026
|
+
};
|
|
4027
|
+
const pinned = [];
|
|
4028
|
+
const body = [];
|
|
4029
|
+
for (const message of messages) if (isPinnedRole(message.role)) pinned.push(message);
|
|
4030
|
+
else body.push(message);
|
|
4031
|
+
const keep = Math.min(keepRecent, body.length);
|
|
4032
|
+
const tail = body.slice(body.length - keep);
|
|
4033
|
+
const head = body.slice(0, body.length - keep);
|
|
4034
|
+
let dropCount = 0;
|
|
4035
|
+
let candidate = [
|
|
4036
|
+
...pinned,
|
|
4037
|
+
...head,
|
|
4038
|
+
...tail
|
|
4039
|
+
];
|
|
4040
|
+
let total = estimateMessagesTokens(candidate);
|
|
4041
|
+
while (total > options.budget && dropCount < head.length) {
|
|
4042
|
+
dropCount += 1;
|
|
4043
|
+
candidate = [
|
|
4044
|
+
...pinned,
|
|
4045
|
+
...head.slice(dropCount),
|
|
4046
|
+
...tail
|
|
4047
|
+
];
|
|
4048
|
+
total = estimateMessagesTokens(candidate);
|
|
4049
|
+
}
|
|
4050
|
+
const dropped = head.slice(0, dropCount);
|
|
4051
|
+
return {
|
|
4052
|
+
messages: candidate,
|
|
4053
|
+
dropped,
|
|
4054
|
+
tokens: total,
|
|
4055
|
+
changed: dropCount > 0
|
|
4056
|
+
};
|
|
4057
|
+
}
|
|
4058
|
+
/**
|
|
4059
|
+
* Drop the oldest messages, including pinned ones, as a last resort.
|
|
4060
|
+
*
|
|
4061
|
+
* Used when even a summary cannot bring the prompt under budget (for example a
|
|
4062
|
+
* single enormous pasted document). The newest message always survives.
|
|
4063
|
+
*/
|
|
4064
|
+
function hardTruncate(messages, budget) {
|
|
4065
|
+
if (messages.length === 0) return {
|
|
4066
|
+
messages: [],
|
|
4067
|
+
dropped: [],
|
|
4068
|
+
tokens: 0,
|
|
4069
|
+
changed: false
|
|
4070
|
+
};
|
|
4071
|
+
let start = 0;
|
|
4072
|
+
let candidate = [...messages];
|
|
4073
|
+
let total = estimateMessagesTokens(candidate);
|
|
4074
|
+
while (total > budget && start < messages.length - 1) {
|
|
4075
|
+
start += 1;
|
|
4076
|
+
candidate = messages.slice(start);
|
|
4077
|
+
total = estimateMessagesTokens(candidate);
|
|
4078
|
+
}
|
|
4079
|
+
return {
|
|
4080
|
+
messages: candidate,
|
|
4081
|
+
dropped: messages.slice(0, start),
|
|
4082
|
+
tokens: total,
|
|
4083
|
+
changed: start > 0
|
|
4084
|
+
};
|
|
4085
|
+
}
|
|
4086
|
+
/** Instructions handed to the model when we ask it to compact a conversation. */
|
|
4087
|
+
const SUMMARIZE_INSTRUCTION = [
|
|
4088
|
+
"You are compacting an ongoing conversation so it can continue without the original history.",
|
|
4089
|
+
"Summarise the transcript below into a dense briefing for the next assistant turn.",
|
|
4090
|
+
"Preserve, in this order of priority:",
|
|
4091
|
+
"1. explicit user requirements, constraints and corrections;",
|
|
4092
|
+
"2. decisions already made, and the reasoning behind them;",
|
|
4093
|
+
"3. concrete facts: file paths, identifiers, commands, numbers, error messages;",
|
|
4094
|
+
"4. unfinished work and the current blocker.",
|
|
4095
|
+
"Drop pleasantries, repetition and superseded attempts.",
|
|
4096
|
+
"Write the briefing only — no preamble, no markdown fence."
|
|
4097
|
+
].join("\n");
|
|
4098
|
+
/** Render a message array as plain text for the summarisation prompt. */
|
|
4099
|
+
function transcriptOf(messages) {
|
|
4100
|
+
const lines = [];
|
|
4101
|
+
for (const message of messages) {
|
|
4102
|
+
const role = message.role === "" ? "unknown" : message.role;
|
|
4103
|
+
lines.push(`### ${role}`);
|
|
4104
|
+
lines.push(renderContent(message.content));
|
|
4105
|
+
if (message["tool_calls"] !== void 0) lines.push(renderContent(message["tool_calls"]));
|
|
4106
|
+
}
|
|
4107
|
+
return lines.join("\n");
|
|
4108
|
+
}
|
|
4109
|
+
function renderContent(content) {
|
|
4110
|
+
if (content === null || content === void 0) return "";
|
|
4111
|
+
if (typeof content === "string") return content;
|
|
4112
|
+
if (Array.isArray(content)) return content.map((part) => renderContent(part)).filter((text) => text !== "").join("\n");
|
|
4113
|
+
if (typeof content === "object") {
|
|
4114
|
+
const record = content;
|
|
4115
|
+
for (const key of [
|
|
4116
|
+
"text",
|
|
4117
|
+
"content",
|
|
4118
|
+
"input"
|
|
4119
|
+
]) if (typeof record[key] === "string") return record[key];
|
|
4120
|
+
if (record["type"] !== void 0 && typeof record["type"] === "string") return `[${record["type"]}]`;
|
|
4121
|
+
return safeStringify(record);
|
|
4122
|
+
}
|
|
4123
|
+
return String(content);
|
|
4124
|
+
}
|
|
4125
|
+
/** Build the synthetic system message that carries a compaction summary. */
|
|
4126
|
+
function summaryMessage(summary) {
|
|
4127
|
+
return {
|
|
4128
|
+
role: "system",
|
|
4129
|
+
content: [
|
|
4130
|
+
"The earlier part of this conversation was compacted to fit the model context window.",
|
|
4131
|
+
"Briefing produced from the dropped turns:",
|
|
4132
|
+
"",
|
|
4133
|
+
summary.trim()
|
|
4134
|
+
].join("\n")
|
|
4135
|
+
};
|
|
4136
|
+
}
|
|
4137
|
+
/**
|
|
4138
|
+
* Compact `messages` to `budget`, summarising the dropped turns when possible.
|
|
4139
|
+
*
|
|
4140
|
+
* The summary is requested with a *bounded* transcript so the compaction call
|
|
4141
|
+
* itself can never overrun the window: if the dropped turns are huge, only the
|
|
4142
|
+
* newest slice of them is summarised, and the oldest are noted as elided.
|
|
4143
|
+
*/
|
|
4144
|
+
async function compactWithSummary(messages, options, deps, signal) {
|
|
4145
|
+
const first = compactMessages(messages, options);
|
|
4146
|
+
if (!first.changed) return {
|
|
4147
|
+
messages: first.messages,
|
|
4148
|
+
tokens: first.tokens,
|
|
4149
|
+
summary: void 0,
|
|
4150
|
+
skipped: void 0
|
|
4151
|
+
};
|
|
4152
|
+
const summaryBudget = Math.max(256, Math.floor(options.budget / 4));
|
|
4153
|
+
let toSummarize = first.dropped;
|
|
4154
|
+
let elided = 0;
|
|
4155
|
+
while (estimateMessagesTokens(toSummarize) > summaryBudget && toSummarize.length > 1) {
|
|
4156
|
+
toSummarize = toSummarize.slice(1);
|
|
4157
|
+
elided += 1;
|
|
4158
|
+
}
|
|
4159
|
+
let summary;
|
|
4160
|
+
let skipped;
|
|
4161
|
+
try {
|
|
4162
|
+
const instruction = elided > 0 ? `${SUMMARIZE_INSTRUCTION}\n\nNote: the ${elided} oldest turn(s) were elided before this transcript.` : SUMMARIZE_INSTRUCTION;
|
|
4163
|
+
const suffix = elided > 0 ? `\n(the ${elided} oldest turn(s) were elided)` : "";
|
|
4164
|
+
const request = [{
|
|
4165
|
+
role: "system",
|
|
4166
|
+
content: instruction
|
|
4167
|
+
}, {
|
|
4168
|
+
role: "user",
|
|
4169
|
+
content: `${transcriptOf(toSummarize)}${suffix}`
|
|
4170
|
+
}];
|
|
4171
|
+
const text = await deps.complete(request, signal);
|
|
4172
|
+
if (text.trim() !== "") summary = text.trim();
|
|
4173
|
+
else skipped = "summariser returned an empty summary";
|
|
4174
|
+
} catch (error) {
|
|
4175
|
+
skipped = `summarisation failed: ${String(error)}`;
|
|
4176
|
+
}
|
|
4177
|
+
if (summary === void 0) return {
|
|
4178
|
+
messages: first.messages,
|
|
4179
|
+
skipped,
|
|
4180
|
+
tokens: first.tokens
|
|
4181
|
+
};
|
|
4182
|
+
const withSummary = injectSummary(first.messages, summary);
|
|
4183
|
+
if (estimateMessagesTokens(withSummary) > options.budget) {
|
|
4184
|
+
const truncated = hardTruncate(withSummary, options.budget);
|
|
4185
|
+
return {
|
|
4186
|
+
messages: truncated.messages,
|
|
4187
|
+
summary,
|
|
4188
|
+
tokens: truncated.tokens
|
|
4189
|
+
};
|
|
4190
|
+
}
|
|
4191
|
+
return {
|
|
4192
|
+
messages: withSummary,
|
|
4193
|
+
summary,
|
|
4194
|
+
tokens: estimateMessagesTokens(withSummary)
|
|
4195
|
+
};
|
|
4196
|
+
}
|
|
4197
|
+
/**
|
|
4198
|
+
* Re-insert a summary as: pinned instructions → summary → surviving tail.
|
|
4199
|
+
*
|
|
4200
|
+
* Order matters. Pinned (system/developer) messages must stay ahead of the
|
|
4201
|
+
* summary so that a later synthetic system message can never override the
|
|
4202
|
+
* harness's own instructions; the tail follows so the newest exchange is the
|
|
4203
|
+
* last thing the model reads.
|
|
4204
|
+
*
|
|
4205
|
+
* `compacted` is always derived from `original` by `compactMessages`, so the
|
|
4206
|
+
* pinned messages it carries are exactly the originals — no need to re-add
|
|
4207
|
+
* them from `original`.
|
|
4208
|
+
*/
|
|
4209
|
+
function injectSummary(compacted, summary) {
|
|
4210
|
+
const pinned = [];
|
|
4211
|
+
const rest = [];
|
|
4212
|
+
for (const message of compacted) if (isPinnedRole(message.role)) pinned.push(message);
|
|
4213
|
+
else rest.push(message);
|
|
4214
|
+
return [
|
|
4215
|
+
...pinned,
|
|
4216
|
+
summaryMessage(summary),
|
|
4217
|
+
...rest
|
|
4218
|
+
];
|
|
4219
|
+
}
|
|
4220
|
+
//#endregion
|
|
3801
4221
|
//#region src/shim.ts
|
|
3802
4222
|
/**
|
|
3803
4223
|
* Loopback OpenAI-compatible endpoint with multi-account failover.
|
|
@@ -3876,7 +4296,8 @@ function writeOpenAIError(res, status, kind, message) {
|
|
|
3876
4296
|
}
|
|
3877
4297
|
/** True when an upstream failure body means the request overran the model's
|
|
3878
4298
|
* context window (OpenAI `context_length_exceeded`, WorkBuddy code 11115 /
|
|
3879
|
-
* "input length too long").
|
|
4299
|
+
* "input length too long"). The shim answers it by compacting the conversation
|
|
4300
|
+
* in place and retrying once; see `recoverFromContextOverrun`. */
|
|
3880
4301
|
function isContextTooLong(body) {
|
|
3881
4302
|
if (body.includes("context_length_exceeded")) return true;
|
|
3882
4303
|
if (body.includes("input length too long")) return true;
|
|
@@ -4038,25 +4459,7 @@ function createWorkBuddyShim(options) {
|
|
|
4038
4459
|
tried.push(account.label);
|
|
4039
4460
|
const result = await client.chatStream(account.credential, prepared, controller.signal);
|
|
4040
4461
|
if (result.ok) {
|
|
4041
|
-
|
|
4042
|
-
pool.noteServed(account.id);
|
|
4043
|
-
refreshBalance(account);
|
|
4044
|
-
res.writeHead(200, {
|
|
4045
|
-
"Content-Type": "text/event-stream",
|
|
4046
|
-
"Cache-Control": "no-cache",
|
|
4047
|
-
"Connection": "keep-alive",
|
|
4048
|
-
"X-Accel-Buffering": "no"
|
|
4049
|
-
});
|
|
4050
|
-
let sawDone = false;
|
|
4051
|
-
const body = Readable.fromWeb(result.response.body);
|
|
4052
|
-
body.on("data", (chunk) => {
|
|
4053
|
-
if (chunk.includes("[DONE]")) sawDone = true;
|
|
4054
|
-
});
|
|
4055
|
-
body.on("error", (error) => {
|
|
4056
|
-
logger?.warn("dsh-workbuddy-xdpool: upstream stream failed mid-flight", error);
|
|
4057
|
-
if (!sawDone && res.writable) res.end("data: [DONE]\n\n");
|
|
4058
|
-
});
|
|
4059
|
-
body.pipe(res);
|
|
4462
|
+
await serveSuccessfulStream(res, account, result, logger, refreshBalance, pool);
|
|
4060
4463
|
return;
|
|
4061
4464
|
}
|
|
4062
4465
|
last = {
|
|
@@ -4084,7 +4487,21 @@ function createWorkBuddyShim(options) {
|
|
|
4084
4487
|
return;
|
|
4085
4488
|
}
|
|
4086
4489
|
if (isContextTooLong(last.message)) {
|
|
4087
|
-
|
|
4490
|
+
const recovered = await recoverFromContextOverrun({
|
|
4491
|
+
raw,
|
|
4492
|
+
modelId,
|
|
4493
|
+
controller,
|
|
4494
|
+
region,
|
|
4495
|
+
logger,
|
|
4496
|
+
client,
|
|
4497
|
+
pool,
|
|
4498
|
+
maxAttempts
|
|
4499
|
+
});
|
|
4500
|
+
if (recovered.ok) {
|
|
4501
|
+
await serveSuccessfulStream(res, recovered.account, recovered.result, logger, refreshBalance, pool);
|
|
4502
|
+
return;
|
|
4503
|
+
}
|
|
4504
|
+
writeOpenAIError(res, 400, "context_length_exceeded", contextOverflowMessage(modelId, recovered.detail));
|
|
4088
4505
|
return;
|
|
4089
4506
|
}
|
|
4090
4507
|
writeOpenAIError(res, KIND_STATUS[last.kind], last.kind, `workbuddy upstream ${last.kind} (http ${last.status}) after ${tried.length} account(s) [${tried.join(" → ")}]: ${last.message.slice(0, 400)}`);
|
|
@@ -4100,6 +4517,168 @@ function createWorkBuddyShim(options) {
|
|
|
4100
4517
|
})
|
|
4101
4518
|
};
|
|
4102
4519
|
}
|
|
4520
|
+
/**
|
|
4521
|
+
* Build the overflow message the Harness must recognize.
|
|
4522
|
+
*
|
|
4523
|
+
* This is deliberately NOT free-form prose. `dsh-compaction-basic` decides
|
|
4524
|
+
* whether to compact-and-retry by running the text that reaches it through
|
|
4525
|
+
* `isContextWindowExceededError()` (`@deepseek-ai/dsh-llm`), whose matcher
|
|
4526
|
+
* accepts only specific phrasings:
|
|
4527
|
+
*
|
|
4528
|
+
* - `context_length_exceeded` / `context window exceeded`
|
|
4529
|
+
* - `maximum context length`
|
|
4530
|
+
* - `<input|prompt|request|messages> too large|long for ... context`
|
|
4531
|
+
* - `<input|prompt|request> exceeds the ... context window`
|
|
4532
|
+
*
|
|
4533
|
+
* The obvious friendly sentence ("the conversation exceeds this model's
|
|
4534
|
+
* context window") matches NONE of them, and neither does the WorkBuddy
|
|
4535
|
+
* upstream's own "input length too long" / code 11115. Emitting either meant
|
|
4536
|
+
* the Harness saw an unclassifiable 400, skipped its recovery path, and
|
|
4537
|
+
* surfaced a dead turn — the bug this function exists to prevent.
|
|
4538
|
+
*
|
|
4539
|
+
* The leading clause carries the machine-matched wording; the trailing clause
|
|
4540
|
+
* is what a human reads. Keep both in sync with
|
|
4541
|
+
* `tests/context-overflow-contract.test.ts`.
|
|
4542
|
+
*/
|
|
4543
|
+
function contextOverflowMessage(modelId, detail = "") {
|
|
4544
|
+
return `This model's maximum context length was exceeded: the prompt is too large for ${modelId === void 0 ? "the model" : `model ${modelId}`}, and the conversation could not be compacted in place${detail === "" ? "" : ` (${detail})`}. Compact the conversation, or start a new chat.`;
|
|
4545
|
+
}
|
|
4546
|
+
/**
|
|
4547
|
+
* Serve one already-successful upstream stream as an SSE response.
|
|
4548
|
+
*
|
|
4549
|
+
* Extracted so the context-overrun recovery path reuses the exact same
|
|
4550
|
+
* bookkeeping (noteServed + background balance refresh) as a first-try hit.
|
|
4551
|
+
*/
|
|
4552
|
+
async function serveSuccessfulStream(res, account, result, logger, refreshBalance, pool) {
|
|
4553
|
+
logger?.info?.(`dsh-workbuddy-xdpool: served by ${account.label}`);
|
|
4554
|
+
pool.noteServed(account.id);
|
|
4555
|
+
refreshBalance(account);
|
|
4556
|
+
res.writeHead(200, {
|
|
4557
|
+
"Content-Type": "text/event-stream",
|
|
4558
|
+
"Cache-Control": "no-cache",
|
|
4559
|
+
"Connection": "keep-alive",
|
|
4560
|
+
"X-Accel-Buffering": "no"
|
|
4561
|
+
});
|
|
4562
|
+
let sawDone = false;
|
|
4563
|
+
const body = Readable.fromWeb(result.response.body);
|
|
4564
|
+
body.on("data", (chunk) => {
|
|
4565
|
+
if (chunk.includes("[DONE]")) sawDone = true;
|
|
4566
|
+
});
|
|
4567
|
+
body.on("error", (error) => {
|
|
4568
|
+
logger?.warn("dsh-workbuddy-xdpool: upstream stream failed mid-flight", error);
|
|
4569
|
+
if (!sawDone && res.writable) res.end("data: [DONE]\n\n");
|
|
4570
|
+
});
|
|
4571
|
+
body.pipe(res);
|
|
4572
|
+
}
|
|
4573
|
+
/**
|
|
4574
|
+
* Compact an over-long conversation and retry it once.
|
|
4575
|
+
*
|
|
4576
|
+
* Strategy, in order:
|
|
4577
|
+
* 1. drop the oldest turns, keeping system messages and the newest exchange;
|
|
4578
|
+
* 2. ask the model to summarise the dropped turns and splice that summary in;
|
|
4579
|
+
* 3. hard-truncate as a last resort.
|
|
4580
|
+
*
|
|
4581
|
+
* Returns `ok: false` only when even a truncated prompt still overran — the
|
|
4582
|
+
* caller then surfaces the original actionable 400.
|
|
4583
|
+
*/
|
|
4584
|
+
async function recoverFromContextOverrun(options) {
|
|
4585
|
+
const { raw, modelId, controller, region, logger, client, pool, maxAttempts } = options;
|
|
4586
|
+
const parsed = client.parseChatBody(raw);
|
|
4587
|
+
if (parsed === void 0) return {
|
|
4588
|
+
ok: false,
|
|
4589
|
+
detail: "request body was not parseable JSON"
|
|
4590
|
+
};
|
|
4591
|
+
const rawMessages = parsed["messages"];
|
|
4592
|
+
if (!Array.isArray(rawMessages)) return {
|
|
4593
|
+
ok: false,
|
|
4594
|
+
detail: "request carried no messages array"
|
|
4595
|
+
};
|
|
4596
|
+
const messages = rawMessages.filter((value) => typeof value === "object" && value !== null && !Array.isArray(value));
|
|
4597
|
+
if (messages.length === 0) return {
|
|
4598
|
+
ok: false,
|
|
4599
|
+
detail: "request carried no usable messages"
|
|
4600
|
+
};
|
|
4601
|
+
const overrunTokens = estimateMessagesTokens(messages);
|
|
4602
|
+
const budget = Math.max(512, Math.floor(overrunTokens / 2));
|
|
4603
|
+
logger?.warn(`dsh-workbuddy-xdpool: context overrun on ${modelId ?? "(no model)"} (~${overrunTokens} tokens); compacting to ~${budget} and retrying once`);
|
|
4604
|
+
let summary;
|
|
4605
|
+
let compacted = messages;
|
|
4606
|
+
let compactionDetail = "";
|
|
4607
|
+
try {
|
|
4608
|
+
const summariser = await pool.acquire(modelId, region);
|
|
4609
|
+
if (summariser === void 0) compactionDetail = "no account available to summarise with";
|
|
4610
|
+
else {
|
|
4611
|
+
const outcome = await compactWithSummary(messages, {
|
|
4612
|
+
budget,
|
|
4613
|
+
keepRecent: 6
|
|
4614
|
+
}, { complete: async (request, signal) => {
|
|
4615
|
+
const body = client.buildChatBody({
|
|
4616
|
+
...parsed,
|
|
4617
|
+
stream: true,
|
|
4618
|
+
max_tokens: Math.max(256, Math.floor(budget / 2))
|
|
4619
|
+
}, request);
|
|
4620
|
+
return await client.completeChat(summariser.credential, body, signal ?? controller.signal);
|
|
4621
|
+
} }, controller.signal);
|
|
4622
|
+
compacted = outcome.messages;
|
|
4623
|
+
summary = outcome.summary;
|
|
4624
|
+
if (outcome.skipped !== void 0) compactionDetail = outcome.skipped;
|
|
4625
|
+
}
|
|
4626
|
+
} catch (error) {
|
|
4627
|
+
compactionDetail = `summarisation failed: ${String(error)}`;
|
|
4628
|
+
}
|
|
4629
|
+
if (estimateMessagesTokens(compacted) > budget) compacted = hardTruncate(compacted, budget).messages;
|
|
4630
|
+
if (summary === void 0 && estimateMessagesTokens(compacted) >= overrunTokens) return {
|
|
4631
|
+
ok: false,
|
|
4632
|
+
detail: compactionDetail === "" ? "compaction could not reduce the prompt" : compactionDetail
|
|
4633
|
+
};
|
|
4634
|
+
const retryBody = client.buildChatBody(parsed, compacted);
|
|
4635
|
+
const tried = [];
|
|
4636
|
+
for (let attempt = 0; attempt < maxAttempts; attempt += 1) {
|
|
4637
|
+
if (controller.signal.aborted) return {
|
|
4638
|
+
ok: false,
|
|
4639
|
+
detail: "client disconnected"
|
|
4640
|
+
};
|
|
4641
|
+
const account = await pool.acquire(modelId, region);
|
|
4642
|
+
if (account === void 0) return {
|
|
4643
|
+
ok: false,
|
|
4644
|
+
detail: "no account available after compaction"
|
|
4645
|
+
};
|
|
4646
|
+
tried.push(account.label);
|
|
4647
|
+
const result = await client.chatStream(account.credential, retryBody, controller.signal);
|
|
4648
|
+
if (result.ok) {
|
|
4649
|
+
logger?.info?.(`dsh-workbuddy-xdpool: recovered from context overrun on ${modelId ?? "(no model)"} (summarised: ${summary === void 0 ? "no" : "yes"})`);
|
|
4650
|
+
return {
|
|
4651
|
+
ok: true,
|
|
4652
|
+
account,
|
|
4653
|
+
result
|
|
4654
|
+
};
|
|
4655
|
+
}
|
|
4656
|
+
if (isContextTooLong(result.message)) return {
|
|
4657
|
+
ok: false,
|
|
4658
|
+
detail: "prompt still exceeded the window after compaction"
|
|
4659
|
+
};
|
|
4660
|
+
if (result.kind === "session_dead") {
|
|
4661
|
+
await pool.refreshAccount(account.id);
|
|
4662
|
+
continue;
|
|
4663
|
+
}
|
|
4664
|
+
if (result.kind === "hard_credit") {
|
|
4665
|
+
pool.penalizeExhausted(account.id);
|
|
4666
|
+
continue;
|
|
4667
|
+
}
|
|
4668
|
+
if (result.kind === "soft_rate") {
|
|
4669
|
+
pool.penalize(account.id, parseRateLimitReset(result.message), modelId);
|
|
4670
|
+
continue;
|
|
4671
|
+
}
|
|
4672
|
+
return {
|
|
4673
|
+
ok: false,
|
|
4674
|
+
detail: `upstream ${result.kind} after compaction`
|
|
4675
|
+
};
|
|
4676
|
+
}
|
|
4677
|
+
return {
|
|
4678
|
+
ok: false,
|
|
4679
|
+
detail: `no account served the compacted request (tried ${tried.length})`
|
|
4680
|
+
};
|
|
4681
|
+
}
|
|
4103
4682
|
//#endregion
|
|
4104
4683
|
//#region src/status.ts
|
|
4105
4684
|
/** Build the status document. Never throws. */
|
package/package.json
CHANGED
|
@@ -2,7 +2,7 @@
|
|
|
2
2
|
"name": "dsh-workbuddy-xdpool",
|
|
3
3
|
"displayName": "DSH WorkBuddy XD Pool",
|
|
4
4
|
"description": "Merge every locally signed-in WorkBuddy account into DeepSeek Harness as one auto-failing-over model pool (multi-account rotation, live credits, daily check-in and model catalog).",
|
|
5
|
-
"version": "1.
|
|
5
|
+
"version": "1.4.0",
|
|
6
6
|
"license": "MIT",
|
|
7
7
|
"author": "XDTrees",
|
|
8
8
|
"homepage": "https://github.com/XDTrees/dsh-workbuddy-xdpool#readme",
|