dsh-workbuddy-xdpool 1.3.0 → 1.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/CHANGELOG.md CHANGED
@@ -4,6 +4,50 @@
4
4
 
5
5
  版本号遵循 [语义化版本](https://semver.org/lang/zh-CN/)。
6
6
 
7
+ ## 1.4.0 (2026-09-24)
8
+
9
+ ### 长对话不再撞死在上下文上限上
10
+
11
+ 以前聊到一定长度,就会弹出一个 `400 context_length_exceeded` 然后被送回主页,只能重开会话。
12
+ 现在会自动收缩上下文继续跑,不再打断你。
13
+
14
+ 这里其实有两层保险,各自管不同的情况:
15
+
16
+ **第一层是 Harness 自带的。** DSH 本来就装了 `compaction-basic`,它会在发请求之前先算一遍 token
17
+ 压力,超了就摘要、重试。问题是它认错误的方式是「看这段文字像不像上下文溢出」,而插件原来报的是
18
+ 一句友好英文 —— `the conversation exceeds <模型名>'s context window` —— 恰好是它唯一认不出来的
19
+ 那种写法。于是它以为这是个普通的坏请求,跳过了恢复,直接把 400 抛给你。
20
+
21
+ (顺带一个坑:上游 WorkBuddy 自己那句 `input length too long`/业务码 `11115`,同样不在它的
22
+ 识别清单里。所以「把上游原话原样透传」这条路也走不通,必须主动改写成它认得的措辞。)
23
+
24
+ 现在溢出时会输出它认得的写法,Harness 的自动摘要就接上了。
25
+
26
+ **第二层是插件自己的。** 就算 Harness 那层没兜住(比如窗口值报得不准,或者你一口气粘了个巨大
27
+ 文件),插件会自己压缩一遍再重试一次:先丢最旧的回合,保留 system 和最近几轮;还放不下就把被丢
28
+ 掉的回合交给模型做个摘要塞回去;再不行才硬截断。摘要这一步哪怕失败也只会降级成直接截断,绝不
29
+ 会反过来让你的这一轮请求失败。
30
+
31
+ ### 「API 密钥无效」不再把所有账号卡死在第一个上
32
+
33
+ 账号池的本意是:某个号不行就换下一个。但错误分类里有个漏洞 —— HTTP 401 只有在响应体里恰好
34
+ 出现两条特定标记之一时,才会被认成「登录态失效」(可刷新、可轮换)。上游换个说法,比如
35
+ 「api密钥无效」、`invalid api key`、`unauthorized`,就掉进了兜底的 `client` 类。
36
+
37
+ 而 `client` 在轮换循环里是**终态**:直接 `break`,不再试后面的号。结果就是一个坏掉的号把整个池
38
+ 钉死在第一个账号上,2/3/4 号全都白装着。
39
+
40
+ 现在 401 和 403 无条件算「登录态失效」—— 刷新 token 然后换下一个号。另外把常见的中英文失效措辞
41
+ 也补进了标记清单,这样即便状态码不是 401(比如被包在 200 信封里)也还能认出来。
42
+
43
+ ### 其它
44
+
45
+ - 顺手修了个压缩逻辑里的真 bug:它用同一个变量同时表示「留下来的」和「被丢掉的」,结果明明已经
46
+ 砍到只剩几条,却报告「没有变化」,导致摘要功能整个跳过、根本没执行。
47
+ - 新增两组测试:`context-overflow-contract.test.ts` 逐字复刻了 Harness 的错误匹配规则(它将来改
48
+ 措辞时这个测试会变红,提醒你同步改);`upstream-error-classification.test.ts` 钉住了上面那个
49
+ 401 分类回归。
50
+
7
51
  ## 1.3.0 (2026-09-23)
8
52
 
9
53
  上一版把八类任务链接了进去,这一版修的是**「接进去了但你看不见」**的问题——功能在跑,界面上却没显示,等于白做。
package/lib/bin.js CHANGED
@@ -29,6 +29,8 @@ const MODELS_CATALOG_PATH = "/v2/enterprises/personal/models";
29
29
  const GLOBAL_CONFIG_PATH = "/v3/config";
30
30
  const JSON_TIMEOUT_MS = 3e4;
31
31
  const ERROR_BODY_LIMIT = 4096;
32
+ /** Cap on the reassembled compaction reply, guarding against a runaway stream. */
33
+ const COMPLETION_TEXT_LIMIT = 65536;
32
34
  /** Insufficient-credit markers, ASCII lowercase plus the original Chinese. */
33
35
  const HARD_CREDIT_MARKERS = [
34
36
  "insufficient credit",
@@ -47,8 +49,31 @@ const HARD_CREDIT_MARKERS = [
47
49
  "额度用尽",
48
50
  "没有积分"
49
51
  ];
50
- /** Session-invalidation markers that mean "sign in again in the WorkBuddy app". */
51
- const SESSION_DEAD_MARKERS = ["Offline user session not found", "12153"];
52
+ /** Session-invalidation markers that mean "this credential is dead; use another".
53
+ * Kept alongside the HTTP-status rule in `classifyUpstreamError`: the status is
54
+ * enough for a direct 401/403, but some failures arrive wrapped in a 200
55
+ * envelope or a 4xx the gateway words differently. Adding the English and
56
+ * Chinese phrasings the upstream actually uses keeps those recoverable too —
57
+ * an unmatched one fell through to `client`, which is terminal in the shim
58
+ * and pinned the pool to the first account (the "API 密钥无效" bug). */
59
+ const SESSION_DEAD_MARKERS = [
60
+ "Offline user session not found",
61
+ "12153",
62
+ "api key is invalid",
63
+ "invalid api key",
64
+ "invalid_api_key",
65
+ "api密钥无效",
66
+ "密钥无效",
67
+ "无效的密钥",
68
+ "unauthorized",
69
+ "token expired",
70
+ "token is invalid",
71
+ "login expired",
72
+ "please login",
73
+ "未登录",
74
+ "登录已失效",
75
+ "重新登录"
76
+ ];
52
77
  /**
53
78
  * Markers for "already checked in today".
54
79
  *
@@ -221,12 +246,82 @@ function envelopeError(status, envelope) {
221
246
  return /* @__PURE__ */ new Error(`workbuddy upstream ${kind} (http ${status}): ${envelope.msg.slice(0, 160)}`);
222
247
  }
223
248
  /**
249
+ * Read an OpenAI-style SSE chat stream and concatenate the assistant text.
250
+ *
251
+ * The upstream always streams (`stream: true` is forced on every chat body),
252
+ * so a non-streaming internal call has to reassemble the deltas itself. Only
253
+ * `choices[0].delta.content` is collected; reasoning deltas are dropped
254
+ * because a compaction summary needs the final answer, not the scratchpad.
255
+ */
256
+ async function readCompletionText(body) {
257
+ const decoder = new TextDecoder();
258
+ const reader = body.getReader();
259
+ let buffer = "";
260
+ let text = "";
261
+ try {
262
+ for (;;) {
263
+ const { done, value } = await reader.read();
264
+ if (done) break;
265
+ buffer += decoder.decode(value, { stream: true });
266
+ let split = buffer.indexOf("\n\n");
267
+ while (split !== -1) {
268
+ const frame = buffer.slice(0, split);
269
+ buffer = buffer.slice(split + 2);
270
+ text += contentOfFrame(frame);
271
+ if (text.length > COMPLETION_TEXT_LIMIT) return text.slice(0, COMPLETION_TEXT_LIMIT);
272
+ split = buffer.indexOf("\n\n");
273
+ }
274
+ }
275
+ if (buffer.trim() !== "") text += contentOfFrame(buffer);
276
+ } finally {
277
+ reader.releaseLock?.();
278
+ }
279
+ return text;
280
+ }
281
+ /** Pull `choices[0].delta.content` (or a non-streaming `message.content`) out of one SSE frame. */
282
+ function contentOfFrame(frame) {
283
+ let out = "";
284
+ for (const rawLine of frame.split(/\r?\n/u)) {
285
+ const line = rawLine.trim();
286
+ if (!line.startsWith("data:")) continue;
287
+ const payload = line.slice(5).trim();
288
+ if (payload === "" || payload === "[DONE]") continue;
289
+ let parsed;
290
+ try {
291
+ parsed = JSON.parse(payload);
292
+ } catch {
293
+ continue;
294
+ }
295
+ if (typeof parsed !== "object" || parsed === null) continue;
296
+ const choices = parsed["choices"];
297
+ if (!Array.isArray(choices) || choices.length === 0) continue;
298
+ const choice = choices[0];
299
+ const delta = choice["delta"];
300
+ if (typeof delta === "object" && delta !== null) {
301
+ const content = delta["content"];
302
+ if (typeof content === "string") out += content;
303
+ }
304
+ const message = choice["message"];
305
+ if (typeof message === "object" && message !== null) {
306
+ const content = message["content"];
307
+ if (typeof content === "string") out += content;
308
+ }
309
+ const data = parsed["data"];
310
+ if (typeof data === "object" && data !== null) {
311
+ const inner = data["content"];
312
+ if (typeof inner === "string") out += inner;
313
+ }
314
+ }
315
+ return out;
316
+ }
317
+ /**
224
318
  * Classify an upstream failure from its HTTP status and body excerpt.
225
319
  * Body markers win over status, because the upstream reuses 400/200 for
226
320
  * several distinct conditions.
227
321
  */
228
322
  function classifyUpstreamError(status, body) {
229
323
  if (status === 402) return "hard_credit";
324
+ if (status === 401 || status === 403) return "session_dead";
230
325
  const lower = body.toLowerCase();
231
326
  for (const marker of HARD_CREDIT_MARKERS) if (lower.includes(marker.toLowerCase()) || body.includes(marker)) return "hard_credit";
232
327
  for (const marker of SESSION_DEAD_MARKERS) if (body.includes(marker)) return "session_dead";
@@ -589,6 +684,51 @@ var WorkBuddyUpstreamClient = class {
589
684
  }
590
685
  return JSON.stringify(obj);
591
686
  }
687
+ /**
688
+ * Parse a raw OpenAI chat body without normalising it.
689
+ *
690
+ * The compactor needs the message array as objects, while `chatStream` only
691
+ * accepts the serialised string form.
692
+ */
693
+ parseChatBody(raw) {
694
+ let body;
695
+ try {
696
+ body = JSON.parse(raw);
697
+ } catch {
698
+ return;
699
+ }
700
+ if (typeof body !== "object" || body === null || Array.isArray(body)) return void 0;
701
+ return body;
702
+ }
703
+ /** Re-serialise `base` with a rewritten `messages` array, still normalised. */
704
+ buildChatBody(base, messages) {
705
+ return this.prepareChatBody(JSON.stringify({
706
+ ...base,
707
+ messages
708
+ }));
709
+ }
710
+ /**
711
+ * Run one NON-streaming completion and return the assistant text.
712
+ *
713
+ * Used only for internal compaction (summarising dropped turns). The chat
714
+ * endpoint itself always streams, so this reassembles the SSE frames into a
715
+ * single string. Throws on any failure: the compactor then falls back to
716
+ * plain truncation rather than failing the user's turn.
717
+ */
718
+ async completeChat(credential, prepared, signal) {
719
+ const response = await this.fetchImpl(`${chatBase(credential)}/v2/chat/completions`, {
720
+ method: "POST",
721
+ headers: chatHeaders(credential),
722
+ body: prepared,
723
+ ...signal === void 0 ? {} : { signal }
724
+ });
725
+ if (!response.ok) {
726
+ const text = (await response.text().catch(() => "")).slice(0, ERROR_BODY_LIMIT);
727
+ throw new Error(`compaction upstream http ${response.status}: ${text}`);
728
+ }
729
+ if (response.body === null) throw new Error("compaction upstream returned no body");
730
+ return await readCompletionText(response.body);
731
+ }
592
732
  /** Forward one chat completion. Never throws for upstream failures. */
593
733
  async chatStream(credential, prepared, signal) {
594
734
  let response;
@@ -3480,6 +3620,17 @@ var WorkBuddyScheduler = class {
3480
3620
  this.logger.info?.(`automation travel ${account.label}: nothing to do (state=${travel.state})`);
3481
3621
  }
3482
3622
  };
3623
+ [
3624
+ "You are compacting an ongoing conversation so it can continue without the original history.",
3625
+ "Summarise the transcript below into a dense briefing for the next assistant turn.",
3626
+ "Preserve, in this order of priority:",
3627
+ "1. explicit user requirements, constraints and corrections;",
3628
+ "2. decisions already made, and the reasoning behind them;",
3629
+ "3. concrete facts: file paths, identifiers, commands, numbers, error messages;",
3630
+ "4. unfinished work and the current blocker.",
3631
+ "Drop pleasantries, repetition and superseded attempts.",
3632
+ "Write the briefing only — no preamble, no markdown fence."
3633
+ ].join("\n");
3483
3634
  //#endregion
3484
3635
  //#region src/status.ts
3485
3636
  /** Format the per-model credit multipliers. */
package/lib/index.d.ts CHANGED
@@ -164,6 +164,32 @@ export declare function expertActualUseEvent(expert: MarketExpert, conversationI
164
164
  */
165
165
  export declare function expertChatEvents(expert: MarketExpert, conversationId: string, requestId: string): Record<string, unknown>[];
166
166
  //#endregion
167
+ //#region src/context-budget.d.ts
168
+ /**
169
+ * Context-window budgeting for the WorkBuddy shim.
170
+ *
171
+ * The upstream answers `context_length_exceeded` (business code 11115) when a
172
+ * request overruns the model's window. Rather than bouncing that back to the
173
+ * user as a dead turn, the shim compacts the conversation on the fly:
174
+ *
175
+ * 1. estimate the prompt cost locally (cheap, no round trip);
176
+ * 2. drop the oldest turns while keeping `system` + the newest exchange;
177
+ * 3. if that still overruns, ask the model itself to summarise the middle of
178
+ * the conversation and splice that summary back in as a system message.
179
+ *
180
+ * Everything here is pure and synchronous-free except `summarizeMessages`,
181
+ * which the caller drives through an injected chat function so this module
182
+ * stays testable without a network.
183
+ *
184
+ * @module dsh-workbuddy-xdpool/context-budget
185
+ */
186
+ /** One OpenAI chat message, narrowed to the fields we must preserve. */
187
+ interface ChatMessage {
188
+ role: string;
189
+ content: unknown;
190
+ [key: string]: unknown;
191
+ }
192
+ //#endregion
167
193
  //#region src/upstream.d.ts
168
194
  /** Upstream failure classes the shim maps onto distinct HTTP answers. */
169
195
  type UpstreamErrorKind = 'hard_credit' | 'soft_rate' | 'session_dead' | 'not_found' | 'server' | 'client';
@@ -388,6 +414,24 @@ export declare class WorkBuddyUpstreamClient {
388
414
  * business code 11128), and flatten `tool_choice` into its string form.
389
415
  */
390
416
  prepareChatBody(raw: string): string;
417
+ /**
418
+ * Parse a raw OpenAI chat body without normalising it.
419
+ *
420
+ * The compactor needs the message array as objects, while `chatStream` only
421
+ * accepts the serialised string form.
422
+ */
423
+ parseChatBody(raw: string): Record<string, unknown> | undefined;
424
+ /** Re-serialise `base` with a rewritten `messages` array, still normalised. */
425
+ buildChatBody(base: Record<string, unknown>, messages: readonly ChatMessage[]): string;
426
+ /**
427
+ * Run one NON-streaming completion and return the assistant text.
428
+ *
429
+ * Used only for internal compaction (summarising dropped turns). The chat
430
+ * endpoint itself always streams, so this reassembles the SSE frames into a
431
+ * single string. Throws on any failure: the compactor then falls back to
432
+ * plain truncation rather than failing the user's turn.
433
+ */
434
+ completeChat(credential: WorkBuddyCredential, prepared: string, signal?: AbortSignal): Promise<string>;
391
435
  /** Forward one chat completion. Never throws for upstream failures. */
392
436
  chatStream(credential: WorkBuddyCredential, prepared: string, signal?: AbortSignal): Promise<ChatStreamResult>;
393
437
  /** POST the token-refresh endpoint; the caller merges the outcome. */
package/lib/index.js CHANGED
@@ -31,6 +31,8 @@ const MODELS_CATALOG_PATH = "/v2/enterprises/personal/models";
31
31
  const GLOBAL_CONFIG_PATH = "/v3/config";
32
32
  const JSON_TIMEOUT_MS = 3e4;
33
33
  const ERROR_BODY_LIMIT = 4096;
34
+ /** Cap on the reassembled compaction reply, guarding against a runaway stream. */
35
+ const COMPLETION_TEXT_LIMIT = 65536;
34
36
  /** Insufficient-credit markers, ASCII lowercase plus the original Chinese. */
35
37
  const HARD_CREDIT_MARKERS = [
36
38
  "insufficient credit",
@@ -49,8 +51,31 @@ const HARD_CREDIT_MARKERS = [
49
51
  "额度用尽",
50
52
  "没有积分"
51
53
  ];
52
- /** Session-invalidation markers that mean "sign in again in the WorkBuddy app". */
53
- const SESSION_DEAD_MARKERS = ["Offline user session not found", "12153"];
54
+ /** Session-invalidation markers that mean "this credential is dead; use another".
55
+ * Kept alongside the HTTP-status rule in `classifyUpstreamError`: the status is
56
+ * enough for a direct 401/403, but some failures arrive wrapped in a 200
57
+ * envelope or a 4xx the gateway words differently. Adding the English and
58
+ * Chinese phrasings the upstream actually uses keeps those recoverable too —
59
+ * an unmatched one fell through to `client`, which is terminal in the shim
60
+ * and pinned the pool to the first account (the "API 密钥无效" bug). */
61
+ const SESSION_DEAD_MARKERS = [
62
+ "Offline user session not found",
63
+ "12153",
64
+ "api key is invalid",
65
+ "invalid api key",
66
+ "invalid_api_key",
67
+ "api密钥无效",
68
+ "密钥无效",
69
+ "无效的密钥",
70
+ "unauthorized",
71
+ "token expired",
72
+ "token is invalid",
73
+ "login expired",
74
+ "please login",
75
+ "未登录",
76
+ "登录已失效",
77
+ "重新登录"
78
+ ];
54
79
  /**
55
80
  * Markers for "already checked in today".
56
81
  *
@@ -223,12 +248,82 @@ function envelopeError(status, envelope) {
223
248
  return /* @__PURE__ */ new Error(`workbuddy upstream ${kind} (http ${status}): ${envelope.msg.slice(0, 160)}`);
224
249
  }
225
250
  /**
251
+ * Read an OpenAI-style SSE chat stream and concatenate the assistant text.
252
+ *
253
+ * The upstream always streams (`stream: true` is forced on every chat body),
254
+ * so a non-streaming internal call has to reassemble the deltas itself. Only
255
+ * `choices[0].delta.content` is collected; reasoning deltas are dropped
256
+ * because a compaction summary needs the final answer, not the scratchpad.
257
+ */
258
+ async function readCompletionText(body) {
259
+ const decoder = new TextDecoder();
260
+ const reader = body.getReader();
261
+ let buffer = "";
262
+ let text = "";
263
+ try {
264
+ for (;;) {
265
+ const { done, value } = await reader.read();
266
+ if (done) break;
267
+ buffer += decoder.decode(value, { stream: true });
268
+ let split = buffer.indexOf("\n\n");
269
+ while (split !== -1) {
270
+ const frame = buffer.slice(0, split);
271
+ buffer = buffer.slice(split + 2);
272
+ text += contentOfFrame(frame);
273
+ if (text.length > COMPLETION_TEXT_LIMIT) return text.slice(0, COMPLETION_TEXT_LIMIT);
274
+ split = buffer.indexOf("\n\n");
275
+ }
276
+ }
277
+ if (buffer.trim() !== "") text += contentOfFrame(buffer);
278
+ } finally {
279
+ reader.releaseLock?.();
280
+ }
281
+ return text;
282
+ }
283
+ /** Pull `choices[0].delta.content` (or a non-streaming `message.content`) out of one SSE frame. */
284
+ function contentOfFrame(frame) {
285
+ let out = "";
286
+ for (const rawLine of frame.split(/\r?\n/u)) {
287
+ const line = rawLine.trim();
288
+ if (!line.startsWith("data:")) continue;
289
+ const payload = line.slice(5).trim();
290
+ if (payload === "" || payload === "[DONE]") continue;
291
+ let parsed;
292
+ try {
293
+ parsed = JSON.parse(payload);
294
+ } catch {
295
+ continue;
296
+ }
297
+ if (typeof parsed !== "object" || parsed === null) continue;
298
+ const choices = parsed["choices"];
299
+ if (!Array.isArray(choices) || choices.length === 0) continue;
300
+ const choice = choices[0];
301
+ const delta = choice["delta"];
302
+ if (typeof delta === "object" && delta !== null) {
303
+ const content = delta["content"];
304
+ if (typeof content === "string") out += content;
305
+ }
306
+ const message = choice["message"];
307
+ if (typeof message === "object" && message !== null) {
308
+ const content = message["content"];
309
+ if (typeof content === "string") out += content;
310
+ }
311
+ const data = parsed["data"];
312
+ if (typeof data === "object" && data !== null) {
313
+ const inner = data["content"];
314
+ if (typeof inner === "string") out += inner;
315
+ }
316
+ }
317
+ return out;
318
+ }
319
+ /**
226
320
  * Classify an upstream failure from its HTTP status and body excerpt.
227
321
  * Body markers win over status, because the upstream reuses 400/200 for
228
322
  * several distinct conditions.
229
323
  */
230
324
  function classifyUpstreamError(status, body) {
231
325
  if (status === 402) return "hard_credit";
326
+ if (status === 401 || status === 403) return "session_dead";
232
327
  const lower = body.toLowerCase();
233
328
  for (const marker of HARD_CREDIT_MARKERS) if (lower.includes(marker.toLowerCase()) || body.includes(marker)) return "hard_credit";
234
329
  for (const marker of SESSION_DEAD_MARKERS) if (body.includes(marker)) return "session_dead";
@@ -696,6 +791,51 @@ var WorkBuddyUpstreamClient = class {
696
791
  }
697
792
  return JSON.stringify(obj);
698
793
  }
794
+ /**
795
+ * Parse a raw OpenAI chat body without normalising it.
796
+ *
797
+ * The compactor needs the message array as objects, while `chatStream` only
798
+ * accepts the serialised string form.
799
+ */
800
+ parseChatBody(raw) {
801
+ let body;
802
+ try {
803
+ body = JSON.parse(raw);
804
+ } catch {
805
+ return;
806
+ }
807
+ if (typeof body !== "object" || body === null || Array.isArray(body)) return void 0;
808
+ return body;
809
+ }
810
+ /** Re-serialise `base` with a rewritten `messages` array, still normalised. */
811
+ buildChatBody(base, messages) {
812
+ return this.prepareChatBody(JSON.stringify({
813
+ ...base,
814
+ messages
815
+ }));
816
+ }
817
+ /**
818
+ * Run one NON-streaming completion and return the assistant text.
819
+ *
820
+ * Used only for internal compaction (summarising dropped turns). The chat
821
+ * endpoint itself always streams, so this reassembles the SSE frames into a
822
+ * single string. Throws on any failure: the compactor then falls back to
823
+ * plain truncation rather than failing the user's turn.
824
+ */
825
+ async completeChat(credential, prepared, signal) {
826
+ const response = await this.fetchImpl(`${chatBase(credential)}/v2/chat/completions`, {
827
+ method: "POST",
828
+ headers: chatHeaders(credential),
829
+ body: prepared,
830
+ ...signal === void 0 ? {} : { signal }
831
+ });
832
+ if (!response.ok) {
833
+ const text = (await response.text().catch(() => "")).slice(0, ERROR_BODY_LIMIT);
834
+ throw new Error(`compaction upstream http ${response.status}: ${text}`);
835
+ }
836
+ if (response.body === null) throw new Error("compaction upstream returned no body");
837
+ return await readCompletionText(response.body);
838
+ }
699
839
  /** Forward one chat completion. Never throws for upstream failures. */
700
840
  async chatStream(credential, prepared, signal) {
701
841
  let response;
@@ -3798,6 +3938,286 @@ var WorkBuddyScheduler = class {
3798
3938
  }
3799
3939
  };
3800
3940
  //#endregion
3941
+ //#region src/context-budget.ts
3942
+ /** Rough character-per-token ratio. CJK is ~1 token/char, latin ~1/4. */
3943
+ const CHARS_PER_TOKEN_LATIN = 4;
3944
+ const CHARS_PER_TOKEN_CJK = 1;
3945
+ /** Fixed per-message overhead the chat template adds (role markers etc.). */
3946
+ const PER_MESSAGE_TOKEN_OVERHEAD = 4;
3947
+ /** Every image/tool part costs at least this much once decoded. */
3948
+ const PER_PART_TOKEN_FLOOR = 16;
3949
+ /**
3950
+ * Estimate the token cost of one message's `content`.
3951
+ *
3952
+ * Deliberately conservative (over-estimates) so we compact slightly early
3953
+ * rather than discovering the overrun upstream.
3954
+ */
3955
+ function estimateContentTokens(content) {
3956
+ if (content === null || content === void 0) return 0;
3957
+ if (typeof content === "string") return estimateTextTokens(content);
3958
+ if (typeof content === "number" || typeof content === "boolean") return PER_PART_TOKEN_FLOOR;
3959
+ if (Array.isArray(content)) {
3960
+ let total = 0;
3961
+ for (const part of content) total += estimateContentTokens(part);
3962
+ return total;
3963
+ }
3964
+ if (typeof content === "object") {
3965
+ const record = content;
3966
+ let total = PER_PART_TOKEN_FLOOR;
3967
+ for (const key of [
3968
+ "text",
3969
+ "image_url",
3970
+ "input",
3971
+ "content"
3972
+ ]) if (key in record) total += estimateContentTokens(record[key]);
3973
+ if (total === PER_PART_TOKEN_FLOOR) total += estimateTextTokens(safeStringify(record));
3974
+ return total;
3975
+ }
3976
+ return 0;
3977
+ }
3978
+ /** Estimate tokens for a plain string, accounting for CJK density. */
3979
+ function estimateTextTokens(text) {
3980
+ if (text === "") return 0;
3981
+ let cjk = 0;
3982
+ for (const char of text) if (isCjk(char.codePointAt(0) ?? 0)) cjk += 1;
3983
+ const latin = text.length - cjk;
3984
+ return Math.ceil(cjk / CHARS_PER_TOKEN_CJK + latin / CHARS_PER_TOKEN_LATIN);
3985
+ }
3986
+ function isCjk(code) {
3987
+ return code >= 12288 && code <= 12351 || code >= 12352 && code <= 12543 || code >= 13312 && code <= 19903 || code >= 19968 && code <= 40959 || code >= 63744 && code <= 64255 || code >= 65280 && code <= 65519 || code >= 131072 && code <= 191471;
3988
+ }
3989
+ function safeStringify(value) {
3990
+ try {
3991
+ return JSON.stringify(value) ?? "";
3992
+ } catch {
3993
+ return String(value);
3994
+ }
3995
+ }
3996
+ /** Estimate the prompt cost of a whole message array. */
3997
+ function estimateMessagesTokens(messages) {
3998
+ let total = 0;
3999
+ for (const message of messages) {
4000
+ total += PER_MESSAGE_TOKEN_OVERHEAD;
4001
+ total += estimateContentTokens(message.content);
4002
+ if (message["tool_calls"] !== void 0) total += estimateContentTokens(message["tool_calls"]);
4003
+ if (message["name"] !== void 0) total += estimateTextTokens(String(message["name"]));
4004
+ }
4005
+ return total;
4006
+ }
4007
+ /** True when `role` carries instructions that must survive compaction. */
4008
+ function isPinnedRole(role) {
4009
+ return role === "system" || role === "developer";
4010
+ }
4011
+ /**
4012
+ * Drop the oldest non-pinned messages until the estimate fits `budget`.
4013
+ *
4014
+ * Pinned (system/developer) messages and the newest `keepRecent` messages are
4015
+ * never dropped here — if those alone overrun the budget, the caller must fall
4016
+ * back to summarisation or give up.
4017
+ */
4018
+ function compactMessages(messages, options) {
4019
+ const keepRecent = Math.max(1, options.keepRecent ?? 4);
4020
+ const estimate = estimateMessagesTokens(messages);
4021
+ if (estimate <= options.budget) return {
4022
+ messages: [...messages],
4023
+ dropped: [],
4024
+ tokens: estimate,
4025
+ changed: false
4026
+ };
4027
+ const pinned = [];
4028
+ const body = [];
4029
+ for (const message of messages) if (isPinnedRole(message.role)) pinned.push(message);
4030
+ else body.push(message);
4031
+ const keep = Math.min(keepRecent, body.length);
4032
+ const tail = body.slice(body.length - keep);
4033
+ const head = body.slice(0, body.length - keep);
4034
+ let dropCount = 0;
4035
+ let candidate = [
4036
+ ...pinned,
4037
+ ...head,
4038
+ ...tail
4039
+ ];
4040
+ let total = estimateMessagesTokens(candidate);
4041
+ while (total > options.budget && dropCount < head.length) {
4042
+ dropCount += 1;
4043
+ candidate = [
4044
+ ...pinned,
4045
+ ...head.slice(dropCount),
4046
+ ...tail
4047
+ ];
4048
+ total = estimateMessagesTokens(candidate);
4049
+ }
4050
+ const dropped = head.slice(0, dropCount);
4051
+ return {
4052
+ messages: candidate,
4053
+ dropped,
4054
+ tokens: total,
4055
+ changed: dropCount > 0
4056
+ };
4057
+ }
4058
+ /**
4059
+ * Drop the oldest messages, including pinned ones, as a last resort.
4060
+ *
4061
+ * Used when even a summary cannot bring the prompt under budget (for example a
4062
+ * single enormous pasted document). The newest message always survives.
4063
+ */
4064
+ function hardTruncate(messages, budget) {
4065
+ if (messages.length === 0) return {
4066
+ messages: [],
4067
+ dropped: [],
4068
+ tokens: 0,
4069
+ changed: false
4070
+ };
4071
+ let start = 0;
4072
+ let candidate = [...messages];
4073
+ let total = estimateMessagesTokens(candidate);
4074
+ while (total > budget && start < messages.length - 1) {
4075
+ start += 1;
4076
+ candidate = messages.slice(start);
4077
+ total = estimateMessagesTokens(candidate);
4078
+ }
4079
+ return {
4080
+ messages: candidate,
4081
+ dropped: messages.slice(0, start),
4082
+ tokens: total,
4083
+ changed: start > 0
4084
+ };
4085
+ }
4086
+ /** Instructions handed to the model when we ask it to compact a conversation. */
4087
+ const SUMMARIZE_INSTRUCTION = [
4088
+ "You are compacting an ongoing conversation so it can continue without the original history.",
4089
+ "Summarise the transcript below into a dense briefing for the next assistant turn.",
4090
+ "Preserve, in this order of priority:",
4091
+ "1. explicit user requirements, constraints and corrections;",
4092
+ "2. decisions already made, and the reasoning behind them;",
4093
+ "3. concrete facts: file paths, identifiers, commands, numbers, error messages;",
4094
+ "4. unfinished work and the current blocker.",
4095
+ "Drop pleasantries, repetition and superseded attempts.",
4096
+ "Write the briefing only — no preamble, no markdown fence."
4097
+ ].join("\n");
4098
+ /** Render a message array as plain text for the summarisation prompt. */
4099
+ function transcriptOf(messages) {
4100
+ const lines = [];
4101
+ for (const message of messages) {
4102
+ const role = message.role === "" ? "unknown" : message.role;
4103
+ lines.push(`### ${role}`);
4104
+ lines.push(renderContent(message.content));
4105
+ if (message["tool_calls"] !== void 0) lines.push(renderContent(message["tool_calls"]));
4106
+ }
4107
+ return lines.join("\n");
4108
+ }
4109
+ function renderContent(content) {
4110
+ if (content === null || content === void 0) return "";
4111
+ if (typeof content === "string") return content;
4112
+ if (Array.isArray(content)) return content.map((part) => renderContent(part)).filter((text) => text !== "").join("\n");
4113
+ if (typeof content === "object") {
4114
+ const record = content;
4115
+ for (const key of [
4116
+ "text",
4117
+ "content",
4118
+ "input"
4119
+ ]) if (typeof record[key] === "string") return record[key];
4120
+ if (record["type"] !== void 0 && typeof record["type"] === "string") return `[${record["type"]}]`;
4121
+ return safeStringify(record);
4122
+ }
4123
+ return String(content);
4124
+ }
4125
+ /** Build the synthetic system message that carries a compaction summary. */
4126
+ function summaryMessage(summary) {
4127
+ return {
4128
+ role: "system",
4129
+ content: [
4130
+ "The earlier part of this conversation was compacted to fit the model context window.",
4131
+ "Briefing produced from the dropped turns:",
4132
+ "",
4133
+ summary.trim()
4134
+ ].join("\n")
4135
+ };
4136
+ }
4137
+ /**
4138
+ * Compact `messages` to `budget`, summarising the dropped turns when possible.
4139
+ *
4140
+ * The summary is requested with a *bounded* transcript so the compaction call
4141
+ * itself can never overrun the window: if the dropped turns are huge, only the
4142
+ * newest slice of them is summarised, and the oldest are noted as elided.
4143
+ */
4144
+ async function compactWithSummary(messages, options, deps, signal) {
4145
+ const first = compactMessages(messages, options);
4146
+ if (!first.changed) return {
4147
+ messages: first.messages,
4148
+ tokens: first.tokens,
4149
+ summary: void 0,
4150
+ skipped: void 0
4151
+ };
4152
+ const summaryBudget = Math.max(256, Math.floor(options.budget / 4));
4153
+ let toSummarize = first.dropped;
4154
+ let elided = 0;
4155
+ while (estimateMessagesTokens(toSummarize) > summaryBudget && toSummarize.length > 1) {
4156
+ toSummarize = toSummarize.slice(1);
4157
+ elided += 1;
4158
+ }
4159
+ let summary;
4160
+ let skipped;
4161
+ try {
4162
+ const instruction = elided > 0 ? `${SUMMARIZE_INSTRUCTION}\n\nNote: the ${elided} oldest turn(s) were elided before this transcript.` : SUMMARIZE_INSTRUCTION;
4163
+ const suffix = elided > 0 ? `\n(the ${elided} oldest turn(s) were elided)` : "";
4164
+ const request = [{
4165
+ role: "system",
4166
+ content: instruction
4167
+ }, {
4168
+ role: "user",
4169
+ content: `${transcriptOf(toSummarize)}${suffix}`
4170
+ }];
4171
+ const text = await deps.complete(request, signal);
4172
+ if (text.trim() !== "") summary = text.trim();
4173
+ else skipped = "summariser returned an empty summary";
4174
+ } catch (error) {
4175
+ skipped = `summarisation failed: ${String(error)}`;
4176
+ }
4177
+ if (summary === void 0) return {
4178
+ messages: first.messages,
4179
+ skipped,
4180
+ tokens: first.tokens
4181
+ };
4182
+ const withSummary = injectSummary(first.messages, summary);
4183
+ if (estimateMessagesTokens(withSummary) > options.budget) {
4184
+ const truncated = hardTruncate(withSummary, options.budget);
4185
+ return {
4186
+ messages: truncated.messages,
4187
+ summary,
4188
+ tokens: truncated.tokens
4189
+ };
4190
+ }
4191
+ return {
4192
+ messages: withSummary,
4193
+ summary,
4194
+ tokens: estimateMessagesTokens(withSummary)
4195
+ };
4196
+ }
4197
+ /**
4198
+ * Re-insert a summary as: pinned instructions → summary → surviving tail.
4199
+ *
4200
+ * Order matters. Pinned (system/developer) messages must stay ahead of the
4201
+ * summary so that a later synthetic system message can never override the
4202
+ * harness's own instructions; the tail follows so the newest exchange is the
4203
+ * last thing the model reads.
4204
+ *
4205
+ * `compacted` is always derived from `original` by `compactMessages`, so the
4206
+ * pinned messages it carries are exactly the originals — no need to re-add
4207
+ * them from `original`.
4208
+ */
4209
+ function injectSummary(compacted, summary) {
4210
+ const pinned = [];
4211
+ const rest = [];
4212
+ for (const message of compacted) if (isPinnedRole(message.role)) pinned.push(message);
4213
+ else rest.push(message);
4214
+ return [
4215
+ ...pinned,
4216
+ summaryMessage(summary),
4217
+ ...rest
4218
+ ];
4219
+ }
4220
+ //#endregion
3801
4221
  //#region src/shim.ts
3802
4222
  /**
3803
4223
  * Loopback OpenAI-compatible endpoint with multi-account failover.
@@ -3876,7 +4296,8 @@ function writeOpenAIError(res, status, kind, message) {
3876
4296
  }
3877
4297
  /** True when an upstream failure body means the request overran the model's
3878
4298
  * context window (OpenAI `context_length_exceeded`, WorkBuddy code 11115 /
3879
- * "input length too long"). Surfaced as a friendly hint, never auto-truncated. */
4299
+ * "input length too long"). The shim answers it by compacting the conversation
4300
+ * in place and retrying once; see `recoverFromContextOverrun`. */
3880
4301
  function isContextTooLong(body) {
3881
4302
  if (body.includes("context_length_exceeded")) return true;
3882
4303
  if (body.includes("input length too long")) return true;
@@ -4038,25 +4459,7 @@ function createWorkBuddyShim(options) {
4038
4459
  tried.push(account.label);
4039
4460
  const result = await client.chatStream(account.credential, prepared, controller.signal);
4040
4461
  if (result.ok) {
4041
- logger?.info?.(`dsh-workbuddy-xdpool: served by ${account.label}`);
4042
- pool.noteServed(account.id);
4043
- refreshBalance(account);
4044
- res.writeHead(200, {
4045
- "Content-Type": "text/event-stream",
4046
- "Cache-Control": "no-cache",
4047
- "Connection": "keep-alive",
4048
- "X-Accel-Buffering": "no"
4049
- });
4050
- let sawDone = false;
4051
- const body = Readable.fromWeb(result.response.body);
4052
- body.on("data", (chunk) => {
4053
- if (chunk.includes("[DONE]")) sawDone = true;
4054
- });
4055
- body.on("error", (error) => {
4056
- logger?.warn("dsh-workbuddy-xdpool: upstream stream failed mid-flight", error);
4057
- if (!sawDone && res.writable) res.end("data: [DONE]\n\n");
4058
- });
4059
- body.pipe(res);
4462
+ await serveSuccessfulStream(res, account, result, logger, refreshBalance, pool);
4060
4463
  return;
4061
4464
  }
4062
4465
  last = {
@@ -4084,7 +4487,21 @@ function createWorkBuddyShim(options) {
4084
4487
  return;
4085
4488
  }
4086
4489
  if (isContextTooLong(last.message)) {
4087
- writeOpenAIError(res, 400, "context_length_exceeded", `${modelId === void 0 ? "the conversation exceeds this model's context window" : `the conversation exceeds ${modelId}'s context window`}. Shorten the conversation, start a new chat, or pick a model with a larger window (e.g. hy4-preview).`);
4490
+ const recovered = await recoverFromContextOverrun({
4491
+ raw,
4492
+ modelId,
4493
+ controller,
4494
+ region,
4495
+ logger,
4496
+ client,
4497
+ pool,
4498
+ maxAttempts
4499
+ });
4500
+ if (recovered.ok) {
4501
+ await serveSuccessfulStream(res, recovered.account, recovered.result, logger, refreshBalance, pool);
4502
+ return;
4503
+ }
4504
+ writeOpenAIError(res, 400, "context_length_exceeded", contextOverflowMessage(modelId, recovered.detail));
4088
4505
  return;
4089
4506
  }
4090
4507
  writeOpenAIError(res, KIND_STATUS[last.kind], last.kind, `workbuddy upstream ${last.kind} (http ${last.status}) after ${tried.length} account(s) [${tried.join(" → ")}]: ${last.message.slice(0, 400)}`);
@@ -4100,6 +4517,168 @@ function createWorkBuddyShim(options) {
4100
4517
  })
4101
4518
  };
4102
4519
  }
4520
+ /**
4521
+ * Build the overflow message the Harness must recognize.
4522
+ *
4523
+ * This is deliberately NOT free-form prose. `dsh-compaction-basic` decides
4524
+ * whether to compact-and-retry by running the text that reaches it through
4525
+ * `isContextWindowExceededError()` (`@deepseek-ai/dsh-llm`), whose matcher
4526
+ * accepts only specific phrasings:
4527
+ *
4528
+ * - `context_length_exceeded` / `context window exceeded`
4529
+ * - `maximum context length`
4530
+ * - `<input|prompt|request|messages> too large|long for ... context`
4531
+ * - `<input|prompt|request> exceeds the ... context window`
4532
+ *
4533
+ * The obvious friendly sentence ("the conversation exceeds this model's
4534
+ * context window") matches NONE of them, and neither does the WorkBuddy
4535
+ * upstream's own "input length too long" / code 11115. Emitting either meant
4536
+ * the Harness saw an unclassifiable 400, skipped its recovery path, and
4537
+ * surfaced a dead turn — the bug this function exists to prevent.
4538
+ *
4539
+ * The leading clause carries the machine-matched wording; the trailing clause
4540
+ * is what a human reads. Keep both in sync with
4541
+ * `tests/context-overflow-contract.test.ts`.
4542
+ */
4543
+ function contextOverflowMessage(modelId, detail = "") {
4544
+ return `This model's maximum context length was exceeded: the prompt is too large for ${modelId === void 0 ? "the model" : `model ${modelId}`}, and the conversation could not be compacted in place${detail === "" ? "" : ` (${detail})`}. Compact the conversation, or start a new chat.`;
4545
+ }
4546
+ /**
4547
+ * Serve one already-successful upstream stream as an SSE response.
4548
+ *
4549
+ * Extracted so the context-overrun recovery path reuses the exact same
4550
+ * bookkeeping (noteServed + background balance refresh) as a first-try hit.
4551
+ */
4552
+ async function serveSuccessfulStream(res, account, result, logger, refreshBalance, pool) {
4553
+ logger?.info?.(`dsh-workbuddy-xdpool: served by ${account.label}`);
4554
+ pool.noteServed(account.id);
4555
+ refreshBalance(account);
4556
+ res.writeHead(200, {
4557
+ "Content-Type": "text/event-stream",
4558
+ "Cache-Control": "no-cache",
4559
+ "Connection": "keep-alive",
4560
+ "X-Accel-Buffering": "no"
4561
+ });
4562
+ let sawDone = false;
4563
+ const body = Readable.fromWeb(result.response.body);
4564
+ body.on("data", (chunk) => {
4565
+ if (chunk.includes("[DONE]")) sawDone = true;
4566
+ });
4567
+ body.on("error", (error) => {
4568
+ logger?.warn("dsh-workbuddy-xdpool: upstream stream failed mid-flight", error);
4569
+ if (!sawDone && res.writable) res.end("data: [DONE]\n\n");
4570
+ });
4571
+ body.pipe(res);
4572
+ }
4573
+ /**
4574
+ * Compact an over-long conversation and retry it once.
4575
+ *
4576
+ * Strategy, in order:
4577
+ * 1. drop the oldest turns, keeping system messages and the newest exchange;
4578
+ * 2. ask the model to summarise the dropped turns and splice that summary in;
4579
+ * 3. hard-truncate as a last resort.
4580
+ *
4581
+ * Returns `ok: false` only when even a truncated prompt still overran — the
4582
+ * caller then surfaces the original actionable 400.
4583
+ */
4584
+ async function recoverFromContextOverrun(options) {
4585
+ const { raw, modelId, controller, region, logger, client, pool, maxAttempts } = options;
4586
+ const parsed = client.parseChatBody(raw);
4587
+ if (parsed === void 0) return {
4588
+ ok: false,
4589
+ detail: "request body was not parseable JSON"
4590
+ };
4591
+ const rawMessages = parsed["messages"];
4592
+ if (!Array.isArray(rawMessages)) return {
4593
+ ok: false,
4594
+ detail: "request carried no messages array"
4595
+ };
4596
+ const messages = rawMessages.filter((value) => typeof value === "object" && value !== null && !Array.isArray(value));
4597
+ if (messages.length === 0) return {
4598
+ ok: false,
4599
+ detail: "request carried no usable messages"
4600
+ };
4601
+ const overrunTokens = estimateMessagesTokens(messages);
4602
+ const budget = Math.max(512, Math.floor(overrunTokens / 2));
4603
+ logger?.warn(`dsh-workbuddy-xdpool: context overrun on ${modelId ?? "(no model)"} (~${overrunTokens} tokens); compacting to ~${budget} and retrying once`);
4604
+ let summary;
4605
+ let compacted = messages;
4606
+ let compactionDetail = "";
4607
+ try {
4608
+ const summariser = await pool.acquire(modelId, region);
4609
+ if (summariser === void 0) compactionDetail = "no account available to summarise with";
4610
+ else {
4611
+ const outcome = await compactWithSummary(messages, {
4612
+ budget,
4613
+ keepRecent: 6
4614
+ }, { complete: async (request, signal) => {
4615
+ const body = client.buildChatBody({
4616
+ ...parsed,
4617
+ stream: true,
4618
+ max_tokens: Math.max(256, Math.floor(budget / 2))
4619
+ }, request);
4620
+ return await client.completeChat(summariser.credential, body, signal ?? controller.signal);
4621
+ } }, controller.signal);
4622
+ compacted = outcome.messages;
4623
+ summary = outcome.summary;
4624
+ if (outcome.skipped !== void 0) compactionDetail = outcome.skipped;
4625
+ }
4626
+ } catch (error) {
4627
+ compactionDetail = `summarisation failed: ${String(error)}`;
4628
+ }
4629
+ if (estimateMessagesTokens(compacted) > budget) compacted = hardTruncate(compacted, budget).messages;
4630
+ if (summary === void 0 && estimateMessagesTokens(compacted) >= overrunTokens) return {
4631
+ ok: false,
4632
+ detail: compactionDetail === "" ? "compaction could not reduce the prompt" : compactionDetail
4633
+ };
4634
+ const retryBody = client.buildChatBody(parsed, compacted);
4635
+ const tried = [];
4636
+ for (let attempt = 0; attempt < maxAttempts; attempt += 1) {
4637
+ if (controller.signal.aborted) return {
4638
+ ok: false,
4639
+ detail: "client disconnected"
4640
+ };
4641
+ const account = await pool.acquire(modelId, region);
4642
+ if (account === void 0) return {
4643
+ ok: false,
4644
+ detail: "no account available after compaction"
4645
+ };
4646
+ tried.push(account.label);
4647
+ const result = await client.chatStream(account.credential, retryBody, controller.signal);
4648
+ if (result.ok) {
4649
+ logger?.info?.(`dsh-workbuddy-xdpool: recovered from context overrun on ${modelId ?? "(no model)"} (summarised: ${summary === void 0 ? "no" : "yes"})`);
4650
+ return {
4651
+ ok: true,
4652
+ account,
4653
+ result
4654
+ };
4655
+ }
4656
+ if (isContextTooLong(result.message)) return {
4657
+ ok: false,
4658
+ detail: "prompt still exceeded the window after compaction"
4659
+ };
4660
+ if (result.kind === "session_dead") {
4661
+ await pool.refreshAccount(account.id);
4662
+ continue;
4663
+ }
4664
+ if (result.kind === "hard_credit") {
4665
+ pool.penalizeExhausted(account.id);
4666
+ continue;
4667
+ }
4668
+ if (result.kind === "soft_rate") {
4669
+ pool.penalize(account.id, parseRateLimitReset(result.message), modelId);
4670
+ continue;
4671
+ }
4672
+ return {
4673
+ ok: false,
4674
+ detail: `upstream ${result.kind} after compaction`
4675
+ };
4676
+ }
4677
+ return {
4678
+ ok: false,
4679
+ detail: `no account served the compacted request (tried ${tried.length})`
4680
+ };
4681
+ }
4103
4682
  //#endregion
4104
4683
  //#region src/status.ts
4105
4684
  /** Build the status document. Never throws. */
package/package.json CHANGED
@@ -2,7 +2,7 @@
2
2
  "name": "dsh-workbuddy-xdpool",
3
3
  "displayName": "DSH WorkBuddy XD Pool",
4
4
  "description": "Merge every locally signed-in WorkBuddy account into DeepSeek Harness as one auto-failing-over model pool (multi-account rotation, live credits, daily check-in and model catalog).",
5
- "version": "1.3.0",
5
+ "version": "1.4.0",
6
6
  "license": "MIT",
7
7
  "author": "XDTrees",
8
8
  "homepage": "https://github.com/XDTrees/dsh-workbuddy-xdpool#readme",