dsh-workbuddy-xdpool 1.2.0 → 1.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/lib/index.js CHANGED
@@ -31,6 +31,8 @@ const MODELS_CATALOG_PATH = "/v2/enterprises/personal/models";
31
31
  const GLOBAL_CONFIG_PATH = "/v3/config";
32
32
  const JSON_TIMEOUT_MS = 3e4;
33
33
  const ERROR_BODY_LIMIT = 4096;
34
+ /** Cap on the reassembled compaction reply, guarding against a runaway stream. */
35
+ const COMPLETION_TEXT_LIMIT = 65536;
34
36
  /** Insufficient-credit markers, ASCII lowercase plus the original Chinese. */
35
37
  const HARD_CREDIT_MARKERS = [
36
38
  "insufficient credit",
@@ -49,8 +51,31 @@ const HARD_CREDIT_MARKERS = [
49
51
  "额度用尽",
50
52
  "没有积分"
51
53
  ];
52
- /** Session-invalidation markers that mean "sign in again in the WorkBuddy app". */
53
- const SESSION_DEAD_MARKERS = ["Offline user session not found", "12153"];
54
+ /** Session-invalidation markers that mean "this credential is dead; use another".
55
+ * Kept alongside the HTTP-status rule in `classifyUpstreamError`: the status is
56
+ * enough for a direct 401/403, but some failures arrive wrapped in a 200
57
+ * envelope or a 4xx the gateway words differently. Adding the English and
58
+ * Chinese phrasings the upstream actually uses keeps those recoverable too —
59
+ * an unmatched one fell through to `client`, which is terminal in the shim
60
+ * and pinned the pool to the first account (the "API 密钥无效" bug). */
61
+ const SESSION_DEAD_MARKERS = [
62
+ "Offline user session not found",
63
+ "12153",
64
+ "api key is invalid",
65
+ "invalid api key",
66
+ "invalid_api_key",
67
+ "api密钥无效",
68
+ "密钥无效",
69
+ "无效的密钥",
70
+ "unauthorized",
71
+ "token expired",
72
+ "token is invalid",
73
+ "login expired",
74
+ "please login",
75
+ "未登录",
76
+ "登录已失效",
77
+ "重新登录"
78
+ ];
54
79
  /**
55
80
  * Markers for "already checked in today".
56
81
  *
@@ -223,12 +248,82 @@ function envelopeError(status, envelope) {
223
248
  return /* @__PURE__ */ new Error(`workbuddy upstream ${kind} (http ${status}): ${envelope.msg.slice(0, 160)}`);
224
249
  }
225
250
  /**
251
+ * Read an OpenAI-style SSE chat stream and concatenate the assistant text.
252
+ *
253
+ * The upstream always streams (`stream: true` is forced on every chat body),
254
+ * so a non-streaming internal call has to reassemble the deltas itself. Only
255
+ * `choices[0].delta.content` is collected; reasoning deltas are dropped
256
+ * because a compaction summary needs the final answer, not the scratchpad.
257
+ */
258
+ async function readCompletionText(body) {
259
+ const decoder = new TextDecoder();
260
+ const reader = body.getReader();
261
+ let buffer = "";
262
+ let text = "";
263
+ try {
264
+ for (;;) {
265
+ const { done, value } = await reader.read();
266
+ if (done) break;
267
+ buffer += decoder.decode(value, { stream: true });
268
+ let split = buffer.indexOf("\n\n");
269
+ while (split !== -1) {
270
+ const frame = buffer.slice(0, split);
271
+ buffer = buffer.slice(split + 2);
272
+ text += contentOfFrame(frame);
273
+ if (text.length > COMPLETION_TEXT_LIMIT) return text.slice(0, COMPLETION_TEXT_LIMIT);
274
+ split = buffer.indexOf("\n\n");
275
+ }
276
+ }
277
+ if (buffer.trim() !== "") text += contentOfFrame(buffer);
278
+ } finally {
279
+ reader.releaseLock?.();
280
+ }
281
+ return text;
282
+ }
283
+ /** Pull `choices[0].delta.content` (or a non-streaming `message.content`) out of one SSE frame. */
284
+ function contentOfFrame(frame) {
285
+ let out = "";
286
+ for (const rawLine of frame.split(/\r?\n/u)) {
287
+ const line = rawLine.trim();
288
+ if (!line.startsWith("data:")) continue;
289
+ const payload = line.slice(5).trim();
290
+ if (payload === "" || payload === "[DONE]") continue;
291
+ let parsed;
292
+ try {
293
+ parsed = JSON.parse(payload);
294
+ } catch {
295
+ continue;
296
+ }
297
+ if (typeof parsed !== "object" || parsed === null) continue;
298
+ const choices = parsed["choices"];
299
+ if (!Array.isArray(choices) || choices.length === 0) continue;
300
+ const choice = choices[0];
301
+ const delta = choice["delta"];
302
+ if (typeof delta === "object" && delta !== null) {
303
+ const content = delta["content"];
304
+ if (typeof content === "string") out += content;
305
+ }
306
+ const message = choice["message"];
307
+ if (typeof message === "object" && message !== null) {
308
+ const content = message["content"];
309
+ if (typeof content === "string") out += content;
310
+ }
311
+ const data = parsed["data"];
312
+ if (typeof data === "object" && data !== null) {
313
+ const inner = data["content"];
314
+ if (typeof inner === "string") out += inner;
315
+ }
316
+ }
317
+ return out;
318
+ }
319
+ /**
226
320
  * Classify an upstream failure from its HTTP status and body excerpt.
227
321
  * Body markers win over status, because the upstream reuses 400/200 for
228
322
  * several distinct conditions.
229
323
  */
230
324
  function classifyUpstreamError(status, body) {
231
325
  if (status === 402) return "hard_credit";
326
+ if (status === 401 || status === 403) return "session_dead";
232
327
  const lower = body.toLowerCase();
233
328
  for (const marker of HARD_CREDIT_MARKERS) if (lower.includes(marker.toLowerCase()) || body.includes(marker)) return "hard_credit";
234
329
  for (const marker of SESSION_DEAD_MARKERS) if (body.includes(marker)) return "session_dead";
@@ -696,6 +791,51 @@ var WorkBuddyUpstreamClient = class {
696
791
  }
697
792
  return JSON.stringify(obj);
698
793
  }
794
+ /**
795
+ * Parse a raw OpenAI chat body without normalising it.
796
+ *
797
+ * The compactor needs the message array as objects, while `chatStream` only
798
+ * accepts the serialised string form.
799
+ */
800
+ parseChatBody(raw) {
801
+ let body;
802
+ try {
803
+ body = JSON.parse(raw);
804
+ } catch {
805
+ return;
806
+ }
807
+ if (typeof body !== "object" || body === null || Array.isArray(body)) return void 0;
808
+ return body;
809
+ }
810
+ /** Re-serialise `base` with a rewritten `messages` array, still normalised. */
811
+ buildChatBody(base, messages) {
812
+ return this.prepareChatBody(JSON.stringify({
813
+ ...base,
814
+ messages
815
+ }));
816
+ }
817
+ /**
818
+ * Run one NON-streaming completion and return the assistant text.
819
+ *
820
+ * Used only for internal compaction (summarising dropped turns). The chat
821
+ * endpoint itself always streams, so this reassembles the SSE frames into a
822
+ * single string. Throws on any failure: the compactor then falls back to
823
+ * plain truncation rather than failing the user's turn.
824
+ */
825
+ async completeChat(credential, prepared, signal) {
826
+ const response = await this.fetchImpl(`${chatBase(credential)}/v2/chat/completions`, {
827
+ method: "POST",
828
+ headers: chatHeaders(credential),
829
+ body: prepared,
830
+ ...signal === void 0 ? {} : { signal }
831
+ });
832
+ if (!response.ok) {
833
+ const text = (await response.text().catch(() => "")).slice(0, ERROR_BODY_LIMIT);
834
+ throw new Error(`compaction upstream http ${response.status}: ${text}`);
835
+ }
836
+ if (response.body === null) throw new Error("compaction upstream returned no body");
837
+ return await readCompletionText(response.body);
838
+ }
699
839
  /** Forward one chat completion. Never throws for upstream failures. */
700
840
  async chatStream(credential, prepared, signal) {
701
841
  let response;
@@ -3031,6 +3171,16 @@ function dayKey(date) {
3031
3171
  return `${date.getFullYear()}-${month}-${day}`;
3032
3172
  }
3033
3173
  /**
3174
+ * The scheduled SLOT `date` falls in, as `YYYY-MM-DDTHH`.
3175
+ *
3176
+ * The per-job run guard keys on this instead of the date, so a job configured
3177
+ * for several hours runs in each of them while a second tick inside the same
3178
+ * hour is still refused.
3179
+ */
3180
+ function slotKey(date) {
3181
+ return `${dayKey(date)}T${String(date.getHours()).padStart(2, "0")}`;
3182
+ }
3183
+ /**
3034
3184
  * Whether `now`'s local hour is one of `hours`.
3035
3185
  *
3036
3186
  * The reference panel computes a `nextFire` instant and sleeps until it; this
@@ -3135,7 +3285,7 @@ var WorkBuddyScheduler = class {
3135
3285
  this.reportHours = options.reportHours ?? [10];
3136
3286
  this.taskHours = options.taskHours ?? [11];
3137
3287
  this.streakHours = options.streakHours ?? [12];
3138
- this.travelHours = options.travelHours ?? [12];
3288
+ this.travelHours = options.travelHours ?? [9, 21];
3139
3289
  }
3140
3290
  /** Apply a new configuration; safe to call while running. */
3141
3291
  /**
@@ -3213,7 +3363,7 @@ var WorkBuddyScheduler = class {
3213
3363
  async runNow(kind, force = false) {
3214
3364
  const today = dayKey(this.now());
3215
3365
  const state = this.states[kind];
3216
- if (!force && state.lastRunDate === today) return { ...state };
3366
+ if (!force && state.lastRunSlot === slotKey(this.now())) return { ...state };
3217
3367
  this.busy = true;
3218
3368
  try {
3219
3369
  await this.runJob(kind, today);
@@ -3308,10 +3458,11 @@ var WorkBuddyScheduler = class {
3308
3458
  this.busy = true;
3309
3459
  try {
3310
3460
  const now = this.now();
3461
+ const slot = slotKey(now);
3311
3462
  const today = dayKey(now);
3312
3463
  for (const kind of JOB_KINDS) {
3313
3464
  if (this.stopped) return;
3314
- if (this.states[kind].lastRunDate === today) continue;
3465
+ if (this.states[kind].lastRunSlot === slot) continue;
3315
3466
  if (!isFireHour(now, this.hoursOf(kind))) continue;
3316
3467
  await this.runJob(kind, today);
3317
3468
  }
@@ -3409,6 +3560,8 @@ var WorkBuddyScheduler = class {
3409
3560
  let credit = 0;
3410
3561
  let energy = 0;
3411
3562
  let claimed = 0;
3563
+ const detail = [];
3564
+ let progressNote;
3412
3565
  const accounts = this.accountsInOrder();
3413
3566
  if (kind === "tasks") await this.sendEventChains(accounts);
3414
3567
  for (let index = 0; index < accounts.length; index++) {
@@ -3421,7 +3574,10 @@ var WorkBuddyScheduler = class {
3421
3574
  try {
3422
3575
  const claim = await this.client.claimDailyCheckin(account.credential);
3423
3576
  this.recordEarnings(account.id, today, { checkinCredit: claim.credit });
3424
- if (claim.credit > 0) this.logger.info?.(`automation checkin ${account.label}: +${claim.credit} credit`);
3577
+ if (claim.credit > 0) {
3578
+ if (claim.credit > 0) detail.push(`签到 +${claim.credit}`);
3579
+ this.logger.info?.(`automation checkin ${account.label}: +${claim.credit} credit`);
3580
+ }
3425
3581
  } catch (error) {
3426
3582
  if (!isAlreadyCheckin(error)) throw error;
3427
3583
  this.logger.info?.(`automation checkin ${account.label}: already done today`);
@@ -3437,6 +3593,7 @@ var WorkBuddyScheduler = class {
3437
3593
  credit += result.credit;
3438
3594
  energy += result.energy;
3439
3595
  claimed += result.claimed;
3596
+ detail.push(...result.titles);
3440
3597
  this.claimableSeen += result.claimableCount;
3441
3598
  this.recordEarnings(account.id, today, {
3442
3599
  credit: result.credit,
@@ -3446,9 +3603,11 @@ var WorkBuddyScheduler = class {
3446
3603
  this.logger.info?.(`automation tasks ${account.label}: ${result.claimed} claimed (+${result.credit} credit, +${result.energy} energy, ${result.claimableCount} claimable seen)`);
3447
3604
  break;
3448
3605
  }
3449
- case "streak":
3450
- await this.redeemStreak(account);
3606
+ case "streak": {
3607
+ const progress = await this.redeemStreak(account);
3608
+ if (progress !== void 0) progressNote = progress;
3451
3609
  break;
3610
+ }
3452
3611
  case "travel": await this.runTravel(account);
3453
3612
  }
3454
3613
  ok++;
@@ -3459,6 +3618,7 @@ var WorkBuddyScheduler = class {
3459
3618
  if (index < accounts.length - 1 && this.delayMs > 0 && !this.stopped) await sleep(this.delayMs);
3460
3619
  }
3461
3620
  const state = this.states[kind];
3621
+ state.lastRunSlot = slotKey(this.now());
3462
3622
  state.lastRunDate = today;
3463
3623
  state.lastRunAtMs = this.now().getTime();
3464
3624
  state.ok = ok;
@@ -3467,6 +3627,8 @@ var WorkBuddyScheduler = class {
3467
3627
  state.energy = energy;
3468
3628
  state.claimed = claimed;
3469
3629
  state.message = this.summarise(kind, ok, failed, claimed, credit, energy);
3630
+ state.detail = detail;
3631
+ state.progress = progressNote;
3470
3632
  this.logger.info?.(`automation ${accountWord}: ${state.message}`);
3471
3633
  }
3472
3634
  /** Compose the one-line summary shown on the card. */
@@ -3669,6 +3831,7 @@ var WorkBuddyScheduler = class {
3669
3831
  let credit = 0;
3670
3832
  let energy = 0;
3671
3833
  let claimed = 0;
3834
+ const titles = [];
3672
3835
  for (const task of claimable) {
3673
3836
  if (this.stopped) break;
3674
3837
  const reward = await this.client.claimTaskReward(credential, task.taskCode);
@@ -3676,6 +3839,7 @@ var WorkBuddyScheduler = class {
3676
3839
  energy += reward.energy;
3677
3840
  if (reward.credit > 0 || reward.energy > 0) {
3678
3841
  claimed++;
3842
+ titles.push(task.title);
3679
3843
  this.logger.info?.(`automation claim ${account.label}: ${task.title} +${reward.credit}c +${reward.energy}e`);
3680
3844
  } else this.logger.info?.(`automation claim ${account.label}: ${task.taskCode} already claimed`);
3681
3845
  if (this.delayMs > 0 && !this.stopped) await sleep(this.delayMs);
@@ -3684,7 +3848,8 @@ var WorkBuddyScheduler = class {
3684
3848
  claimed,
3685
3849
  credit,
3686
3850
  energy,
3687
- claimableCount: claimable.length
3851
+ claimableCount: claimable.length,
3852
+ titles
3688
3853
  };
3689
3854
  }
3690
3855
  /**
@@ -3724,6 +3889,11 @@ var WorkBuddyScheduler = class {
3724
3889
  }
3725
3890
  if (this.delayMs > 0 && !this.stopped) await sleep(this.delayMs);
3726
3891
  }
3892
+ const pendingTier = status.tiers.find((tier) => tier.status === "locked");
3893
+ if (pendingTier !== void 0) {
3894
+ const remaining = Math.max(0, pendingTier.days - status.days);
3895
+ return remaining > 0 ? `${pendingTier.tier} in ${remaining}d` : `${pendingTier.tier} ready`;
3896
+ }
3727
3897
  }
3728
3898
  /**
3729
3899
  * One trip through the buddy travel loop for an account.
@@ -3768,6 +3938,286 @@ var WorkBuddyScheduler = class {
3768
3938
  }
3769
3939
  };
3770
3940
  //#endregion
3941
+ //#region src/context-budget.ts
3942
+ /** Rough character-per-token ratio. CJK is ~1 token/char, latin ~1/4. */
3943
+ const CHARS_PER_TOKEN_LATIN = 4;
3944
+ const CHARS_PER_TOKEN_CJK = 1;
3945
+ /** Fixed per-message overhead the chat template adds (role markers etc.). */
3946
+ const PER_MESSAGE_TOKEN_OVERHEAD = 4;
3947
+ /** Every image/tool part costs at least this much once decoded. */
3948
+ const PER_PART_TOKEN_FLOOR = 16;
3949
+ /**
3950
+ * Estimate the token cost of one message's `content`.
3951
+ *
3952
+ * Deliberately conservative (over-estimates) so we compact slightly early
3953
+ * rather than discovering the overrun upstream.
3954
+ */
3955
+ function estimateContentTokens(content) {
3956
+ if (content === null || content === void 0) return 0;
3957
+ if (typeof content === "string") return estimateTextTokens(content);
3958
+ if (typeof content === "number" || typeof content === "boolean") return PER_PART_TOKEN_FLOOR;
3959
+ if (Array.isArray(content)) {
3960
+ let total = 0;
3961
+ for (const part of content) total += estimateContentTokens(part);
3962
+ return total;
3963
+ }
3964
+ if (typeof content === "object") {
3965
+ const record = content;
3966
+ let total = PER_PART_TOKEN_FLOOR;
3967
+ for (const key of [
3968
+ "text",
3969
+ "image_url",
3970
+ "input",
3971
+ "content"
3972
+ ]) if (key in record) total += estimateContentTokens(record[key]);
3973
+ if (total === PER_PART_TOKEN_FLOOR) total += estimateTextTokens(safeStringify(record));
3974
+ return total;
3975
+ }
3976
+ return 0;
3977
+ }
3978
+ /** Estimate tokens for a plain string, accounting for CJK density. */
3979
+ function estimateTextTokens(text) {
3980
+ if (text === "") return 0;
3981
+ let cjk = 0;
3982
+ for (const char of text) if (isCjk(char.codePointAt(0) ?? 0)) cjk += 1;
3983
+ const latin = text.length - cjk;
3984
+ return Math.ceil(cjk / CHARS_PER_TOKEN_CJK + latin / CHARS_PER_TOKEN_LATIN);
3985
+ }
3986
+ function isCjk(code) {
3987
+ return code >= 12288 && code <= 12351 || code >= 12352 && code <= 12543 || code >= 13312 && code <= 19903 || code >= 19968 && code <= 40959 || code >= 63744 && code <= 64255 || code >= 65280 && code <= 65519 || code >= 131072 && code <= 191471;
3988
+ }
3989
+ function safeStringify(value) {
3990
+ try {
3991
+ return JSON.stringify(value) ?? "";
3992
+ } catch {
3993
+ return String(value);
3994
+ }
3995
+ }
3996
+ /** Estimate the prompt cost of a whole message array. */
3997
+ function estimateMessagesTokens(messages) {
3998
+ let total = 0;
3999
+ for (const message of messages) {
4000
+ total += PER_MESSAGE_TOKEN_OVERHEAD;
4001
+ total += estimateContentTokens(message.content);
4002
+ if (message["tool_calls"] !== void 0) total += estimateContentTokens(message["tool_calls"]);
4003
+ if (message["name"] !== void 0) total += estimateTextTokens(String(message["name"]));
4004
+ }
4005
+ return total;
4006
+ }
4007
+ /** True when `role` carries instructions that must survive compaction. */
4008
+ function isPinnedRole(role) {
4009
+ return role === "system" || role === "developer";
4010
+ }
4011
+ /**
4012
+ * Drop the oldest non-pinned messages until the estimate fits `budget`.
4013
+ *
4014
+ * Pinned (system/developer) messages and the newest `keepRecent` messages are
4015
+ * never dropped here — if those alone overrun the budget, the caller must fall
4016
+ * back to summarisation or give up.
4017
+ */
4018
+ function compactMessages(messages, options) {
4019
+ const keepRecent = Math.max(1, options.keepRecent ?? 4);
4020
+ const estimate = estimateMessagesTokens(messages);
4021
+ if (estimate <= options.budget) return {
4022
+ messages: [...messages],
4023
+ dropped: [],
4024
+ tokens: estimate,
4025
+ changed: false
4026
+ };
4027
+ const pinned = [];
4028
+ const body = [];
4029
+ for (const message of messages) if (isPinnedRole(message.role)) pinned.push(message);
4030
+ else body.push(message);
4031
+ const keep = Math.min(keepRecent, body.length);
4032
+ const tail = body.slice(body.length - keep);
4033
+ const head = body.slice(0, body.length - keep);
4034
+ let dropCount = 0;
4035
+ let candidate = [
4036
+ ...pinned,
4037
+ ...head,
4038
+ ...tail
4039
+ ];
4040
+ let total = estimateMessagesTokens(candidate);
4041
+ while (total > options.budget && dropCount < head.length) {
4042
+ dropCount += 1;
4043
+ candidate = [
4044
+ ...pinned,
4045
+ ...head.slice(dropCount),
4046
+ ...tail
4047
+ ];
4048
+ total = estimateMessagesTokens(candidate);
4049
+ }
4050
+ const dropped = head.slice(0, dropCount);
4051
+ return {
4052
+ messages: candidate,
4053
+ dropped,
4054
+ tokens: total,
4055
+ changed: dropCount > 0
4056
+ };
4057
+ }
4058
+ /**
4059
+ * Drop the oldest messages, including pinned ones, as a last resort.
4060
+ *
4061
+ * Used when even a summary cannot bring the prompt under budget (for example a
4062
+ * single enormous pasted document). The newest message always survives.
4063
+ */
4064
+ function hardTruncate(messages, budget) {
4065
+ if (messages.length === 0) return {
4066
+ messages: [],
4067
+ dropped: [],
4068
+ tokens: 0,
4069
+ changed: false
4070
+ };
4071
+ let start = 0;
4072
+ let candidate = [...messages];
4073
+ let total = estimateMessagesTokens(candidate);
4074
+ while (total > budget && start < messages.length - 1) {
4075
+ start += 1;
4076
+ candidate = messages.slice(start);
4077
+ total = estimateMessagesTokens(candidate);
4078
+ }
4079
+ return {
4080
+ messages: candidate,
4081
+ dropped: messages.slice(0, start),
4082
+ tokens: total,
4083
+ changed: start > 0
4084
+ };
4085
+ }
4086
+ /** Instructions handed to the model when we ask it to compact a conversation. */
4087
+ const SUMMARIZE_INSTRUCTION = [
4088
+ "You are compacting an ongoing conversation so it can continue without the original history.",
4089
+ "Summarise the transcript below into a dense briefing for the next assistant turn.",
4090
+ "Preserve, in this order of priority:",
4091
+ "1. explicit user requirements, constraints and corrections;",
4092
+ "2. decisions already made, and the reasoning behind them;",
4093
+ "3. concrete facts: file paths, identifiers, commands, numbers, error messages;",
4094
+ "4. unfinished work and the current blocker.",
4095
+ "Drop pleasantries, repetition and superseded attempts.",
4096
+ "Write the briefing only — no preamble, no markdown fence."
4097
+ ].join("\n");
4098
+ /** Render a message array as plain text for the summarisation prompt. */
4099
+ function transcriptOf(messages) {
4100
+ const lines = [];
4101
+ for (const message of messages) {
4102
+ const role = message.role === "" ? "unknown" : message.role;
4103
+ lines.push(`### ${role}`);
4104
+ lines.push(renderContent(message.content));
4105
+ if (message["tool_calls"] !== void 0) lines.push(renderContent(message["tool_calls"]));
4106
+ }
4107
+ return lines.join("\n");
4108
+ }
4109
+ function renderContent(content) {
4110
+ if (content === null || content === void 0) return "";
4111
+ if (typeof content === "string") return content;
4112
+ if (Array.isArray(content)) return content.map((part) => renderContent(part)).filter((text) => text !== "").join("\n");
4113
+ if (typeof content === "object") {
4114
+ const record = content;
4115
+ for (const key of [
4116
+ "text",
4117
+ "content",
4118
+ "input"
4119
+ ]) if (typeof record[key] === "string") return record[key];
4120
+ if (record["type"] !== void 0 && typeof record["type"] === "string") return `[${record["type"]}]`;
4121
+ return safeStringify(record);
4122
+ }
4123
+ return String(content);
4124
+ }
4125
+ /** Build the synthetic system message that carries a compaction summary. */
4126
+ function summaryMessage(summary) {
4127
+ return {
4128
+ role: "system",
4129
+ content: [
4130
+ "The earlier part of this conversation was compacted to fit the model context window.",
4131
+ "Briefing produced from the dropped turns:",
4132
+ "",
4133
+ summary.trim()
4134
+ ].join("\n")
4135
+ };
4136
+ }
4137
+ /**
4138
+ * Compact `messages` to `budget`, summarising the dropped turns when possible.
4139
+ *
4140
+ * The summary is requested with a *bounded* transcript so the compaction call
4141
+ * itself can never overrun the window: if the dropped turns are huge, only the
4142
+ * newest slice of them is summarised, and the oldest are noted as elided.
4143
+ */
4144
+ async function compactWithSummary(messages, options, deps, signal) {
4145
+ const first = compactMessages(messages, options);
4146
+ if (!first.changed) return {
4147
+ messages: first.messages,
4148
+ tokens: first.tokens,
4149
+ summary: void 0,
4150
+ skipped: void 0
4151
+ };
4152
+ const summaryBudget = Math.max(256, Math.floor(options.budget / 4));
4153
+ let toSummarize = first.dropped;
4154
+ let elided = 0;
4155
+ while (estimateMessagesTokens(toSummarize) > summaryBudget && toSummarize.length > 1) {
4156
+ toSummarize = toSummarize.slice(1);
4157
+ elided += 1;
4158
+ }
4159
+ let summary;
4160
+ let skipped;
4161
+ try {
4162
+ const instruction = elided > 0 ? `${SUMMARIZE_INSTRUCTION}\n\nNote: the ${elided} oldest turn(s) were elided before this transcript.` : SUMMARIZE_INSTRUCTION;
4163
+ const suffix = elided > 0 ? `\n(the ${elided} oldest turn(s) were elided)` : "";
4164
+ const request = [{
4165
+ role: "system",
4166
+ content: instruction
4167
+ }, {
4168
+ role: "user",
4169
+ content: `${transcriptOf(toSummarize)}${suffix}`
4170
+ }];
4171
+ const text = await deps.complete(request, signal);
4172
+ if (text.trim() !== "") summary = text.trim();
4173
+ else skipped = "summariser returned an empty summary";
4174
+ } catch (error) {
4175
+ skipped = `summarisation failed: ${String(error)}`;
4176
+ }
4177
+ if (summary === void 0) return {
4178
+ messages: first.messages,
4179
+ skipped,
4180
+ tokens: first.tokens
4181
+ };
4182
+ const withSummary = injectSummary(first.messages, summary);
4183
+ if (estimateMessagesTokens(withSummary) > options.budget) {
4184
+ const truncated = hardTruncate(withSummary, options.budget);
4185
+ return {
4186
+ messages: truncated.messages,
4187
+ summary,
4188
+ tokens: truncated.tokens
4189
+ };
4190
+ }
4191
+ return {
4192
+ messages: withSummary,
4193
+ summary,
4194
+ tokens: estimateMessagesTokens(withSummary)
4195
+ };
4196
+ }
4197
+ /**
4198
+ * Re-insert a summary as: pinned instructions → summary → surviving tail.
4199
+ *
4200
+ * Order matters. Pinned (system/developer) messages must stay ahead of the
4201
+ * summary so that a later synthetic system message can never override the
4202
+ * harness's own instructions; the tail follows so the newest exchange is the
4203
+ * last thing the model reads.
4204
+ *
4205
+ * `compacted` is always derived from `original` by `compactMessages`, so the
4206
+ * pinned messages it carries are exactly the originals — no need to re-add
4207
+ * them from `original`.
4208
+ */
4209
+ function injectSummary(compacted, summary) {
4210
+ const pinned = [];
4211
+ const rest = [];
4212
+ for (const message of compacted) if (isPinnedRole(message.role)) pinned.push(message);
4213
+ else rest.push(message);
4214
+ return [
4215
+ ...pinned,
4216
+ summaryMessage(summary),
4217
+ ...rest
4218
+ ];
4219
+ }
4220
+ //#endregion
3771
4221
  //#region src/shim.ts
3772
4222
  /**
3773
4223
  * Loopback OpenAI-compatible endpoint with multi-account failover.
@@ -3846,7 +4296,8 @@ function writeOpenAIError(res, status, kind, message) {
3846
4296
  }
3847
4297
  /** True when an upstream failure body means the request overran the model's
3848
4298
  * context window (OpenAI `context_length_exceeded`, WorkBuddy code 11115 /
3849
- * "input length too long"). Surfaced as a friendly hint, never auto-truncated. */
4299
+ * "input length too long"). The shim answers it by compacting the conversation
4300
+ * in place and retrying once; see `recoverFromContextOverrun`. */
3850
4301
  function isContextTooLong(body) {
3851
4302
  if (body.includes("context_length_exceeded")) return true;
3852
4303
  if (body.includes("input length too long")) return true;
@@ -4008,25 +4459,7 @@ function createWorkBuddyShim(options) {
4008
4459
  tried.push(account.label);
4009
4460
  const result = await client.chatStream(account.credential, prepared, controller.signal);
4010
4461
  if (result.ok) {
4011
- logger?.info?.(`dsh-workbuddy-xdpool: served by ${account.label}`);
4012
- pool.noteServed(account.id);
4013
- refreshBalance(account);
4014
- res.writeHead(200, {
4015
- "Content-Type": "text/event-stream",
4016
- "Cache-Control": "no-cache",
4017
- "Connection": "keep-alive",
4018
- "X-Accel-Buffering": "no"
4019
- });
4020
- let sawDone = false;
4021
- const body = Readable.fromWeb(result.response.body);
4022
- body.on("data", (chunk) => {
4023
- if (chunk.includes("[DONE]")) sawDone = true;
4024
- });
4025
- body.on("error", (error) => {
4026
- logger?.warn("dsh-workbuddy-xdpool: upstream stream failed mid-flight", error);
4027
- if (!sawDone && res.writable) res.end("data: [DONE]\n\n");
4028
- });
4029
- body.pipe(res);
4462
+ await serveSuccessfulStream(res, account, result, logger, refreshBalance, pool);
4030
4463
  return;
4031
4464
  }
4032
4465
  last = {
@@ -4054,7 +4487,21 @@ function createWorkBuddyShim(options) {
4054
4487
  return;
4055
4488
  }
4056
4489
  if (isContextTooLong(last.message)) {
4057
- writeOpenAIError(res, 400, "context_length_exceeded", `${modelId === void 0 ? "the conversation exceeds this model's context window" : `the conversation exceeds ${modelId}'s context window`}. Shorten the conversation, start a new chat, or pick a model with a larger window (e.g. hy4-preview).`);
4490
+ const recovered = await recoverFromContextOverrun({
4491
+ raw,
4492
+ modelId,
4493
+ controller,
4494
+ region,
4495
+ logger,
4496
+ client,
4497
+ pool,
4498
+ maxAttempts
4499
+ });
4500
+ if (recovered.ok) {
4501
+ await serveSuccessfulStream(res, recovered.account, recovered.result, logger, refreshBalance, pool);
4502
+ return;
4503
+ }
4504
+ writeOpenAIError(res, 400, "context_length_exceeded", contextOverflowMessage(modelId, recovered.detail));
4058
4505
  return;
4059
4506
  }
4060
4507
  writeOpenAIError(res, KIND_STATUS[last.kind], last.kind, `workbuddy upstream ${last.kind} (http ${last.status}) after ${tried.length} account(s) [${tried.join(" → ")}]: ${last.message.slice(0, 400)}`);
@@ -4070,6 +4517,168 @@ function createWorkBuddyShim(options) {
4070
4517
  })
4071
4518
  };
4072
4519
  }
4520
+ /**
4521
+ * Build the overflow message the Harness must recognize.
4522
+ *
4523
+ * This is deliberately NOT free-form prose. `dsh-compaction-basic` decides
4524
+ * whether to compact-and-retry by running the text that reaches it through
4525
+ * `isContextWindowExceededError()` (`@deepseek-ai/dsh-llm`), whose matcher
4526
+ * accepts only specific phrasings:
4527
+ *
4528
+ * - `context_length_exceeded` / `context window exceeded`
4529
+ * - `maximum context length`
4530
+ * - `<input|prompt|request|messages> too large|long for ... context`
4531
+ * - `<input|prompt|request> exceeds the ... context window`
4532
+ *
4533
+ * The obvious friendly sentence ("the conversation exceeds this model's
4534
+ * context window") matches NONE of them, and neither does the WorkBuddy
4535
+ * upstream's own "input length too long" / code 11115. Emitting either meant
4536
+ * the Harness saw an unclassifiable 400, skipped its recovery path, and
4537
+ * surfaced a dead turn — the bug this function exists to prevent.
4538
+ *
4539
+ * The leading clause carries the machine-matched wording; the trailing clause
4540
+ * is what a human reads. Keep both in sync with
4541
+ * `tests/context-overflow-contract.test.ts`.
4542
+ */
4543
+ function contextOverflowMessage(modelId, detail = "") {
4544
+ return `This model's maximum context length was exceeded: the prompt is too large for ${modelId === void 0 ? "the model" : `model ${modelId}`}, and the conversation could not be compacted in place${detail === "" ? "" : ` (${detail})`}. Compact the conversation, or start a new chat.`;
4545
+ }
4546
+ /**
4547
+ * Serve one already-successful upstream stream as an SSE response.
4548
+ *
4549
+ * Extracted so the context-overrun recovery path reuses the exact same
4550
+ * bookkeeping (noteServed + background balance refresh) as a first-try hit.
4551
+ */
4552
+ async function serveSuccessfulStream(res, account, result, logger, refreshBalance, pool) {
4553
+ logger?.info?.(`dsh-workbuddy-xdpool: served by ${account.label}`);
4554
+ pool.noteServed(account.id);
4555
+ refreshBalance(account);
4556
+ res.writeHead(200, {
4557
+ "Content-Type": "text/event-stream",
4558
+ "Cache-Control": "no-cache",
4559
+ "Connection": "keep-alive",
4560
+ "X-Accel-Buffering": "no"
4561
+ });
4562
+ let sawDone = false;
4563
+ const body = Readable.fromWeb(result.response.body);
4564
+ body.on("data", (chunk) => {
4565
+ if (chunk.includes("[DONE]")) sawDone = true;
4566
+ });
4567
+ body.on("error", (error) => {
4568
+ logger?.warn("dsh-workbuddy-xdpool: upstream stream failed mid-flight", error);
4569
+ if (!sawDone && res.writable) res.end("data: [DONE]\n\n");
4570
+ });
4571
+ body.pipe(res);
4572
+ }
4573
+ /**
4574
+ * Compact an over-long conversation and retry it once.
4575
+ *
4576
+ * Strategy, in order:
4577
+ * 1. drop the oldest turns, keeping system messages and the newest exchange;
4578
+ * 2. ask the model to summarise the dropped turns and splice that summary in;
4579
+ * 3. hard-truncate as a last resort.
4580
+ *
4581
+ * Returns `ok: false` only when even a truncated prompt still overran — the
4582
+ * caller then surfaces the original actionable 400.
4583
+ */
4584
+ async function recoverFromContextOverrun(options) {
4585
+ const { raw, modelId, controller, region, logger, client, pool, maxAttempts } = options;
4586
+ const parsed = client.parseChatBody(raw);
4587
+ if (parsed === void 0) return {
4588
+ ok: false,
4589
+ detail: "request body was not parseable JSON"
4590
+ };
4591
+ const rawMessages = parsed["messages"];
4592
+ if (!Array.isArray(rawMessages)) return {
4593
+ ok: false,
4594
+ detail: "request carried no messages array"
4595
+ };
4596
+ const messages = rawMessages.filter((value) => typeof value === "object" && value !== null && !Array.isArray(value));
4597
+ if (messages.length === 0) return {
4598
+ ok: false,
4599
+ detail: "request carried no usable messages"
4600
+ };
4601
+ const overrunTokens = estimateMessagesTokens(messages);
4602
+ const budget = Math.max(512, Math.floor(overrunTokens / 2));
4603
+ logger?.warn(`dsh-workbuddy-xdpool: context overrun on ${modelId ?? "(no model)"} (~${overrunTokens} tokens); compacting to ~${budget} and retrying once`);
4604
+ let summary;
4605
+ let compacted = messages;
4606
+ let compactionDetail = "";
4607
+ try {
4608
+ const summariser = await pool.acquire(modelId, region);
4609
+ if (summariser === void 0) compactionDetail = "no account available to summarise with";
4610
+ else {
4611
+ const outcome = await compactWithSummary(messages, {
4612
+ budget,
4613
+ keepRecent: 6
4614
+ }, { complete: async (request, signal) => {
4615
+ const body = client.buildChatBody({
4616
+ ...parsed,
4617
+ stream: true,
4618
+ max_tokens: Math.max(256, Math.floor(budget / 2))
4619
+ }, request);
4620
+ return await client.completeChat(summariser.credential, body, signal ?? controller.signal);
4621
+ } }, controller.signal);
4622
+ compacted = outcome.messages;
4623
+ summary = outcome.summary;
4624
+ if (outcome.skipped !== void 0) compactionDetail = outcome.skipped;
4625
+ }
4626
+ } catch (error) {
4627
+ compactionDetail = `summarisation failed: ${String(error)}`;
4628
+ }
4629
+ if (estimateMessagesTokens(compacted) > budget) compacted = hardTruncate(compacted, budget).messages;
4630
+ if (summary === void 0 && estimateMessagesTokens(compacted) >= overrunTokens) return {
4631
+ ok: false,
4632
+ detail: compactionDetail === "" ? "compaction could not reduce the prompt" : compactionDetail
4633
+ };
4634
+ const retryBody = client.buildChatBody(parsed, compacted);
4635
+ const tried = [];
4636
+ for (let attempt = 0; attempt < maxAttempts; attempt += 1) {
4637
+ if (controller.signal.aborted) return {
4638
+ ok: false,
4639
+ detail: "client disconnected"
4640
+ };
4641
+ const account = await pool.acquire(modelId, region);
4642
+ if (account === void 0) return {
4643
+ ok: false,
4644
+ detail: "no account available after compaction"
4645
+ };
4646
+ tried.push(account.label);
4647
+ const result = await client.chatStream(account.credential, retryBody, controller.signal);
4648
+ if (result.ok) {
4649
+ logger?.info?.(`dsh-workbuddy-xdpool: recovered from context overrun on ${modelId ?? "(no model)"} (summarised: ${summary === void 0 ? "no" : "yes"})`);
4650
+ return {
4651
+ ok: true,
4652
+ account,
4653
+ result
4654
+ };
4655
+ }
4656
+ if (isContextTooLong(result.message)) return {
4657
+ ok: false,
4658
+ detail: "prompt still exceeded the window after compaction"
4659
+ };
4660
+ if (result.kind === "session_dead") {
4661
+ await pool.refreshAccount(account.id);
4662
+ continue;
4663
+ }
4664
+ if (result.kind === "hard_credit") {
4665
+ pool.penalizeExhausted(account.id);
4666
+ continue;
4667
+ }
4668
+ if (result.kind === "soft_rate") {
4669
+ pool.penalize(account.id, parseRateLimitReset(result.message), modelId);
4670
+ continue;
4671
+ }
4672
+ return {
4673
+ ok: false,
4674
+ detail: `upstream ${result.kind} after compaction`
4675
+ };
4676
+ }
4677
+ return {
4678
+ ok: false,
4679
+ detail: `no account served the compacted request (tried ${tried.length})`
4680
+ };
4681
+ }
4073
4682
  //#endregion
4074
4683
  //#region src/status.ts
4075
4684
  /** Build the status document. Never throws. */