dsh-workbuddy-xdpool 1.2.0 → 1.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +91 -0
- package/README.en.md +112 -45
- package/README.md +104 -53
- package/assets/automation-panel.png +0 -0
- package/assets/settings-card.png +0 -0
- package/lib/bin.js +190 -9
- package/lib/client.js +23 -7
- package/lib/index.d.ts +81 -0
- package/lib/index.js +639 -30
- package/package.json +1 -1
package/lib/index.js
CHANGED
|
@@ -31,6 +31,8 @@ const MODELS_CATALOG_PATH = "/v2/enterprises/personal/models";
|
|
|
31
31
|
const GLOBAL_CONFIG_PATH = "/v3/config";
|
|
32
32
|
const JSON_TIMEOUT_MS = 3e4;
|
|
33
33
|
const ERROR_BODY_LIMIT = 4096;
|
|
34
|
+
/** Cap on the reassembled compaction reply, guarding against a runaway stream. */
|
|
35
|
+
const COMPLETION_TEXT_LIMIT = 65536;
|
|
34
36
|
/** Insufficient-credit markers, ASCII lowercase plus the original Chinese. */
|
|
35
37
|
const HARD_CREDIT_MARKERS = [
|
|
36
38
|
"insufficient credit",
|
|
@@ -49,8 +51,31 @@ const HARD_CREDIT_MARKERS = [
|
|
|
49
51
|
"额度用尽",
|
|
50
52
|
"没有积分"
|
|
51
53
|
];
|
|
52
|
-
/** Session-invalidation markers that mean "
|
|
53
|
-
|
|
54
|
+
/** Session-invalidation markers that mean "this credential is dead; use another".
|
|
55
|
+
* Kept alongside the HTTP-status rule in `classifyUpstreamError`: the status is
|
|
56
|
+
* enough for a direct 401/403, but some failures arrive wrapped in a 200
|
|
57
|
+
* envelope or a 4xx the gateway words differently. Adding the English and
|
|
58
|
+
* Chinese phrasings the upstream actually uses keeps those recoverable too —
|
|
59
|
+
* an unmatched one fell through to `client`, which is terminal in the shim
|
|
60
|
+
* and pinned the pool to the first account (the "API 密钥无效" bug). */
|
|
61
|
+
const SESSION_DEAD_MARKERS = [
|
|
62
|
+
"Offline user session not found",
|
|
63
|
+
"12153",
|
|
64
|
+
"api key is invalid",
|
|
65
|
+
"invalid api key",
|
|
66
|
+
"invalid_api_key",
|
|
67
|
+
"api密钥无效",
|
|
68
|
+
"密钥无效",
|
|
69
|
+
"无效的密钥",
|
|
70
|
+
"unauthorized",
|
|
71
|
+
"token expired",
|
|
72
|
+
"token is invalid",
|
|
73
|
+
"login expired",
|
|
74
|
+
"please login",
|
|
75
|
+
"未登录",
|
|
76
|
+
"登录已失效",
|
|
77
|
+
"重新登录"
|
|
78
|
+
];
|
|
54
79
|
/**
|
|
55
80
|
* Markers for "already checked in today".
|
|
56
81
|
*
|
|
@@ -223,12 +248,82 @@ function envelopeError(status, envelope) {
|
|
|
223
248
|
return /* @__PURE__ */ new Error(`workbuddy upstream ${kind} (http ${status}): ${envelope.msg.slice(0, 160)}`);
|
|
224
249
|
}
|
|
225
250
|
/**
|
|
251
|
+
* Read an OpenAI-style SSE chat stream and concatenate the assistant text.
|
|
252
|
+
*
|
|
253
|
+
* The upstream always streams (`stream: true` is forced on every chat body),
|
|
254
|
+
* so a non-streaming internal call has to reassemble the deltas itself. Only
|
|
255
|
+
* `choices[0].delta.content` is collected; reasoning deltas are dropped
|
|
256
|
+
* because a compaction summary needs the final answer, not the scratchpad.
|
|
257
|
+
*/
|
|
258
|
+
async function readCompletionText(body) {
|
|
259
|
+
const decoder = new TextDecoder();
|
|
260
|
+
const reader = body.getReader();
|
|
261
|
+
let buffer = "";
|
|
262
|
+
let text = "";
|
|
263
|
+
try {
|
|
264
|
+
for (;;) {
|
|
265
|
+
const { done, value } = await reader.read();
|
|
266
|
+
if (done) break;
|
|
267
|
+
buffer += decoder.decode(value, { stream: true });
|
|
268
|
+
let split = buffer.indexOf("\n\n");
|
|
269
|
+
while (split !== -1) {
|
|
270
|
+
const frame = buffer.slice(0, split);
|
|
271
|
+
buffer = buffer.slice(split + 2);
|
|
272
|
+
text += contentOfFrame(frame);
|
|
273
|
+
if (text.length > COMPLETION_TEXT_LIMIT) return text.slice(0, COMPLETION_TEXT_LIMIT);
|
|
274
|
+
split = buffer.indexOf("\n\n");
|
|
275
|
+
}
|
|
276
|
+
}
|
|
277
|
+
if (buffer.trim() !== "") text += contentOfFrame(buffer);
|
|
278
|
+
} finally {
|
|
279
|
+
reader.releaseLock?.();
|
|
280
|
+
}
|
|
281
|
+
return text;
|
|
282
|
+
}
|
|
283
|
+
/** Pull `choices[0].delta.content` (or a non-streaming `message.content`) out of one SSE frame. */
|
|
284
|
+
function contentOfFrame(frame) {
|
|
285
|
+
let out = "";
|
|
286
|
+
for (const rawLine of frame.split(/\r?\n/u)) {
|
|
287
|
+
const line = rawLine.trim();
|
|
288
|
+
if (!line.startsWith("data:")) continue;
|
|
289
|
+
const payload = line.slice(5).trim();
|
|
290
|
+
if (payload === "" || payload === "[DONE]") continue;
|
|
291
|
+
let parsed;
|
|
292
|
+
try {
|
|
293
|
+
parsed = JSON.parse(payload);
|
|
294
|
+
} catch {
|
|
295
|
+
continue;
|
|
296
|
+
}
|
|
297
|
+
if (typeof parsed !== "object" || parsed === null) continue;
|
|
298
|
+
const choices = parsed["choices"];
|
|
299
|
+
if (!Array.isArray(choices) || choices.length === 0) continue;
|
|
300
|
+
const choice = choices[0];
|
|
301
|
+
const delta = choice["delta"];
|
|
302
|
+
if (typeof delta === "object" && delta !== null) {
|
|
303
|
+
const content = delta["content"];
|
|
304
|
+
if (typeof content === "string") out += content;
|
|
305
|
+
}
|
|
306
|
+
const message = choice["message"];
|
|
307
|
+
if (typeof message === "object" && message !== null) {
|
|
308
|
+
const content = message["content"];
|
|
309
|
+
if (typeof content === "string") out += content;
|
|
310
|
+
}
|
|
311
|
+
const data = parsed["data"];
|
|
312
|
+
if (typeof data === "object" && data !== null) {
|
|
313
|
+
const inner = data["content"];
|
|
314
|
+
if (typeof inner === "string") out += inner;
|
|
315
|
+
}
|
|
316
|
+
}
|
|
317
|
+
return out;
|
|
318
|
+
}
|
|
319
|
+
/**
|
|
226
320
|
* Classify an upstream failure from its HTTP status and body excerpt.
|
|
227
321
|
* Body markers win over status, because the upstream reuses 400/200 for
|
|
228
322
|
* several distinct conditions.
|
|
229
323
|
*/
|
|
230
324
|
function classifyUpstreamError(status, body) {
|
|
231
325
|
if (status === 402) return "hard_credit";
|
|
326
|
+
if (status === 401 || status === 403) return "session_dead";
|
|
232
327
|
const lower = body.toLowerCase();
|
|
233
328
|
for (const marker of HARD_CREDIT_MARKERS) if (lower.includes(marker.toLowerCase()) || body.includes(marker)) return "hard_credit";
|
|
234
329
|
for (const marker of SESSION_DEAD_MARKERS) if (body.includes(marker)) return "session_dead";
|
|
@@ -696,6 +791,51 @@ var WorkBuddyUpstreamClient = class {
|
|
|
696
791
|
}
|
|
697
792
|
return JSON.stringify(obj);
|
|
698
793
|
}
|
|
794
|
+
/**
|
|
795
|
+
* Parse a raw OpenAI chat body without normalising it.
|
|
796
|
+
*
|
|
797
|
+
* The compactor needs the message array as objects, while `chatStream` only
|
|
798
|
+
* accepts the serialised string form.
|
|
799
|
+
*/
|
|
800
|
+
parseChatBody(raw) {
|
|
801
|
+
let body;
|
|
802
|
+
try {
|
|
803
|
+
body = JSON.parse(raw);
|
|
804
|
+
} catch {
|
|
805
|
+
return;
|
|
806
|
+
}
|
|
807
|
+
if (typeof body !== "object" || body === null || Array.isArray(body)) return void 0;
|
|
808
|
+
return body;
|
|
809
|
+
}
|
|
810
|
+
/** Re-serialise `base` with a rewritten `messages` array, still normalised. */
|
|
811
|
+
buildChatBody(base, messages) {
|
|
812
|
+
return this.prepareChatBody(JSON.stringify({
|
|
813
|
+
...base,
|
|
814
|
+
messages
|
|
815
|
+
}));
|
|
816
|
+
}
|
|
817
|
+
/**
|
|
818
|
+
* Run one NON-streaming completion and return the assistant text.
|
|
819
|
+
*
|
|
820
|
+
* Used only for internal compaction (summarising dropped turns). The chat
|
|
821
|
+
* endpoint itself always streams, so this reassembles the SSE frames into a
|
|
822
|
+
* single string. Throws on any failure: the compactor then falls back to
|
|
823
|
+
* plain truncation rather than failing the user's turn.
|
|
824
|
+
*/
|
|
825
|
+
async completeChat(credential, prepared, signal) {
|
|
826
|
+
const response = await this.fetchImpl(`${chatBase(credential)}/v2/chat/completions`, {
|
|
827
|
+
method: "POST",
|
|
828
|
+
headers: chatHeaders(credential),
|
|
829
|
+
body: prepared,
|
|
830
|
+
...signal === void 0 ? {} : { signal }
|
|
831
|
+
});
|
|
832
|
+
if (!response.ok) {
|
|
833
|
+
const text = (await response.text().catch(() => "")).slice(0, ERROR_BODY_LIMIT);
|
|
834
|
+
throw new Error(`compaction upstream http ${response.status}: ${text}`);
|
|
835
|
+
}
|
|
836
|
+
if (response.body === null) throw new Error("compaction upstream returned no body");
|
|
837
|
+
return await readCompletionText(response.body);
|
|
838
|
+
}
|
|
699
839
|
/** Forward one chat completion. Never throws for upstream failures. */
|
|
700
840
|
async chatStream(credential, prepared, signal) {
|
|
701
841
|
let response;
|
|
@@ -3031,6 +3171,16 @@ function dayKey(date) {
|
|
|
3031
3171
|
return `${date.getFullYear()}-${month}-${day}`;
|
|
3032
3172
|
}
|
|
3033
3173
|
/**
|
|
3174
|
+
* The scheduled SLOT `date` falls in, as `YYYY-MM-DDTHH`.
|
|
3175
|
+
*
|
|
3176
|
+
* The per-job run guard keys on this instead of the date, so a job configured
|
|
3177
|
+
* for several hours runs in each of them while a second tick inside the same
|
|
3178
|
+
* hour is still refused.
|
|
3179
|
+
*/
|
|
3180
|
+
function slotKey(date) {
|
|
3181
|
+
return `${dayKey(date)}T${String(date.getHours()).padStart(2, "0")}`;
|
|
3182
|
+
}
|
|
3183
|
+
/**
|
|
3034
3184
|
* Whether `now`'s local hour is one of `hours`.
|
|
3035
3185
|
*
|
|
3036
3186
|
* The reference panel computes a `nextFire` instant and sleeps until it; this
|
|
@@ -3135,7 +3285,7 @@ var WorkBuddyScheduler = class {
|
|
|
3135
3285
|
this.reportHours = options.reportHours ?? [10];
|
|
3136
3286
|
this.taskHours = options.taskHours ?? [11];
|
|
3137
3287
|
this.streakHours = options.streakHours ?? [12];
|
|
3138
|
-
this.travelHours = options.travelHours ?? [
|
|
3288
|
+
this.travelHours = options.travelHours ?? [9, 21];
|
|
3139
3289
|
}
|
|
3140
3290
|
/** Apply a new configuration; safe to call while running. */
|
|
3141
3291
|
/**
|
|
@@ -3213,7 +3363,7 @@ var WorkBuddyScheduler = class {
|
|
|
3213
3363
|
async runNow(kind, force = false) {
|
|
3214
3364
|
const today = dayKey(this.now());
|
|
3215
3365
|
const state = this.states[kind];
|
|
3216
|
-
if (!force && state.
|
|
3366
|
+
if (!force && state.lastRunSlot === slotKey(this.now())) return { ...state };
|
|
3217
3367
|
this.busy = true;
|
|
3218
3368
|
try {
|
|
3219
3369
|
await this.runJob(kind, today);
|
|
@@ -3308,10 +3458,11 @@ var WorkBuddyScheduler = class {
|
|
|
3308
3458
|
this.busy = true;
|
|
3309
3459
|
try {
|
|
3310
3460
|
const now = this.now();
|
|
3461
|
+
const slot = slotKey(now);
|
|
3311
3462
|
const today = dayKey(now);
|
|
3312
3463
|
for (const kind of JOB_KINDS) {
|
|
3313
3464
|
if (this.stopped) return;
|
|
3314
|
-
if (this.states[kind].
|
|
3465
|
+
if (this.states[kind].lastRunSlot === slot) continue;
|
|
3315
3466
|
if (!isFireHour(now, this.hoursOf(kind))) continue;
|
|
3316
3467
|
await this.runJob(kind, today);
|
|
3317
3468
|
}
|
|
@@ -3409,6 +3560,8 @@ var WorkBuddyScheduler = class {
|
|
|
3409
3560
|
let credit = 0;
|
|
3410
3561
|
let energy = 0;
|
|
3411
3562
|
let claimed = 0;
|
|
3563
|
+
const detail = [];
|
|
3564
|
+
let progressNote;
|
|
3412
3565
|
const accounts = this.accountsInOrder();
|
|
3413
3566
|
if (kind === "tasks") await this.sendEventChains(accounts);
|
|
3414
3567
|
for (let index = 0; index < accounts.length; index++) {
|
|
@@ -3421,7 +3574,10 @@ var WorkBuddyScheduler = class {
|
|
|
3421
3574
|
try {
|
|
3422
3575
|
const claim = await this.client.claimDailyCheckin(account.credential);
|
|
3423
3576
|
this.recordEarnings(account.id, today, { checkinCredit: claim.credit });
|
|
3424
|
-
if (claim.credit > 0)
|
|
3577
|
+
if (claim.credit > 0) {
|
|
3578
|
+
if (claim.credit > 0) detail.push(`签到 +${claim.credit}`);
|
|
3579
|
+
this.logger.info?.(`automation checkin ${account.label}: +${claim.credit} credit`);
|
|
3580
|
+
}
|
|
3425
3581
|
} catch (error) {
|
|
3426
3582
|
if (!isAlreadyCheckin(error)) throw error;
|
|
3427
3583
|
this.logger.info?.(`automation checkin ${account.label}: already done today`);
|
|
@@ -3437,6 +3593,7 @@ var WorkBuddyScheduler = class {
|
|
|
3437
3593
|
credit += result.credit;
|
|
3438
3594
|
energy += result.energy;
|
|
3439
3595
|
claimed += result.claimed;
|
|
3596
|
+
detail.push(...result.titles);
|
|
3440
3597
|
this.claimableSeen += result.claimableCount;
|
|
3441
3598
|
this.recordEarnings(account.id, today, {
|
|
3442
3599
|
credit: result.credit,
|
|
@@ -3446,9 +3603,11 @@ var WorkBuddyScheduler = class {
|
|
|
3446
3603
|
this.logger.info?.(`automation tasks ${account.label}: ${result.claimed} claimed (+${result.credit} credit, +${result.energy} energy, ${result.claimableCount} claimable seen)`);
|
|
3447
3604
|
break;
|
|
3448
3605
|
}
|
|
3449
|
-
case "streak":
|
|
3450
|
-
await this.redeemStreak(account);
|
|
3606
|
+
case "streak": {
|
|
3607
|
+
const progress = await this.redeemStreak(account);
|
|
3608
|
+
if (progress !== void 0) progressNote = progress;
|
|
3451
3609
|
break;
|
|
3610
|
+
}
|
|
3452
3611
|
case "travel": await this.runTravel(account);
|
|
3453
3612
|
}
|
|
3454
3613
|
ok++;
|
|
@@ -3459,6 +3618,7 @@ var WorkBuddyScheduler = class {
|
|
|
3459
3618
|
if (index < accounts.length - 1 && this.delayMs > 0 && !this.stopped) await sleep(this.delayMs);
|
|
3460
3619
|
}
|
|
3461
3620
|
const state = this.states[kind];
|
|
3621
|
+
state.lastRunSlot = slotKey(this.now());
|
|
3462
3622
|
state.lastRunDate = today;
|
|
3463
3623
|
state.lastRunAtMs = this.now().getTime();
|
|
3464
3624
|
state.ok = ok;
|
|
@@ -3467,6 +3627,8 @@ var WorkBuddyScheduler = class {
|
|
|
3467
3627
|
state.energy = energy;
|
|
3468
3628
|
state.claimed = claimed;
|
|
3469
3629
|
state.message = this.summarise(kind, ok, failed, claimed, credit, energy);
|
|
3630
|
+
state.detail = detail;
|
|
3631
|
+
state.progress = progressNote;
|
|
3470
3632
|
this.logger.info?.(`automation ${accountWord}: ${state.message}`);
|
|
3471
3633
|
}
|
|
3472
3634
|
/** Compose the one-line summary shown on the card. */
|
|
@@ -3669,6 +3831,7 @@ var WorkBuddyScheduler = class {
|
|
|
3669
3831
|
let credit = 0;
|
|
3670
3832
|
let energy = 0;
|
|
3671
3833
|
let claimed = 0;
|
|
3834
|
+
const titles = [];
|
|
3672
3835
|
for (const task of claimable) {
|
|
3673
3836
|
if (this.stopped) break;
|
|
3674
3837
|
const reward = await this.client.claimTaskReward(credential, task.taskCode);
|
|
@@ -3676,6 +3839,7 @@ var WorkBuddyScheduler = class {
|
|
|
3676
3839
|
energy += reward.energy;
|
|
3677
3840
|
if (reward.credit > 0 || reward.energy > 0) {
|
|
3678
3841
|
claimed++;
|
|
3842
|
+
titles.push(task.title);
|
|
3679
3843
|
this.logger.info?.(`automation claim ${account.label}: ${task.title} +${reward.credit}c +${reward.energy}e`);
|
|
3680
3844
|
} else this.logger.info?.(`automation claim ${account.label}: ${task.taskCode} already claimed`);
|
|
3681
3845
|
if (this.delayMs > 0 && !this.stopped) await sleep(this.delayMs);
|
|
@@ -3684,7 +3848,8 @@ var WorkBuddyScheduler = class {
|
|
|
3684
3848
|
claimed,
|
|
3685
3849
|
credit,
|
|
3686
3850
|
energy,
|
|
3687
|
-
claimableCount: claimable.length
|
|
3851
|
+
claimableCount: claimable.length,
|
|
3852
|
+
titles
|
|
3688
3853
|
};
|
|
3689
3854
|
}
|
|
3690
3855
|
/**
|
|
@@ -3724,6 +3889,11 @@ var WorkBuddyScheduler = class {
|
|
|
3724
3889
|
}
|
|
3725
3890
|
if (this.delayMs > 0 && !this.stopped) await sleep(this.delayMs);
|
|
3726
3891
|
}
|
|
3892
|
+
const pendingTier = status.tiers.find((tier) => tier.status === "locked");
|
|
3893
|
+
if (pendingTier !== void 0) {
|
|
3894
|
+
const remaining = Math.max(0, pendingTier.days - status.days);
|
|
3895
|
+
return remaining > 0 ? `${pendingTier.tier} in ${remaining}d` : `${pendingTier.tier} ready`;
|
|
3896
|
+
}
|
|
3727
3897
|
}
|
|
3728
3898
|
/**
|
|
3729
3899
|
* One trip through the buddy travel loop for an account.
|
|
@@ -3768,6 +3938,286 @@ var WorkBuddyScheduler = class {
|
|
|
3768
3938
|
}
|
|
3769
3939
|
};
|
|
3770
3940
|
//#endregion
|
|
3941
|
+
//#region src/context-budget.ts
|
|
3942
|
+
/** Rough character-per-token ratio. CJK is ~1 token/char, latin ~1/4. */
|
|
3943
|
+
const CHARS_PER_TOKEN_LATIN = 4;
|
|
3944
|
+
const CHARS_PER_TOKEN_CJK = 1;
|
|
3945
|
+
/** Fixed per-message overhead the chat template adds (role markers etc.). */
|
|
3946
|
+
const PER_MESSAGE_TOKEN_OVERHEAD = 4;
|
|
3947
|
+
/** Every image/tool part costs at least this much once decoded. */
|
|
3948
|
+
const PER_PART_TOKEN_FLOOR = 16;
|
|
3949
|
+
/**
|
|
3950
|
+
* Estimate the token cost of one message's `content`.
|
|
3951
|
+
*
|
|
3952
|
+
* Deliberately conservative (over-estimates) so we compact slightly early
|
|
3953
|
+
* rather than discovering the overrun upstream.
|
|
3954
|
+
*/
|
|
3955
|
+
function estimateContentTokens(content) {
|
|
3956
|
+
if (content === null || content === void 0) return 0;
|
|
3957
|
+
if (typeof content === "string") return estimateTextTokens(content);
|
|
3958
|
+
if (typeof content === "number" || typeof content === "boolean") return PER_PART_TOKEN_FLOOR;
|
|
3959
|
+
if (Array.isArray(content)) {
|
|
3960
|
+
let total = 0;
|
|
3961
|
+
for (const part of content) total += estimateContentTokens(part);
|
|
3962
|
+
return total;
|
|
3963
|
+
}
|
|
3964
|
+
if (typeof content === "object") {
|
|
3965
|
+
const record = content;
|
|
3966
|
+
let total = PER_PART_TOKEN_FLOOR;
|
|
3967
|
+
for (const key of [
|
|
3968
|
+
"text",
|
|
3969
|
+
"image_url",
|
|
3970
|
+
"input",
|
|
3971
|
+
"content"
|
|
3972
|
+
]) if (key in record) total += estimateContentTokens(record[key]);
|
|
3973
|
+
if (total === PER_PART_TOKEN_FLOOR) total += estimateTextTokens(safeStringify(record));
|
|
3974
|
+
return total;
|
|
3975
|
+
}
|
|
3976
|
+
return 0;
|
|
3977
|
+
}
|
|
3978
|
+
/** Estimate tokens for a plain string, accounting for CJK density. */
|
|
3979
|
+
function estimateTextTokens(text) {
|
|
3980
|
+
if (text === "") return 0;
|
|
3981
|
+
let cjk = 0;
|
|
3982
|
+
for (const char of text) if (isCjk(char.codePointAt(0) ?? 0)) cjk += 1;
|
|
3983
|
+
const latin = text.length - cjk;
|
|
3984
|
+
return Math.ceil(cjk / CHARS_PER_TOKEN_CJK + latin / CHARS_PER_TOKEN_LATIN);
|
|
3985
|
+
}
|
|
3986
|
+
function isCjk(code) {
|
|
3987
|
+
return code >= 12288 && code <= 12351 || code >= 12352 && code <= 12543 || code >= 13312 && code <= 19903 || code >= 19968 && code <= 40959 || code >= 63744 && code <= 64255 || code >= 65280 && code <= 65519 || code >= 131072 && code <= 191471;
|
|
3988
|
+
}
|
|
3989
|
+
function safeStringify(value) {
|
|
3990
|
+
try {
|
|
3991
|
+
return JSON.stringify(value) ?? "";
|
|
3992
|
+
} catch {
|
|
3993
|
+
return String(value);
|
|
3994
|
+
}
|
|
3995
|
+
}
|
|
3996
|
+
/** Estimate the prompt cost of a whole message array. */
|
|
3997
|
+
function estimateMessagesTokens(messages) {
|
|
3998
|
+
let total = 0;
|
|
3999
|
+
for (const message of messages) {
|
|
4000
|
+
total += PER_MESSAGE_TOKEN_OVERHEAD;
|
|
4001
|
+
total += estimateContentTokens(message.content);
|
|
4002
|
+
if (message["tool_calls"] !== void 0) total += estimateContentTokens(message["tool_calls"]);
|
|
4003
|
+
if (message["name"] !== void 0) total += estimateTextTokens(String(message["name"]));
|
|
4004
|
+
}
|
|
4005
|
+
return total;
|
|
4006
|
+
}
|
|
4007
|
+
/** True when `role` carries instructions that must survive compaction. */
|
|
4008
|
+
function isPinnedRole(role) {
|
|
4009
|
+
return role === "system" || role === "developer";
|
|
4010
|
+
}
|
|
4011
|
+
/**
|
|
4012
|
+
* Drop the oldest non-pinned messages until the estimate fits `budget`.
|
|
4013
|
+
*
|
|
4014
|
+
* Pinned (system/developer) messages and the newest `keepRecent` messages are
|
|
4015
|
+
* never dropped here — if those alone overrun the budget, the caller must fall
|
|
4016
|
+
* back to summarisation or give up.
|
|
4017
|
+
*/
|
|
4018
|
+
function compactMessages(messages, options) {
|
|
4019
|
+
const keepRecent = Math.max(1, options.keepRecent ?? 4);
|
|
4020
|
+
const estimate = estimateMessagesTokens(messages);
|
|
4021
|
+
if (estimate <= options.budget) return {
|
|
4022
|
+
messages: [...messages],
|
|
4023
|
+
dropped: [],
|
|
4024
|
+
tokens: estimate,
|
|
4025
|
+
changed: false
|
|
4026
|
+
};
|
|
4027
|
+
const pinned = [];
|
|
4028
|
+
const body = [];
|
|
4029
|
+
for (const message of messages) if (isPinnedRole(message.role)) pinned.push(message);
|
|
4030
|
+
else body.push(message);
|
|
4031
|
+
const keep = Math.min(keepRecent, body.length);
|
|
4032
|
+
const tail = body.slice(body.length - keep);
|
|
4033
|
+
const head = body.slice(0, body.length - keep);
|
|
4034
|
+
let dropCount = 0;
|
|
4035
|
+
let candidate = [
|
|
4036
|
+
...pinned,
|
|
4037
|
+
...head,
|
|
4038
|
+
...tail
|
|
4039
|
+
];
|
|
4040
|
+
let total = estimateMessagesTokens(candidate);
|
|
4041
|
+
while (total > options.budget && dropCount < head.length) {
|
|
4042
|
+
dropCount += 1;
|
|
4043
|
+
candidate = [
|
|
4044
|
+
...pinned,
|
|
4045
|
+
...head.slice(dropCount),
|
|
4046
|
+
...tail
|
|
4047
|
+
];
|
|
4048
|
+
total = estimateMessagesTokens(candidate);
|
|
4049
|
+
}
|
|
4050
|
+
const dropped = head.slice(0, dropCount);
|
|
4051
|
+
return {
|
|
4052
|
+
messages: candidate,
|
|
4053
|
+
dropped,
|
|
4054
|
+
tokens: total,
|
|
4055
|
+
changed: dropCount > 0
|
|
4056
|
+
};
|
|
4057
|
+
}
|
|
4058
|
+
/**
|
|
4059
|
+
* Drop the oldest messages, including pinned ones, as a last resort.
|
|
4060
|
+
*
|
|
4061
|
+
* Used when even a summary cannot bring the prompt under budget (for example a
|
|
4062
|
+
* single enormous pasted document). The newest message always survives.
|
|
4063
|
+
*/
|
|
4064
|
+
function hardTruncate(messages, budget) {
|
|
4065
|
+
if (messages.length === 0) return {
|
|
4066
|
+
messages: [],
|
|
4067
|
+
dropped: [],
|
|
4068
|
+
tokens: 0,
|
|
4069
|
+
changed: false
|
|
4070
|
+
};
|
|
4071
|
+
let start = 0;
|
|
4072
|
+
let candidate = [...messages];
|
|
4073
|
+
let total = estimateMessagesTokens(candidate);
|
|
4074
|
+
while (total > budget && start < messages.length - 1) {
|
|
4075
|
+
start += 1;
|
|
4076
|
+
candidate = messages.slice(start);
|
|
4077
|
+
total = estimateMessagesTokens(candidate);
|
|
4078
|
+
}
|
|
4079
|
+
return {
|
|
4080
|
+
messages: candidate,
|
|
4081
|
+
dropped: messages.slice(0, start),
|
|
4082
|
+
tokens: total,
|
|
4083
|
+
changed: start > 0
|
|
4084
|
+
};
|
|
4085
|
+
}
|
|
4086
|
+
/** Instructions handed to the model when we ask it to compact a conversation. */
|
|
4087
|
+
const SUMMARIZE_INSTRUCTION = [
|
|
4088
|
+
"You are compacting an ongoing conversation so it can continue without the original history.",
|
|
4089
|
+
"Summarise the transcript below into a dense briefing for the next assistant turn.",
|
|
4090
|
+
"Preserve, in this order of priority:",
|
|
4091
|
+
"1. explicit user requirements, constraints and corrections;",
|
|
4092
|
+
"2. decisions already made, and the reasoning behind them;",
|
|
4093
|
+
"3. concrete facts: file paths, identifiers, commands, numbers, error messages;",
|
|
4094
|
+
"4. unfinished work and the current blocker.",
|
|
4095
|
+
"Drop pleasantries, repetition and superseded attempts.",
|
|
4096
|
+
"Write the briefing only — no preamble, no markdown fence."
|
|
4097
|
+
].join("\n");
|
|
4098
|
+
/** Render a message array as plain text for the summarisation prompt. */
|
|
4099
|
+
function transcriptOf(messages) {
|
|
4100
|
+
const lines = [];
|
|
4101
|
+
for (const message of messages) {
|
|
4102
|
+
const role = message.role === "" ? "unknown" : message.role;
|
|
4103
|
+
lines.push(`### ${role}`);
|
|
4104
|
+
lines.push(renderContent(message.content));
|
|
4105
|
+
if (message["tool_calls"] !== void 0) lines.push(renderContent(message["tool_calls"]));
|
|
4106
|
+
}
|
|
4107
|
+
return lines.join("\n");
|
|
4108
|
+
}
|
|
4109
|
+
function renderContent(content) {
|
|
4110
|
+
if (content === null || content === void 0) return "";
|
|
4111
|
+
if (typeof content === "string") return content;
|
|
4112
|
+
if (Array.isArray(content)) return content.map((part) => renderContent(part)).filter((text) => text !== "").join("\n");
|
|
4113
|
+
if (typeof content === "object") {
|
|
4114
|
+
const record = content;
|
|
4115
|
+
for (const key of [
|
|
4116
|
+
"text",
|
|
4117
|
+
"content",
|
|
4118
|
+
"input"
|
|
4119
|
+
]) if (typeof record[key] === "string") return record[key];
|
|
4120
|
+
if (record["type"] !== void 0 && typeof record["type"] === "string") return `[${record["type"]}]`;
|
|
4121
|
+
return safeStringify(record);
|
|
4122
|
+
}
|
|
4123
|
+
return String(content);
|
|
4124
|
+
}
|
|
4125
|
+
/** Build the synthetic system message that carries a compaction summary. */
|
|
4126
|
+
function summaryMessage(summary) {
|
|
4127
|
+
return {
|
|
4128
|
+
role: "system",
|
|
4129
|
+
content: [
|
|
4130
|
+
"The earlier part of this conversation was compacted to fit the model context window.",
|
|
4131
|
+
"Briefing produced from the dropped turns:",
|
|
4132
|
+
"",
|
|
4133
|
+
summary.trim()
|
|
4134
|
+
].join("\n")
|
|
4135
|
+
};
|
|
4136
|
+
}
|
|
4137
|
+
/**
|
|
4138
|
+
* Compact `messages` to `budget`, summarising the dropped turns when possible.
|
|
4139
|
+
*
|
|
4140
|
+
* The summary is requested with a *bounded* transcript so the compaction call
|
|
4141
|
+
* itself can never overrun the window: if the dropped turns are huge, only the
|
|
4142
|
+
* newest slice of them is summarised, and the oldest are noted as elided.
|
|
4143
|
+
*/
|
|
4144
|
+
async function compactWithSummary(messages, options, deps, signal) {
|
|
4145
|
+
const first = compactMessages(messages, options);
|
|
4146
|
+
if (!first.changed) return {
|
|
4147
|
+
messages: first.messages,
|
|
4148
|
+
tokens: first.tokens,
|
|
4149
|
+
summary: void 0,
|
|
4150
|
+
skipped: void 0
|
|
4151
|
+
};
|
|
4152
|
+
const summaryBudget = Math.max(256, Math.floor(options.budget / 4));
|
|
4153
|
+
let toSummarize = first.dropped;
|
|
4154
|
+
let elided = 0;
|
|
4155
|
+
while (estimateMessagesTokens(toSummarize) > summaryBudget && toSummarize.length > 1) {
|
|
4156
|
+
toSummarize = toSummarize.slice(1);
|
|
4157
|
+
elided += 1;
|
|
4158
|
+
}
|
|
4159
|
+
let summary;
|
|
4160
|
+
let skipped;
|
|
4161
|
+
try {
|
|
4162
|
+
const instruction = elided > 0 ? `${SUMMARIZE_INSTRUCTION}\n\nNote: the ${elided} oldest turn(s) were elided before this transcript.` : SUMMARIZE_INSTRUCTION;
|
|
4163
|
+
const suffix = elided > 0 ? `\n(the ${elided} oldest turn(s) were elided)` : "";
|
|
4164
|
+
const request = [{
|
|
4165
|
+
role: "system",
|
|
4166
|
+
content: instruction
|
|
4167
|
+
}, {
|
|
4168
|
+
role: "user",
|
|
4169
|
+
content: `${transcriptOf(toSummarize)}${suffix}`
|
|
4170
|
+
}];
|
|
4171
|
+
const text = await deps.complete(request, signal);
|
|
4172
|
+
if (text.trim() !== "") summary = text.trim();
|
|
4173
|
+
else skipped = "summariser returned an empty summary";
|
|
4174
|
+
} catch (error) {
|
|
4175
|
+
skipped = `summarisation failed: ${String(error)}`;
|
|
4176
|
+
}
|
|
4177
|
+
if (summary === void 0) return {
|
|
4178
|
+
messages: first.messages,
|
|
4179
|
+
skipped,
|
|
4180
|
+
tokens: first.tokens
|
|
4181
|
+
};
|
|
4182
|
+
const withSummary = injectSummary(first.messages, summary);
|
|
4183
|
+
if (estimateMessagesTokens(withSummary) > options.budget) {
|
|
4184
|
+
const truncated = hardTruncate(withSummary, options.budget);
|
|
4185
|
+
return {
|
|
4186
|
+
messages: truncated.messages,
|
|
4187
|
+
summary,
|
|
4188
|
+
tokens: truncated.tokens
|
|
4189
|
+
};
|
|
4190
|
+
}
|
|
4191
|
+
return {
|
|
4192
|
+
messages: withSummary,
|
|
4193
|
+
summary,
|
|
4194
|
+
tokens: estimateMessagesTokens(withSummary)
|
|
4195
|
+
};
|
|
4196
|
+
}
|
|
4197
|
+
/**
|
|
4198
|
+
* Re-insert a summary as: pinned instructions → summary → surviving tail.
|
|
4199
|
+
*
|
|
4200
|
+
* Order matters. Pinned (system/developer) messages must stay ahead of the
|
|
4201
|
+
* summary so that a later synthetic system message can never override the
|
|
4202
|
+
* harness's own instructions; the tail follows so the newest exchange is the
|
|
4203
|
+
* last thing the model reads.
|
|
4204
|
+
*
|
|
4205
|
+
* `compacted` is always derived from `original` by `compactMessages`, so the
|
|
4206
|
+
* pinned messages it carries are exactly the originals — no need to re-add
|
|
4207
|
+
* them from `original`.
|
|
4208
|
+
*/
|
|
4209
|
+
function injectSummary(compacted, summary) {
|
|
4210
|
+
const pinned = [];
|
|
4211
|
+
const rest = [];
|
|
4212
|
+
for (const message of compacted) if (isPinnedRole(message.role)) pinned.push(message);
|
|
4213
|
+
else rest.push(message);
|
|
4214
|
+
return [
|
|
4215
|
+
...pinned,
|
|
4216
|
+
summaryMessage(summary),
|
|
4217
|
+
...rest
|
|
4218
|
+
];
|
|
4219
|
+
}
|
|
4220
|
+
//#endregion
|
|
3771
4221
|
//#region src/shim.ts
|
|
3772
4222
|
/**
|
|
3773
4223
|
* Loopback OpenAI-compatible endpoint with multi-account failover.
|
|
@@ -3846,7 +4296,8 @@ function writeOpenAIError(res, status, kind, message) {
|
|
|
3846
4296
|
}
|
|
3847
4297
|
/** True when an upstream failure body means the request overran the model's
|
|
3848
4298
|
* context window (OpenAI `context_length_exceeded`, WorkBuddy code 11115 /
|
|
3849
|
-
* "input length too long").
|
|
4299
|
+
* "input length too long"). The shim answers it by compacting the conversation
|
|
4300
|
+
* in place and retrying once; see `recoverFromContextOverrun`. */
|
|
3850
4301
|
function isContextTooLong(body) {
|
|
3851
4302
|
if (body.includes("context_length_exceeded")) return true;
|
|
3852
4303
|
if (body.includes("input length too long")) return true;
|
|
@@ -4008,25 +4459,7 @@ function createWorkBuddyShim(options) {
|
|
|
4008
4459
|
tried.push(account.label);
|
|
4009
4460
|
const result = await client.chatStream(account.credential, prepared, controller.signal);
|
|
4010
4461
|
if (result.ok) {
|
|
4011
|
-
|
|
4012
|
-
pool.noteServed(account.id);
|
|
4013
|
-
refreshBalance(account);
|
|
4014
|
-
res.writeHead(200, {
|
|
4015
|
-
"Content-Type": "text/event-stream",
|
|
4016
|
-
"Cache-Control": "no-cache",
|
|
4017
|
-
"Connection": "keep-alive",
|
|
4018
|
-
"X-Accel-Buffering": "no"
|
|
4019
|
-
});
|
|
4020
|
-
let sawDone = false;
|
|
4021
|
-
const body = Readable.fromWeb(result.response.body);
|
|
4022
|
-
body.on("data", (chunk) => {
|
|
4023
|
-
if (chunk.includes("[DONE]")) sawDone = true;
|
|
4024
|
-
});
|
|
4025
|
-
body.on("error", (error) => {
|
|
4026
|
-
logger?.warn("dsh-workbuddy-xdpool: upstream stream failed mid-flight", error);
|
|
4027
|
-
if (!sawDone && res.writable) res.end("data: [DONE]\n\n");
|
|
4028
|
-
});
|
|
4029
|
-
body.pipe(res);
|
|
4462
|
+
await serveSuccessfulStream(res, account, result, logger, refreshBalance, pool);
|
|
4030
4463
|
return;
|
|
4031
4464
|
}
|
|
4032
4465
|
last = {
|
|
@@ -4054,7 +4487,21 @@ function createWorkBuddyShim(options) {
|
|
|
4054
4487
|
return;
|
|
4055
4488
|
}
|
|
4056
4489
|
if (isContextTooLong(last.message)) {
|
|
4057
|
-
|
|
4490
|
+
const recovered = await recoverFromContextOverrun({
|
|
4491
|
+
raw,
|
|
4492
|
+
modelId,
|
|
4493
|
+
controller,
|
|
4494
|
+
region,
|
|
4495
|
+
logger,
|
|
4496
|
+
client,
|
|
4497
|
+
pool,
|
|
4498
|
+
maxAttempts
|
|
4499
|
+
});
|
|
4500
|
+
if (recovered.ok) {
|
|
4501
|
+
await serveSuccessfulStream(res, recovered.account, recovered.result, logger, refreshBalance, pool);
|
|
4502
|
+
return;
|
|
4503
|
+
}
|
|
4504
|
+
writeOpenAIError(res, 400, "context_length_exceeded", contextOverflowMessage(modelId, recovered.detail));
|
|
4058
4505
|
return;
|
|
4059
4506
|
}
|
|
4060
4507
|
writeOpenAIError(res, KIND_STATUS[last.kind], last.kind, `workbuddy upstream ${last.kind} (http ${last.status}) after ${tried.length} account(s) [${tried.join(" → ")}]: ${last.message.slice(0, 400)}`);
|
|
@@ -4070,6 +4517,168 @@ function createWorkBuddyShim(options) {
|
|
|
4070
4517
|
})
|
|
4071
4518
|
};
|
|
4072
4519
|
}
|
|
4520
|
+
/**
|
|
4521
|
+
* Build the overflow message the Harness must recognize.
|
|
4522
|
+
*
|
|
4523
|
+
* This is deliberately NOT free-form prose. `dsh-compaction-basic` decides
|
|
4524
|
+
* whether to compact-and-retry by running the text that reaches it through
|
|
4525
|
+
* `isContextWindowExceededError()` (`@deepseek-ai/dsh-llm`), whose matcher
|
|
4526
|
+
* accepts only specific phrasings:
|
|
4527
|
+
*
|
|
4528
|
+
* - `context_length_exceeded` / `context window exceeded`
|
|
4529
|
+
* - `maximum context length`
|
|
4530
|
+
* - `<input|prompt|request|messages> too large|long for ... context`
|
|
4531
|
+
* - `<input|prompt|request> exceeds the ... context window`
|
|
4532
|
+
*
|
|
4533
|
+
* The obvious friendly sentence ("the conversation exceeds this model's
|
|
4534
|
+
* context window") matches NONE of them, and neither does the WorkBuddy
|
|
4535
|
+
* upstream's own "input length too long" / code 11115. Emitting either meant
|
|
4536
|
+
* the Harness saw an unclassifiable 400, skipped its recovery path, and
|
|
4537
|
+
* surfaced a dead turn — the bug this function exists to prevent.
|
|
4538
|
+
*
|
|
4539
|
+
* The leading clause carries the machine-matched wording; the trailing clause
|
|
4540
|
+
* is what a human reads. Keep both in sync with
|
|
4541
|
+
* `tests/context-overflow-contract.test.ts`.
|
|
4542
|
+
*/
|
|
4543
|
+
function contextOverflowMessage(modelId, detail = "") {
|
|
4544
|
+
return `This model's maximum context length was exceeded: the prompt is too large for ${modelId === void 0 ? "the model" : `model ${modelId}`}, and the conversation could not be compacted in place${detail === "" ? "" : ` (${detail})`}. Compact the conversation, or start a new chat.`;
|
|
4545
|
+
}
|
|
4546
|
+
/**
|
|
4547
|
+
* Serve one already-successful upstream stream as an SSE response.
|
|
4548
|
+
*
|
|
4549
|
+
* Extracted so the context-overrun recovery path reuses the exact same
|
|
4550
|
+
* bookkeeping (noteServed + background balance refresh) as a first-try hit.
|
|
4551
|
+
*/
|
|
4552
|
+
async function serveSuccessfulStream(res, account, result, logger, refreshBalance, pool) {
|
|
4553
|
+
logger?.info?.(`dsh-workbuddy-xdpool: served by ${account.label}`);
|
|
4554
|
+
pool.noteServed(account.id);
|
|
4555
|
+
refreshBalance(account);
|
|
4556
|
+
res.writeHead(200, {
|
|
4557
|
+
"Content-Type": "text/event-stream",
|
|
4558
|
+
"Cache-Control": "no-cache",
|
|
4559
|
+
"Connection": "keep-alive",
|
|
4560
|
+
"X-Accel-Buffering": "no"
|
|
4561
|
+
});
|
|
4562
|
+
let sawDone = false;
|
|
4563
|
+
const body = Readable.fromWeb(result.response.body);
|
|
4564
|
+
body.on("data", (chunk) => {
|
|
4565
|
+
if (chunk.includes("[DONE]")) sawDone = true;
|
|
4566
|
+
});
|
|
4567
|
+
body.on("error", (error) => {
|
|
4568
|
+
logger?.warn("dsh-workbuddy-xdpool: upstream stream failed mid-flight", error);
|
|
4569
|
+
if (!sawDone && res.writable) res.end("data: [DONE]\n\n");
|
|
4570
|
+
});
|
|
4571
|
+
body.pipe(res);
|
|
4572
|
+
}
|
|
4573
|
+
/**
|
|
4574
|
+
* Compact an over-long conversation and retry it once.
|
|
4575
|
+
*
|
|
4576
|
+
* Strategy, in order:
|
|
4577
|
+
* 1. drop the oldest turns, keeping system messages and the newest exchange;
|
|
4578
|
+
* 2. ask the model to summarise the dropped turns and splice that summary in;
|
|
4579
|
+
* 3. hard-truncate as a last resort.
|
|
4580
|
+
*
|
|
4581
|
+
* Returns `ok: false` only when even a truncated prompt still overran — the
|
|
4582
|
+
* caller then surfaces the original actionable 400.
|
|
4583
|
+
*/
|
|
4584
|
+
async function recoverFromContextOverrun(options) {
|
|
4585
|
+
const { raw, modelId, controller, region, logger, client, pool, maxAttempts } = options;
|
|
4586
|
+
const parsed = client.parseChatBody(raw);
|
|
4587
|
+
if (parsed === void 0) return {
|
|
4588
|
+
ok: false,
|
|
4589
|
+
detail: "request body was not parseable JSON"
|
|
4590
|
+
};
|
|
4591
|
+
const rawMessages = parsed["messages"];
|
|
4592
|
+
if (!Array.isArray(rawMessages)) return {
|
|
4593
|
+
ok: false,
|
|
4594
|
+
detail: "request carried no messages array"
|
|
4595
|
+
};
|
|
4596
|
+
const messages = rawMessages.filter((value) => typeof value === "object" && value !== null && !Array.isArray(value));
|
|
4597
|
+
if (messages.length === 0) return {
|
|
4598
|
+
ok: false,
|
|
4599
|
+
detail: "request carried no usable messages"
|
|
4600
|
+
};
|
|
4601
|
+
const overrunTokens = estimateMessagesTokens(messages);
|
|
4602
|
+
const budget = Math.max(512, Math.floor(overrunTokens / 2));
|
|
4603
|
+
logger?.warn(`dsh-workbuddy-xdpool: context overrun on ${modelId ?? "(no model)"} (~${overrunTokens} tokens); compacting to ~${budget} and retrying once`);
|
|
4604
|
+
let summary;
|
|
4605
|
+
let compacted = messages;
|
|
4606
|
+
let compactionDetail = "";
|
|
4607
|
+
try {
|
|
4608
|
+
const summariser = await pool.acquire(modelId, region);
|
|
4609
|
+
if (summariser === void 0) compactionDetail = "no account available to summarise with";
|
|
4610
|
+
else {
|
|
4611
|
+
const outcome = await compactWithSummary(messages, {
|
|
4612
|
+
budget,
|
|
4613
|
+
keepRecent: 6
|
|
4614
|
+
}, { complete: async (request, signal) => {
|
|
4615
|
+
const body = client.buildChatBody({
|
|
4616
|
+
...parsed,
|
|
4617
|
+
stream: true,
|
|
4618
|
+
max_tokens: Math.max(256, Math.floor(budget / 2))
|
|
4619
|
+
}, request);
|
|
4620
|
+
return await client.completeChat(summariser.credential, body, signal ?? controller.signal);
|
|
4621
|
+
} }, controller.signal);
|
|
4622
|
+
compacted = outcome.messages;
|
|
4623
|
+
summary = outcome.summary;
|
|
4624
|
+
if (outcome.skipped !== void 0) compactionDetail = outcome.skipped;
|
|
4625
|
+
}
|
|
4626
|
+
} catch (error) {
|
|
4627
|
+
compactionDetail = `summarisation failed: ${String(error)}`;
|
|
4628
|
+
}
|
|
4629
|
+
if (estimateMessagesTokens(compacted) > budget) compacted = hardTruncate(compacted, budget).messages;
|
|
4630
|
+
if (summary === void 0 && estimateMessagesTokens(compacted) >= overrunTokens) return {
|
|
4631
|
+
ok: false,
|
|
4632
|
+
detail: compactionDetail === "" ? "compaction could not reduce the prompt" : compactionDetail
|
|
4633
|
+
};
|
|
4634
|
+
const retryBody = client.buildChatBody(parsed, compacted);
|
|
4635
|
+
const tried = [];
|
|
4636
|
+
for (let attempt = 0; attempt < maxAttempts; attempt += 1) {
|
|
4637
|
+
if (controller.signal.aborted) return {
|
|
4638
|
+
ok: false,
|
|
4639
|
+
detail: "client disconnected"
|
|
4640
|
+
};
|
|
4641
|
+
const account = await pool.acquire(modelId, region);
|
|
4642
|
+
if (account === void 0) return {
|
|
4643
|
+
ok: false,
|
|
4644
|
+
detail: "no account available after compaction"
|
|
4645
|
+
};
|
|
4646
|
+
tried.push(account.label);
|
|
4647
|
+
const result = await client.chatStream(account.credential, retryBody, controller.signal);
|
|
4648
|
+
if (result.ok) {
|
|
4649
|
+
logger?.info?.(`dsh-workbuddy-xdpool: recovered from context overrun on ${modelId ?? "(no model)"} (summarised: ${summary === void 0 ? "no" : "yes"})`);
|
|
4650
|
+
return {
|
|
4651
|
+
ok: true,
|
|
4652
|
+
account,
|
|
4653
|
+
result
|
|
4654
|
+
};
|
|
4655
|
+
}
|
|
4656
|
+
if (isContextTooLong(result.message)) return {
|
|
4657
|
+
ok: false,
|
|
4658
|
+
detail: "prompt still exceeded the window after compaction"
|
|
4659
|
+
};
|
|
4660
|
+
if (result.kind === "session_dead") {
|
|
4661
|
+
await pool.refreshAccount(account.id);
|
|
4662
|
+
continue;
|
|
4663
|
+
}
|
|
4664
|
+
if (result.kind === "hard_credit") {
|
|
4665
|
+
pool.penalizeExhausted(account.id);
|
|
4666
|
+
continue;
|
|
4667
|
+
}
|
|
4668
|
+
if (result.kind === "soft_rate") {
|
|
4669
|
+
pool.penalize(account.id, parseRateLimitReset(result.message), modelId);
|
|
4670
|
+
continue;
|
|
4671
|
+
}
|
|
4672
|
+
return {
|
|
4673
|
+
ok: false,
|
|
4674
|
+
detail: `upstream ${result.kind} after compaction`
|
|
4675
|
+
};
|
|
4676
|
+
}
|
|
4677
|
+
return {
|
|
4678
|
+
ok: false,
|
|
4679
|
+
detail: `no account served the compacted request (tried ${tried.length})`
|
|
4680
|
+
};
|
|
4681
|
+
}
|
|
4073
4682
|
//#endregion
|
|
4074
4683
|
//#region src/status.ts
|
|
4075
4684
|
/** Build the status document. Never throws. */
|