dsh-local-models 0.3.0 → 0.3.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/lib/index.js +88 -6
- package/package.json +1 -1
- package/skills/dsh-local-models-ops/SKILL.md +24 -0
package/lib/index.js
CHANGED
|
@@ -89,11 +89,25 @@ const MMPROJ_CPU = process.env.LOCAL_MODELS_MMPROJ_CPU !== "0";
|
|
|
89
89
|
* the advertised maxTokens becomes the adapter's defaultMaxTokens, which
|
|
90
90
|
* dsh-compaction-basic uses as its reserved output `O` in `W - O - headroom`.
|
|
91
91
|
* Advertising the whole window (the old 131K) leaves no message budget at
|
|
92
|
-
* `W = O` — proactive compaction warns once and never fires.
|
|
93
|
-
*
|
|
94
|
-
*
|
|
95
|
-
*
|
|
96
|
-
*
|
|
92
|
+
* `W = O` — proactive compaction warns once and never fires.
|
|
93
|
+
*
|
|
94
|
+
* NOTE — the 32K cap alone does NOT buy a late threshold under the core
|
|
95
|
+
* defaults (headroomTokens 65536, thresholdRatio 0.8, retainRatio 0.16; see
|
|
96
|
+
* compactionBudgetFor() below). For a 131072 window with O = 32768 the
|
|
97
|
+
* effective threshold is min(0.8 * W, W - O - headroom) = 32768, i.e. the
|
|
98
|
+
* session starts compacting at ~32-38K pressure tokens (the meter prices
|
|
99
|
+
* tools + system on top of the surface, so the visible trigger lands a few K
|
|
100
|
+
* above the raw threshold) — and with xhigh + preserveThinking every turn
|
|
101
|
+
* adds 10-20K thinking tokens, so compaction fires again on the next step:
|
|
102
|
+
* the "compaction loop". The honest fix for mid-size windows is a per-model
|
|
103
|
+
* headroom override on the agent-preset row that mounts the running engine
|
|
104
|
+
* (`preset-standard` → `plugins` → `compaction` group → `compaction-basic`;
|
|
105
|
+
* headroom ~6-16K moves the 131K threshold to ~82-92K) — a patch on the
|
|
106
|
+
* top-level `compaction-basic` row is dead config (that row is disabled),
|
|
107
|
+
* and a smaller O would truncate the xhigh thinking blocks this cap
|
|
108
|
+
* exists to protect. Heavy xhigh thinking that needs more must raise
|
|
109
|
+
* per-request maxTokens explicitly (which honestly moves the compaction
|
|
110
|
+
* threshold earlier instead of silently disabling it).
|
|
97
111
|
*/
|
|
98
112
|
const MAX_OUTPUT_TOKENS = 32768;
|
|
99
113
|
/** Default output share of the window: never offer more than half as max
|
|
@@ -101,6 +115,71 @@ const MAX_OUTPUT_TOKENS = 32768;
|
|
|
101
115
|
function defaultMaxOutput(contextWindow) {
|
|
102
116
|
return Math.max(1, Math.min(MAX_OUTPUT_TOKENS, Math.floor(contextWindow / 2)));
|
|
103
117
|
}
|
|
118
|
+
/**
|
|
119
|
+
* dsh-compaction-basic defaults this plugin's advertisement is priced
|
|
120
|
+
* against. Duplicated here (not imported) because the core package is not a
|
|
121
|
+
* dependency of this plugin — keep in sync with upstream's resolveConfig().
|
|
122
|
+
*/
|
|
123
|
+
export const COMPACTION_DEFAULTS = {
|
|
124
|
+
thresholdRatio: 0.8,
|
|
125
|
+
headroomTokens: 65536,
|
|
126
|
+
retainRatio: 0.16,
|
|
127
|
+
};
|
|
128
|
+
/**
|
|
129
|
+
* Price one advertised window the way dsh-compaction-basic's
|
|
130
|
+
* resolveCompactSpec() does, so the tab/skill can tell the user WHEN
|
|
131
|
+
* proactive compaction will actually fire — before they hit the loop.
|
|
132
|
+
*
|
|
133
|
+
* messageBudget = W - O (history available to messages)
|
|
134
|
+
* pressure = messageBudget - headroom
|
|
135
|
+
* threshold = min(floor(W * ratio), pressure) (fire at/above this)
|
|
136
|
+
* retain = floor(messageBudget * retainRatio) (tail kept per compact)
|
|
137
|
+
*
|
|
138
|
+
* `viable` is false when the pressure budget is <= 0: proactive compaction
|
|
139
|
+
* is then disabled entirely (core warns once and only recovers on overflow).
|
|
140
|
+
* That is the 96K-window cliff: W = 98304, O = 32768 leaves exactly 0.
|
|
141
|
+
*
|
|
142
|
+
* `recommendedHeadroom` is the headroom override that lands the threshold at
|
|
143
|
+
* ~70% of the window (clamped to [4096, 65536] so a summary + one retry
|
|
144
|
+
* always fit): the value to put in a `modelPolicies` entry for this route.
|
|
145
|
+
* Pure; `maxTokens` defaults to what buildProviderProfile would advertise.
|
|
146
|
+
*/
|
|
147
|
+
export function compactionBudgetFor(contextWindow, opts = {}) {
|
|
148
|
+
const W = Number.isInteger(contextWindow) && contextWindow > 0 ? contextWindow : 8192;
|
|
149
|
+
const O = Number.isInteger(opts.maxTokens) && opts.maxTokens > 0
|
|
150
|
+
? opts.maxTokens
|
|
151
|
+
: defaultMaxOutput(W);
|
|
152
|
+
const ratio = typeof opts.thresholdRatio === "number" && opts.thresholdRatio > 0 && opts.thresholdRatio <= 1
|
|
153
|
+
? opts.thresholdRatio
|
|
154
|
+
: COMPACTION_DEFAULTS.thresholdRatio;
|
|
155
|
+
const headroom = Number.isInteger(opts.headroomTokens) && opts.headroomTokens >= 0
|
|
156
|
+
? opts.headroomTokens
|
|
157
|
+
: COMPACTION_DEFAULTS.headroomTokens;
|
|
158
|
+
const retainRatio = typeof opts.retainRatio === "number" && opts.retainRatio > 0 && opts.retainRatio < ratio
|
|
159
|
+
? opts.retainRatio
|
|
160
|
+
: COMPACTION_DEFAULTS.retainRatio;
|
|
161
|
+
const messageBudget = W - O;
|
|
162
|
+
const pressureBudget = messageBudget - headroom;
|
|
163
|
+
const thresholdTokens = Math.floor(Math.min(W * ratio, pressureBudget));
|
|
164
|
+
const retainTokens = Math.floor(messageBudget * retainRatio);
|
|
165
|
+
// Headroom that would put the threshold at ~70% of the window: the
|
|
166
|
+
// pressure budget must cover 0.7 * W, i.e. headroom <= message - 0.7W.
|
|
167
|
+
const recommendedHeadroom = Math.max(
|
|
168
|
+
4096,
|
|
169
|
+
Math.min(COMPACTION_DEFAULTS.headroomTokens, messageBudget - Math.floor(W * 0.7)),
|
|
170
|
+
);
|
|
171
|
+
return {
|
|
172
|
+
contextWindow: W,
|
|
173
|
+
maxTokens: O,
|
|
174
|
+
headroomTokens: headroom,
|
|
175
|
+
messageBudget,
|
|
176
|
+
pressureBudget,
|
|
177
|
+
thresholdTokens,
|
|
178
|
+
retainTokens,
|
|
179
|
+
viable: pressureBudget > 0 && retainTokens < thresholdTokens,
|
|
180
|
+
recommendedHeadroom,
|
|
181
|
+
};
|
|
182
|
+
}
|
|
104
183
|
/** Fixed-MTP ceiling. The tab offers 0-7 and the API clamps here: upstream
|
|
105
184
|
* accepts any `--spec-draft-n-max` and clamps the effective depth to the
|
|
106
185
|
* model's own nextn depth at load. Depth 3 is the measured 16 GiB sweet spot
|
|
@@ -1634,7 +1713,10 @@ export function buildProviderProfile(st, requestedRoute) {
|
|
|
1634
1713
|
// series is a heavy thinker and at xhigh its thinking block alone
|
|
1635
1714
|
// blows past small ceilings, so 32K is the default — but never more
|
|
1636
1715
|
// than half the window, or compaction's `W - O - headroom` budget
|
|
1637
|
-
// collapses and proactive compaction never fires.
|
|
1716
|
+
// collapses and proactive compaction never fires. NOTE: under the
|
|
1717
|
+
// core defaults (headroom 65536) a 131072 window still thresholds
|
|
1718
|
+
// at ~32K — see compactionBudgetFor(); mid-size windows need a
|
|
1719
|
+
// per-model headroom override, not a smaller O. Need more room
|
|
1638
1720
|
// for a monster thinking block? Raise per-request maxTokens
|
|
1639
1721
|
// explicitly instead.
|
|
1640
1722
|
maxTokens: defaultMaxOutput(contextWindow),
|
package/package.json
CHANGED
|
@@ -105,6 +105,30 @@ The registered model advertises `maxTokens: min(32768, floor(ctx/2))` — a
|
|
|
105
105
|
message budget and proactive compaction never fires. Heavy-thinking models
|
|
106
106
|
at xhigh that need more than 32K must raise per-request maxTokens
|
|
107
107
|
explicitly (which honestly moves the compaction threshold earlier).
|
|
108
|
+
Compaction threshold under the core defaults (headroomTokens 65536,
|
|
109
|
+
thresholdRatio 0.8, retainRatio 0.16) is `min(0.8*W, W - O - headroom)` —
|
|
110
|
+
so a 131072 window with O = 32768 thresholds at **32768**, firing at
|
|
111
|
+
~32-38K pressure tokens and looping with xhigh + preserveThinking (every
|
|
112
|
+
turn re-adds 10-20K thinking tokens). A 98304 window is worse: pressure
|
|
113
|
+
exactly 0, proactive compaction disabled entirely. The fix is a per-model
|
|
114
|
+
headroom override — but placement matters: the engines that run live
|
|
115
|
+
inside the agent-preset scopes (`preset-standard` → `plugins` →
|
|
116
|
+
`compaction` group → `compaction-basic`), NOT the top-level
|
|
117
|
+
`compaction-basic` row (disabled there; a patch on it is dead config that
|
|
118
|
+
dump-config will happily show). So the override must ride a wholesale
|
|
119
|
+
`config` on the preset row (row config is replaced whole, not merged):
|
|
120
|
+
copy the installed preset file
|
|
121
|
+
(`dsh-web-app/presets/standard.patch.yml` → `insert[0].config`), add
|
|
122
|
+
`modelPolicies` to the nested engine, paste as `- id: preset-standard`.
|
|
123
|
+
Compute values with `compactionBudgetFor(ctx).recommendedHeadroom`
|
|
124
|
+
(~6.5K for 131K → ~92K threshold; ~4K floor for 96K → ~61K). Because it
|
|
125
|
+
is wholesale, the row freezes: re-generate after any dsh upgrade that
|
|
126
|
+
touches the preset file, or upstream changes to that preset are silently
|
|
127
|
+
dropped. Only patch presets actually used (`standard` here); `ptc` /
|
|
128
|
+
`cordis` / `minimal` need the same treatment if adopted.
|
|
129
|
+
|
|
130
|
+
Shortcut with zero config edits: run the bigger window instead — the
|
|
131
|
+
200K/250K profiles threshold at ~106K/158K under defaults.
|
|
108
132
|
For opencode against the spawned server:
|
|
109
133
|
- the opencode config resolves per directory (project `opencode.json` beats
|
|
110
134
|
nothing; the global `~/.config/opencode/opencode.json(c)` is authoritative
|