@genee/omp-opsx-addon 0.8.0 → 0.10.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +440 -71
- package/index.ts +1106 -235
- package/lib/agent-defs.ts +80 -89
- package/lib/complexity-router.ts +336 -0
- package/lib/concurrency-control.ts +1246 -0
- package/lib/concurrency-multiproc.worker.ts +376 -0
- package/lib/concurrency-state.ts +397 -0
- package/lib/concurrency-tuner.ts +773 -0
- package/lib/concurrency-writer.ts +894 -0
- package/lib/continuity-guard.ts +117 -0
- package/lib/family-filter.ts +115 -2
- package/lib/model-roles.ts +43 -41
- package/lib/model-selector.ts +300 -36
- package/lib/model-speed.ts +189 -0
- package/lib/pick-model-render.ts +414 -0
- package/lib/pipe-core.ts +303 -93
- package/lib/pipe-push.ts +212 -26
- package/lib/provider-variants.ts +154 -0
- package/lib/reachability-refresh.ts +10 -2
- package/lib/selection-filters.ts +42 -1
- package/lib/stall-detector.ts +185 -0
- package/lib/system-prompt.ts +28 -11
- package/lib/tiers-data.ts +47 -5
- package/lib/tiers-updater.ts +51 -8
- package/lib/unified-config.ts +735 -68
- package/lib/usage-render.ts +57 -4
- package/lib/usage-widget.ts +18 -4
- package/package.json +1 -1
- package/skills/opsx-orchestration-protocol/SKILL.md +87 -0
package/lib/model-selector.ts
CHANGED
|
@@ -2,15 +2,25 @@
|
|
|
2
2
|
* Model selector — picks the best `Model<Api>` for a given role.
|
|
3
3
|
*
|
|
4
4
|
* Sorts by (in order):
|
|
5
|
-
* 1.
|
|
6
|
-
*
|
|
7
|
-
*
|
|
8
|
-
*
|
|
9
|
-
*
|
|
5
|
+
* 1. Coverage class — hard partition, primary key: monthly-plan covered (0)
|
|
6
|
+
* before standard (1) before pay-as-you-go (2). PAYG candidates only
|
|
7
|
+
* reach the top when no plan/standard candidate is available;
|
|
8
|
+
* `paygLastResort: false` collapses this key. Classification is generic
|
|
9
|
+
* (zero hardcoded provider names — see `classifyCoverage`).
|
|
10
|
+
* 2. Effective gap = tierGap + surgePenalty (closest to expected wins).
|
|
11
|
+
* Surge penalty applies only during GLM-5 peak pricing windows — the
|
|
12
|
+
* PAYG-vs-plan trade-off is the partition's job, not a penalty.
|
|
13
|
+
* 3. Rank (higher tier wins).
|
|
10
14
|
* 4. Remaining fraction (healthier first).
|
|
11
|
-
* 5. Cost (
|
|
12
|
-
*
|
|
13
|
-
*
|
|
15
|
+
* 5. Cost band — quantized costInput: max(0, floor(log2)+1) power-of-two
|
|
16
|
+
* bands; band 0 = free/missing price plus clamped <$1 positive prices.
|
|
17
|
+
* Monotone in costInput, so cross-band order ≡ raw cost order (zero drift).
|
|
18
|
+
* 6. Speed inside the band (tokPerS desc; measured before unmeasured;
|
|
19
|
+
* sentinels/unmeasured = -Infinity). `aggressive` mode moves speed ahead
|
|
20
|
+
* of the band (crosses price within the same class + tier axis).
|
|
21
|
+
* 7. Cost raw value (cheaper first) — in-band order and speed-equal fallback.
|
|
22
|
+
* 8. Reputation (higher first).
|
|
23
|
+
* 9. Name (stable tie-break).
|
|
14
24
|
*/
|
|
15
25
|
|
|
16
26
|
import type { Model, Api } from '@oh-my-pi/pi-catalog/types';
|
|
@@ -21,8 +31,10 @@ import {
|
|
|
21
31
|
scoreModelWithProvider,
|
|
22
32
|
tierNameToRank,
|
|
23
33
|
} from './model-tiers.js';
|
|
24
|
-
import type
|
|
34
|
+
import { type ProviderHealth, canonicalizeProvider, quotaPolicyFor } from './usage-resolver.js';
|
|
35
|
+
import { FREE_BAND, SENTINEL_BAND, type SpeedAwareRuntime, type SpeedIndex, speedIndexKey } from './model-speed.js';
|
|
25
36
|
import { getPeakStatus, isGLM5Model, type PeakStatus } from './peak-detector.js';
|
|
37
|
+
import type { QualityPreference } from './complexity-router.js';
|
|
26
38
|
|
|
27
39
|
export interface ScoredCandidate {
|
|
28
40
|
model: Model<Api>;
|
|
@@ -34,12 +46,14 @@ export interface ScoredCandidate {
|
|
|
34
46
|
remainingFraction?: number;
|
|
35
47
|
tierGap: number;
|
|
36
48
|
costInput: number;
|
|
37
|
-
/** 0 =
|
|
38
|
-
|
|
49
|
+
/** Coverage partition: 0 = monthly-plan covered, 1 = standard, 2 = pay-as-you-go. Primary sort key; lower preferred. */
|
|
50
|
+
coverageClass: 0 | 1 | 2;
|
|
39
51
|
/** Coding reputation 0-100 from model family (SWE-bench ranking). Higher preferred. */
|
|
40
52
|
reputation: number;
|
|
41
53
|
/** Surge pricing penalty (0 = normal, 1+ = penalized). Added to tierGap for sorting. */
|
|
42
54
|
surgePenalty: number;
|
|
55
|
+
/** provider_auto sentinel: heuristic scoring bypassed for the declared tier; dedup-exempt; costInput = +∞ (cost-neutral). */
|
|
56
|
+
isProviderAuto: boolean;
|
|
43
57
|
}
|
|
44
58
|
export interface SelectorResult {
|
|
45
59
|
role: string;
|
|
@@ -62,6 +76,26 @@ export interface SelectorInput {
|
|
|
62
76
|
planProviders?: Set<string>;
|
|
63
77
|
/** Pre-computed peak status (for testing). If omitted, computed from current time. */
|
|
64
78
|
peakStatus?: PeakStatus;
|
|
79
|
+
/** Keep pay-as-you-go candidates as last resort (coverage partition active). Default true. */
|
|
80
|
+
paygLastResort?: boolean;
|
|
81
|
+
/** provider → 服务端路由哨兵模型 id(键已 canonicalizeProvider 规范化,值与 catalog id 大小写不敏感匹配)。 */
|
|
82
|
+
providerAuto?: Record<string, string>;
|
|
83
|
+
/** provider → 哨兵声明档位(解析层已过 TIER_NAMES 校验);未声明的哨兵按 mid。 */
|
|
84
|
+
providerAutoTiers?: Record<string, TierName>;
|
|
85
|
+
/** 速度感知 runtime(mode + 已载入索引);缺省/禁用时排序与 W1 定型序逐字节一致。 */
|
|
86
|
+
speedAware?: SpeedAwareRuntime;
|
|
87
|
+
/**
|
|
88
|
+
* 质量带旋钮(W4,design D6):balanced/缺省 → 既有分支逐字节(零漂移的
|
|
89
|
+
* 结构保证是「不进入」新分支,而非行为等价的新路径);cost → tier 轴槽位
|
|
90
|
+
* 内部容差带量化;quality → 质量分前缀分层。槽位不移动、既有键相对次序
|
|
91
|
+
* 不改动(model-selection-ordering requirement 为跨引用权威约束)。
|
|
92
|
+
*/
|
|
93
|
+
qualityPreference?: QualityPreference;
|
|
94
|
+
/**
|
|
95
|
+
* 数据层标注段(W4 路由,调用方组装,如 `band=complex, stall+1`):纯追加
|
|
96
|
+
* 到 pickedReason 与 decision 的 picked 行;缺省零追加(W3 speed 段同款)。
|
|
97
|
+
*/
|
|
98
|
+
note?: string;
|
|
65
99
|
}
|
|
66
100
|
|
|
67
101
|
function globToRegex(pattern: string): RegExp {
|
|
@@ -114,10 +148,54 @@ export const DEFAULT_PLAN_PROVIDERS = new Set([
|
|
|
114
148
|
'kimi',
|
|
115
149
|
'openai-codex',
|
|
116
150
|
'cursor',
|
|
151
|
+
'ark-coding-plan',
|
|
117
152
|
]);
|
|
118
153
|
|
|
119
|
-
|
|
120
|
-
|
|
154
|
+
/** Coverage partition classes: 0 = monthly-plan covered, 1 = standard, 2 = pay-as-you-go. */
|
|
155
|
+
export type CoverageClass = 0 | 1 | 2;
|
|
156
|
+
|
|
157
|
+
/**
|
|
158
|
+
* Coverage class of a provider: plan-set member → 0 (plan);
|
|
159
|
+
* `quotaPolicyFor() === 'balance'` (pay-as-you-go) → 2; everything else → 1.
|
|
160
|
+
* Generic by design — no provider names here. Both sides are canonicalized so
|
|
161
|
+
* `ArkCodingPlan` / `arkcodingplan` spellings hit `ark-coding-plan`. An
|
|
162
|
+
* undefined plan set is treated as empty.
|
|
163
|
+
*/
|
|
164
|
+
export function classifyCoverage(
|
|
165
|
+
provider: string,
|
|
166
|
+
planProviders?: ReadonlySet<string>,
|
|
167
|
+
): CoverageClass {
|
|
168
|
+
const canonical = canonicalizeProvider(provider);
|
|
169
|
+
if (planProviders) {
|
|
170
|
+
for (const entry of planProviders) {
|
|
171
|
+
if (canonicalizeProvider(entry) === canonical) return 0;
|
|
172
|
+
}
|
|
173
|
+
}
|
|
174
|
+
return quotaPolicyFor(canonical) === 'balance' ? 2 : 1;
|
|
175
|
+
}
|
|
176
|
+
|
|
177
|
+
/** Human-readable coverage class (`class=plan|standard|payg`) for decision logs and /pick-model choices rendering. */
|
|
178
|
+
export const COVERAGE_LABEL: Record<CoverageClass, 'plan' | 'standard' | 'payg'> = {
|
|
179
|
+
0: 'plan',
|
|
180
|
+
1: 'standard',
|
|
181
|
+
2: 'payg',
|
|
182
|
+
};
|
|
183
|
+
|
|
184
|
+
export function mergePlanProviders(
|
|
185
|
+
extra?: readonly string[],
|
|
186
|
+
remove?: readonly string[],
|
|
187
|
+
): Set<string> {
|
|
188
|
+
const out = new Set<string>();
|
|
189
|
+
for (const entry of [...DEFAULT_PLAN_PROVIDERS, ...(extra ?? [])]) {
|
|
190
|
+
out.add(canonicalizeProvider(entry));
|
|
191
|
+
}
|
|
192
|
+
if (remove?.length) {
|
|
193
|
+
const removed = new Set(remove.map(canonicalizeProvider));
|
|
194
|
+
for (const entry of out) {
|
|
195
|
+
if (removed.has(entry)) out.delete(entry);
|
|
196
|
+
}
|
|
197
|
+
}
|
|
198
|
+
return out;
|
|
121
199
|
}
|
|
122
200
|
|
|
123
201
|
function isAllowlisted(selector: string, allowlist: string[] | undefined): boolean {
|
|
@@ -166,7 +244,16 @@ function isNewer(a: [number, number], b: [number, number]): boolean {
|
|
|
166
244
|
* Prefers latest generation first, then reputation as tiebreaker. */
|
|
167
245
|
function dedupByProviderTier(candidates: ScoredCandidate[]): ScoredCandidate[] {
|
|
168
246
|
const best = new Map<string, ScoredCandidate>();
|
|
247
|
+
const autos: ScoredCandidate[] = [];
|
|
169
248
|
for (const c of candidates) {
|
|
249
|
+
// Sentinels skip the (provider, tier) contest entirely: version-less ids
|
|
250
|
+
// (`default` / `ark-code-latest`) extract [0,0] and would always be
|
|
251
|
+
// evicted by any concrete model. At most one sentinel per provider
|
|
252
|
+
// (provider_auto maps provider → single id), so no flooding risk.
|
|
253
|
+
if (c.isProviderAuto) {
|
|
254
|
+
autos.push(c);
|
|
255
|
+
continue;
|
|
256
|
+
}
|
|
170
257
|
const key = `${c.model.provider}:${c.tier}`;
|
|
171
258
|
const existing = best.get(key);
|
|
172
259
|
if (!existing) {
|
|
@@ -185,20 +272,84 @@ function dedupByProviderTier(candidates: ScoredCandidate[]): ScoredCandidate[] {
|
|
|
185
272
|
}
|
|
186
273
|
}
|
|
187
274
|
}
|
|
188
|
-
return [...best.values()];
|
|
275
|
+
return [...best.values(), ...autos];
|
|
276
|
+
}
|
|
277
|
+
|
|
278
|
+
/**
|
|
279
|
+
* Cost band quantization (frozen slot — change: speed-aware-selection):
|
|
280
|
+
* `max(0, floor(log2(c)) + 1)` for c > 0. The clamp keeps every band ≥ 0 and
|
|
281
|
+
* monotone in costInput, so any cross-band pair orders exactly like the raw
|
|
282
|
+
* costInput comparison — the unquantized W1 order never regresses. c ≤ 0
|
|
283
|
+
* (free / missing price) and clamp-collected 0 < c < 1 positive prices share
|
|
284
|
+
* band 0 (raw costInput still orders inside: free, then $0.4, then $0.99);
|
|
285
|
+
* +∞ (provider-auto sentinel) lands in the last band, behind everything.
|
|
286
|
+
*/
|
|
287
|
+
export function costBand(costInput: number): number {
|
|
288
|
+
if (costInput <= 0) return FREE_BAND;
|
|
289
|
+
if (!Number.isFinite(costInput)) return SENTINEL_BAND;
|
|
290
|
+
return Math.max(0, Math.floor(Math.log2(costInput)) + 1);
|
|
291
|
+
}
|
|
292
|
+
|
|
293
|
+
/**
|
|
294
|
+
* Quality-knob cost-mode tolerance (W4, design D6; module constant — W3 band
|
|
295
|
+
* 先例, not configurable): candidates within T of the per-class minimal
|
|
296
|
+
* effective tier-axis value share band 0 and compare de-axised (Azure「带内
|
|
297
|
+
* 等质,选最便宜」的容差带类比,披露于 README); beyond T is band 1, sorted by
|
|
298
|
+
* the full frozen chain. Band 0 always precedes band 1 (cross-band quality
|
|
299
|
+
* order preserved).
|
|
300
|
+
*/
|
|
301
|
+
export const QUALITY_COST_TOLERANCE = 1;
|
|
302
|
+
|
|
303
|
+
/**
|
|
304
|
+
* Per-candidate speed sort key: measured tokPerS, else -Infinity (identity
|
|
305
|
+
* lower bound — measured beats unmeasured inside a band; unmeasured
|
|
306
|
+
* candidates keep their relative W1 order via band/costInput). The sentinel
|
|
307
|
+
* guard lives on the candidate side, not the data side: a black-box
|
|
308
|
+
* provider-auto route must not score even if model_perf happens to hold a
|
|
309
|
+
* row for its id.
|
|
310
|
+
*/
|
|
311
|
+
function speedKeyOf(c: ScoredCandidate, index: SpeedIndex): number {
|
|
312
|
+
if (c.isProviderAuto) return Number.NEGATIVE_INFINITY;
|
|
313
|
+
return index.get(speedIndexKey(`${c.model.provider}/${c.model.id}`))?.tokPerS ?? Number.NEGATIVE_INFINITY;
|
|
189
314
|
}
|
|
190
315
|
|
|
191
316
|
export function selectModel(input: SelectorInput): SelectorResult {
|
|
192
|
-
const { role, models, tierOverrides, health, allowlist, planProviders, peakStatus } = input;
|
|
317
|
+
const { role, models, tierOverrides, health, allowlist, planProviders, peakStatus, paygLastResort, providerAuto, providerAutoTiers, speedAware, qualityPreference, note } = input;
|
|
193
318
|
const roleTiers = input.roleTiers ?? DEFAULT_ROLE_TIER;
|
|
194
319
|
const expectedTier: TierName = roleTiers[role] ?? DEFAULT_ROLE_TIER[role] ?? 'mid';
|
|
195
320
|
const expectedRank = tierNameToRank(expectedTier);
|
|
196
321
|
|
|
197
322
|
const allCandidates: ScoredCandidate[] = models.map((model) => {
|
|
198
323
|
const selector = `${model.provider}/${model.id}`;
|
|
199
|
-
const score = scoreModelWithProvider(model.provider, model.id, model.name, tierOverrides);
|
|
200
324
|
const h = health?.get(model.provider);
|
|
201
325
|
const quota = candidateQuota(model, h);
|
|
326
|
+
const autoKey = providerAuto ? canonicalizeProvider(model.provider) : undefined;
|
|
327
|
+
const autoId = providerAuto && autoKey ? providerAuto[autoKey] : undefined;
|
|
328
|
+
if (autoId !== undefined && model.id.toLowerCase() === autoId.toLowerCase()) {
|
|
329
|
+
// provider_auto sentinel: skip the local heuristic entirely — the
|
|
330
|
+
// declared tier carries rank/tierGap (its only legitimate edge),
|
|
331
|
+
// reputation stays 0 ("no local standing"), and +∞ costInput
|
|
332
|
+
// yields to every real-cost candidate within the same coverage
|
|
333
|
+
// class + tier: the catalog's all-zero cost must not fake a win.
|
|
334
|
+
const tier = (autoKey ? providerAutoTiers?.[autoKey] : undefined) ?? 'mid';
|
|
335
|
+
const rank = tierNameToRank(tier);
|
|
336
|
+
return {
|
|
337
|
+
model,
|
|
338
|
+
selector,
|
|
339
|
+
rank,
|
|
340
|
+
tier,
|
|
341
|
+
reason: `provider-auto (tier=${tier})`,
|
|
342
|
+
exhausted: quota.exhausted,
|
|
343
|
+
remainingFraction: quota.remainingFraction,
|
|
344
|
+
tierGap: Math.abs(rank - expectedRank),
|
|
345
|
+
costInput: Number.POSITIVE_INFINITY,
|
|
346
|
+
coverageClass: classifyCoverage(model.provider, planProviders),
|
|
347
|
+
reputation: 0,
|
|
348
|
+
surgePenalty: 0,
|
|
349
|
+
isProviderAuto: true,
|
|
350
|
+
};
|
|
351
|
+
}
|
|
352
|
+
const score = scoreModelWithProvider(model.provider, model.id, model.name, tierOverrides);
|
|
202
353
|
return {
|
|
203
354
|
model,
|
|
204
355
|
selector,
|
|
@@ -209,9 +360,10 @@ export function selectModel(input: SelectorInput): SelectorResult {
|
|
|
209
360
|
remainingFraction: quota.remainingFraction,
|
|
210
361
|
tierGap: Math.abs(score.rank - expectedRank),
|
|
211
362
|
costInput: model.cost?.input ?? 0,
|
|
212
|
-
|
|
363
|
+
coverageClass: classifyCoverage(model.provider, planProviders),
|
|
213
364
|
reputation: score.reputation,
|
|
214
365
|
surgePenalty: 0,
|
|
366
|
+
isProviderAuto: false,
|
|
215
367
|
};
|
|
216
368
|
});
|
|
217
369
|
|
|
@@ -221,46 +373,148 @@ export function selectModel(input: SelectorInput): SelectorResult {
|
|
|
221
373
|
// from flooding the candidate pool against the latest (M3).
|
|
222
374
|
const deduped = dedupByProviderTier(filtered);
|
|
223
375
|
// ── surge pricing penalty ──────────────────────────────────────
|
|
224
|
-
//
|
|
376
|
+
// GLM-5 peak-time pricing adjustment only: the PAYG-vs-plan trade-off is
|
|
377
|
+
// handled by the coverage-class partition, not by surge penalties.
|
|
225
378
|
const peak = peakStatus ?? getPeakStatus();
|
|
226
|
-
const hasMonthlyAvailable = deduped.some(
|
|
227
|
-
(c) => planProviders?.has(c.model.provider) && !c.exhausted,
|
|
228
|
-
);
|
|
229
379
|
for (const c of deduped) {
|
|
230
|
-
const isDS = c.model.provider === 'deepseek';
|
|
231
380
|
const isGLM5 = c.model.provider === 'zhipu-coding-plan' && isGLM5Model(c.model.id);
|
|
232
|
-
|
|
233
|
-
if (isDS && peak.deepseekPeak) penalty = Math.max(penalty, 2);
|
|
234
|
-
if (isGLM5 && peak.glm5Peak) penalty = Math.max(penalty, 2);
|
|
235
|
-
// When any monthly-plan provider has available quota, heavily
|
|
236
|
-
// penalize DeepSeek (PAYG) to preserve plan budget.
|
|
237
|
-
if (isDS && hasMonthlyAvailable) penalty = Math.max(penalty, 3);
|
|
238
|
-
c.surgePenalty = penalty;
|
|
381
|
+
c.surgePenalty = isGLM5 && peak.glm5Peak ? 2 : 0;
|
|
239
382
|
}
|
|
240
383
|
|
|
241
384
|
|
|
242
385
|
|
|
386
|
+
// ── W4 quality_preference knob (design D6) ─────────────────────
|
|
387
|
+
// balanced/absent → knobActive stays false and every structure below is
|
|
388
|
+
// inert: the frozen chain in the sort lambda runs byte-identically (the
|
|
389
|
+
// zero-drift guarantee is "not entering" the knob path).
|
|
390
|
+
const knobActive = qualityPreference === 'cost' || qualityPreference === 'quality';
|
|
391
|
+
// Cost mode anchor: per coverage class (fixed 0|1|2 slots), the minimal
|
|
392
|
+
// effective tier-axis value in the deduped pool (W3 costBand 量化的同构
|
|
393
|
+
// 先例 — slot-internal quantization, one pre-scan).
|
|
394
|
+
const tierAnchors: (number | undefined)[] = [undefined, undefined, undefined];
|
|
395
|
+
if (qualityPreference === 'cost') {
|
|
396
|
+
for (const c of deduped) {
|
|
397
|
+
const eff = c.tierGap + c.surgePenalty;
|
|
398
|
+
const anchor = tierAnchors[c.coverageClass];
|
|
399
|
+
if (anchor === undefined || eff < anchor) tierAnchors[c.coverageClass] = eff;
|
|
400
|
+
}
|
|
401
|
+
}
|
|
402
|
+
// Tier-axis band: ≤ T above the class anchor → band 0 (de-axised), else band 1.
|
|
403
|
+
const tierBandOf = (c: ScoredCandidate): number => {
|
|
404
|
+
const anchor = tierAnchors[c.coverageClass] ?? 0;
|
|
405
|
+
return c.tierGap + c.surgePenalty - anchor <= QUALITY_COST_TOLERANCE ? 0 : 1;
|
|
406
|
+
};
|
|
407
|
+
// Frozen-chain tail shared by both knob branches: the keys AFTER the tier
|
|
408
|
+
// axis in the W1-W3 order (remainingFraction → [cost band → speed] →
|
|
409
|
+
// costInput raw asc → reputation → selector). includeReputation=false for
|
|
410
|
+
// the quality prefix (reputation already consumed by the prefix keys).
|
|
411
|
+
// Allocated only when the knob is active — the legacy path allocates nothing.
|
|
412
|
+
const knobTail = knobActive
|
|
413
|
+
? (x: ScoredCandidate, y: ScoredCandidate, includeReputation: boolean): number => {
|
|
414
|
+
const xr = x.remainingFraction ?? 1;
|
|
415
|
+
const yr = y.remainingFraction ?? 1;
|
|
416
|
+
if (xr !== yr) return yr - xr;
|
|
417
|
+
if (speedAware) {
|
|
418
|
+
const bandX = costBand(x.costInput);
|
|
419
|
+
const bandY = costBand(y.costInput);
|
|
420
|
+
const speedX = speedKeyOf(x, speedAware.index);
|
|
421
|
+
const speedY = speedKeyOf(y, speedAware.index);
|
|
422
|
+
if (speedAware.mode === 'aggressive') {
|
|
423
|
+
if (speedX !== speedY) return speedY - speedX;
|
|
424
|
+
if (bandX !== bandY) return bandX - bandY;
|
|
425
|
+
} else {
|
|
426
|
+
if (bandX !== bandY) return bandX - bandY;
|
|
427
|
+
if (speedX !== speedY) return speedY - speedX;
|
|
428
|
+
}
|
|
429
|
+
}
|
|
430
|
+
if (x.costInput !== y.costInput) return x.costInput - y.costInput;
|
|
431
|
+
if (includeReputation && x.reputation !== y.reputation) return y.reputation - x.reputation;
|
|
432
|
+
return x.selector.localeCompare(y.selector);
|
|
433
|
+
}
|
|
434
|
+
: null;
|
|
435
|
+
|
|
243
436
|
let picked: ScoredCandidate | null = null;
|
|
244
437
|
if (deduped.length > 0) {
|
|
245
438
|
const sorted = [...deduped].sort((a, b) => {
|
|
439
|
+
// Coverage class is the primary key: a hard partition — PAYG only
|
|
440
|
+
// wins when no plan/standard candidate is available. When
|
|
441
|
+
// paygLastResort is false the partition is collapsed (every
|
|
442
|
+
// candidate compares equal on this key).
|
|
443
|
+
if ((paygLastResort ?? true) && a.coverageClass !== b.coverageClass) {
|
|
444
|
+
return a.coverageClass - b.coverageClass;
|
|
445
|
+
}
|
|
446
|
+
if (knobTail && qualityPreference === 'quality') {
|
|
447
|
+
// 质量分前缀分层(design D6):(effGap, rank, reputation desc)
|
|
448
|
+
// 提入前缀(coverageClass 已在上方消费)——成本/速度无法越过一个
|
|
449
|
+
// 轴位或声誉差距(Azure Quality「无视成本」)。rank/reputation 已
|
|
450
|
+
// 由前缀消费,分区内不再重复;分区内剩余键保持原链子序列。
|
|
451
|
+
const effA = a.tierGap + a.surgePenalty;
|
|
452
|
+
const effB = b.tierGap + b.surgePenalty;
|
|
453
|
+
if (effA !== effB) return effA - effB;
|
|
454
|
+
if (a.rank !== b.rank) return b.rank - a.rank;
|
|
455
|
+
if (a.reputation !== b.reputation) return b.reputation - a.reputation;
|
|
456
|
+
return knobTail(a, b, false);
|
|
457
|
+
}
|
|
458
|
+
if (knobTail && qualityPreference === 'cost') {
|
|
459
|
+
const bandDelta = tierBandOf(a) - tierBandOf(b);
|
|
460
|
+
if (bandDelta !== 0) return bandDelta; // 带 0 恒先于带 1(跨带质量序保持)
|
|
461
|
+
if (tierBandOf(a) === 0) {
|
|
462
|
+
// 带 0 内去轴化(Azure「带内等质,选最便宜」字面落地,README
|
|
463
|
+
// 披露):tierGap/surge/rank 不参与——带内质量等价,成本/速度
|
|
464
|
+
// 决胜。哨兵按声明档正常参与量化(effGap 用声明档,无特判)。
|
|
465
|
+
return knobTail(a, b, true);
|
|
466
|
+
}
|
|
467
|
+
// 带 1 内:全冻结链自洽(落到下方既有链)。
|
|
468
|
+
}
|
|
246
469
|
const effA = a.tierGap + a.surgePenalty;
|
|
247
470
|
const effB = b.tierGap + b.surgePenalty;
|
|
248
471
|
if (effA !== effB) return effA - effB;
|
|
249
472
|
if (a.rank !== b.rank) return b.rank - a.rank;
|
|
250
|
-
// Same effective tier: prefer monthly plan, then cheapest model
|
|
251
|
-
if (a.planPriority !== b.planPriority) return a.planPriority - b.planPriority;
|
|
252
473
|
const ar = a.remainingFraction ?? 1;
|
|
253
474
|
const br = b.remainingFraction ?? 1;
|
|
254
475
|
if (ar !== br) return br - ar;
|
|
255
|
-
if (
|
|
476
|
+
if (speedAware) {
|
|
477
|
+
// Speed-aware selection fills the frozen cost-band + speed
|
|
478
|
+
// slots (slots never move). Keys are plain per-candidate
|
|
479
|
+
// scalars — this lexicographic chain is a strict weak
|
|
480
|
+
// ordering; "both measured" pairwise gating is NOT (three
|
|
481
|
+
// same-band candidates cycle under it, design D4).
|
|
482
|
+
const bandA = costBand(a.costInput);
|
|
483
|
+
const bandB = costBand(b.costInput);
|
|
484
|
+
const speedA = speedKeyOf(a, speedAware.index);
|
|
485
|
+
const speedB = speedKeyOf(b, speedAware.index);
|
|
486
|
+
if (speedAware.mode === 'aggressive') {
|
|
487
|
+
// Speed crosses price bands within the same class + tier axis.
|
|
488
|
+
if (speedA !== speedB) return speedB - speedA;
|
|
489
|
+
if (bandA !== bandB) return bandA - bandB;
|
|
490
|
+
} else {
|
|
491
|
+
if (bandA !== bandB) return bandA - bandB;
|
|
492
|
+
if (speedA !== speedB) return speedB - speedA;
|
|
493
|
+
}
|
|
494
|
+
// Same band + same speed (incl. both unmeasured): raw cost wins.
|
|
495
|
+
if (a.costInput !== b.costInput) return a.costInput - b.costInput;
|
|
496
|
+
} else if (a.costInput !== b.costInput) {
|
|
497
|
+
// No speed runtime: byte-identical to the frozen W1 order.
|
|
498
|
+
return a.costInput - b.costInput;
|
|
499
|
+
}
|
|
256
500
|
if (a.reputation !== b.reputation) return b.reputation - a.reputation;
|
|
257
501
|
return a.selector.localeCompare(b.selector);
|
|
258
502
|
});
|
|
259
503
|
picked = sorted[0] ?? null;
|
|
260
504
|
}
|
|
261
|
-
|
|
505
|
+
// Data-layer observability only (rendering never carries pickedReason):
|
|
506
|
+
// annotate when the picked candidate is measured — never for sentinels,
|
|
507
|
+
// never when speed awareness is off.
|
|
508
|
+
const pickedSpeed = picked && speedAware && !picked.isProviderAuto
|
|
509
|
+
? speedAware.index.get(speedIndexKey(`${picked.model.provider}/${picked.model.id}`))
|
|
510
|
+
: undefined;
|
|
511
|
+
const speedNote = pickedSpeed ? ` speed=${pickedSpeed.tokPerS.toFixed(1)}tok/s` : '';
|
|
512
|
+
// W4 data-layer annotation (caller-composed, e.g. `band=complex, stall+1`):
|
|
513
|
+
// pure append — absent note keeps the W3 format byte-identical.
|
|
514
|
+
const routeNote = note?.trim() ? ` ${note.trim()}` : '';
|
|
515
|
+
const decision = buildDecisionLog(role, expectedTier, allCandidates, deduped, picked, speedNote, routeNote);
|
|
262
516
|
const pickedReason = picked
|
|
263
|
-
? `${picked.selector} (tier=${picked.tier}, gap=${picked.tierGap}${picked.surgePenalty > 0 ? `+${picked.surgePenalty}` : ''})`
|
|
517
|
+
? `${picked.selector} (tier=${picked.tier}, gap=${picked.tierGap}${picked.surgePenalty > 0 ? `+${picked.surgePenalty}` : ''}, class=${COVERAGE_LABEL[picked.coverageClass]}${picked.isProviderAuto ? ', auto' : ''}${pickedSpeed ? `, speed=${pickedSpeed.tokPerS.toFixed(1)}tok/s` : ''}${routeNote ? `,${routeNote}` : ''})`
|
|
264
518
|
: null;
|
|
265
519
|
|
|
266
520
|
return { role, expectedTier, picked: picked?.model ?? null, candidates: deduped, allCandidates, decision, pickedReason };
|
|
@@ -269,6 +523,7 @@ export function selectModel(input: SelectorInput): SelectorResult {
|
|
|
269
523
|
function buildDecisionLog(
|
|
270
524
|
role: string, expectedTier: TierName,
|
|
271
525
|
all: ScoredCandidate[], filtered: ScoredCandidate[], picked: ScoredCandidate | null,
|
|
526
|
+
speedNote: string, routeNote: string = '',
|
|
272
527
|
): string {
|
|
273
528
|
const total = all.length;
|
|
274
529
|
const exhausted = all.filter((c) => c.exhausted).length;
|
|
@@ -277,7 +532,7 @@ function buildDecisionLog(
|
|
|
277
532
|
return [
|
|
278
533
|
`role=${role}: expected tier=${expectedTier}`,
|
|
279
534
|
`${total} available, ${exhausted} exhausted, ${filtered.length} candidates`,
|
|
280
|
-
`picked=${picked.selector} tier=${picked.tier} gap=${picked.tierGap}`,
|
|
535
|
+
`picked=${picked.selector} tier=${picked.tier} class=${COVERAGE_LABEL[picked.coverageClass]} gap=${picked.tierGap}${picked.isProviderAuto ? ' auto' : ''}${speedNote}${routeNote}`,
|
|
281
536
|
`reason=${picked.reason}`,
|
|
282
537
|
].join(' | ');
|
|
283
538
|
}
|
|
@@ -290,10 +545,19 @@ export function selectAllRoles(
|
|
|
290
545
|
allowlist: string[] | undefined,
|
|
291
546
|
planProviders?: Set<string>,
|
|
292
547
|
peakStatus?: PeakStatus,
|
|
548
|
+
paygLastResort?: boolean,
|
|
549
|
+
providerAuto?: Record<string, string>,
|
|
550
|
+
providerAutoTiers?: Record<string, TierName>,
|
|
551
|
+
/** Single trailing object (no more loose scalars); absent = W1 byte-stable order. */
|
|
552
|
+
speedAware?: SpeedAwareRuntime,
|
|
553
|
+
/** W4 质量带旋钮(透传 selectModel;缺省 = 既有分支逐字节)。 */
|
|
554
|
+
qualityPreference?: QualityPreference,
|
|
555
|
+
/** W4 数据层标注段(透传 selectModel;缺省零追加)。 */
|
|
556
|
+
note?: string,
|
|
293
557
|
): Record<string, SelectorResult> {
|
|
294
558
|
const out: Record<string, SelectorResult> = {};
|
|
295
559
|
for (const role of roles) {
|
|
296
|
-
out[role] = selectModel({ role, models, roleTiers, tierOverrides, health, allowlist, planProviders, peakStatus });
|
|
560
|
+
out[role] = selectModel({ role, models, roleTiers, tierOverrides, health, allowlist, planProviders, peakStatus, paygLastResort, providerAuto, providerAutoTiers, speedAware, qualityPreference, note });
|
|
297
561
|
}
|
|
298
562
|
return out;
|
|
299
563
|
}
|
|
@@ -0,0 +1,189 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Speed-aware selection — read-only consumer of the host `agent.db`
|
|
3
|
+
* `model_perf` table (change: speed-aware-selection).
|
|
4
|
+
*
|
|
5
|
+
* Data plane: `model_perf(model_key TEXT PRIMARY KEY, samples REAL,
|
|
6
|
+
* output_tokens REAL, gen_ms REAL, ttft_samples REAL, ttft_ms REAL,
|
|
7
|
+
* updated_at INTEGER)`. Cumulative counters per model; tok/s =
|
|
8
|
+
* `output_tokens / gen_ms × 1000`, ttft = `ttft_ms / ttft_samples` (0 when no
|
|
9
|
+
* TTFT data). The provider segment of `model_key` keeps its catalog alias
|
|
10
|
+
* verbatim (`ArkCodingPlan/...`), so index keys are normalized through
|
|
11
|
+
* `canonicalizeProvider` + lowercase id — the same convention the selector
|
|
12
|
+
* uses on the candidate side.
|
|
13
|
+
*
|
|
14
|
+
* Read path is event-driven only (no timer, no background task): the first
|
|
15
|
+
* enabled selection opens the DB once, and a module-level TTL cache (300s)
|
|
16
|
+
* serves every selection in between. Read failures (missing db / missing
|
|
17
|
+
* table / corrupt file) degrade to an empty index with the same negative
|
|
18
|
+
* TTL — silently, by design: an optional data source must never pollute the
|
|
19
|
+
* selection path's log. Zero writers: the DB opens `readonly` and every
|
|
20
|
+
* statement is a SELECT. This module deliberately does NOT import
|
|
21
|
+
* concurrency-control (independent read-only consumer, no file coupling).
|
|
22
|
+
*/
|
|
23
|
+
|
|
24
|
+
import { Database } from 'bun:sqlite';
|
|
25
|
+
import { homedir } from 'node:os';
|
|
26
|
+
import { join } from 'node:path';
|
|
27
|
+
import { canonicalizeProvider } from './usage-resolver.js';
|
|
28
|
+
|
|
29
|
+
/** Aggregated speed entry for one model (tokPerS rounded by the consumer). */
|
|
30
|
+
export interface SpeedEntry {
|
|
31
|
+
/** Output tokens per second: sum(output_tokens)/sum(gen_ms)×1000. */
|
|
32
|
+
tokPerS: number;
|
|
33
|
+
/** Mean time-to-first-token in ms; 0 when the row carries no TTFT data. */
|
|
34
|
+
ttftMs: number;
|
|
35
|
+
/** Sample count (post min_samples filter — always ≥ the threshold). */
|
|
36
|
+
samples: number;
|
|
37
|
+
}
|
|
38
|
+
|
|
39
|
+
/** Normalized model key → measured speed. */
|
|
40
|
+
export type SpeedIndex = ReadonlyMap<string, SpeedEntry>;
|
|
41
|
+
|
|
42
|
+
/** Comparison modes for the speed slot (see model-selector). */
|
|
43
|
+
export const SPEED_MODES = ['tie-break', 'aggressive'] as const;
|
|
44
|
+
export type SpeedMode = (typeof SPEED_MODES)[number];
|
|
45
|
+
|
|
46
|
+
export interface SpeedAwareConfig {
|
|
47
|
+
enabled: boolean;
|
|
48
|
+
mode: SpeedMode;
|
|
49
|
+
minSamples: number;
|
|
50
|
+
}
|
|
51
|
+
|
|
52
|
+
/** Everything the selector needs: the mode plus the loaded index. */
|
|
53
|
+
export interface SpeedAwareRuntime {
|
|
54
|
+
mode: SpeedMode;
|
|
55
|
+
index: SpeedIndex;
|
|
56
|
+
}
|
|
57
|
+
|
|
58
|
+
/** Rows below this sample count stay unmeasured (load-time filter). */
|
|
59
|
+
export const DEFAULT_MIN_SAMPLES = 20;
|
|
60
|
+
/** Module-level cache lifetime; expired → reload on the next selection. */
|
|
61
|
+
export const SPEED_CACHE_TTL_MS = 300_000;
|
|
62
|
+
|
|
63
|
+
/** Cost-band quantization constants (selector imports these — frozen slots). */
|
|
64
|
+
export const FREE_BAND = 0;
|
|
65
|
+
export const SENTINEL_BAND = Number.MAX_SAFE_INTEGER;
|
|
66
|
+
|
|
67
|
+
/**
|
|
68
|
+
* Normalize a model key to the index/candidate lookup convention:
|
|
69
|
+
* `canonicalizeProvider(providerSegment)/idSegment.toLowerCase()`.
|
|
70
|
+
* `model_key` rows keep the provider alias verbatim (`ArkCodingPlan/...`),
|
|
71
|
+
* candidates carry the catalog spelling — both fold onto the same key.
|
|
72
|
+
*/
|
|
73
|
+
export function speedIndexKey(modelKey: string): string {
|
|
74
|
+
const slash = modelKey.indexOf('/');
|
|
75
|
+
if (slash < 0) return modelKey.toLowerCase();
|
|
76
|
+
return `${canonicalizeProvider(modelKey.slice(0, slash))}/${modelKey.slice(slash + 1).toLowerCase()}`;
|
|
77
|
+
}
|
|
78
|
+
|
|
79
|
+
export interface LoadSpeedIndexOptions {
|
|
80
|
+
/** Rows with samples below this are skipped (defaults to DEFAULT_MIN_SAMPLES). */
|
|
81
|
+
minSamples?: number;
|
|
82
|
+
}
|
|
83
|
+
|
|
84
|
+
/** Defensive numeric column read: number → finite as-is, bigint → number, else 0. */
|
|
85
|
+
function toNumber(value: unknown): number {
|
|
86
|
+
if (typeof value === 'number' && Number.isFinite(value)) return value;
|
|
87
|
+
if (typeof value === 'bigint') {
|
|
88
|
+
const n = Number(value);
|
|
89
|
+
return Number.isFinite(n) ? n : 0;
|
|
90
|
+
}
|
|
91
|
+
return 0;
|
|
92
|
+
}
|
|
93
|
+
|
|
94
|
+
/**
|
|
95
|
+
* Load the whole `model_perf` table into a normalized speed index.
|
|
96
|
+
* Opens the DB `readonly` and closes it in a `finally` — throws on missing
|
|
97
|
+
* db/table/corrupt file (the caller decides whether to degrade silently).
|
|
98
|
+
*/
|
|
99
|
+
export function loadSpeedIndex(dbPath: string, opts?: LoadSpeedIndexOptions): SpeedIndex {
|
|
100
|
+
const minSamples = opts?.minSamples ?? DEFAULT_MIN_SAMPLES;
|
|
101
|
+
const db = new Database(dbPath, { readonly: true });
|
|
102
|
+
loadCountForTest++;
|
|
103
|
+
try {
|
|
104
|
+
// SELECT * + name-based reads: hosts with an older schema (no ttft
|
|
105
|
+
// columns) degrade naturally — undefined → 0, no fallback query.
|
|
106
|
+
const rows = db.query('SELECT * FROM model_perf').all() as Array<Record<string, unknown>>;
|
|
107
|
+
const out = new Map<string, SpeedEntry>();
|
|
108
|
+
for (const row of rows) {
|
|
109
|
+
const modelKey = row.model_key;
|
|
110
|
+
if (typeof modelKey !== 'string' || modelKey === '') continue;
|
|
111
|
+
const genMs = toNumber(row.gen_ms);
|
|
112
|
+
if (genMs <= 0) continue;
|
|
113
|
+
const samples = toNumber(row.samples);
|
|
114
|
+
if (samples < minSamples) continue;
|
|
115
|
+
const ttftSamples = toNumber(row.ttft_samples);
|
|
116
|
+
const ttftMs = toNumber(row.ttft_ms);
|
|
117
|
+
out.set(speedIndexKey(modelKey), {
|
|
118
|
+
tokPerS: (toNumber(row.output_tokens) / genMs) * 1000,
|
|
119
|
+
ttftMs: ttftSamples > 0 ? ttftMs / ttftSamples : 0,
|
|
120
|
+
samples,
|
|
121
|
+
});
|
|
122
|
+
}
|
|
123
|
+
return out;
|
|
124
|
+
} finally {
|
|
125
|
+
db.close();
|
|
126
|
+
}
|
|
127
|
+
}
|
|
128
|
+
|
|
129
|
+
// ── module-level TTL singleton (event-driven; no timers) ────────────────────
|
|
130
|
+
|
|
131
|
+
let cache: { index: SpeedIndex; loadedAt: number } | null = null;
|
|
132
|
+
let dbPathForTest: string | null = null;
|
|
133
|
+
let loadCountForTest = 0;
|
|
134
|
+
|
|
135
|
+
function refreshCache(minSamples: number, now: number): SpeedIndex {
|
|
136
|
+
let index: SpeedIndex;
|
|
137
|
+
try {
|
|
138
|
+
index = loadSpeedIndex(dbPathForTest ?? join(homedir(), '.omp', 'agent', 'agent.db'), { minSamples });
|
|
139
|
+
} catch {
|
|
140
|
+
// Missing db / missing table / corrupt file → empty index, same TTL
|
|
141
|
+
// as a successful load (negative cache: no reopen storm), silent.
|
|
142
|
+
index = new Map();
|
|
143
|
+
}
|
|
144
|
+
cache = { index, loadedAt: now };
|
|
145
|
+
return index;
|
|
146
|
+
}
|
|
147
|
+
|
|
148
|
+
function currentCachedIndex(minSamples: number): SpeedIndex {
|
|
149
|
+
const now = Date.now();
|
|
150
|
+
if (cache && now - cache.loadedAt < SPEED_CACHE_TTL_MS) return cache.index;
|
|
151
|
+
return refreshCache(minSamples, now);
|
|
152
|
+
}
|
|
153
|
+
|
|
154
|
+
/**
|
|
155
|
+
* Assembly-layer hook: `undefined` config or `enabled: false` returns
|
|
156
|
+
* `undefined` with ZERO IO (the disabled path never touches the DB);
|
|
157
|
+
* otherwise returns the mode plus the TTL-cached speed index.
|
|
158
|
+
*/
|
|
159
|
+
export function resolveSpeedAware(cfg?: SpeedAwareConfig): SpeedAwareRuntime | undefined {
|
|
160
|
+
if (!cfg?.enabled) return undefined;
|
|
161
|
+
return { mode: cfg.mode, index: currentCachedIndex(cfg.minSamples) };
|
|
162
|
+
}
|
|
163
|
+
|
|
164
|
+
/**
|
|
165
|
+
* Frozen programmatic signature (programme smart-model-selection W3):
|
|
166
|
+
* measured entry for a model key (`provider/id`), or null when unmeasured.
|
|
167
|
+
* Walks the same default TTL singleton the selection path uses.
|
|
168
|
+
*/
|
|
169
|
+
export function speedScore(modelKey: string): SpeedEntry | null {
|
|
170
|
+
return currentCachedIndex(DEFAULT_MIN_SAMPLES).get(speedIndexKey(modelKey)) ?? null;
|
|
171
|
+
}
|
|
172
|
+
|
|
173
|
+
// ── test seams (precedent: index.ts `_setOpsxGlobalHomeForTest`) ────────────
|
|
174
|
+
|
|
175
|
+
/** Point the singleton at a fixture db; `null` restores the derived default. */
|
|
176
|
+
export function setSpeedDbPathForTest(path: string | null): void {
|
|
177
|
+
dbPathForTest = path;
|
|
178
|
+
}
|
|
179
|
+
|
|
180
|
+
/** Drop the TTL cache and zero the load counter (fresh baseline for zero-read assertions). */
|
|
181
|
+
export function resetSpeedCacheForTest(): void {
|
|
182
|
+
cache = null;
|
|
183
|
+
loadCountForTest = 0;
|
|
184
|
+
}
|
|
185
|
+
|
|
186
|
+
/** Successful db opens since module load; zero-read assertions for disabled paths. */
|
|
187
|
+
export function speedLoadCountForTest(): number {
|
|
188
|
+
return loadCountForTest;
|
|
189
|
+
}
|