pi-smart-router 0.2.0 → 0.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +19 -0
- package/config/benchmark-profiles.json +145 -0
- package/config/models.yaml.example +5 -0
- package/config/routing-calibration.json.example +14 -2
- package/dist/config/pi-model-mapper.d.ts +12 -2
- package/dist/config/pi-model-mapper.d.ts.map +1 -1
- package/dist/config/pi-model-mapper.js +91 -6
- package/dist/config/pi-model-mapper.js.map +1 -1
- package/dist/domain/matching/hydra-input.d.ts +6 -5
- package/dist/domain/matching/hydra-input.d.ts.map +1 -1
- package/dist/domain/matching/hydra-input.js +73 -6
- package/dist/domain/matching/hydra-input.js.map +1 -1
- package/dist/domain/pipeline/router-pipeline.d.ts +9 -0
- package/dist/domain/pipeline/router-pipeline.d.ts.map +1 -1
- package/dist/domain/pipeline/router-pipeline.js +39 -4
- package/dist/domain/pipeline/router-pipeline.js.map +1 -1
- package/dist/domain/routing/isotonic-calibrator.d.ts +56 -0
- package/dist/domain/routing/isotonic-calibrator.d.ts.map +1 -0
- package/dist/domain/routing/isotonic-calibrator.js +187 -0
- package/dist/domain/routing/isotonic-calibrator.js.map +1 -0
- package/dist/domain/routing/p-success-classifier.d.ts +53 -7
- package/dist/domain/routing/p-success-classifier.d.ts.map +1 -1
- package/dist/domain/routing/p-success-classifier.js +205 -21
- package/dist/domain/routing/p-success-classifier.js.map +1 -1
- package/dist/domain/types/entities.d.ts +6 -0
- package/dist/domain/types/entities.d.ts.map +1 -1
- package/dist/infrastructure/telemetry/routing-telemetry.d.ts.map +1 -1
- package/dist/infrastructure/telemetry/routing-telemetry.js +4 -0
- package/dist/infrastructure/telemetry/routing-telemetry.js.map +1 -1
- package/package.json +6 -3
- package/specs/001-build-smart-router/contracts/telemetry-contrib.schema.json +29 -1
- package/src/config/pi-model-mapper.ts +110 -6
- package/src/domain/matching/hydra-input.ts +86 -7
- package/src/domain/pipeline/router-pipeline.ts +54 -3
- package/src/domain/routing/isotonic-calibrator.ts +255 -0
- package/src/domain/routing/p-success-classifier.ts +299 -26
- package/src/domain/types/entities.ts +6 -0
- package/src/infrastructure/telemetry/routing-telemetry.ts +4 -0
|
@@ -143,9 +143,37 @@
|
|
|
143
143
|
"model_override",
|
|
144
144
|
"compaction_pin_break",
|
|
145
145
|
"feedback_good",
|
|
146
|
-
"feedback_bad"
|
|
146
|
+
"feedback_bad",
|
|
147
|
+
"tool_failure_chain",
|
|
148
|
+
"stop_reason_invalid",
|
|
149
|
+
"reprompt_detected",
|
|
150
|
+
"high_edit_distance",
|
|
151
|
+
"provider_failover",
|
|
152
|
+
"stop_reason_length",
|
|
153
|
+
"infra_error"
|
|
147
154
|
]
|
|
148
155
|
}
|
|
156
|
+
},
|
|
157
|
+
"tool_failure_chain_count": {
|
|
158
|
+
"type": ["integer", "null"],
|
|
159
|
+
"minimum": 0,
|
|
160
|
+
"description": "Consecutive identical tool failures at routing time (privacy-safe scalar)."
|
|
161
|
+
},
|
|
162
|
+
"stop_reason_invalid": {
|
|
163
|
+
"type": ["boolean", "null"],
|
|
164
|
+
"description": "True when provider stop_reason indicates invalid task completion."
|
|
165
|
+
},
|
|
166
|
+
"reprompt_rate": {
|
|
167
|
+
"type": ["number", "null"],
|
|
168
|
+
"minimum": 0,
|
|
169
|
+
"maximum": 1,
|
|
170
|
+
"description": "Normalized session re-prompt rate without storing prompt text."
|
|
171
|
+
},
|
|
172
|
+
"edit_distance_proxy": {
|
|
173
|
+
"type": ["number", "null"],
|
|
174
|
+
"minimum": 0,
|
|
175
|
+
"maximum": 1,
|
|
176
|
+
"description": "Normalized prompt-length delta proxy for edit/re-prompt detection."
|
|
149
177
|
}
|
|
150
178
|
}
|
|
151
179
|
}
|
|
@@ -2,10 +2,17 @@
|
|
|
2
2
|
* Pi model registry → ModelProfile mapper.
|
|
3
3
|
*
|
|
4
4
|
* Maps pi `Model` objects (provider + id) to router fleet entries using
|
|
5
|
-
* pattern-based lookup for known families.
|
|
6
|
-
*
|
|
5
|
+
* pattern-based lookup for known families. When `config/benchmark-profiles.json`
|
|
6
|
+
* contains a row for the model id (SP-134/136 ingest output), capability
|
|
7
|
+
* vectors are grounded in benchmark scores instead of regex defaults.
|
|
8
|
+
* Unknown models or missing benchmark rows receive conservative pattern defaults.
|
|
7
9
|
*/
|
|
8
10
|
|
|
11
|
+
import { existsSync, readFileSync } from 'node:fs';
|
|
12
|
+
import { resolve } from 'node:path';
|
|
13
|
+
|
|
14
|
+
import { z } from 'zod';
|
|
15
|
+
|
|
9
16
|
import type {
|
|
10
17
|
ModelCapabilities,
|
|
11
18
|
ModelLimits,
|
|
@@ -15,6 +22,103 @@ import type {
|
|
|
15
22
|
Tier,
|
|
16
23
|
} from '../domain/types/entities.js';
|
|
17
24
|
|
|
25
|
+
/** Checked-in ingest artifact from `npm run routing:ingest-benchmarks` (SP-134). */
|
|
26
|
+
export const DEFAULT_BENCHMARK_PROFILES_PATH = resolve('config', 'benchmark-profiles.json');
|
|
27
|
+
|
|
28
|
+
const benchmarkCapabilitiesSchema = z.object({
|
|
29
|
+
reasoning: z.number().min(0).max(1),
|
|
30
|
+
code_gen: z.number().min(0).max(1),
|
|
31
|
+
tool_use: z.number().min(0).max(1),
|
|
32
|
+
});
|
|
33
|
+
|
|
34
|
+
const benchmarkModelRowSchema = z.object({
|
|
35
|
+
model_id: z.string().min(1),
|
|
36
|
+
capabilities: benchmarkCapabilitiesSchema,
|
|
37
|
+
});
|
|
38
|
+
|
|
39
|
+
const benchmarkProfilesArtifactSchema = z.object({
|
|
40
|
+
version: z.literal(1),
|
|
41
|
+
models: z.array(benchmarkModelRowSchema).min(1),
|
|
42
|
+
});
|
|
43
|
+
|
|
44
|
+
let benchmarkProfilesPathOverride: string | null | undefined;
|
|
45
|
+
let benchmarkCapabilitiesByModelId: ReadonlyMap<string, ModelCapabilities> | null | undefined;
|
|
46
|
+
|
|
47
|
+
/**
|
|
48
|
+
* Test hook — override benchmark artifact path (null disables benchmark grounding).
|
|
49
|
+
*/
|
|
50
|
+
export function setBenchmarkProfilesPathForTests(filePath: string | null): void {
|
|
51
|
+
benchmarkProfilesPathOverride = filePath;
|
|
52
|
+
benchmarkCapabilitiesByModelId = undefined;
|
|
53
|
+
}
|
|
54
|
+
|
|
55
|
+
/** Test hook — clear cached benchmark artifact between cases. */
|
|
56
|
+
export function resetBenchmarkProfilesCacheForTests(): void {
|
|
57
|
+
benchmarkProfilesPathOverride = undefined;
|
|
58
|
+
benchmarkCapabilitiesByModelId = undefined;
|
|
59
|
+
}
|
|
60
|
+
|
|
61
|
+
function resolveBenchmarkProfilesPath(): string | null {
|
|
62
|
+
if (benchmarkProfilesPathOverride === null) {
|
|
63
|
+
return null;
|
|
64
|
+
}
|
|
65
|
+
return benchmarkProfilesPathOverride ?? DEFAULT_BENCHMARK_PROFILES_PATH;
|
|
66
|
+
}
|
|
67
|
+
|
|
68
|
+
function loadBenchmarkCapabilitiesMap(): ReadonlyMap<string, ModelCapabilities> | null {
|
|
69
|
+
if (benchmarkCapabilitiesByModelId !== undefined) {
|
|
70
|
+
return benchmarkCapabilitiesByModelId;
|
|
71
|
+
}
|
|
72
|
+
|
|
73
|
+
const filePath = resolveBenchmarkProfilesPath();
|
|
74
|
+
if (filePath === null || !existsSync(filePath)) {
|
|
75
|
+
benchmarkCapabilitiesByModelId = null;
|
|
76
|
+
return null;
|
|
77
|
+
}
|
|
78
|
+
|
|
79
|
+
try {
|
|
80
|
+
const raw = readFileSync(filePath, 'utf8');
|
|
81
|
+
const parsed: unknown = JSON.parse(raw);
|
|
82
|
+
const result = benchmarkProfilesArtifactSchema.safeParse(parsed);
|
|
83
|
+
if (!result.success) {
|
|
84
|
+
throw new Error(result.error.message);
|
|
85
|
+
}
|
|
86
|
+
|
|
87
|
+
const map = new Map<string, ModelCapabilities>();
|
|
88
|
+
for (const row of result.data.models) {
|
|
89
|
+
map.set(row.model_id, { ...row.capabilities });
|
|
90
|
+
}
|
|
91
|
+
benchmarkCapabilitiesByModelId = map;
|
|
92
|
+
return map;
|
|
93
|
+
} catch (err: unknown) {
|
|
94
|
+
console.warn('benchmark profiles artifact invalid; using regex capability defaults', {
|
|
95
|
+
path: filePath,
|
|
96
|
+
error: err instanceof Error ? err.message : String(err),
|
|
97
|
+
});
|
|
98
|
+
benchmarkCapabilitiesByModelId = null;
|
|
99
|
+
return null;
|
|
100
|
+
}
|
|
101
|
+
}
|
|
102
|
+
|
|
103
|
+
function lookupBenchmarkCapabilities(modelId: string): ModelCapabilities | undefined {
|
|
104
|
+
return loadBenchmarkCapabilitiesMap()?.get(modelId);
|
|
105
|
+
}
|
|
106
|
+
|
|
107
|
+
function withBenchmarkCapabilities(
|
|
108
|
+
defaults: ModelFamilyDefaults,
|
|
109
|
+
modelId: string,
|
|
110
|
+
): ModelFamilyDefaults {
|
|
111
|
+
const grounded = lookupBenchmarkCapabilities(modelId);
|
|
112
|
+
if (grounded === undefined) {
|
|
113
|
+
return defaults;
|
|
114
|
+
}
|
|
115
|
+
|
|
116
|
+
return {
|
|
117
|
+
...defaults,
|
|
118
|
+
capabilities: { ...grounded },
|
|
119
|
+
};
|
|
120
|
+
}
|
|
121
|
+
|
|
18
122
|
/** Pi registry `Model.cost` shape — per-token USD rates. */
|
|
19
123
|
export interface PiRegistryCost {
|
|
20
124
|
readonly input: number;
|
|
@@ -262,19 +366,19 @@ export function mapPiModelToProfile(input: PiModelInput): ModelProfile {
|
|
|
262
366
|
id: input.id,
|
|
263
367
|
...(input.name !== undefined ? { name: input.name } : {}),
|
|
264
368
|
};
|
|
265
|
-
return buildProfile(localInput, LOCAL_DEFAULTS);
|
|
369
|
+
return buildProfile(localInput, withBenchmarkCapabilities(LOCAL_DEFAULTS, input.id));
|
|
266
370
|
}
|
|
267
371
|
|
|
268
372
|
if (input.id === OPAQUE_FLEET_DEFAULT_ID) {
|
|
269
|
-
return buildProfile(input, CURSOR_AUTO_DEFAULTS);
|
|
373
|
+
return buildProfile(input, withBenchmarkCapabilities(CURSOR_AUTO_DEFAULTS, input.id));
|
|
270
374
|
}
|
|
271
375
|
|
|
272
376
|
const matched = matchPatternRules(input.id);
|
|
273
377
|
if (matched) {
|
|
274
|
-
return buildProfile(input, matched);
|
|
378
|
+
return buildProfile(input, withBenchmarkCapabilities(matched, input.id));
|
|
275
379
|
}
|
|
276
380
|
|
|
277
|
-
return buildProfile(input, UNKNOWN_DEFAULTS);
|
|
381
|
+
return buildProfile(input, withBenchmarkCapabilities(UNKNOWN_DEFAULTS, input.id));
|
|
278
382
|
}
|
|
279
383
|
|
|
280
384
|
/**
|
|
@@ -1,13 +1,26 @@
|
|
|
1
1
|
/**
|
|
2
|
-
* HyDRA embedding input builder — SP-112, GitHub #60.
|
|
2
|
+
* HyDRA embedding input builder — SP-112, SP-138, GitHub #60, #76.
|
|
3
3
|
*
|
|
4
|
-
* Prefixes
|
|
5
|
-
* turn context, not only the latest user string.
|
|
4
|
+
* Prefixes the latest user turn with session metadata so requirement vectors
|
|
5
|
+
* reflect turn context, not only the latest user string. Prior assistant
|
|
6
|
+
* responses are excluded from the encoder body per HyDRA reference (#76).
|
|
6
7
|
*/
|
|
7
8
|
|
|
8
|
-
import type { RoutingRequest } from '../types/index.js';
|
|
9
|
+
import type { Message, RoutingRequest } from '../types/index.js';
|
|
9
10
|
import type { TriageResult } from '../triage/triage-engine.js';
|
|
10
11
|
|
|
12
|
+
const FAILURE_PATTERNS = [
|
|
13
|
+
'error',
|
|
14
|
+
'fail',
|
|
15
|
+
'exception',
|
|
16
|
+
'timed out',
|
|
17
|
+
'timeout',
|
|
18
|
+
'econnrefused',
|
|
19
|
+
'enotfound',
|
|
20
|
+
'econnreset',
|
|
21
|
+
'epipe',
|
|
22
|
+
] as const;
|
|
23
|
+
|
|
11
24
|
function resolveMessageCount(request: RoutingRequest): number {
|
|
12
25
|
return request.messages?.length ?? 0;
|
|
13
26
|
}
|
|
@@ -28,10 +41,73 @@ function resolveTurnType(request: RoutingRequest): string {
|
|
|
28
41
|
return request.turn_type ?? 'unknown';
|
|
29
42
|
}
|
|
30
43
|
|
|
44
|
+
function resolveCompactionFlag(request: RoutingRequest): 0 | 1 {
|
|
45
|
+
return request.compaction_flag ? 1 : 0;
|
|
46
|
+
}
|
|
47
|
+
|
|
48
|
+
function looksLikeToolFailure(content: string): boolean {
|
|
49
|
+
const lower = content.toLowerCase();
|
|
50
|
+
return FAILURE_PATTERNS.some((pattern) => lower.includes(pattern));
|
|
51
|
+
}
|
|
52
|
+
|
|
53
|
+
/**
|
|
54
|
+
* Observational loop-pressure flag: 1 when the latest tool message looks
|
|
55
|
+
* like a failure. Mirrors loop-escalation heuristics without session pin state.
|
|
56
|
+
*/
|
|
57
|
+
function resolveLoopPressure(request: RoutingRequest): 0 | 1 {
|
|
58
|
+
const messages = request.messages;
|
|
59
|
+
if (!messages || messages.length === 0) {
|
|
60
|
+
return 0;
|
|
61
|
+
}
|
|
62
|
+
|
|
63
|
+
for (let i = messages.length - 1; i >= 0; i--) {
|
|
64
|
+
const message = messages[i]!;
|
|
65
|
+
if (message.role === 'tool') {
|
|
66
|
+
return looksLikeToolFailure(message.content) ? 1 : 0;
|
|
67
|
+
}
|
|
68
|
+
}
|
|
69
|
+
|
|
70
|
+
return 0;
|
|
71
|
+
}
|
|
72
|
+
|
|
73
|
+
function messageHasAttachmentIndicator(message: Message): boolean {
|
|
74
|
+
if (message.tool_blocks && message.tool_blocks.length > 0) {
|
|
75
|
+
return true;
|
|
76
|
+
}
|
|
77
|
+
|
|
78
|
+
if (message.role !== 'user') {
|
|
79
|
+
return false;
|
|
80
|
+
}
|
|
81
|
+
|
|
82
|
+
const lower = message.content.toLowerCase();
|
|
83
|
+
return lower.includes('data:image');
|
|
84
|
+
}
|
|
85
|
+
|
|
86
|
+
function resolveAttachmentFlag(request: RoutingRequest): 0 | 1 {
|
|
87
|
+
return request.messages?.some((message) => messageHasAttachmentIndicator(message)) ? 1 : 0;
|
|
88
|
+
}
|
|
89
|
+
|
|
90
|
+
/**
|
|
91
|
+
* Latest user turn text only — excludes prior assistant responses (#76).
|
|
92
|
+
*/
|
|
93
|
+
function resolveHydraPromptText(request: RoutingRequest): string {
|
|
94
|
+
const messages = request.messages;
|
|
95
|
+
if (messages) {
|
|
96
|
+
for (let i = messages.length - 1; i >= 0; i--) {
|
|
97
|
+
const message = messages[i]!;
|
|
98
|
+
if (message.role === 'user' && message.content.trim()) {
|
|
99
|
+
return message.content;
|
|
100
|
+
}
|
|
101
|
+
}
|
|
102
|
+
}
|
|
103
|
+
|
|
104
|
+
return request.prompt_text;
|
|
105
|
+
}
|
|
106
|
+
|
|
31
107
|
/**
|
|
32
|
-
* Build HyDRA encoder input: metadata prefix +
|
|
108
|
+
* Build HyDRA encoder input: seven-flag metadata prefix + latest user text.
|
|
33
109
|
*
|
|
34
|
-
* Format: `[turns:N|tools:0|tokens:N|type
|
|
110
|
+
* Format: `[turns:N|tools:0|tokens:N|type:...|compact:0|loop:0|attach:0] {text}`
|
|
35
111
|
*
|
|
36
112
|
* Metadata affects capability prediction only; tier selection uses the
|
|
37
113
|
* cluster/feature gate (SP-103) separately.
|
|
@@ -47,7 +123,10 @@ export function buildHydraInput(
|
|
|
47
123
|
`tools:${resolveHasToolContext(request) ? 1 : 0}`,
|
|
48
124
|
`tokens:${resolveEstimatedInputTokens(request)}`,
|
|
49
125
|
`type:${resolveTurnType(request)}`,
|
|
126
|
+
`compact:${resolveCompactionFlag(request)}`,
|
|
127
|
+
`loop:${resolveLoopPressure(request)}`,
|
|
128
|
+
`attach:${resolveAttachmentFlag(request)}`,
|
|
50
129
|
].join('|');
|
|
51
130
|
|
|
52
|
-
return `[${flags}] ${request
|
|
131
|
+
return `[${flags}] ${resolveHydraPromptText(request)}`;
|
|
53
132
|
}
|
|
@@ -58,6 +58,11 @@ import {
|
|
|
58
58
|
buildTierFeatures,
|
|
59
59
|
scoreLowIntensity,
|
|
60
60
|
} from '../routing/tier-features.js';
|
|
61
|
+
import {
|
|
62
|
+
applyIsotonicCalibratorTimed,
|
|
63
|
+
resolveIsotonicCalibrator,
|
|
64
|
+
type IsotonicCalibratorArtifact,
|
|
65
|
+
} from '../routing/isotonic-calibrator.js';
|
|
61
66
|
import {
|
|
62
67
|
predictPSuccessCheapTimed,
|
|
63
68
|
resolvePSuccessWeights,
|
|
@@ -168,6 +173,9 @@ export interface PipelineOptions {
|
|
|
168
173
|
/** Preloaded P(success) weights for tests; lazy-loads artifact when omitted (SP-105). */
|
|
169
174
|
readonly pSuccessWeights?: PSuccessWeights;
|
|
170
175
|
readonly pSuccessWeightsPath?: string;
|
|
176
|
+
/** Preloaded isotonic calibrator for tests; lazy-loads bundle when omitted (SP-133). */
|
|
177
|
+
readonly isotonicCalibrator?: IsotonicCalibratorArtifact | null;
|
|
178
|
+
readonly routingCalibrationPath?: string;
|
|
171
179
|
/** SAAR pin policy (SP-123). Must match sessionPinner.saarConfig when enabled. */
|
|
172
180
|
readonly saarConfig?: SaarConfig;
|
|
173
181
|
}
|
|
@@ -194,11 +202,15 @@ export class RouterPipeline {
|
|
|
194
202
|
private currentTierHintReasonCode: string | null = null;
|
|
195
203
|
private currentLowIntensityScore: number | null = null;
|
|
196
204
|
private currentPSuccessCheap: number | null = null;
|
|
205
|
+
private currentPSuccessRaw: number | null = null;
|
|
206
|
+
private currentPSuccessCalibrated: number | null = null;
|
|
197
207
|
private currentPSuccessAlpha: number | null = null;
|
|
198
208
|
private currentExpectedCostByTier: ExpectedCostBreakdown[] | null = null;
|
|
199
209
|
private currentLocalEligibleReason: string | null = null;
|
|
200
210
|
private pSuccessWeightsLoaded = false;
|
|
201
211
|
private cachedPSuccessWeights: PSuccessWeights | null = null;
|
|
212
|
+
private isotonicCalibratorLoaded = false;
|
|
213
|
+
private cachedIsotonicCalibrator: IsotonicCalibratorArtifact | null = null;
|
|
202
214
|
private currentContextFitRejected: readonly CandidateScore[] = [];
|
|
203
215
|
private currentContextFitViableCount = 0;
|
|
204
216
|
private contextOverflowPreferredProvider: string | null = null;
|
|
@@ -243,6 +255,8 @@ export class RouterPipeline {
|
|
|
243
255
|
this.currentTierHintReasonCode = null;
|
|
244
256
|
this.currentLowIntensityScore = null;
|
|
245
257
|
this.currentPSuccessCheap = null;
|
|
258
|
+
this.currentPSuccessRaw = null;
|
|
259
|
+
this.currentPSuccessCalibrated = null;
|
|
246
260
|
this.currentPSuccessAlpha = null;
|
|
247
261
|
this.currentExpectedCostByTier = null;
|
|
248
262
|
this.currentLocalEligibleReason = null;
|
|
@@ -341,6 +355,8 @@ export class RouterPipeline {
|
|
|
341
355
|
tier_hint_reason_code: this.currentTierHintReasonCode,
|
|
342
356
|
low_intensity_score: this.currentLowIntensityScore,
|
|
343
357
|
p_success_cheap: this.currentPSuccessCheap,
|
|
358
|
+
p_success_raw: this.currentPSuccessRaw,
|
|
359
|
+
p_success_calibrated: this.currentPSuccessCalibrated,
|
|
344
360
|
p_success_alpha: this.currentPSuccessAlpha,
|
|
345
361
|
local_eligible_reason: this.currentLocalEligibleReason,
|
|
346
362
|
};
|
|
@@ -1027,7 +1043,13 @@ export class RouterPipeline {
|
|
|
1027
1043
|
const weights = this.resolvePSuccessWeights();
|
|
1028
1044
|
const pFeatures = tierFeaturesToPSuccessFeatures(tierFeatures);
|
|
1029
1045
|
const pResult = predictPSuccessCheapTimed(pFeatures, weights);
|
|
1030
|
-
|
|
1046
|
+
const calibrator = this.resolveIsotonicCalibrator();
|
|
1047
|
+
const calibratedResult = applyIsotonicCalibratorTimed(pResult.probability, calibrator);
|
|
1048
|
+
const pSuccessForGate = calibratedResult.calibrated;
|
|
1049
|
+
|
|
1050
|
+
this.currentPSuccessRaw = pResult.probability;
|
|
1051
|
+
this.currentPSuccessCalibrated = pSuccessForGate;
|
|
1052
|
+
this.currentPSuccessCheap = pSuccessForGate;
|
|
1031
1053
|
|
|
1032
1054
|
const structuralHint = this.resolveTierHint(
|
|
1033
1055
|
score,
|
|
@@ -1041,10 +1063,14 @@ export class RouterPipeline {
|
|
|
1041
1063
|
? (() => {
|
|
1042
1064
|
const selection = this.selectExpectedCostTierHint(
|
|
1043
1065
|
request,
|
|
1044
|
-
|
|
1066
|
+
pSuccessForGate,
|
|
1045
1067
|
alpha,
|
|
1046
1068
|
);
|
|
1047
|
-
this.logExpectedCostExplain(
|
|
1069
|
+
this.logExpectedCostExplain(pSuccessForGate, alpha, selection, {
|
|
1070
|
+
p_success_raw: pResult.probability,
|
|
1071
|
+
p_success_calibrated: pSuccessForGate,
|
|
1072
|
+
calibration_applied: calibratedResult.calibration_applied,
|
|
1073
|
+
});
|
|
1048
1074
|
return {
|
|
1049
1075
|
tierHint: selection.tierHint,
|
|
1050
1076
|
reasonCode: selection.reasonCode,
|
|
@@ -1106,10 +1132,18 @@ export class RouterPipeline {
|
|
|
1106
1132
|
tierCosts: readonly ExpectedCostBreakdown[];
|
|
1107
1133
|
rationale: string;
|
|
1108
1134
|
},
|
|
1135
|
+
calibration?: {
|
|
1136
|
+
readonly p_success_raw: number;
|
|
1137
|
+
readonly p_success_calibrated: number;
|
|
1138
|
+
readonly calibration_applied: boolean;
|
|
1139
|
+
},
|
|
1109
1140
|
): void {
|
|
1110
1141
|
console.info('Expected-cost tier gate', {
|
|
1111
1142
|
reason: selection.reasonCode,
|
|
1112
1143
|
p_success_cheap: pSuccessCheap,
|
|
1144
|
+
p_success_raw: calibration?.p_success_raw ?? pSuccessCheap,
|
|
1145
|
+
p_success_calibrated: calibration?.p_success_calibrated ?? pSuccessCheap,
|
|
1146
|
+
calibration_applied: calibration?.calibration_applied ?? false,
|
|
1113
1147
|
alpha,
|
|
1114
1148
|
chosen_tier: selection.tierHint,
|
|
1115
1149
|
rationale: selection.rationale,
|
|
@@ -1140,6 +1174,23 @@ export class RouterPipeline {
|
|
|
1140
1174
|
return this.cachedPSuccessWeights!;
|
|
1141
1175
|
}
|
|
1142
1176
|
|
|
1177
|
+
private resolveIsotonicCalibrator(): IsotonicCalibratorArtifact | null {
|
|
1178
|
+
if (this.options.isotonicCalibrator !== undefined) {
|
|
1179
|
+
return this.options.isotonicCalibrator;
|
|
1180
|
+
}
|
|
1181
|
+
|
|
1182
|
+
if (!this.isotonicCalibratorLoaded) {
|
|
1183
|
+
this.cachedIsotonicCalibrator = resolveIsotonicCalibrator({
|
|
1184
|
+
...(this.options.routingCalibrationPath !== undefined
|
|
1185
|
+
? { filePath: this.options.routingCalibrationPath }
|
|
1186
|
+
: {}),
|
|
1187
|
+
});
|
|
1188
|
+
this.isotonicCalibratorLoaded = true;
|
|
1189
|
+
}
|
|
1190
|
+
|
|
1191
|
+
return this.cachedIsotonicCalibrator;
|
|
1192
|
+
}
|
|
1193
|
+
|
|
1143
1194
|
private resolveTierHint(
|
|
1144
1195
|
score: number,
|
|
1145
1196
|
highThreshold: number,
|
|
@@ -0,0 +1,255 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Online isotonic P(success) calibrator for low_intensity gate (SP-133).
|
|
3
|
+
*
|
|
4
|
+
* Loads monotonic knot tables from routing-calibration bundle at serve time.
|
|
5
|
+
* O(log n) binary search lookup with <5ms budget; falls back to raw logistic
|
|
6
|
+
* when the artifact is missing or under-trained.
|
|
7
|
+
*/
|
|
8
|
+
|
|
9
|
+
import { existsSync, readFileSync } from 'node:fs';
|
|
10
|
+
import { resolve } from 'node:path';
|
|
11
|
+
|
|
12
|
+
import { z } from 'zod';
|
|
13
|
+
|
|
14
|
+
import { MIN_TRAINING_SAMPLES } from './p-success-classifier.js';
|
|
15
|
+
|
|
16
|
+
export const ISOTONIC_CALIBRATOR_ARTIFACT_VERSION = 1 as const;
|
|
17
|
+
export const ISOTONIC_LOOKUP_BUDGET_MS = 5;
|
|
18
|
+
export const DEFAULT_ROUTING_CALIBRATION_PATH = resolve('config', 'routing-calibration.json');
|
|
19
|
+
|
|
20
|
+
export interface IsotonicCalibratorArtifact {
|
|
21
|
+
readonly version: typeof ISOTONIC_CALIBRATOR_ARTIFACT_VERSION;
|
|
22
|
+
readonly min_training_samples: number;
|
|
23
|
+
readonly x_knots: readonly number[];
|
|
24
|
+
readonly y_knots: readonly number[];
|
|
25
|
+
readonly trained_sample_count: number;
|
|
26
|
+
readonly holdout_ece_raw: number | null;
|
|
27
|
+
readonly holdout_ece_calibrated: number | null;
|
|
28
|
+
}
|
|
29
|
+
|
|
30
|
+
export interface IsotonicCalibratorResult {
|
|
31
|
+
readonly calibrated: number;
|
|
32
|
+
readonly elapsed_ms: number;
|
|
33
|
+
readonly within_budget: boolean;
|
|
34
|
+
readonly calibration_applied: boolean;
|
|
35
|
+
}
|
|
36
|
+
|
|
37
|
+
export interface LoadIsotonicCalibratorOptions {
|
|
38
|
+
readonly filePath?: string;
|
|
39
|
+
}
|
|
40
|
+
|
|
41
|
+
export class IsotonicCalibratorLoaderError extends Error {
|
|
42
|
+
override readonly name = 'IsotonicCalibratorLoaderError';
|
|
43
|
+
|
|
44
|
+
constructor(message: string, options?: ErrorOptions) {
|
|
45
|
+
super(message, options);
|
|
46
|
+
}
|
|
47
|
+
}
|
|
48
|
+
|
|
49
|
+
const IsotonicCalibratorArtifactSchema = z.object({
|
|
50
|
+
version: z.literal(ISOTONIC_CALIBRATOR_ARTIFACT_VERSION),
|
|
51
|
+
min_training_samples: z.number().int().min(0),
|
|
52
|
+
x_knots: z.array(z.number().finite().min(0).max(1)).min(2),
|
|
53
|
+
y_knots: z.array(z.number().finite().min(0).max(1)).min(2),
|
|
54
|
+
trained_sample_count: z.number().int().min(0),
|
|
55
|
+
holdout_ece_raw: z.number().finite().min(0).max(1).nullable(),
|
|
56
|
+
holdout_ece_calibrated: z.number().finite().min(0).max(1).nullable(),
|
|
57
|
+
});
|
|
58
|
+
|
|
59
|
+
function clamp01(value: number): number {
|
|
60
|
+
if (value <= 0) return 0;
|
|
61
|
+
if (value >= 1) return 1;
|
|
62
|
+
return value;
|
|
63
|
+
}
|
|
64
|
+
|
|
65
|
+
function formatZodIssues(error: { issues: readonly { path: readonly PropertyKey[]; message: string }[] }): string {
|
|
66
|
+
return error.issues
|
|
67
|
+
.map((issue) => ` - ${issue.path.join('.')}: ${issue.message}`)
|
|
68
|
+
.join('\n');
|
|
69
|
+
}
|
|
70
|
+
|
|
71
|
+
/** True when artifact has enough labeled samples for serve-time calibration. */
|
|
72
|
+
export function isIsotonicCalibratorTrained(artifact: IsotonicCalibratorArtifact): boolean {
|
|
73
|
+
return artifact.trained_sample_count >= artifact.min_training_samples;
|
|
74
|
+
}
|
|
75
|
+
|
|
76
|
+
export function createDefaultIsotonicCalibratorArtifact(): IsotonicCalibratorArtifact {
|
|
77
|
+
return {
|
|
78
|
+
version: ISOTONIC_CALIBRATOR_ARTIFACT_VERSION,
|
|
79
|
+
min_training_samples: MIN_TRAINING_SAMPLES,
|
|
80
|
+
x_knots: [0, 1],
|
|
81
|
+
y_knots: [0, 1],
|
|
82
|
+
trained_sample_count: 0,
|
|
83
|
+
holdout_ece_raw: null,
|
|
84
|
+
holdout_ece_calibrated: null,
|
|
85
|
+
};
|
|
86
|
+
}
|
|
87
|
+
|
|
88
|
+
/** Piecewise-constant isotonic lookup with O(log n) binary search on knots. */
|
|
89
|
+
export function applyIsotonicLookup(
|
|
90
|
+
rawScore: number,
|
|
91
|
+
xKnots: readonly number[],
|
|
92
|
+
yKnots: readonly number[],
|
|
93
|
+
): number {
|
|
94
|
+
if (xKnots.length === 0 || yKnots.length === 0 || xKnots.length !== yKnots.length) {
|
|
95
|
+
return clamp01(rawScore);
|
|
96
|
+
}
|
|
97
|
+
|
|
98
|
+
const score = clamp01(rawScore);
|
|
99
|
+
if (score <= xKnots[0]!) {
|
|
100
|
+
return clamp01(yKnots[0]!);
|
|
101
|
+
}
|
|
102
|
+
|
|
103
|
+
const lastIndex = xKnots.length - 1;
|
|
104
|
+
if (score >= xKnots[lastIndex]!) {
|
|
105
|
+
return clamp01(yKnots[lastIndex]!);
|
|
106
|
+
}
|
|
107
|
+
|
|
108
|
+
let low = 0;
|
|
109
|
+
let high = lastIndex;
|
|
110
|
+
while (low < high) {
|
|
111
|
+
const mid = Math.floor((low + high + 1) / 2);
|
|
112
|
+
if (xKnots[mid]! <= score) {
|
|
113
|
+
low = mid;
|
|
114
|
+
} else {
|
|
115
|
+
high = mid - 1;
|
|
116
|
+
}
|
|
117
|
+
}
|
|
118
|
+
|
|
119
|
+
return clamp01(yKnots[low]!);
|
|
120
|
+
}
|
|
121
|
+
|
|
122
|
+
/**
|
|
123
|
+
* Apply isotonic calibration to a raw logistic P(success) score.
|
|
124
|
+
* Returns the raw score when the artifact is missing or under-trained.
|
|
125
|
+
*/
|
|
126
|
+
export function applyIsotonicCalibrator(
|
|
127
|
+
rawScore: number,
|
|
128
|
+
artifact: IsotonicCalibratorArtifact | null,
|
|
129
|
+
): number {
|
|
130
|
+
if (artifact === null || !isIsotonicCalibratorTrained(artifact)) {
|
|
131
|
+
return clamp01(rawScore);
|
|
132
|
+
}
|
|
133
|
+
|
|
134
|
+
return applyIsotonicLookup(rawScore, artifact.x_knots, artifact.y_knots);
|
|
135
|
+
}
|
|
136
|
+
|
|
137
|
+
/** Apply isotonic calibration with elapsed timing guard for the online routing budget. */
|
|
138
|
+
export function applyIsotonicCalibratorTimed(
|
|
139
|
+
rawScore: number,
|
|
140
|
+
artifact: IsotonicCalibratorArtifact | null,
|
|
141
|
+
budgetMs: number = ISOTONIC_LOOKUP_BUDGET_MS,
|
|
142
|
+
): IsotonicCalibratorResult {
|
|
143
|
+
const start = performance.now();
|
|
144
|
+
const calibration_applied = artifact !== null && isIsotonicCalibratorTrained(artifact);
|
|
145
|
+
const calibrated = applyIsotonicCalibrator(rawScore, artifact);
|
|
146
|
+
const elapsed_ms = performance.now() - start;
|
|
147
|
+
|
|
148
|
+
if (elapsed_ms > budgetMs) {
|
|
149
|
+
console.warn('Isotonic P(success) lookup exceeded latency budget', {
|
|
150
|
+
elapsed_ms,
|
|
151
|
+
budget_ms: budgetMs,
|
|
152
|
+
knot_count: artifact?.x_knots.length ?? 0,
|
|
153
|
+
});
|
|
154
|
+
}
|
|
155
|
+
|
|
156
|
+
return {
|
|
157
|
+
calibrated,
|
|
158
|
+
elapsed_ms,
|
|
159
|
+
within_budget: elapsed_ms <= budgetMs,
|
|
160
|
+
calibration_applied,
|
|
161
|
+
};
|
|
162
|
+
}
|
|
163
|
+
|
|
164
|
+
/** Parse and validate an isotonic calibrator artifact from JSON text. */
|
|
165
|
+
export function parseIsotonicCalibratorJson(raw: string): IsotonicCalibratorArtifact {
|
|
166
|
+
let parsed: unknown;
|
|
167
|
+
try {
|
|
168
|
+
parsed = JSON.parse(raw);
|
|
169
|
+
} catch (err: unknown) {
|
|
170
|
+
const message = err instanceof Error ? err.message : String(err);
|
|
171
|
+
throw new IsotonicCalibratorLoaderError(`Failed to parse JSON: ${message}`, { cause: err });
|
|
172
|
+
}
|
|
173
|
+
|
|
174
|
+
const result = IsotonicCalibratorArtifactSchema.safeParse(parsed);
|
|
175
|
+
if (!result.success) {
|
|
176
|
+
throw new IsotonicCalibratorLoaderError(
|
|
177
|
+
`Invalid isotonic calibrator artifact:\n${formatZodIssues(result.error)}`,
|
|
178
|
+
{ cause: result.error },
|
|
179
|
+
);
|
|
180
|
+
}
|
|
181
|
+
|
|
182
|
+
return result.data;
|
|
183
|
+
}
|
|
184
|
+
|
|
185
|
+
/** Extract isotonic_calibrator from a routing-calibration bundle object. */
|
|
186
|
+
export function parseIsotonicCalibratorFromBundle(parsed: unknown): IsotonicCalibratorArtifact {
|
|
187
|
+
if (typeof parsed !== 'object' || parsed === null) {
|
|
188
|
+
throw new IsotonicCalibratorLoaderError('Routing calibration bundle must be a JSON object');
|
|
189
|
+
}
|
|
190
|
+
|
|
191
|
+
const isotonic = (parsed as Record<string, unknown>).isotonic_calibrator;
|
|
192
|
+
if (isotonic === undefined) {
|
|
193
|
+
throw new IsotonicCalibratorLoaderError('Routing calibration bundle missing isotonic_calibrator');
|
|
194
|
+
}
|
|
195
|
+
|
|
196
|
+
const result = IsotonicCalibratorArtifactSchema.safeParse(isotonic);
|
|
197
|
+
if (!result.success) {
|
|
198
|
+
throw new IsotonicCalibratorLoaderError(
|
|
199
|
+
`Invalid isotonic_calibrator in routing calibration bundle:\n${formatZodIssues(result.error)}`,
|
|
200
|
+
{ cause: result.error },
|
|
201
|
+
);
|
|
202
|
+
}
|
|
203
|
+
|
|
204
|
+
return result.data;
|
|
205
|
+
}
|
|
206
|
+
|
|
207
|
+
/**
|
|
208
|
+
* Load isotonic calibrator from routing-calibration bundle on disk.
|
|
209
|
+
* Returns null when the bundle file is missing.
|
|
210
|
+
*/
|
|
211
|
+
export function loadIsotonicCalibrator(
|
|
212
|
+
options?: LoadIsotonicCalibratorOptions,
|
|
213
|
+
): IsotonicCalibratorArtifact | null {
|
|
214
|
+
const filePath = options?.filePath ?? DEFAULT_ROUTING_CALIBRATION_PATH;
|
|
215
|
+
|
|
216
|
+
if (!existsSync(filePath)) {
|
|
217
|
+
return null;
|
|
218
|
+
}
|
|
219
|
+
|
|
220
|
+
let raw: string;
|
|
221
|
+
try {
|
|
222
|
+
raw = readFileSync(filePath, 'utf8');
|
|
223
|
+
} catch (err: unknown) {
|
|
224
|
+
const message = err instanceof Error ? err.message : String(err);
|
|
225
|
+
throw new IsotonicCalibratorLoaderError(`Failed to read routing calibration file: ${message}`, {
|
|
226
|
+
cause: err,
|
|
227
|
+
});
|
|
228
|
+
}
|
|
229
|
+
|
|
230
|
+
let parsed: unknown;
|
|
231
|
+
try {
|
|
232
|
+
parsed = JSON.parse(raw);
|
|
233
|
+
} catch (err: unknown) {
|
|
234
|
+
const message = err instanceof Error ? err.message : String(err);
|
|
235
|
+
throw new IsotonicCalibratorLoaderError(`Failed to parse routing calibration JSON: ${message}`, {
|
|
236
|
+
cause: err,
|
|
237
|
+
});
|
|
238
|
+
}
|
|
239
|
+
|
|
240
|
+
return parseIsotonicCalibratorFromBundle(parsed);
|
|
241
|
+
}
|
|
242
|
+
|
|
243
|
+
/** Resolve calibrator for online inference — missing or invalid artifacts fall back safely. */
|
|
244
|
+
export function resolveIsotonicCalibrator(
|
|
245
|
+
options?: LoadIsotonicCalibratorOptions,
|
|
246
|
+
): IsotonicCalibratorArtifact | null {
|
|
247
|
+
try {
|
|
248
|
+
return loadIsotonicCalibrator(options);
|
|
249
|
+
} catch (err: unknown) {
|
|
250
|
+
console.warn('Isotonic calibrator artifact invalid; using raw logistic fallback', {
|
|
251
|
+
error: err instanceof Error ? err.message : String(err),
|
|
252
|
+
});
|
|
253
|
+
return null;
|
|
254
|
+
}
|
|
255
|
+
}
|