pi-smart-router 0.2.0 → 0.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (38) hide show
  1. package/README.md +19 -0
  2. package/config/benchmark-profiles.json +145 -0
  3. package/config/models.yaml.example +5 -0
  4. package/config/routing-calibration.json.example +14 -2
  5. package/dist/config/pi-model-mapper.d.ts +12 -2
  6. package/dist/config/pi-model-mapper.d.ts.map +1 -1
  7. package/dist/config/pi-model-mapper.js +91 -6
  8. package/dist/config/pi-model-mapper.js.map +1 -1
  9. package/dist/domain/matching/hydra-input.d.ts +6 -5
  10. package/dist/domain/matching/hydra-input.d.ts.map +1 -1
  11. package/dist/domain/matching/hydra-input.js +73 -6
  12. package/dist/domain/matching/hydra-input.js.map +1 -1
  13. package/dist/domain/pipeline/router-pipeline.d.ts +9 -0
  14. package/dist/domain/pipeline/router-pipeline.d.ts.map +1 -1
  15. package/dist/domain/pipeline/router-pipeline.js +39 -4
  16. package/dist/domain/pipeline/router-pipeline.js.map +1 -1
  17. package/dist/domain/routing/isotonic-calibrator.d.ts +56 -0
  18. package/dist/domain/routing/isotonic-calibrator.d.ts.map +1 -0
  19. package/dist/domain/routing/isotonic-calibrator.js +187 -0
  20. package/dist/domain/routing/isotonic-calibrator.js.map +1 -0
  21. package/dist/domain/routing/p-success-classifier.d.ts +53 -7
  22. package/dist/domain/routing/p-success-classifier.d.ts.map +1 -1
  23. package/dist/domain/routing/p-success-classifier.js +205 -21
  24. package/dist/domain/routing/p-success-classifier.js.map +1 -1
  25. package/dist/domain/types/entities.d.ts +6 -0
  26. package/dist/domain/types/entities.d.ts.map +1 -1
  27. package/dist/infrastructure/telemetry/routing-telemetry.d.ts.map +1 -1
  28. package/dist/infrastructure/telemetry/routing-telemetry.js +4 -0
  29. package/dist/infrastructure/telemetry/routing-telemetry.js.map +1 -1
  30. package/package.json +6 -3
  31. package/specs/001-build-smart-router/contracts/telemetry-contrib.schema.json +29 -1
  32. package/src/config/pi-model-mapper.ts +110 -6
  33. package/src/domain/matching/hydra-input.ts +86 -7
  34. package/src/domain/pipeline/router-pipeline.ts +54 -3
  35. package/src/domain/routing/isotonic-calibrator.ts +255 -0
  36. package/src/domain/routing/p-success-classifier.ts +299 -26
  37. package/src/domain/types/entities.ts +6 -0
  38. package/src/infrastructure/telemetry/routing-telemetry.ts +4 -0
@@ -143,9 +143,37 @@
143
143
  "model_override",
144
144
  "compaction_pin_break",
145
145
  "feedback_good",
146
- "feedback_bad"
146
+ "feedback_bad",
147
+ "tool_failure_chain",
148
+ "stop_reason_invalid",
149
+ "reprompt_detected",
150
+ "high_edit_distance",
151
+ "provider_failover",
152
+ "stop_reason_length",
153
+ "infra_error"
147
154
  ]
148
155
  }
156
+ },
157
+ "tool_failure_chain_count": {
158
+ "type": ["integer", "null"],
159
+ "minimum": 0,
160
+ "description": "Consecutive identical tool failures at routing time (privacy-safe scalar)."
161
+ },
162
+ "stop_reason_invalid": {
163
+ "type": ["boolean", "null"],
164
+ "description": "True when provider stop_reason indicates invalid task completion."
165
+ },
166
+ "reprompt_rate": {
167
+ "type": ["number", "null"],
168
+ "minimum": 0,
169
+ "maximum": 1,
170
+ "description": "Normalized session re-prompt rate without storing prompt text."
171
+ },
172
+ "edit_distance_proxy": {
173
+ "type": ["number", "null"],
174
+ "minimum": 0,
175
+ "maximum": 1,
176
+ "description": "Normalized prompt-length delta proxy for edit/re-prompt detection."
149
177
  }
150
178
  }
151
179
  }
@@ -2,10 +2,17 @@
2
2
  * Pi model registry → ModelProfile mapper.
3
3
  *
4
4
  * Maps pi `Model` objects (provider + id) to router fleet entries using
5
- * pattern-based lookup for known families. Unknown models receive conservative
6
- * economical-cloud defaults.
5
+ * pattern-based lookup for known families. When `config/benchmark-profiles.json`
6
+ * contains a row for the model id (SP-134/136 ingest output), capability
7
+ * vectors are grounded in benchmark scores instead of regex defaults.
8
+ * Unknown models or missing benchmark rows receive conservative pattern defaults.
7
9
  */
8
10
 
11
+ import { existsSync, readFileSync } from 'node:fs';
12
+ import { resolve } from 'node:path';
13
+
14
+ import { z } from 'zod';
15
+
9
16
  import type {
10
17
  ModelCapabilities,
11
18
  ModelLimits,
@@ -15,6 +22,103 @@ import type {
15
22
  Tier,
16
23
  } from '../domain/types/entities.js';
17
24
 
25
+ /** Checked-in ingest artifact from `npm run routing:ingest-benchmarks` (SP-134). */
26
+ export const DEFAULT_BENCHMARK_PROFILES_PATH = resolve('config', 'benchmark-profiles.json');
27
+
28
+ const benchmarkCapabilitiesSchema = z.object({
29
+ reasoning: z.number().min(0).max(1),
30
+ code_gen: z.number().min(0).max(1),
31
+ tool_use: z.number().min(0).max(1),
32
+ });
33
+
34
+ const benchmarkModelRowSchema = z.object({
35
+ model_id: z.string().min(1),
36
+ capabilities: benchmarkCapabilitiesSchema,
37
+ });
38
+
39
+ const benchmarkProfilesArtifactSchema = z.object({
40
+ version: z.literal(1),
41
+ models: z.array(benchmarkModelRowSchema).min(1),
42
+ });
43
+
44
+ let benchmarkProfilesPathOverride: string | null | undefined;
45
+ let benchmarkCapabilitiesByModelId: ReadonlyMap<string, ModelCapabilities> | null | undefined;
46
+
47
+ /**
48
+ * Test hook — override benchmark artifact path (null disables benchmark grounding).
49
+ */
50
+ export function setBenchmarkProfilesPathForTests(filePath: string | null): void {
51
+ benchmarkProfilesPathOverride = filePath;
52
+ benchmarkCapabilitiesByModelId = undefined;
53
+ }
54
+
55
+ /** Test hook — clear cached benchmark artifact between cases. */
56
+ export function resetBenchmarkProfilesCacheForTests(): void {
57
+ benchmarkProfilesPathOverride = undefined;
58
+ benchmarkCapabilitiesByModelId = undefined;
59
+ }
60
+
61
+ function resolveBenchmarkProfilesPath(): string | null {
62
+ if (benchmarkProfilesPathOverride === null) {
63
+ return null;
64
+ }
65
+ return benchmarkProfilesPathOverride ?? DEFAULT_BENCHMARK_PROFILES_PATH;
66
+ }
67
+
68
+ function loadBenchmarkCapabilitiesMap(): ReadonlyMap<string, ModelCapabilities> | null {
69
+ if (benchmarkCapabilitiesByModelId !== undefined) {
70
+ return benchmarkCapabilitiesByModelId;
71
+ }
72
+
73
+ const filePath = resolveBenchmarkProfilesPath();
74
+ if (filePath === null || !existsSync(filePath)) {
75
+ benchmarkCapabilitiesByModelId = null;
76
+ return null;
77
+ }
78
+
79
+ try {
80
+ const raw = readFileSync(filePath, 'utf8');
81
+ const parsed: unknown = JSON.parse(raw);
82
+ const result = benchmarkProfilesArtifactSchema.safeParse(parsed);
83
+ if (!result.success) {
84
+ throw new Error(result.error.message);
85
+ }
86
+
87
+ const map = new Map<string, ModelCapabilities>();
88
+ for (const row of result.data.models) {
89
+ map.set(row.model_id, { ...row.capabilities });
90
+ }
91
+ benchmarkCapabilitiesByModelId = map;
92
+ return map;
93
+ } catch (err: unknown) {
94
+ console.warn('benchmark profiles artifact invalid; using regex capability defaults', {
95
+ path: filePath,
96
+ error: err instanceof Error ? err.message : String(err),
97
+ });
98
+ benchmarkCapabilitiesByModelId = null;
99
+ return null;
100
+ }
101
+ }
102
+
103
+ function lookupBenchmarkCapabilities(modelId: string): ModelCapabilities | undefined {
104
+ return loadBenchmarkCapabilitiesMap()?.get(modelId);
105
+ }
106
+
107
+ function withBenchmarkCapabilities(
108
+ defaults: ModelFamilyDefaults,
109
+ modelId: string,
110
+ ): ModelFamilyDefaults {
111
+ const grounded = lookupBenchmarkCapabilities(modelId);
112
+ if (grounded === undefined) {
113
+ return defaults;
114
+ }
115
+
116
+ return {
117
+ ...defaults,
118
+ capabilities: { ...grounded },
119
+ };
120
+ }
121
+
18
122
  /** Pi registry `Model.cost` shape — per-token USD rates. */
19
123
  export interface PiRegistryCost {
20
124
  readonly input: number;
@@ -262,19 +366,19 @@ export function mapPiModelToProfile(input: PiModelInput): ModelProfile {
262
366
  id: input.id,
263
367
  ...(input.name !== undefined ? { name: input.name } : {}),
264
368
  };
265
- return buildProfile(localInput, LOCAL_DEFAULTS);
369
+ return buildProfile(localInput, withBenchmarkCapabilities(LOCAL_DEFAULTS, input.id));
266
370
  }
267
371
 
268
372
  if (input.id === OPAQUE_FLEET_DEFAULT_ID) {
269
- return buildProfile(input, CURSOR_AUTO_DEFAULTS);
373
+ return buildProfile(input, withBenchmarkCapabilities(CURSOR_AUTO_DEFAULTS, input.id));
270
374
  }
271
375
 
272
376
  const matched = matchPatternRules(input.id);
273
377
  if (matched) {
274
- return buildProfile(input, matched);
378
+ return buildProfile(input, withBenchmarkCapabilities(matched, input.id));
275
379
  }
276
380
 
277
- return buildProfile(input, UNKNOWN_DEFAULTS);
381
+ return buildProfile(input, withBenchmarkCapabilities(UNKNOWN_DEFAULTS, input.id));
278
382
  }
279
383
 
280
384
  /**
@@ -1,13 +1,26 @@
1
1
  /**
2
- * HyDRA embedding input builder — SP-112, GitHub #60.
2
+ * HyDRA embedding input builder — SP-112, SP-138, GitHub #60, #76.
3
3
  *
4
- * Prefixes prompt text with session metadata so requirement vectors reflect
5
- * turn context, not only the latest user string.
4
+ * Prefixes the latest user turn with session metadata so requirement vectors
5
+ * reflect turn context, not only the latest user string. Prior assistant
6
+ * responses are excluded from the encoder body per HyDRA reference (#76).
6
7
  */
7
8
 
8
- import type { RoutingRequest } from '../types/index.js';
9
+ import type { Message, RoutingRequest } from '../types/index.js';
9
10
  import type { TriageResult } from '../triage/triage-engine.js';
10
11
 
12
+ const FAILURE_PATTERNS = [
13
+ 'error',
14
+ 'fail',
15
+ 'exception',
16
+ 'timed out',
17
+ 'timeout',
18
+ 'econnrefused',
19
+ 'enotfound',
20
+ 'econnreset',
21
+ 'epipe',
22
+ ] as const;
23
+
11
24
  function resolveMessageCount(request: RoutingRequest): number {
12
25
  return request.messages?.length ?? 0;
13
26
  }
@@ -28,10 +41,73 @@ function resolveTurnType(request: RoutingRequest): string {
28
41
  return request.turn_type ?? 'unknown';
29
42
  }
30
43
 
44
+ function resolveCompactionFlag(request: RoutingRequest): 0 | 1 {
45
+ return request.compaction_flag ? 1 : 0;
46
+ }
47
+
48
+ function looksLikeToolFailure(content: string): boolean {
49
+ const lower = content.toLowerCase();
50
+ return FAILURE_PATTERNS.some((pattern) => lower.includes(pattern));
51
+ }
52
+
53
+ /**
54
+ * Observational loop-pressure flag: 1 when the latest tool message looks
55
+ * like a failure. Mirrors loop-escalation heuristics without session pin state.
56
+ */
57
+ function resolveLoopPressure(request: RoutingRequest): 0 | 1 {
58
+ const messages = request.messages;
59
+ if (!messages || messages.length === 0) {
60
+ return 0;
61
+ }
62
+
63
+ for (let i = messages.length - 1; i >= 0; i--) {
64
+ const message = messages[i]!;
65
+ if (message.role === 'tool') {
66
+ return looksLikeToolFailure(message.content) ? 1 : 0;
67
+ }
68
+ }
69
+
70
+ return 0;
71
+ }
72
+
73
+ function messageHasAttachmentIndicator(message: Message): boolean {
74
+ if (message.tool_blocks && message.tool_blocks.length > 0) {
75
+ return true;
76
+ }
77
+
78
+ if (message.role !== 'user') {
79
+ return false;
80
+ }
81
+
82
+ const lower = message.content.toLowerCase();
83
+ return lower.includes('data:image');
84
+ }
85
+
86
+ function resolveAttachmentFlag(request: RoutingRequest): 0 | 1 {
87
+ return request.messages?.some((message) => messageHasAttachmentIndicator(message)) ? 1 : 0;
88
+ }
89
+
90
+ /**
91
+ * Latest user turn text only — excludes prior assistant responses (#76).
92
+ */
93
+ function resolveHydraPromptText(request: RoutingRequest): string {
94
+ const messages = request.messages;
95
+ if (messages) {
96
+ for (let i = messages.length - 1; i >= 0; i--) {
97
+ const message = messages[i]!;
98
+ if (message.role === 'user' && message.content.trim()) {
99
+ return message.content;
100
+ }
101
+ }
102
+ }
103
+
104
+ return request.prompt_text;
105
+ }
106
+
31
107
  /**
32
- * Build HyDRA encoder input: metadata prefix + prompt text.
108
+ * Build HyDRA encoder input: seven-flag metadata prefix + latest user text.
33
109
  *
34
- * Format: `[turns:N|tools:0|tokens:N|type:...] {prompt_text}`
110
+ * Format: `[turns:N|tools:0|tokens:N|type:...|compact:0|loop:0|attach:0] {text}`
35
111
  *
36
112
  * Metadata affects capability prediction only; tier selection uses the
37
113
  * cluster/feature gate (SP-103) separately.
@@ -47,7 +123,10 @@ export function buildHydraInput(
47
123
  `tools:${resolveHasToolContext(request) ? 1 : 0}`,
48
124
  `tokens:${resolveEstimatedInputTokens(request)}`,
49
125
  `type:${resolveTurnType(request)}`,
126
+ `compact:${resolveCompactionFlag(request)}`,
127
+ `loop:${resolveLoopPressure(request)}`,
128
+ `attach:${resolveAttachmentFlag(request)}`,
50
129
  ].join('|');
51
130
 
52
- return `[${flags}] ${request.prompt_text}`;
131
+ return `[${flags}] ${resolveHydraPromptText(request)}`;
53
132
  }
@@ -58,6 +58,11 @@ import {
58
58
  buildTierFeatures,
59
59
  scoreLowIntensity,
60
60
  } from '../routing/tier-features.js';
61
+ import {
62
+ applyIsotonicCalibratorTimed,
63
+ resolveIsotonicCalibrator,
64
+ type IsotonicCalibratorArtifact,
65
+ } from '../routing/isotonic-calibrator.js';
61
66
  import {
62
67
  predictPSuccessCheapTimed,
63
68
  resolvePSuccessWeights,
@@ -168,6 +173,9 @@ export interface PipelineOptions {
168
173
  /** Preloaded P(success) weights for tests; lazy-loads artifact when omitted (SP-105). */
169
174
  readonly pSuccessWeights?: PSuccessWeights;
170
175
  readonly pSuccessWeightsPath?: string;
176
+ /** Preloaded isotonic calibrator for tests; lazy-loads bundle when omitted (SP-133). */
177
+ readonly isotonicCalibrator?: IsotonicCalibratorArtifact | null;
178
+ readonly routingCalibrationPath?: string;
171
179
  /** SAAR pin policy (SP-123). Must match sessionPinner.saarConfig when enabled. */
172
180
  readonly saarConfig?: SaarConfig;
173
181
  }
@@ -194,11 +202,15 @@ export class RouterPipeline {
194
202
  private currentTierHintReasonCode: string | null = null;
195
203
  private currentLowIntensityScore: number | null = null;
196
204
  private currentPSuccessCheap: number | null = null;
205
+ private currentPSuccessRaw: number | null = null;
206
+ private currentPSuccessCalibrated: number | null = null;
197
207
  private currentPSuccessAlpha: number | null = null;
198
208
  private currentExpectedCostByTier: ExpectedCostBreakdown[] | null = null;
199
209
  private currentLocalEligibleReason: string | null = null;
200
210
  private pSuccessWeightsLoaded = false;
201
211
  private cachedPSuccessWeights: PSuccessWeights | null = null;
212
+ private isotonicCalibratorLoaded = false;
213
+ private cachedIsotonicCalibrator: IsotonicCalibratorArtifact | null = null;
202
214
  private currentContextFitRejected: readonly CandidateScore[] = [];
203
215
  private currentContextFitViableCount = 0;
204
216
  private contextOverflowPreferredProvider: string | null = null;
@@ -243,6 +255,8 @@ export class RouterPipeline {
243
255
  this.currentTierHintReasonCode = null;
244
256
  this.currentLowIntensityScore = null;
245
257
  this.currentPSuccessCheap = null;
258
+ this.currentPSuccessRaw = null;
259
+ this.currentPSuccessCalibrated = null;
246
260
  this.currentPSuccessAlpha = null;
247
261
  this.currentExpectedCostByTier = null;
248
262
  this.currentLocalEligibleReason = null;
@@ -341,6 +355,8 @@ export class RouterPipeline {
341
355
  tier_hint_reason_code: this.currentTierHintReasonCode,
342
356
  low_intensity_score: this.currentLowIntensityScore,
343
357
  p_success_cheap: this.currentPSuccessCheap,
358
+ p_success_raw: this.currentPSuccessRaw,
359
+ p_success_calibrated: this.currentPSuccessCalibrated,
344
360
  p_success_alpha: this.currentPSuccessAlpha,
345
361
  local_eligible_reason: this.currentLocalEligibleReason,
346
362
  };
@@ -1027,7 +1043,13 @@ export class RouterPipeline {
1027
1043
  const weights = this.resolvePSuccessWeights();
1028
1044
  const pFeatures = tierFeaturesToPSuccessFeatures(tierFeatures);
1029
1045
  const pResult = predictPSuccessCheapTimed(pFeatures, weights);
1030
- this.currentPSuccessCheap = pResult.probability;
1046
+ const calibrator = this.resolveIsotonicCalibrator();
1047
+ const calibratedResult = applyIsotonicCalibratorTimed(pResult.probability, calibrator);
1048
+ const pSuccessForGate = calibratedResult.calibrated;
1049
+
1050
+ this.currentPSuccessRaw = pResult.probability;
1051
+ this.currentPSuccessCalibrated = pSuccessForGate;
1052
+ this.currentPSuccessCheap = pSuccessForGate;
1031
1053
 
1032
1054
  const structuralHint = this.resolveTierHint(
1033
1055
  score,
@@ -1041,10 +1063,14 @@ export class RouterPipeline {
1041
1063
  ? (() => {
1042
1064
  const selection = this.selectExpectedCostTierHint(
1043
1065
  request,
1044
- pResult.probability,
1066
+ pSuccessForGate,
1045
1067
  alpha,
1046
1068
  );
1047
- this.logExpectedCostExplain(pResult.probability, alpha, selection);
1069
+ this.logExpectedCostExplain(pSuccessForGate, alpha, selection, {
1070
+ p_success_raw: pResult.probability,
1071
+ p_success_calibrated: pSuccessForGate,
1072
+ calibration_applied: calibratedResult.calibration_applied,
1073
+ });
1048
1074
  return {
1049
1075
  tierHint: selection.tierHint,
1050
1076
  reasonCode: selection.reasonCode,
@@ -1106,10 +1132,18 @@ export class RouterPipeline {
1106
1132
  tierCosts: readonly ExpectedCostBreakdown[];
1107
1133
  rationale: string;
1108
1134
  },
1135
+ calibration?: {
1136
+ readonly p_success_raw: number;
1137
+ readonly p_success_calibrated: number;
1138
+ readonly calibration_applied: boolean;
1139
+ },
1109
1140
  ): void {
1110
1141
  console.info('Expected-cost tier gate', {
1111
1142
  reason: selection.reasonCode,
1112
1143
  p_success_cheap: pSuccessCheap,
1144
+ p_success_raw: calibration?.p_success_raw ?? pSuccessCheap,
1145
+ p_success_calibrated: calibration?.p_success_calibrated ?? pSuccessCheap,
1146
+ calibration_applied: calibration?.calibration_applied ?? false,
1113
1147
  alpha,
1114
1148
  chosen_tier: selection.tierHint,
1115
1149
  rationale: selection.rationale,
@@ -1140,6 +1174,23 @@ export class RouterPipeline {
1140
1174
  return this.cachedPSuccessWeights!;
1141
1175
  }
1142
1176
 
1177
+ private resolveIsotonicCalibrator(): IsotonicCalibratorArtifact | null {
1178
+ if (this.options.isotonicCalibrator !== undefined) {
1179
+ return this.options.isotonicCalibrator;
1180
+ }
1181
+
1182
+ if (!this.isotonicCalibratorLoaded) {
1183
+ this.cachedIsotonicCalibrator = resolveIsotonicCalibrator({
1184
+ ...(this.options.routingCalibrationPath !== undefined
1185
+ ? { filePath: this.options.routingCalibrationPath }
1186
+ : {}),
1187
+ });
1188
+ this.isotonicCalibratorLoaded = true;
1189
+ }
1190
+
1191
+ return this.cachedIsotonicCalibrator;
1192
+ }
1193
+
1143
1194
  private resolveTierHint(
1144
1195
  score: number,
1145
1196
  highThreshold: number,
@@ -0,0 +1,255 @@
1
+ /**
2
+ * Online isotonic P(success) calibrator for low_intensity gate (SP-133).
3
+ *
4
+ * Loads monotonic knot tables from routing-calibration bundle at serve time.
5
+ * O(log n) binary search lookup with <5ms budget; falls back to raw logistic
6
+ * when the artifact is missing or under-trained.
7
+ */
8
+
9
+ import { existsSync, readFileSync } from 'node:fs';
10
+ import { resolve } from 'node:path';
11
+
12
+ import { z } from 'zod';
13
+
14
+ import { MIN_TRAINING_SAMPLES } from './p-success-classifier.js';
15
+
16
+ export const ISOTONIC_CALIBRATOR_ARTIFACT_VERSION = 1 as const;
17
+ export const ISOTONIC_LOOKUP_BUDGET_MS = 5;
18
+ export const DEFAULT_ROUTING_CALIBRATION_PATH = resolve('config', 'routing-calibration.json');
19
+
20
+ export interface IsotonicCalibratorArtifact {
21
+ readonly version: typeof ISOTONIC_CALIBRATOR_ARTIFACT_VERSION;
22
+ readonly min_training_samples: number;
23
+ readonly x_knots: readonly number[];
24
+ readonly y_knots: readonly number[];
25
+ readonly trained_sample_count: number;
26
+ readonly holdout_ece_raw: number | null;
27
+ readonly holdout_ece_calibrated: number | null;
28
+ }
29
+
30
+ export interface IsotonicCalibratorResult {
31
+ readonly calibrated: number;
32
+ readonly elapsed_ms: number;
33
+ readonly within_budget: boolean;
34
+ readonly calibration_applied: boolean;
35
+ }
36
+
37
+ export interface LoadIsotonicCalibratorOptions {
38
+ readonly filePath?: string;
39
+ }
40
+
41
+ export class IsotonicCalibratorLoaderError extends Error {
42
+ override readonly name = 'IsotonicCalibratorLoaderError';
43
+
44
+ constructor(message: string, options?: ErrorOptions) {
45
+ super(message, options);
46
+ }
47
+ }
48
+
49
+ const IsotonicCalibratorArtifactSchema = z.object({
50
+ version: z.literal(ISOTONIC_CALIBRATOR_ARTIFACT_VERSION),
51
+ min_training_samples: z.number().int().min(0),
52
+ x_knots: z.array(z.number().finite().min(0).max(1)).min(2),
53
+ y_knots: z.array(z.number().finite().min(0).max(1)).min(2),
54
+ trained_sample_count: z.number().int().min(0),
55
+ holdout_ece_raw: z.number().finite().min(0).max(1).nullable(),
56
+ holdout_ece_calibrated: z.number().finite().min(0).max(1).nullable(),
57
+ });
58
+
59
+ function clamp01(value: number): number {
60
+ if (value <= 0) return 0;
61
+ if (value >= 1) return 1;
62
+ return value;
63
+ }
64
+
65
+ function formatZodIssues(error: { issues: readonly { path: readonly PropertyKey[]; message: string }[] }): string {
66
+ return error.issues
67
+ .map((issue) => ` - ${issue.path.join('.')}: ${issue.message}`)
68
+ .join('\n');
69
+ }
70
+
71
+ /** True when artifact has enough labeled samples for serve-time calibration. */
72
+ export function isIsotonicCalibratorTrained(artifact: IsotonicCalibratorArtifact): boolean {
73
+ return artifact.trained_sample_count >= artifact.min_training_samples;
74
+ }
75
+
76
+ export function createDefaultIsotonicCalibratorArtifact(): IsotonicCalibratorArtifact {
77
+ return {
78
+ version: ISOTONIC_CALIBRATOR_ARTIFACT_VERSION,
79
+ min_training_samples: MIN_TRAINING_SAMPLES,
80
+ x_knots: [0, 1],
81
+ y_knots: [0, 1],
82
+ trained_sample_count: 0,
83
+ holdout_ece_raw: null,
84
+ holdout_ece_calibrated: null,
85
+ };
86
+ }
87
+
88
+ /** Piecewise-constant isotonic lookup with O(log n) binary search on knots. */
89
+ export function applyIsotonicLookup(
90
+ rawScore: number,
91
+ xKnots: readonly number[],
92
+ yKnots: readonly number[],
93
+ ): number {
94
+ if (xKnots.length === 0 || yKnots.length === 0 || xKnots.length !== yKnots.length) {
95
+ return clamp01(rawScore);
96
+ }
97
+
98
+ const score = clamp01(rawScore);
99
+ if (score <= xKnots[0]!) {
100
+ return clamp01(yKnots[0]!);
101
+ }
102
+
103
+ const lastIndex = xKnots.length - 1;
104
+ if (score >= xKnots[lastIndex]!) {
105
+ return clamp01(yKnots[lastIndex]!);
106
+ }
107
+
108
+ let low = 0;
109
+ let high = lastIndex;
110
+ while (low < high) {
111
+ const mid = Math.floor((low + high + 1) / 2);
112
+ if (xKnots[mid]! <= score) {
113
+ low = mid;
114
+ } else {
115
+ high = mid - 1;
116
+ }
117
+ }
118
+
119
+ return clamp01(yKnots[low]!);
120
+ }
121
+
122
+ /**
123
+ * Apply isotonic calibration to a raw logistic P(success) score.
124
+ * Returns the raw score when the artifact is missing or under-trained.
125
+ */
126
+ export function applyIsotonicCalibrator(
127
+ rawScore: number,
128
+ artifact: IsotonicCalibratorArtifact | null,
129
+ ): number {
130
+ if (artifact === null || !isIsotonicCalibratorTrained(artifact)) {
131
+ return clamp01(rawScore);
132
+ }
133
+
134
+ return applyIsotonicLookup(rawScore, artifact.x_knots, artifact.y_knots);
135
+ }
136
+
137
+ /** Apply isotonic calibration with elapsed timing guard for the online routing budget. */
138
+ export function applyIsotonicCalibratorTimed(
139
+ rawScore: number,
140
+ artifact: IsotonicCalibratorArtifact | null,
141
+ budgetMs: number = ISOTONIC_LOOKUP_BUDGET_MS,
142
+ ): IsotonicCalibratorResult {
143
+ const start = performance.now();
144
+ const calibration_applied = artifact !== null && isIsotonicCalibratorTrained(artifact);
145
+ const calibrated = applyIsotonicCalibrator(rawScore, artifact);
146
+ const elapsed_ms = performance.now() - start;
147
+
148
+ if (elapsed_ms > budgetMs) {
149
+ console.warn('Isotonic P(success) lookup exceeded latency budget', {
150
+ elapsed_ms,
151
+ budget_ms: budgetMs,
152
+ knot_count: artifact?.x_knots.length ?? 0,
153
+ });
154
+ }
155
+
156
+ return {
157
+ calibrated,
158
+ elapsed_ms,
159
+ within_budget: elapsed_ms <= budgetMs,
160
+ calibration_applied,
161
+ };
162
+ }
163
+
164
+ /** Parse and validate an isotonic calibrator artifact from JSON text. */
165
+ export function parseIsotonicCalibratorJson(raw: string): IsotonicCalibratorArtifact {
166
+ let parsed: unknown;
167
+ try {
168
+ parsed = JSON.parse(raw);
169
+ } catch (err: unknown) {
170
+ const message = err instanceof Error ? err.message : String(err);
171
+ throw new IsotonicCalibratorLoaderError(`Failed to parse JSON: ${message}`, { cause: err });
172
+ }
173
+
174
+ const result = IsotonicCalibratorArtifactSchema.safeParse(parsed);
175
+ if (!result.success) {
176
+ throw new IsotonicCalibratorLoaderError(
177
+ `Invalid isotonic calibrator artifact:\n${formatZodIssues(result.error)}`,
178
+ { cause: result.error },
179
+ );
180
+ }
181
+
182
+ return result.data;
183
+ }
184
+
185
+ /** Extract isotonic_calibrator from a routing-calibration bundle object. */
186
+ export function parseIsotonicCalibratorFromBundle(parsed: unknown): IsotonicCalibratorArtifact {
187
+ if (typeof parsed !== 'object' || parsed === null) {
188
+ throw new IsotonicCalibratorLoaderError('Routing calibration bundle must be a JSON object');
189
+ }
190
+
191
+ const isotonic = (parsed as Record<string, unknown>).isotonic_calibrator;
192
+ if (isotonic === undefined) {
193
+ throw new IsotonicCalibratorLoaderError('Routing calibration bundle missing isotonic_calibrator');
194
+ }
195
+
196
+ const result = IsotonicCalibratorArtifactSchema.safeParse(isotonic);
197
+ if (!result.success) {
198
+ throw new IsotonicCalibratorLoaderError(
199
+ `Invalid isotonic_calibrator in routing calibration bundle:\n${formatZodIssues(result.error)}`,
200
+ { cause: result.error },
201
+ );
202
+ }
203
+
204
+ return result.data;
205
+ }
206
+
207
+ /**
208
+ * Load isotonic calibrator from routing-calibration bundle on disk.
209
+ * Returns null when the bundle file is missing.
210
+ */
211
+ export function loadIsotonicCalibrator(
212
+ options?: LoadIsotonicCalibratorOptions,
213
+ ): IsotonicCalibratorArtifact | null {
214
+ const filePath = options?.filePath ?? DEFAULT_ROUTING_CALIBRATION_PATH;
215
+
216
+ if (!existsSync(filePath)) {
217
+ return null;
218
+ }
219
+
220
+ let raw: string;
221
+ try {
222
+ raw = readFileSync(filePath, 'utf8');
223
+ } catch (err: unknown) {
224
+ const message = err instanceof Error ? err.message : String(err);
225
+ throw new IsotonicCalibratorLoaderError(`Failed to read routing calibration file: ${message}`, {
226
+ cause: err,
227
+ });
228
+ }
229
+
230
+ let parsed: unknown;
231
+ try {
232
+ parsed = JSON.parse(raw);
233
+ } catch (err: unknown) {
234
+ const message = err instanceof Error ? err.message : String(err);
235
+ throw new IsotonicCalibratorLoaderError(`Failed to parse routing calibration JSON: ${message}`, {
236
+ cause: err,
237
+ });
238
+ }
239
+
240
+ return parseIsotonicCalibratorFromBundle(parsed);
241
+ }
242
+
243
+ /** Resolve calibrator for online inference — missing or invalid artifacts fall back safely. */
244
+ export function resolveIsotonicCalibrator(
245
+ options?: LoadIsotonicCalibratorOptions,
246
+ ): IsotonicCalibratorArtifact | null {
247
+ try {
248
+ return loadIsotonicCalibrator(options);
249
+ } catch (err: unknown) {
250
+ console.warn('Isotonic calibrator artifact invalid; using raw logistic fallback', {
251
+ error: err instanceof Error ? err.message : String(err),
252
+ });
253
+ return null;
254
+ }
255
+ }