pi-smart-router 0.2.0 → 0.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +19 -0
- package/config/benchmark-profiles.json +145 -0
- package/config/models.yaml.example +5 -0
- package/config/routing-calibration.json.example +14 -2
- package/dist/config/pi-model-mapper.d.ts +12 -2
- package/dist/config/pi-model-mapper.d.ts.map +1 -1
- package/dist/config/pi-model-mapper.js +91 -6
- package/dist/config/pi-model-mapper.js.map +1 -1
- package/dist/domain/matching/hydra-input.d.ts +6 -5
- package/dist/domain/matching/hydra-input.d.ts.map +1 -1
- package/dist/domain/matching/hydra-input.js +73 -6
- package/dist/domain/matching/hydra-input.js.map +1 -1
- package/dist/domain/pipeline/router-pipeline.d.ts +9 -0
- package/dist/domain/pipeline/router-pipeline.d.ts.map +1 -1
- package/dist/domain/pipeline/router-pipeline.js +39 -4
- package/dist/domain/pipeline/router-pipeline.js.map +1 -1
- package/dist/domain/routing/isotonic-calibrator.d.ts +56 -0
- package/dist/domain/routing/isotonic-calibrator.d.ts.map +1 -0
- package/dist/domain/routing/isotonic-calibrator.js +187 -0
- package/dist/domain/routing/isotonic-calibrator.js.map +1 -0
- package/dist/domain/routing/p-success-classifier.d.ts +53 -7
- package/dist/domain/routing/p-success-classifier.d.ts.map +1 -1
- package/dist/domain/routing/p-success-classifier.js +205 -21
- package/dist/domain/routing/p-success-classifier.js.map +1 -1
- package/dist/domain/types/entities.d.ts +6 -0
- package/dist/domain/types/entities.d.ts.map +1 -1
- package/dist/infrastructure/telemetry/routing-telemetry.d.ts.map +1 -1
- package/dist/infrastructure/telemetry/routing-telemetry.js +4 -0
- package/dist/infrastructure/telemetry/routing-telemetry.js.map +1 -1
- package/package.json +6 -3
- package/specs/001-build-smart-router/contracts/telemetry-contrib.schema.json +29 -1
- package/src/config/pi-model-mapper.ts +110 -6
- package/src/domain/matching/hydra-input.ts +86 -7
- package/src/domain/pipeline/router-pipeline.ts +54 -3
- package/src/domain/routing/isotonic-calibrator.ts +255 -0
- package/src/domain/routing/p-success-classifier.ts +299 -26
- package/src/domain/types/entities.ts +6 -0
- package/src/infrastructure/telemetry/routing-telemetry.ts +4 -0
|
@@ -23,19 +23,86 @@ export const MIN_TRAINING_SAMPLES = 30;
|
|
|
23
23
|
/** Neutral probability when training data is insufficient or artifact missing. */
|
|
24
24
|
export const NEUTRAL_P_SUCCESS = 0.5;
|
|
25
25
|
|
|
26
|
-
/**
|
|
27
|
-
export const
|
|
26
|
+
/** Behavioral outcome signals that mark cheap-tier training failure (SP-062). */
|
|
27
|
+
export const BEHAVIORAL_FAILURE_OUTCOME_SIGNALS: readonly OutcomeSignalType[] = [
|
|
28
28
|
'model_override',
|
|
29
29
|
'feedback_bad',
|
|
30
30
|
] as const;
|
|
31
31
|
|
|
32
|
-
/**
|
|
33
|
-
export const
|
|
32
|
+
/** Verifier-grade failure proxies derived from privacy-safe telemetry scalars (SP-131). */
|
|
33
|
+
export const VERIFIER_FAILURE_OUTCOME_SIGNALS = [
|
|
34
|
+
'tool_failure_chain',
|
|
35
|
+
'stop_reason_invalid',
|
|
36
|
+
'reprompt_detected',
|
|
37
|
+
'high_edit_distance',
|
|
38
|
+
] as const;
|
|
39
|
+
|
|
40
|
+
export type VerifierFailureOutcomeSignal = (typeof VERIFIER_FAILURE_OUTCOME_SIGNALS)[number];
|
|
41
|
+
|
|
42
|
+
/** Execution telemetry failure signals (SP-104 review follow-up, SP-131). */
|
|
43
|
+
export const EXECUTION_FAILURE_OUTCOME_SIGNALS = [
|
|
34
44
|
'provider_failover',
|
|
35
45
|
'stop_reason_length',
|
|
36
46
|
'infra_error',
|
|
37
47
|
] as const;
|
|
38
48
|
|
|
49
|
+
export type ExecutionFailureOutcomeSignal = (typeof EXECUTION_FAILURE_OUTCOME_SIGNALS)[number];
|
|
50
|
+
|
|
51
|
+
/** Union of all training label signals consumed by P(success) export and aggregate paths. */
|
|
52
|
+
export type TrainingOutcomeSignal =
|
|
53
|
+
| OutcomeSignalType
|
|
54
|
+
| VerifierFailureOutcomeSignal
|
|
55
|
+
| ExecutionFailureOutcomeSignal;
|
|
56
|
+
|
|
57
|
+
/** Outcome signals that mark a routing attempt as unsuccessful for cheap-tier training. */
|
|
58
|
+
export const FAILURE_OUTCOME_SIGNALS: readonly TrainingOutcomeSignal[] = [
|
|
59
|
+
...BEHAVIORAL_FAILURE_OUTCOME_SIGNALS,
|
|
60
|
+
...VERIFIER_FAILURE_OUTCOME_SIGNALS,
|
|
61
|
+
...EXECUTION_FAILURE_OUTCOME_SIGNALS,
|
|
62
|
+
] as const;
|
|
63
|
+
|
|
64
|
+
/** @deprecated Use EXECUTION_FAILURE_OUTCOME_SIGNALS */
|
|
65
|
+
export const FUTURE_FAILURE_SIGNALS = EXECUTION_FAILURE_OUTCOME_SIGNALS;
|
|
66
|
+
|
|
67
|
+
/** Privacy-safe scalar failure proxies exported in calibration contrib rows (SP-131). */
|
|
68
|
+
export const P_SUCCESS_FAILURE_PROXY_FIELDS = [
|
|
69
|
+
'tool_failure_chain_count',
|
|
70
|
+
'stop_reason_invalid',
|
|
71
|
+
'reprompt_rate',
|
|
72
|
+
'edit_distance_proxy',
|
|
73
|
+
] as const;
|
|
74
|
+
|
|
75
|
+
export type PSuccessFailureProxyField = (typeof P_SUCCESS_FAILURE_PROXY_FIELDS)[number];
|
|
76
|
+
|
|
77
|
+
export interface PSuccessFailureProxies {
|
|
78
|
+
readonly tool_failure_chain_count: number | null;
|
|
79
|
+
readonly stop_reason_invalid: boolean | null;
|
|
80
|
+
readonly reprompt_rate: number | null;
|
|
81
|
+
readonly edit_distance_proxy: number | null;
|
|
82
|
+
}
|
|
83
|
+
|
|
84
|
+
/** Minimum identical tool failures before labeling a tool-failure chain (loop escalation uses 3). */
|
|
85
|
+
export const TOOL_FAILURE_CHAIN_LABEL_THRESHOLD = 2;
|
|
86
|
+
|
|
87
|
+
/** Normalized re-prompt rate at or above this marks failure. */
|
|
88
|
+
export const REPROMPT_RATE_FAILURE_THRESHOLD = 0.5;
|
|
89
|
+
|
|
90
|
+
/** Normalized edit-distance proxy at or above this marks likely re-prompt / correction. */
|
|
91
|
+
export const EDIT_DISTANCE_PROXY_FAILURE_THRESHOLD = 0.6;
|
|
92
|
+
|
|
93
|
+
const PROMPT_LENGTH_NORM_FOR_EDIT_PROXY = 8_000;
|
|
94
|
+
|
|
95
|
+
/** Provider stop reasons treated as invalid task completion for training labels. */
|
|
96
|
+
export const INVALID_STOP_REASONS = new Set([
|
|
97
|
+
'length',
|
|
98
|
+
'max_tokens',
|
|
99
|
+
'content_filter',
|
|
100
|
+
'tool_calls',
|
|
101
|
+
'function_call',
|
|
102
|
+
'unknown',
|
|
103
|
+
'error',
|
|
104
|
+
]);
|
|
105
|
+
|
|
39
106
|
export const P_SUCCESS_FEATURE_NAMES = [
|
|
40
107
|
'prompt_length_norm',
|
|
41
108
|
'estimated_input_tokens_norm',
|
|
@@ -108,13 +175,18 @@ export interface LabeledTrainingSample {
|
|
|
108
175
|
readonly request_id: string;
|
|
109
176
|
readonly features: PSuccessFeatures;
|
|
110
177
|
readonly success: boolean;
|
|
111
|
-
readonly outcome_signals: readonly
|
|
178
|
+
readonly outcome_signals: readonly TrainingOutcomeSignal[];
|
|
179
|
+
readonly failure_proxies: PSuccessFailureProxies;
|
|
112
180
|
}
|
|
113
181
|
|
|
114
182
|
export interface DatasetExportJoinRow extends Record<string, unknown> {
|
|
115
183
|
readonly request_id: string;
|
|
116
184
|
readonly success_label: boolean | null;
|
|
117
|
-
readonly outcome_signals: readonly
|
|
185
|
+
readonly outcome_signals: readonly TrainingOutcomeSignal[];
|
|
186
|
+
readonly tool_failure_chain_count: number | null;
|
|
187
|
+
readonly stop_reason_invalid: boolean | null;
|
|
188
|
+
readonly reprompt_rate: number | null;
|
|
189
|
+
readonly edit_distance_proxy: number | null;
|
|
118
190
|
}
|
|
119
191
|
|
|
120
192
|
const PROMPT_LENGTH_NORM = 8_000;
|
|
@@ -182,13 +254,164 @@ export function featuresToVector(features: PSuccessFeatures): number[] {
|
|
|
182
254
|
return P_SUCCESS_FEATURE_NAMES.map((name) => features[name]);
|
|
183
255
|
}
|
|
184
256
|
|
|
185
|
-
|
|
257
|
+
function isTrainingOutcomeSignal(value: string): value is TrainingOutcomeSignal {
|
|
258
|
+
return (
|
|
259
|
+
(BEHAVIORAL_FAILURE_OUTCOME_SIGNALS as readonly string[]).includes(value) ||
|
|
260
|
+
(VERIFIER_FAILURE_OUTCOME_SIGNALS as readonly string[]).includes(value) ||
|
|
261
|
+
(EXECUTION_FAILURE_OUTCOME_SIGNALS as readonly string[]).includes(value) ||
|
|
262
|
+
value === 'compaction_pin_break' ||
|
|
263
|
+
value === 'feedback_good'
|
|
264
|
+
);
|
|
265
|
+
}
|
|
266
|
+
|
|
267
|
+
function intOrNull(value: unknown): number | null {
|
|
268
|
+
return typeof value === 'number' && Number.isFinite(value) ? Math.max(0, Math.trunc(value)) : null;
|
|
269
|
+
}
|
|
270
|
+
|
|
271
|
+
function boolOrNull(value: unknown): boolean | null {
|
|
272
|
+
return typeof value === 'boolean' ? value : null;
|
|
273
|
+
}
|
|
274
|
+
|
|
275
|
+
function rateOrNull(value: unknown): number | null {
|
|
276
|
+
if (typeof value !== 'number' || !Number.isFinite(value)) {
|
|
277
|
+
return null;
|
|
278
|
+
}
|
|
279
|
+
return clamp01(value);
|
|
280
|
+
}
|
|
281
|
+
|
|
282
|
+
function normalizeStopReasonInvalid(record: Record<string, unknown>): boolean | null {
|
|
283
|
+
const explicit = boolOrNull(record.stop_reason_invalid);
|
|
284
|
+
if (explicit !== null) {
|
|
285
|
+
return explicit;
|
|
286
|
+
}
|
|
287
|
+
|
|
288
|
+
const stopReason = record.stop_reason;
|
|
289
|
+
if (typeof stopReason !== 'string' || stopReason.length === 0) {
|
|
290
|
+
return null;
|
|
291
|
+
}
|
|
292
|
+
|
|
293
|
+
return INVALID_STOP_REASONS.has(stopReason.toLowerCase());
|
|
294
|
+
}
|
|
295
|
+
|
|
296
|
+
function normalizeRepromptRate(record: Record<string, unknown>): number | null {
|
|
297
|
+
const explicit = rateOrNull(record.reprompt_rate);
|
|
298
|
+
if (explicit !== null) {
|
|
299
|
+
return explicit;
|
|
300
|
+
}
|
|
301
|
+
|
|
302
|
+
const repromptCount = intOrNull(record.reprompt_count);
|
|
303
|
+
if (repromptCount === null) {
|
|
304
|
+
return null;
|
|
305
|
+
}
|
|
306
|
+
|
|
307
|
+
const turnIndex = intOrNull(record.turn_index_in_session);
|
|
308
|
+
const denominator = turnIndex !== null && turnIndex > 0 ? turnIndex : repromptCount + 1;
|
|
309
|
+
return clamp01(repromptCount / denominator);
|
|
310
|
+
}
|
|
311
|
+
|
|
312
|
+
function normalizeEditDistanceProxy(record: Record<string, unknown>): number | null {
|
|
313
|
+
const explicit = rateOrNull(record.edit_distance_proxy);
|
|
314
|
+
if (explicit !== null) {
|
|
315
|
+
return explicit;
|
|
316
|
+
}
|
|
317
|
+
|
|
318
|
+
const delta = intOrNull(record.prompt_length_delta);
|
|
319
|
+
if (delta !== null) {
|
|
320
|
+
return clamp01(Math.abs(delta) / PROMPT_LENGTH_NORM_FOR_EDIT_PROXY);
|
|
321
|
+
}
|
|
322
|
+
|
|
323
|
+
const current = intOrNull(record.prompt_length_chars);
|
|
324
|
+
const prior = intOrNull(record.prior_prompt_length_chars);
|
|
325
|
+
if (current === null || prior === null) {
|
|
326
|
+
return null;
|
|
327
|
+
}
|
|
328
|
+
|
|
329
|
+
const maxLength = Math.max(current, prior, 1);
|
|
330
|
+
return clamp01(Math.abs(current - prior) / maxLength);
|
|
331
|
+
}
|
|
332
|
+
|
|
333
|
+
function normalizeToolFailureChainCount(record: Record<string, unknown>): number | null {
|
|
334
|
+
const explicit = intOrNull(record.tool_failure_chain_count);
|
|
335
|
+
if (explicit !== null) {
|
|
336
|
+
return explicit;
|
|
337
|
+
}
|
|
338
|
+
|
|
339
|
+
return intOrNull(record.consecutive_tool_failures);
|
|
340
|
+
}
|
|
341
|
+
|
|
342
|
+
/** Extract privacy-safe failure proxy scalars from a contrib or export row. */
|
|
343
|
+
export function extractFailureProxies(record: Record<string, unknown>): PSuccessFailureProxies {
|
|
344
|
+
return {
|
|
345
|
+
tool_failure_chain_count: normalizeToolFailureChainCount(record),
|
|
346
|
+
stop_reason_invalid: normalizeStopReasonInvalid(record),
|
|
347
|
+
reprompt_rate: normalizeRepromptRate(record),
|
|
348
|
+
edit_distance_proxy: normalizeEditDistanceProxy(record),
|
|
349
|
+
};
|
|
350
|
+
}
|
|
351
|
+
|
|
352
|
+
/** Map failure proxy scalars to verifier-grade training outcome signals. */
|
|
353
|
+
export function deriveVerifierFailureSignals(
|
|
354
|
+
proxies: PSuccessFailureProxies,
|
|
355
|
+
): readonly VerifierFailureOutcomeSignal[] {
|
|
356
|
+
const signals: VerifierFailureOutcomeSignal[] = [];
|
|
357
|
+
|
|
358
|
+
if (
|
|
359
|
+
proxies.tool_failure_chain_count !== null &&
|
|
360
|
+
proxies.tool_failure_chain_count >= TOOL_FAILURE_CHAIN_LABEL_THRESHOLD
|
|
361
|
+
) {
|
|
362
|
+
signals.push('tool_failure_chain');
|
|
363
|
+
}
|
|
364
|
+
|
|
365
|
+
if (proxies.stop_reason_invalid === true) {
|
|
366
|
+
signals.push('stop_reason_invalid');
|
|
367
|
+
}
|
|
368
|
+
|
|
369
|
+
if (
|
|
370
|
+
proxies.reprompt_rate !== null &&
|
|
371
|
+
proxies.reprompt_rate >= REPROMPT_RATE_FAILURE_THRESHOLD
|
|
372
|
+
) {
|
|
373
|
+
signals.push('reprompt_detected');
|
|
374
|
+
}
|
|
375
|
+
|
|
376
|
+
if (
|
|
377
|
+
proxies.edit_distance_proxy !== null &&
|
|
378
|
+
proxies.edit_distance_proxy >= EDIT_DISTANCE_PROXY_FAILURE_THRESHOLD
|
|
379
|
+
) {
|
|
380
|
+
signals.push('high_edit_distance');
|
|
381
|
+
}
|
|
382
|
+
|
|
383
|
+
return signals;
|
|
384
|
+
}
|
|
385
|
+
|
|
386
|
+
function mergeTrainingOutcomeSignals(
|
|
387
|
+
...groups: ReadonlyArray<readonly TrainingOutcomeSignal[]>
|
|
388
|
+
): readonly TrainingOutcomeSignal[] {
|
|
389
|
+
return [...new Set(groups.flat())];
|
|
390
|
+
}
|
|
391
|
+
|
|
392
|
+
function parseTrainingOutcomeSignals(raw: unknown): readonly TrainingOutcomeSignal[] {
|
|
393
|
+
if (!Array.isArray(raw)) {
|
|
394
|
+
return [];
|
|
395
|
+
}
|
|
396
|
+
|
|
397
|
+
return raw.filter(
|
|
398
|
+
(value): value is TrainingOutcomeSignal =>
|
|
399
|
+
typeof value === 'string' && isTrainingOutcomeSignal(value),
|
|
400
|
+
);
|
|
401
|
+
}
|
|
402
|
+
|
|
403
|
+
/** Derive success label from behavioral outcomes plus optional failure proxies. */
|
|
186
404
|
export function deriveSuccessLabel(
|
|
187
405
|
outcomes: readonly RoutingOutcomeRecord[],
|
|
188
|
-
|
|
189
|
-
|
|
406
|
+
options?: { readonly failureProxies?: PSuccessFailureProxies },
|
|
407
|
+
): { success: boolean | null; outcome_signals: readonly TrainingOutcomeSignal[] } {
|
|
408
|
+
const behavioralSignals = [...new Set(outcomes.map((entry) => entry.signal_type))];
|
|
409
|
+
const verifierSignals = options?.failureProxies
|
|
410
|
+
? deriveVerifierFailureSignals(options.failureProxies)
|
|
411
|
+
: [];
|
|
412
|
+
const signals = mergeTrainingOutcomeSignals(behavioralSignals, verifierSignals);
|
|
190
413
|
|
|
191
|
-
if (signals.some((signal) => FAILURE_OUTCOME_SIGNALS.includes(signal))) {
|
|
414
|
+
if (signals.some((signal) => (FAILURE_OUTCOME_SIGNALS as readonly string[]).includes(signal))) {
|
|
192
415
|
return { success: false, outcome_signals: signals };
|
|
193
416
|
}
|
|
194
417
|
|
|
@@ -204,6 +427,57 @@ export function deriveSuccessLabel(
|
|
|
204
427
|
return { success: true, outcome_signals: signals };
|
|
205
428
|
}
|
|
206
429
|
|
|
430
|
+
/** Derive training labels from a privacy-safe export/contrib row (no prompt text). */
|
|
431
|
+
export function deriveSuccessLabelFromExportRow(
|
|
432
|
+
record: Record<string, unknown>,
|
|
433
|
+
): {
|
|
434
|
+
success: boolean | null;
|
|
435
|
+
outcome_signals: readonly TrainingOutcomeSignal[];
|
|
436
|
+
failure_proxies: PSuccessFailureProxies;
|
|
437
|
+
} {
|
|
438
|
+
const failure_proxies = extractFailureProxies(record);
|
|
439
|
+
const existingSignals = parseTrainingOutcomeSignals(record.outcome_signals);
|
|
440
|
+
const verifierSignals = deriveVerifierFailureSignals(failure_proxies);
|
|
441
|
+
const mergedSignals = mergeTrainingOutcomeSignals(existingSignals, verifierSignals);
|
|
442
|
+
|
|
443
|
+
if (mergedSignals.some((signal) => (FAILURE_OUTCOME_SIGNALS as readonly string[]).includes(signal))) {
|
|
444
|
+
return { success: false, outcome_signals: mergedSignals, failure_proxies };
|
|
445
|
+
}
|
|
446
|
+
|
|
447
|
+
if (mergedSignals.includes('feedback_good')) {
|
|
448
|
+
return { success: true, outcome_signals: mergedSignals, failure_proxies };
|
|
449
|
+
}
|
|
450
|
+
|
|
451
|
+
if (typeof record.success_label === 'boolean') {
|
|
452
|
+
return {
|
|
453
|
+
success: record.success_label,
|
|
454
|
+
outcome_signals: mergedSignals,
|
|
455
|
+
failure_proxies,
|
|
456
|
+
};
|
|
457
|
+
}
|
|
458
|
+
|
|
459
|
+
if (mergedSignals.length === 0) {
|
|
460
|
+
return { success: null, outcome_signals: mergedSignals, failure_proxies };
|
|
461
|
+
}
|
|
462
|
+
|
|
463
|
+
return { success: true, outcome_signals: mergedSignals, failure_proxies };
|
|
464
|
+
}
|
|
465
|
+
|
|
466
|
+
/** Attach normalized failure proxy fields for calibration export rows. */
|
|
467
|
+
export function attachFailureProxiesToExport(
|
|
468
|
+
exportRecord: Record<string, unknown>,
|
|
469
|
+
): Record<string, unknown> {
|
|
470
|
+
const proxies = extractFailureProxies(exportRecord);
|
|
471
|
+
|
|
472
|
+
return {
|
|
473
|
+
...exportRecord,
|
|
474
|
+
tool_failure_chain_count: proxies.tool_failure_chain_count,
|
|
475
|
+
stop_reason_invalid: proxies.stop_reason_invalid,
|
|
476
|
+
reprompt_rate: proxies.reprompt_rate,
|
|
477
|
+
edit_distance_proxy: proxies.edit_distance_proxy,
|
|
478
|
+
};
|
|
479
|
+
}
|
|
480
|
+
|
|
207
481
|
/** Group outcomes by request_id for export joins. */
|
|
208
482
|
export function indexOutcomesByRequestId(
|
|
209
483
|
outcomes: readonly RoutingOutcomeRecord[],
|
|
@@ -228,13 +502,15 @@ export function joinDatasetWithOutcomes(
|
|
|
228
502
|
|
|
229
503
|
return datasetRecords.map((record) => {
|
|
230
504
|
const linked = outcomesByRequest.get(record.request_id) ?? [];
|
|
231
|
-
const
|
|
505
|
+
const failure_proxies = extractFailureProxies(record as unknown as Record<string, unknown>);
|
|
506
|
+
const { success, outcome_signals } = deriveSuccessLabel(linked, { failureProxies: failure_proxies });
|
|
232
507
|
|
|
233
508
|
return {
|
|
234
509
|
request_id: record.request_id,
|
|
235
510
|
features: extractPSuccessFeatures(record),
|
|
236
511
|
success: success ?? true,
|
|
237
512
|
outcome_signals,
|
|
513
|
+
failure_proxies,
|
|
238
514
|
};
|
|
239
515
|
});
|
|
240
516
|
}
|
|
@@ -244,13 +520,19 @@ export function attachOutcomeLabelsToExport(
|
|
|
244
520
|
exportRecord: Record<string, unknown>,
|
|
245
521
|
outcomes: readonly RoutingOutcomeRecord[],
|
|
246
522
|
): DatasetExportJoinRow {
|
|
247
|
-
const
|
|
523
|
+
const withProxies = attachFailureProxiesToExport(exportRecord);
|
|
524
|
+
const failure_proxies = extractFailureProxies(withProxies);
|
|
525
|
+
const { success, outcome_signals } = deriveSuccessLabel(outcomes, { failureProxies: failure_proxies });
|
|
248
526
|
|
|
249
527
|
return {
|
|
250
|
-
...
|
|
528
|
+
...withProxies,
|
|
251
529
|
request_id: String(exportRecord.request_id ?? ''),
|
|
252
530
|
success_label: success,
|
|
253
531
|
outcome_signals,
|
|
532
|
+
tool_failure_chain_count: failure_proxies.tool_failure_chain_count,
|
|
533
|
+
stop_reason_invalid: failure_proxies.stop_reason_invalid,
|
|
534
|
+
reprompt_rate: failure_proxies.reprompt_rate,
|
|
535
|
+
edit_distance_proxy: failure_proxies.edit_distance_proxy,
|
|
254
536
|
};
|
|
255
537
|
}
|
|
256
538
|
|
|
@@ -325,24 +607,15 @@ export function parseTrainingExportLine(line: string): LabeledTrainingSample | n
|
|
|
325
607
|
return null;
|
|
326
608
|
}
|
|
327
609
|
|
|
328
|
-
const
|
|
329
|
-
let success = true;
|
|
330
|
-
if (typeof successLabel === 'boolean') {
|
|
331
|
-
success = successLabel;
|
|
332
|
-
}
|
|
333
|
-
|
|
334
|
-
const rawSignals = parsed.outcome_signals;
|
|
335
|
-
const outcome_signals = Array.isArray(rawSignals)
|
|
336
|
-
? rawSignals.filter((value): value is OutcomeSignalType => typeof value === 'string')
|
|
337
|
-
: [];
|
|
338
|
-
|
|
610
|
+
const labeled = deriveSuccessLabelFromExportRow(parsed);
|
|
339
611
|
const record = parsed as unknown as RoutingDatasetRecord;
|
|
340
612
|
|
|
341
613
|
return {
|
|
342
614
|
request_id: requestId,
|
|
343
615
|
features: extractPSuccessFeatures(record),
|
|
344
|
-
success: success === false ? false : true,
|
|
345
|
-
outcome_signals,
|
|
616
|
+
success: labeled.success === false ? false : true,
|
|
617
|
+
outcome_signals: labeled.outcome_signals,
|
|
618
|
+
failure_proxies: labeled.failure_proxies,
|
|
346
619
|
};
|
|
347
620
|
}
|
|
348
621
|
|
|
@@ -192,6 +192,8 @@ export interface LowIntensityBreakdown {
|
|
|
192
192
|
readonly tier_hint_reason_code: string | null;
|
|
193
193
|
readonly tier_selection_reason_code: string | null;
|
|
194
194
|
readonly p_success_cheap: number | null;
|
|
195
|
+
readonly p_success_raw: number | null;
|
|
196
|
+
readonly p_success_calibrated: number | null;
|
|
195
197
|
readonly p_success_alpha: number | null;
|
|
196
198
|
readonly rejected_tiers: readonly RejectedTierEntry[];
|
|
197
199
|
}
|
|
@@ -259,6 +261,10 @@ export interface RoutingFeatureSidecar {
|
|
|
259
261
|
readonly low_intensity_score: number | null;
|
|
260
262
|
/** P(success) cheap-tier probability from low_intensity gate (SP-105). */
|
|
261
263
|
readonly p_success_cheap: number | null;
|
|
264
|
+
/** Raw logistic P(success) before isotonic calibration (SP-133). */
|
|
265
|
+
readonly p_success_raw: number | null;
|
|
266
|
+
/** Isotonic-calibrated P(success) used for gate thresholding (SP-133). */
|
|
267
|
+
readonly p_success_calibrated: number | null;
|
|
262
268
|
/** Operator alpha threshold used for P(success) routing (SP-105). */
|
|
263
269
|
readonly p_success_alpha: number | null;
|
|
264
270
|
/** Context-fit gate observability (SP-110). */
|
|
@@ -388,6 +388,8 @@ function buildLowIntensityBreakdown(
|
|
|
388
388
|
tier_hint_reason_code: features.tier_hint_reason_code,
|
|
389
389
|
tier_selection_reason_code: resolveTierSelectionReasonCode(features),
|
|
390
390
|
p_success_cheap: features.p_success_cheap,
|
|
391
|
+
p_success_raw: features.p_success_raw,
|
|
392
|
+
p_success_calibrated: features.p_success_calibrated,
|
|
391
393
|
p_success_alpha: features.p_success_alpha,
|
|
392
394
|
rejected_tiers: extractRejectedTiers(features.candidates),
|
|
393
395
|
};
|
|
@@ -722,6 +724,8 @@ function emptyFeatureSidecar() {
|
|
|
722
724
|
tier_hint_reason_code: null,
|
|
723
725
|
low_intensity_score: null,
|
|
724
726
|
p_success_cheap: null,
|
|
727
|
+
p_success_raw: null,
|
|
728
|
+
p_success_calibrated: null,
|
|
725
729
|
p_success_alpha: null,
|
|
726
730
|
local_eligible_reason: null,
|
|
727
731
|
};
|