open-multi-agent-kit 1.0.0 → 1.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (244) hide show
  1. package/CHANGELOG.md +6 -1
  2. package/README.md +2 -2
  3. package/dist/cli/help.d.ts.map +1 -1
  4. package/dist/cli/help.js +1 -0
  5. package/dist/cli/help.js.map +1 -1
  6. package/dist/coordination/awareness.d.ts +32 -0
  7. package/dist/coordination/awareness.d.ts.map +1 -0
  8. package/dist/coordination/awareness.js +55 -0
  9. package/dist/coordination/awareness.js.map +1 -0
  10. package/dist/coordination/broker.d.ts +79 -0
  11. package/dist/coordination/broker.d.ts.map +1 -0
  12. package/dist/coordination/broker.js +195 -0
  13. package/dist/coordination/broker.js.map +1 -0
  14. package/dist/coordination/index.d.ts +16 -0
  15. package/dist/coordination/index.d.ts.map +1 -0
  16. package/dist/coordination/index.js +16 -0
  17. package/dist/coordination/index.js.map +1 -0
  18. package/dist/coordination/integration.d.ts +59 -0
  19. package/dist/coordination/integration.d.ts.map +1 -0
  20. package/dist/coordination/integration.js +126 -0
  21. package/dist/coordination/integration.js.map +1 -0
  22. package/dist/coordination/operation.d.ts +106 -0
  23. package/dist/coordination/operation.d.ts.map +1 -0
  24. package/dist/coordination/operation.js +179 -0
  25. package/dist/coordination/operation.js.map +1 -0
  26. package/dist/coordination/resource.d.ts +33 -0
  27. package/dist/coordination/resource.d.ts.map +1 -0
  28. package/dist/coordination/resource.js +89 -0
  29. package/dist/coordination/resource.js.map +1 -0
  30. package/dist/coordination/session.d.ts +74 -0
  31. package/dist/coordination/session.d.ts.map +1 -0
  32. package/dist/coordination/session.js +143 -0
  33. package/dist/coordination/session.js.map +1 -0
  34. package/dist/coordination/types.d.ts +68 -0
  35. package/dist/coordination/types.d.ts.map +1 -0
  36. package/dist/coordination/types.js +26 -0
  37. package/dist/coordination/types.js.map +1 -0
  38. package/dist/core/agent-session.d.ts.map +1 -1
  39. package/dist/core/agent-session.js +28 -7
  40. package/dist/core/agent-session.js.map +1 -1
  41. package/dist/core/compaction/control-state.d.ts +41 -0
  42. package/dist/core/compaction/control-state.d.ts.map +1 -0
  43. package/dist/core/compaction/control-state.js +85 -0
  44. package/dist/core/compaction/control-state.js.map +1 -0
  45. package/dist/core/compaction/fallback.d.ts +60 -0
  46. package/dist/core/compaction/fallback.d.ts.map +1 -0
  47. package/dist/core/compaction/fallback.js +117 -0
  48. package/dist/core/compaction/fallback.js.map +1 -0
  49. package/dist/core/compaction/index.d.ts +2 -0
  50. package/dist/core/compaction/index.d.ts.map +1 -1
  51. package/dist/core/compaction/index.js +2 -0
  52. package/dist/core/compaction/index.js.map +1 -1
  53. package/dist/core/context-budget-headroom-candidates.d.ts +16 -1
  54. package/dist/core/context-budget-headroom-candidates.d.ts.map +1 -1
  55. package/dist/core/context-budget-headroom-candidates.js +31 -13
  56. package/dist/core/context-budget-headroom-candidates.js.map +1 -1
  57. package/dist/core/context-budget-headroom-types.d.ts +7 -1
  58. package/dist/core/context-budget-headroom-types.d.ts.map +1 -1
  59. package/dist/core/context-budget-headroom-types.js +0 -1
  60. package/dist/core/context-budget-headroom-types.js.map +1 -1
  61. package/dist/core/context-budget-headroom.d.ts +10 -0
  62. package/dist/core/context-budget-headroom.d.ts.map +1 -1
  63. package/dist/core/context-budget-headroom.js +11 -2
  64. package/dist/core/context-budget-headroom.js.map +1 -1
  65. package/dist/core/context-budget-v2-global-pass.d.ts +21 -0
  66. package/dist/core/context-budget-v2-global-pass.d.ts.map +1 -0
  67. package/dist/core/context-budget-v2-global-pass.js +203 -0
  68. package/dist/core/context-budget-v2-global-pass.js.map +1 -0
  69. package/dist/core/context-budget-v2-planned-items.d.ts +12 -0
  70. package/dist/core/context-budget-v2-planned-items.d.ts.map +1 -1
  71. package/dist/core/context-budget-v2-planned-items.js +21 -2
  72. package/dist/core/context-budget-v2-planned-items.js.map +1 -1
  73. package/dist/core/context-budget-v2-planner.d.ts.map +1 -1
  74. package/dist/core/context-budget-v2-planner.js +18 -19
  75. package/dist/core/context-budget-v2-planner.js.map +1 -1
  76. package/dist/core/context-budget-v2-scoring.d.ts +7 -1
  77. package/dist/core/context-budget-v2-scoring.d.ts.map +1 -1
  78. package/dist/core/context-budget-v2-scoring.js.map +1 -1
  79. package/dist/core/context-budget-v2-selection.d.ts +3 -0
  80. package/dist/core/context-budget-v2-selection.d.ts.map +1 -1
  81. package/dist/core/context-budget-v2-selection.js +6 -2
  82. package/dist/core/context-budget-v2-selection.js.map +1 -1
  83. package/dist/core/context-budget-v2-types.d.ts +6 -1
  84. package/dist/core/context-budget-v2-types.d.ts.map +1 -1
  85. package/dist/core/context-budget-v2-types.js +6 -1
  86. package/dist/core/context-budget-v2-types.js.map +1 -1
  87. package/dist/core/provider-default-models.d.ts +1 -0
  88. package/dist/core/provider-default-models.d.ts.map +1 -1
  89. package/dist/core/provider-default-models.js +1 -0
  90. package/dist/core/provider-default-models.js.map +1 -1
  91. package/dist/core/provider-display-names.d.ts.map +1 -1
  92. package/dist/core/provider-display-names.js +1 -0
  93. package/dist/core/provider-display-names.js.map +1 -1
  94. package/dist/core/reasoning-router-resolver.d.ts +7 -0
  95. package/dist/core/reasoning-router-resolver.d.ts.map +1 -1
  96. package/dist/core/reasoning-router-resolver.js +21 -3
  97. package/dist/core/reasoning-router-resolver.js.map +1 -1
  98. package/dist/core/reasoning-router-v4.d.ts +8 -5
  99. package/dist/core/reasoning-router-v4.d.ts.map +1 -1
  100. package/dist/core/reasoning-router-v4.js +8 -5
  101. package/dist/core/reasoning-router-v4.js.map +1 -1
  102. package/dist/core/session-compaction-service.d.ts +15 -0
  103. package/dist/core/session-compaction-service.d.ts.map +1 -1
  104. package/dist/core/session-compaction-service.js +17 -4
  105. package/dist/core/session-compaction-service.js.map +1 -1
  106. package/dist/core/todo-runtime-state.d.ts +12 -0
  107. package/dist/core/todo-runtime-state.d.ts.map +1 -1
  108. package/dist/core/todo-runtime-state.js +23 -0
  109. package/dist/core/todo-runtime-state.js.map +1 -1
  110. package/dist/index.d.ts +2 -0
  111. package/dist/index.d.ts.map +1 -1
  112. package/dist/index.js +2 -0
  113. package/dist/index.js.map +1 -1
  114. package/dist/metacognition/calibration-selective.d.ts +78 -0
  115. package/dist/metacognition/calibration-selective.d.ts.map +1 -0
  116. package/dist/metacognition/calibration-selective.js +161 -0
  117. package/dist/metacognition/calibration-selective.js.map +1 -0
  118. package/dist/metacognition/calibration.d.ts +62 -0
  119. package/dist/metacognition/calibration.d.ts.map +1 -0
  120. package/dist/metacognition/calibration.js +122 -0
  121. package/dist/metacognition/calibration.js.map +1 -0
  122. package/dist/metacognition/checkpoint.d.ts +60 -0
  123. package/dist/metacognition/checkpoint.d.ts.map +1 -0
  124. package/dist/metacognition/checkpoint.js +57 -0
  125. package/dist/metacognition/checkpoint.js.map +1 -0
  126. package/dist/metacognition/context7.d.ts +48 -0
  127. package/dist/metacognition/context7.d.ts.map +1 -0
  128. package/dist/metacognition/context7.js +154 -0
  129. package/dist/metacognition/context7.js.map +1 -0
  130. package/dist/metacognition/decision.d.ts +44 -0
  131. package/dist/metacognition/decision.d.ts.map +1 -0
  132. package/dist/metacognition/decision.js +111 -0
  133. package/dist/metacognition/decision.js.map +1 -0
  134. package/dist/metacognition/evaluation.d.ts +73 -0
  135. package/dist/metacognition/evaluation.d.ts.map +1 -0
  136. package/dist/metacognition/evaluation.js +97 -0
  137. package/dist/metacognition/evaluation.js.map +1 -0
  138. package/dist/metacognition/index.d.ts +31 -0
  139. package/dist/metacognition/index.d.ts.map +1 -0
  140. package/dist/metacognition/index.js +31 -0
  141. package/dist/metacognition/index.js.map +1 -0
  142. package/dist/metacognition/knowledge-action.d.ts +43 -0
  143. package/dist/metacognition/knowledge-action.d.ts.map +1 -0
  144. package/dist/metacognition/knowledge-action.js +60 -0
  145. package/dist/metacognition/knowledge-action.js.map +1 -0
  146. package/dist/metacognition/knowledge.d.ts +84 -0
  147. package/dist/metacognition/knowledge.d.ts.map +1 -0
  148. package/dist/metacognition/knowledge.js +165 -0
  149. package/dist/metacognition/knowledge.js.map +1 -0
  150. package/dist/metacognition/obligations.d.ts +82 -0
  151. package/dist/metacognition/obligations.d.ts.map +1 -0
  152. package/dist/metacognition/obligations.js +139 -0
  153. package/dist/metacognition/obligations.js.map +1 -0
  154. package/dist/metacognition/observation-validity.d.ts +83 -0
  155. package/dist/metacognition/observation-validity.d.ts.map +1 -0
  156. package/dist/metacognition/observation-validity.js +118 -0
  157. package/dist/metacognition/observation-validity.js.map +1 -0
  158. package/dist/metacognition/observe.d.ts +45 -0
  159. package/dist/metacognition/observe.d.ts.map +1 -0
  160. package/dist/metacognition/observe.js +59 -0
  161. package/dist/metacognition/observe.js.map +1 -0
  162. package/dist/metacognition/policy.d.ts +53 -0
  163. package/dist/metacognition/policy.d.ts.map +1 -0
  164. package/dist/metacognition/policy.js +155 -0
  165. package/dist/metacognition/policy.js.map +1 -0
  166. package/dist/metacognition/predictions.d.ts +78 -0
  167. package/dist/metacognition/predictions.d.ts.map +1 -0
  168. package/dist/metacognition/predictions.js +181 -0
  169. package/dist/metacognition/predictions.js.map +1 -0
  170. package/dist/metacognition/retrieval.d.ts +53 -0
  171. package/dist/metacognition/retrieval.d.ts.map +1 -0
  172. package/dist/metacognition/retrieval.js +170 -0
  173. package/dist/metacognition/retrieval.js.map +1 -0
  174. package/dist/metacognition/risk.d.ts +48 -0
  175. package/dist/metacognition/risk.d.ts.map +1 -0
  176. package/dist/metacognition/risk.js +165 -0
  177. package/dist/metacognition/risk.js.map +1 -0
  178. package/dist/metacognition/route-economics.d.ts +95 -0
  179. package/dist/metacognition/route-economics.d.ts.map +1 -0
  180. package/dist/metacognition/route-economics.js +110 -0
  181. package/dist/metacognition/route-economics.js.map +1 -0
  182. package/dist/metacognition/runtime-bridge.d.ts +54 -0
  183. package/dist/metacognition/runtime-bridge.d.ts.map +1 -0
  184. package/dist/metacognition/runtime-bridge.js +140 -0
  185. package/dist/metacognition/runtime-bridge.js.map +1 -0
  186. package/dist/metacognition/skills.d.ts +42 -0
  187. package/dist/metacognition/skills.d.ts.map +1 -0
  188. package/dist/metacognition/skills.js +220 -0
  189. package/dist/metacognition/skills.js.map +1 -0
  190. package/dist/metacognition/state.d.ts +110 -0
  191. package/dist/metacognition/state.d.ts.map +1 -0
  192. package/dist/metacognition/state.js +52 -0
  193. package/dist/metacognition/state.js.map +1 -0
  194. package/dist/metacognition/validation.d.ts +19 -0
  195. package/dist/metacognition/validation.d.ts.map +1 -0
  196. package/dist/metacognition/validation.js +58 -0
  197. package/dist/metacognition/validation.js.map +1 -0
  198. package/dist/metacognition/verification.d.ts +93 -0
  199. package/dist/metacognition/verification.d.ts.map +1 -0
  200. package/dist/metacognition/verification.js +132 -0
  201. package/dist/metacognition/verification.js.map +1 -0
  202. package/dist/metacognition/verifier.d.ts +49 -0
  203. package/dist/metacognition/verifier.d.ts.map +1 -0
  204. package/dist/metacognition/verifier.js +88 -0
  205. package/dist/metacognition/verifier.js.map +1 -0
  206. package/dist/observation/identity.d.ts +14 -0
  207. package/dist/observation/identity.d.ts.map +1 -0
  208. package/dist/observation/identity.js +29 -0
  209. package/dist/observation/identity.js.map +1 -0
  210. package/dist/observation/index.d.ts +13 -0
  211. package/dist/observation/index.d.ts.map +1 -0
  212. package/dist/observation/index.js +13 -0
  213. package/dist/observation/index.js.map +1 -0
  214. package/dist/observation/observe-mode.d.ts +46 -0
  215. package/dist/observation/observe-mode.d.ts.map +1 -0
  216. package/dist/observation/observe-mode.js +83 -0
  217. package/dist/observation/observe-mode.js.map +1 -0
  218. package/dist/observation/store.d.ts +53 -0
  219. package/dist/observation/store.d.ts.map +1 -0
  220. package/dist/observation/store.js +126 -0
  221. package/dist/observation/store.js.map +1 -0
  222. package/dist/observation/types.d.ts +72 -0
  223. package/dist/observation/types.d.ts.map +1 -0
  224. package/dist/observation/types.js +9 -0
  225. package/dist/observation/types.js.map +1 -0
  226. package/dist/observation/view.d.ts +39 -0
  227. package/dist/observation/view.d.ts.map +1 -0
  228. package/dist/observation/view.js +173 -0
  229. package/dist/observation/view.js.map +1 -0
  230. package/docs/metacognition.md +51 -0
  231. package/docs/model-catalog-refresh.md +59 -0
  232. package/docs/providers.md +53 -0
  233. package/docs/runtime-algorithms.md +107 -3
  234. package/examples/extensions/custom-provider-anthropic/package-lock.json +2 -2
  235. package/examples/extensions/custom-provider-anthropic/package.json +1 -1
  236. package/examples/extensions/custom-provider-gitlab-duo/package.json +1 -1
  237. package/examples/extensions/gondolin/package-lock.json +2 -2
  238. package/examples/extensions/gondolin/package.json +1 -1
  239. package/examples/extensions/sandbox/package-lock.json +2 -2
  240. package/examples/extensions/sandbox/package.json +1 -1
  241. package/examples/extensions/with-deps/package-lock.json +2 -2
  242. package/examples/extensions/with-deps/package.json +1 -1
  243. package/npm-shrinkwrap.json +18 -18
  244. package/package.json +6 -6
@@ -0,0 +1,78 @@
1
+ /**
2
+ * Probability calibration and selective execution (Jev audit algorithm A3).
3
+ *
4
+ * A provider's `confidence` is a statistic over its own option distribution,
5
+ * not a calibrated probability of any event we care about. Before it can gate
6
+ * execution, the predicted event has to be named — "the chosen candidate was
7
+ * appropriate" is a different event from "the driver executed it" and from
8
+ * "the task succeeded" — and the score has to be measured on held-out data.
9
+ *
10
+ * These are the measurement primitives, deliberately small: distribution
11
+ * features, a post-hoc temperature rule, scoring rules, and a selective gate
12
+ * that always reports coverage next to risk. None of them makes a score
13
+ * trustworthy; they make it measurable.
14
+ */
15
+ /** Floor for numeric zeros in temperature scaling; recorded so results are reproducible. */
16
+ export declare const TEMPERATURE_EPSILON = 1e-12;
17
+ /** Clip for log loss so one confident miss does not return Infinity. */
18
+ export declare const LOG_LOSS_CLIP = 1e-15;
19
+ export interface DistributionFeatures {
20
+ readonly maxProbability: number;
21
+ /** p(1) − p(2). Undefined for a single candidate: there is nothing to compare. */
22
+ readonly margin: number | undefined;
23
+ /** Undefined for a single candidate, where entropy carries no information. */
24
+ readonly normalizedEntropy: number | undefined;
25
+ readonly candidateCount: number;
26
+ readonly logCandidateCount: number;
27
+ }
28
+ /**
29
+ * Summary features of a candidate distribution.
30
+ *
31
+ * A single candidate can show probability 1 while telling you nothing about
32
+ * whether that candidate is correct, so its margin and entropy are undefined
33
+ * rather than maximally confident.
34
+ */
35
+ export declare function distributionFeatures(probabilities: readonly number[]): DistributionFeatures;
36
+ /**
37
+ * Post-hoc temperature rule: p̃(a) ∝ max(p(a), ε)^(1/T).
38
+ *
39
+ * T is fitted on a calibration split, never on the data being scored. This
40
+ * does not assume the input came from a softmax; whether it helps has to be
41
+ * measured separately.
42
+ */
43
+ export declare function temperatureScale(probabilities: readonly number[], temperature: number, epsilon?: number): number[];
44
+ export declare function brierScore(forecasts: readonly number[], outcomes: readonly number[]): number;
45
+ export declare function negativeLogLoss(forecasts: readonly number[], outcomes: readonly number[], clip?: number): number;
46
+ export interface CalibrationBin {
47
+ readonly lower: number;
48
+ readonly upper: number;
49
+ readonly count: number;
50
+ readonly accuracy: number | undefined;
51
+ readonly confidence: number | undefined;
52
+ }
53
+ export interface CalibrationReport {
54
+ readonly error: number;
55
+ readonly bins: readonly CalibrationBin[];
56
+ readonly samples: number;
57
+ }
58
+ /**
59
+ * Expected calibration error with its bin occupancy.
60
+ *
61
+ * ECE is sensitive to binning, so the bins and their counts are returned with
62
+ * it: a near-zero error over three samples is not evidence of calibration.
63
+ */
64
+ export declare function expectedCalibrationError(forecasts: readonly number[], outcomes: readonly number[], binCount: number): CalibrationReport;
65
+ export interface SelectiveExecutionReport {
66
+ readonly admitted: number;
67
+ readonly coverage: number;
68
+ /** Undefined when nothing was admitted: that is unmeasurable, not risk-free. */
69
+ readonly risk: number | undefined;
70
+ }
71
+ /**
72
+ * Selective execution gate: a(τ) = 1[q̂ ≥ τ] · 1[G = 1].
73
+ *
74
+ * Risk and coverage are returned together because refusing everything drives
75
+ * measured risk to zero while delivering nothing.
76
+ */
77
+ export declare function selectiveExecution(scores: readonly number[], appropriate: readonly number[], threshold: number, policyGates: readonly boolean[]): SelectiveExecutionReport;
78
+ //# sourceMappingURL=calibration-selective.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"calibration-selective.d.ts","sourceRoot":"","sources":["../../src/metacognition/calibration-selective.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;GAaG;AAKH,4FAA4F;AAC5F,eAAO,MAAM,mBAAmB,QAAQ,CAAC;AACzC,wEAAwE;AACxE,eAAO,MAAM,aAAa,QAAQ,CAAC;AAEnC,MAAM,WAAW,oBAAoB;IACpC,QAAQ,CAAC,cAAc,EAAE,MAAM,CAAC;IAChC,oFAAkF;IAClF,QAAQ,CAAC,MAAM,EAAE,MAAM,GAAG,SAAS,CAAC;IACpC,8EAA8E;IAC9E,QAAQ,CAAC,iBAAiB,EAAE,MAAM,GAAG,SAAS,CAAC;IAC/C,QAAQ,CAAC,cAAc,EAAE,MAAM,CAAC;IAChC,QAAQ,CAAC,iBAAiB,EAAE,MAAM,CAAC;CACnC;AAYD;;;;;;GAMG;AACH,wBAAgB,oBAAoB,CAAC,aAAa,EAAE,SAAS,MAAM,EAAE,GAAG,oBAAoB,CAwB3F;AAED;;;;;;GAMG;AACH,wBAAgB,gBAAgB,CAC/B,aAAa,EAAE,SAAS,MAAM,EAAE,EAChC,WAAW,EAAE,MAAM,EACnB,OAAO,GAAE,MAA4B,GACnC,MAAM,EAAE,CAWV;AAYD,wBAAgB,UAAU,CAAC,SAAS,EAAE,SAAS,MAAM,EAAE,EAAE,QAAQ,EAAE,SAAS,MAAM,EAAE,GAAG,MAAM,CAK5F;AAED,wBAAgB,eAAe,CAC9B,SAAS,EAAE,SAAS,MAAM,EAAE,EAC5B,QAAQ,EAAE,SAAS,MAAM,EAAE,EAC3B,IAAI,GAAE,MAAsB,GAC1B,MAAM,CASR;AAED,MAAM,WAAW,cAAc;IAC9B,QAAQ,CAAC,KAAK,EAAE,MAAM,CAAC;IACvB,QAAQ,CAAC,KAAK,EAAE,MAAM,CAAC;IACvB,QAAQ,CAAC,KAAK,EAAE,MAAM,CAAC;IACvB,QAAQ,CAAC,QAAQ,EAAE,MAAM,GAAG,SAAS,CAAC;IACtC,QAAQ,CAAC,UAAU,EAAE,MAAM,GAAG,SAAS,CAAC;CACxC;AAED,MAAM,WAAW,iBAAiB;IACjC,QAAQ,CAAC,KAAK,EAAE,MAAM,CAAC;IACvB,QAAQ,CAAC,IAAI,EAAE,SAAS,cAAc,EAAE,CAAC;IACzC,QAAQ,CAAC,OAAO,EAAE,MAAM,CAAC;CACzB;AAED;;;;;GAKG;AACH,wBAAgB,wBAAwB,CACvC,SAAS,EAAE,SAAS,MAAM,EAAE,EAC5B,QAAQ,EAAE,SAAS,MAAM,EAAE,EAC3B,QAAQ,EAAE,MAAM,GACd,iBAAiB,CAsBnB;AAED,MAAM,WAAW,wBAAwB;IACxC,QAAQ,CAAC,QAAQ,EAAE,MAAM,CAAC;IAC1B,QAAQ,CAAC,QAAQ,EAAE,MAAM,CAAC;IAC1B,gFAAgF;IAChF,QAAQ,CAAC,IAAI,EAAE,MAAM,GAAG,SAAS,CAAC;CAClC;AAED;;;;;GAKG;AACH,wBAAgB,kBAAkB,CACjC,MAAM,EAAE,SAAS,MAAM,EAAE,EACzB,WAAW,EAAE,SAAS,MAAM,EAAE,EAC9B,SAAS,EAAE,MAAM,EACjB,WAAW,EAAE,SAAS,OAAO,EAAE,GAC7B,wBAAwB,CAmB1B","sourcesContent":["/**\n * Probability calibration and selective execution (Jev audit algorithm A3).\n *\n * A provider's `confidence` is a statistic over its own option distribution,\n * not a calibrated probability of any event we care about. Before it can gate\n * execution, the predicted event has to be named — \"the chosen candidate was\n * appropriate\" is a different event from \"the driver executed it\" and from\n * \"the task succeeded\" — and the score has to be measured on held-out data.\n *\n * These are the measurement primitives, deliberately small: distribution\n * features, a post-hoc temperature rule, scoring rules, and a selective gate\n * that always reports coverage next to risk. None of them makes a score\n * trustworthy; they make it measurable.\n */\n\nimport { ensure } from \"./validation.ts\";\n\nconst SUM_TOLERANCE = 1e-9;\n/** Floor for numeric zeros in temperature scaling; recorded so results are reproducible. */\nexport const TEMPERATURE_EPSILON = 1e-12;\n/** Clip for log loss so one confident miss does not return Infinity. */\nexport const LOG_LOSS_CLIP = 1e-15;\n\nexport interface DistributionFeatures {\n\treadonly maxProbability: number;\n\t/** p(1) − p(2). Undefined for a single candidate: there is nothing to compare. */\n\treadonly margin: number | undefined;\n\t/** Undefined for a single candidate, where entropy carries no information. */\n\treadonly normalizedEntropy: number | undefined;\n\treadonly candidateCount: number;\n\treadonly logCandidateCount: number;\n}\n\nfunction assertDistribution(probabilities: readonly number[]): void {\n\tensure(Array.isArray(probabilities) && probabilities.length > 0, \"distribution must be non-empty\");\n\tlet total = 0;\n\tfor (const p of probabilities) {\n\t\tensure(typeof p === \"number\" && Number.isFinite(p) && p >= 0 && p <= 1, \"each probability must be in [0,1]\");\n\t\ttotal += p;\n\t}\n\tensure(Math.abs(total - 1) <= SUM_TOLERANCE, \"distribution must sum to one\");\n}\n\n/**\n * Summary features of a candidate distribution.\n *\n * A single candidate can show probability 1 while telling you nothing about\n * whether that candidate is correct, so its margin and entropy are undefined\n * rather than maximally confident.\n */\nexport function distributionFeatures(probabilities: readonly number[]): DistributionFeatures {\n\tassertDistribution(probabilities);\n\tconst sorted = [...probabilities].sort((a, b) => b - a);\n\tconst count = probabilities.length;\n\tconst maxProbability = sorted[0] ?? 0;\n\tif (count === 1) {\n\t\treturn {\n\t\t\tmaxProbability,\n\t\t\tmargin: undefined,\n\t\t\tnormalizedEntropy: undefined,\n\t\t\tcandidateCount: 1,\n\t\t\tlogCandidateCount: 0,\n\t\t};\n\t}\n\t// 0·log0 is defined as 0; skipping the term is exactly that definition.\n\tlet entropy = 0;\n\tfor (const p of probabilities) if (p > 0) entropy -= p * Math.log(p);\n\treturn {\n\t\tmaxProbability,\n\t\tmargin: maxProbability - (sorted[1] ?? 0),\n\t\tnormalizedEntropy: entropy / Math.log(count),\n\t\tcandidateCount: count,\n\t\tlogCandidateCount: Math.log(count),\n\t};\n}\n\n/**\n * Post-hoc temperature rule: p̃(a) ∝ max(p(a), ε)^(1/T).\n *\n * T is fitted on a calibration split, never on the data being scored. This\n * does not assume the input came from a softmax; whether it helps has to be\n * measured separately.\n */\nexport function temperatureScale(\n\tprobabilities: readonly number[],\n\ttemperature: number,\n\tepsilon: number = TEMPERATURE_EPSILON,\n): number[] {\n\tassertDistribution(probabilities);\n\tensure(\n\t\ttypeof temperature === \"number\" && Number.isFinite(temperature) && temperature > 0,\n\t\t\"temperature must be positive\",\n\t);\n\tensure(typeof epsilon === \"number\" && epsilon > 0 && epsilon < 1, \"epsilon must be in (0,1)\");\n\tconst raised = probabilities.map((p) => Math.max(p, epsilon) ** (1 / temperature));\n\tconst total = raised.reduce((sum, value) => sum + value, 0);\n\tensure(total > 0 && Number.isFinite(total), \"temperature scaling produced a degenerate distribution\");\n\treturn raised.map((value) => value / total);\n}\n\nfunction assertBinaryPaired(forecasts: readonly number[], outcomes: readonly number[]): void {\n\tensure(Array.isArray(forecasts) && Array.isArray(outcomes), \"forecasts and outcomes must be arrays\");\n\tensure(forecasts.length === outcomes.length, \"forecasts and outcomes must align\");\n\tensure(forecasts.length > 0, \"at least one paired observation is required\");\n\tfor (const q of forecasts) {\n\t\tensure(typeof q === \"number\" && Number.isFinite(q) && q >= 0 && q <= 1, \"each forecast must be in [0,1]\");\n\t}\n\tfor (const y of outcomes) ensure(y === 0 || y === 1, \"each outcome must be 0 or 1\");\n}\n\nexport function brierScore(forecasts: readonly number[], outcomes: readonly number[]): number {\n\tassertBinaryPaired(forecasts, outcomes);\n\tlet total = 0;\n\tfor (const [i, q] of forecasts.entries()) total += (q - outcomes[i]!) ** 2;\n\treturn total / forecasts.length;\n}\n\nexport function negativeLogLoss(\n\tforecasts: readonly number[],\n\toutcomes: readonly number[],\n\tclip: number = LOG_LOSS_CLIP,\n): number {\n\tassertBinaryPaired(forecasts, outcomes);\n\tensure(typeof clip === \"number\" && clip > 0 && clip < 0.5, \"clip must be in (0,0.5)\");\n\tlet total = 0;\n\tfor (const [i, raw] of forecasts.entries()) {\n\t\tconst q = Math.min(1 - clip, Math.max(clip, raw));\n\t\ttotal -= outcomes[i] === 1 ? Math.log(q) : Math.log(1 - q);\n\t}\n\treturn total / forecasts.length;\n}\n\nexport interface CalibrationBin {\n\treadonly lower: number;\n\treadonly upper: number;\n\treadonly count: number;\n\treadonly accuracy: number | undefined;\n\treadonly confidence: number | undefined;\n}\n\nexport interface CalibrationReport {\n\treadonly error: number;\n\treadonly bins: readonly CalibrationBin[];\n\treadonly samples: number;\n}\n\n/**\n * Expected calibration error with its bin occupancy.\n *\n * ECE is sensitive to binning, so the bins and their counts are returned with\n * it: a near-zero error over three samples is not evidence of calibration.\n */\nexport function expectedCalibrationError(\n\tforecasts: readonly number[],\n\toutcomes: readonly number[],\n\tbinCount: number,\n): CalibrationReport {\n\tassertBinaryPaired(forecasts, outcomes);\n\tensure(Number.isSafeInteger(binCount) && binCount > 0, \"binCount must be a positive integer\");\n\tconst sums = Array.from({ length: binCount }, () => ({ count: 0, outcome: 0, forecast: 0 }));\n\tfor (const [i, q] of forecasts.entries()) {\n\t\tconst index = Math.min(binCount - 1, Math.floor(q * binCount));\n\t\tconst bin = sums[index]!;\n\t\tbin.count += 1;\n\t\tbin.outcome += outcomes[i]!;\n\t\tbin.forecast += q;\n\t}\n\tconst samples = forecasts.length;\n\tlet error = 0;\n\tconst bins = sums.map((bin, index) => {\n\t\tconst accuracy = bin.count > 0 ? bin.outcome / bin.count : undefined;\n\t\tconst confidence = bin.count > 0 ? bin.forecast / bin.count : undefined;\n\t\tif (accuracy !== undefined && confidence !== undefined) {\n\t\t\terror += (bin.count / samples) * Math.abs(accuracy - confidence);\n\t\t}\n\t\treturn { lower: index / binCount, upper: (index + 1) / binCount, count: bin.count, accuracy, confidence };\n\t});\n\treturn { error, bins, samples };\n}\n\nexport interface SelectiveExecutionReport {\n\treadonly admitted: number;\n\treadonly coverage: number;\n\t/** Undefined when nothing was admitted: that is unmeasurable, not risk-free. */\n\treadonly risk: number | undefined;\n}\n\n/**\n * Selective execution gate: a(τ) = 1[q̂ ≥ τ] · 1[G = 1].\n *\n * Risk and coverage are returned together because refusing everything drives\n * measured risk to zero while delivering nothing.\n */\nexport function selectiveExecution(\n\tscores: readonly number[],\n\tappropriate: readonly number[],\n\tthreshold: number,\n\tpolicyGates: readonly boolean[],\n): SelectiveExecutionReport {\n\tassertBinaryPaired(scores, appropriate);\n\tensure(policyGates.length === scores.length, \"policy gates must align with scores\");\n\tensure(\n\t\ttypeof threshold === \"number\" && Number.isFinite(threshold) && threshold >= 0 && threshold <= 1,\n\t\t\"threshold must be in [0,1]\",\n\t);\n\tlet admitted = 0;\n\tlet errors = 0;\n\tfor (const [i, score] of scores.entries()) {\n\t\tif (score < threshold || policyGates[i] !== true) continue;\n\t\tadmitted += 1;\n\t\tif (appropriate[i] === 0) errors += 1;\n\t}\n\treturn {\n\t\tadmitted,\n\t\tcoverage: admitted / scores.length,\n\t\trisk: admitted > 0 ? errors / admitted : undefined,\n\t};\n}\n"]}
@@ -0,0 +1,161 @@
1
+ /**
2
+ * Probability calibration and selective execution (Jev audit algorithm A3).
3
+ *
4
+ * A provider's `confidence` is a statistic over its own option distribution,
5
+ * not a calibrated probability of any event we care about. Before it can gate
6
+ * execution, the predicted event has to be named — "the chosen candidate was
7
+ * appropriate" is a different event from "the driver executed it" and from
8
+ * "the task succeeded" — and the score has to be measured on held-out data.
9
+ *
10
+ * These are the measurement primitives, deliberately small: distribution
11
+ * features, a post-hoc temperature rule, scoring rules, and a selective gate
12
+ * that always reports coverage next to risk. None of them makes a score
13
+ * trustworthy; they make it measurable.
14
+ */
15
+ import { ensure } from "./validation.js";
16
+ const SUM_TOLERANCE = 1e-9;
17
+ /** Floor for numeric zeros in temperature scaling; recorded so results are reproducible. */
18
+ export const TEMPERATURE_EPSILON = 1e-12;
19
+ /** Clip for log loss so one confident miss does not return Infinity. */
20
+ export const LOG_LOSS_CLIP = 1e-15;
21
+ function assertDistribution(probabilities) {
22
+ ensure(Array.isArray(probabilities) && probabilities.length > 0, "distribution must be non-empty");
23
+ let total = 0;
24
+ for (const p of probabilities) {
25
+ ensure(typeof p === "number" && Number.isFinite(p) && p >= 0 && p <= 1, "each probability must be in [0,1]");
26
+ total += p;
27
+ }
28
+ ensure(Math.abs(total - 1) <= SUM_TOLERANCE, "distribution must sum to one");
29
+ }
30
+ /**
31
+ * Summary features of a candidate distribution.
32
+ *
33
+ * A single candidate can show probability 1 while telling you nothing about
34
+ * whether that candidate is correct, so its margin and entropy are undefined
35
+ * rather than maximally confident.
36
+ */
37
+ export function distributionFeatures(probabilities) {
38
+ assertDistribution(probabilities);
39
+ const sorted = [...probabilities].sort((a, b) => b - a);
40
+ const count = probabilities.length;
41
+ const maxProbability = sorted[0] ?? 0;
42
+ if (count === 1) {
43
+ return {
44
+ maxProbability,
45
+ margin: undefined,
46
+ normalizedEntropy: undefined,
47
+ candidateCount: 1,
48
+ logCandidateCount: 0,
49
+ };
50
+ }
51
+ // 0·log0 is defined as 0; skipping the term is exactly that definition.
52
+ let entropy = 0;
53
+ for (const p of probabilities)
54
+ if (p > 0)
55
+ entropy -= p * Math.log(p);
56
+ return {
57
+ maxProbability,
58
+ margin: maxProbability - (sorted[1] ?? 0),
59
+ normalizedEntropy: entropy / Math.log(count),
60
+ candidateCount: count,
61
+ logCandidateCount: Math.log(count),
62
+ };
63
+ }
64
+ /**
65
+ * Post-hoc temperature rule: p̃(a) ∝ max(p(a), ε)^(1/T).
66
+ *
67
+ * T is fitted on a calibration split, never on the data being scored. This
68
+ * does not assume the input came from a softmax; whether it helps has to be
69
+ * measured separately.
70
+ */
71
+ export function temperatureScale(probabilities, temperature, epsilon = TEMPERATURE_EPSILON) {
72
+ assertDistribution(probabilities);
73
+ ensure(typeof temperature === "number" && Number.isFinite(temperature) && temperature > 0, "temperature must be positive");
74
+ ensure(typeof epsilon === "number" && epsilon > 0 && epsilon < 1, "epsilon must be in (0,1)");
75
+ const raised = probabilities.map((p) => Math.max(p, epsilon) ** (1 / temperature));
76
+ const total = raised.reduce((sum, value) => sum + value, 0);
77
+ ensure(total > 0 && Number.isFinite(total), "temperature scaling produced a degenerate distribution");
78
+ return raised.map((value) => value / total);
79
+ }
80
+ function assertBinaryPaired(forecasts, outcomes) {
81
+ ensure(Array.isArray(forecasts) && Array.isArray(outcomes), "forecasts and outcomes must be arrays");
82
+ ensure(forecasts.length === outcomes.length, "forecasts and outcomes must align");
83
+ ensure(forecasts.length > 0, "at least one paired observation is required");
84
+ for (const q of forecasts) {
85
+ ensure(typeof q === "number" && Number.isFinite(q) && q >= 0 && q <= 1, "each forecast must be in [0,1]");
86
+ }
87
+ for (const y of outcomes)
88
+ ensure(y === 0 || y === 1, "each outcome must be 0 or 1");
89
+ }
90
+ export function brierScore(forecasts, outcomes) {
91
+ assertBinaryPaired(forecasts, outcomes);
92
+ let total = 0;
93
+ for (const [i, q] of forecasts.entries())
94
+ total += (q - outcomes[i]) ** 2;
95
+ return total / forecasts.length;
96
+ }
97
+ export function negativeLogLoss(forecasts, outcomes, clip = LOG_LOSS_CLIP) {
98
+ assertBinaryPaired(forecasts, outcomes);
99
+ ensure(typeof clip === "number" && clip > 0 && clip < 0.5, "clip must be in (0,0.5)");
100
+ let total = 0;
101
+ for (const [i, raw] of forecasts.entries()) {
102
+ const q = Math.min(1 - clip, Math.max(clip, raw));
103
+ total -= outcomes[i] === 1 ? Math.log(q) : Math.log(1 - q);
104
+ }
105
+ return total / forecasts.length;
106
+ }
107
+ /**
108
+ * Expected calibration error with its bin occupancy.
109
+ *
110
+ * ECE is sensitive to binning, so the bins and their counts are returned with
111
+ * it: a near-zero error over three samples is not evidence of calibration.
112
+ */
113
+ export function expectedCalibrationError(forecasts, outcomes, binCount) {
114
+ assertBinaryPaired(forecasts, outcomes);
115
+ ensure(Number.isSafeInteger(binCount) && binCount > 0, "binCount must be a positive integer");
116
+ const sums = Array.from({ length: binCount }, () => ({ count: 0, outcome: 0, forecast: 0 }));
117
+ for (const [i, q] of forecasts.entries()) {
118
+ const index = Math.min(binCount - 1, Math.floor(q * binCount));
119
+ const bin = sums[index];
120
+ bin.count += 1;
121
+ bin.outcome += outcomes[i];
122
+ bin.forecast += q;
123
+ }
124
+ const samples = forecasts.length;
125
+ let error = 0;
126
+ const bins = sums.map((bin, index) => {
127
+ const accuracy = bin.count > 0 ? bin.outcome / bin.count : undefined;
128
+ const confidence = bin.count > 0 ? bin.forecast / bin.count : undefined;
129
+ if (accuracy !== undefined && confidence !== undefined) {
130
+ error += (bin.count / samples) * Math.abs(accuracy - confidence);
131
+ }
132
+ return { lower: index / binCount, upper: (index + 1) / binCount, count: bin.count, accuracy, confidence };
133
+ });
134
+ return { error, bins, samples };
135
+ }
136
+ /**
137
+ * Selective execution gate: a(τ) = 1[q̂ ≥ τ] · 1[G = 1].
138
+ *
139
+ * Risk and coverage are returned together because refusing everything drives
140
+ * measured risk to zero while delivering nothing.
141
+ */
142
+ export function selectiveExecution(scores, appropriate, threshold, policyGates) {
143
+ assertBinaryPaired(scores, appropriate);
144
+ ensure(policyGates.length === scores.length, "policy gates must align with scores");
145
+ ensure(typeof threshold === "number" && Number.isFinite(threshold) && threshold >= 0 && threshold <= 1, "threshold must be in [0,1]");
146
+ let admitted = 0;
147
+ let errors = 0;
148
+ for (const [i, score] of scores.entries()) {
149
+ if (score < threshold || policyGates[i] !== true)
150
+ continue;
151
+ admitted += 1;
152
+ if (appropriate[i] === 0)
153
+ errors += 1;
154
+ }
155
+ return {
156
+ admitted,
157
+ coverage: admitted / scores.length,
158
+ risk: admitted > 0 ? errors / admitted : undefined,
159
+ };
160
+ }
161
+ //# sourceMappingURL=calibration-selective.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"calibration-selective.js","sourceRoot":"","sources":["../../src/metacognition/calibration-selective.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;GAaG;AAEH,OAAO,EAAE,MAAM,EAAE,MAAM,iBAAiB,CAAC;AAEzC,MAAM,aAAa,GAAG,IAAI,CAAC;AAC3B,4FAA4F;AAC5F,MAAM,CAAC,MAAM,mBAAmB,GAAG,KAAK,CAAC;AACzC,wEAAwE;AACxE,MAAM,CAAC,MAAM,aAAa,GAAG,KAAK,CAAC;AAYnC,SAAS,kBAAkB,CAAC,aAAgC,EAAQ;IACnE,MAAM,CAAC,KAAK,CAAC,OAAO,CAAC,aAAa,CAAC,IAAI,aAAa,CAAC,MAAM,GAAG,CAAC,EAAE,gCAAgC,CAAC,CAAC;IACnG,IAAI,KAAK,GAAG,CAAC,CAAC;IACd,KAAK,MAAM,CAAC,IAAI,aAAa,EAAE,CAAC;QAC/B,MAAM,CAAC,OAAO,CAAC,KAAK,QAAQ,IAAI,MAAM,CAAC,QAAQ,CAAC,CAAC,CAAC,IAAI,CAAC,IAAI,CAAC,IAAI,CAAC,IAAI,CAAC,EAAE,mCAAmC,CAAC,CAAC;QAC7G,KAAK,IAAI,CAAC,CAAC;IACZ,CAAC;IACD,MAAM,CAAC,IAAI,CAAC,GAAG,CAAC,KAAK,GAAG,CAAC,CAAC,IAAI,aAAa,EAAE,8BAA8B,CAAC,CAAC;AAAA,CAC7E;AAED;;;;;;GAMG;AACH,MAAM,UAAU,oBAAoB,CAAC,aAAgC,EAAwB;IAC5F,kBAAkB,CAAC,aAAa,CAAC,CAAC;IAClC,MAAM,MAAM,GAAG,CAAC,GAAG,aAAa,CAAC,CAAC,IAAI,CAAC,CAAC,CAAC,EAAE,CAAC,EAAE,EAAE,CAAC,CAAC,GAAG,CAAC,CAAC,CAAC;IACxD,MAAM,KAAK,GAAG,aAAa,CAAC,MAAM,CAAC;IACnC,MAAM,cAAc,GAAG,MAAM,CAAC,CAAC,CAAC,IAAI,CAAC,CAAC;IACtC,IAAI,KAAK,KAAK,CAAC,EAAE,CAAC;QACjB,OAAO;YACN,cAAc;YACd,MAAM,EAAE,SAAS;YACjB,iBAAiB,EAAE,SAAS;YAC5B,cAAc,EAAE,CAAC;YACjB,iBAAiB,EAAE,CAAC;SACpB,CAAC;IACH,CAAC;IACD,yEAAwE;IACxE,IAAI,OAAO,GAAG,CAAC,CAAC;IAChB,KAAK,MAAM,CAAC,IAAI,aAAa;QAAE,IAAI,CAAC,GAAG,CAAC;YAAE,OAAO,IAAI,CAAC,GAAG,IAAI,CAAC,GAAG,CAAC,CAAC,CAAC,CAAC;IACrE,OAAO;QACN,cAAc;QACd,MAAM,EAAE,cAAc,GAAG,CAAC,MAAM,CAAC,CAAC,CAAC,IAAI,CAAC,CAAC;QACzC,iBAAiB,EAAE,OAAO,GAAG,IAAI,CAAC,GAAG,CAAC,KAAK,CAAC;QAC5C,cAAc,EAAE,KAAK;QACrB,iBAAiB,EAAE,IAAI,CAAC,GAAG,CAAC,KAAK,CAAC;KAClC,CAAC;AAAA,CACF;AAED;;;;;;GAMG;AACH,MAAM,UAAU,gBAAgB,CAC/B,aAAgC,EAChC,WAAmB,EACnB,OAAO,GAAW,mBAAmB,EAC1B;IACX,kBAAkB,CAAC,aAAa,CAAC,CAAC;IAClC,MAAM,CACL,OAAO,WAAW,KAAK,QAAQ,IAAI,MAAM,CAAC,QAAQ,CAAC,WAAW,CAAC,IAAI,WAAW,GAAG,CAAC,EAClF,8BAA8B,CAC9B,CAAC;IACF,MAAM,CAAC,OAAO,OAAO,KAAK,QAAQ,IAAI,OAAO,GAAG,CAAC,IAAI,OAAO,GAAG,CAAC,EAAE,0BAA0B,CAAC,CAAC;IAC9F,MAAM,MAAM,GAAG,aAAa,CAAC,GAAG,CAAC,CAAC,CAAC,EAAE,EAAE,CAAC,IAAI,CAAC,GAAG,CAAC,CAAC,EAAE,OAAO,CAAC,IAAI,CAAC,CAAC,GAAG,WAAW,CAAC,CAAC,CAAC;IACnF,MAAM,KAAK,GAAG,MAAM,CAAC,MAAM,CAAC,CAAC,GAAG,EAAE,KAAK,EAAE,EAAE,CAAC,GAAG,GAAG,KAAK,EAAE,CAAC,CAAC,CAAC;IAC5D,MAAM,CAAC,KAAK,GAAG,CAAC,IAAI,MAAM,CAAC,QAAQ,CAAC,KAAK,CAAC,EAAE,wDAAwD,CAAC,CAAC;IACtG,OAAO,MAAM,CAAC,GAAG,CAAC,CAAC,KAAK,EAAE,EAAE,CAAC,KAAK,GAAG,KAAK,CAAC,CAAC;AAAA,CAC5C;AAED,SAAS,kBAAkB,CAAC,SAA4B,EAAE,QAA2B,EAAQ;IAC5F,MAAM,CAAC,KAAK,CAAC,OAAO,CAAC,SAAS,CAAC,IAAI,KAAK,CAAC,OAAO,CAAC,QAAQ,CAAC,EAAE,uCAAuC,CAAC,CAAC;IACrG,MAAM,CAAC,SAAS,CAAC,MAAM,KAAK,QAAQ,CAAC,MAAM,EAAE,mCAAmC,CAAC,CAAC;IAClF,MAAM,CAAC,SAAS,CAAC,MAAM,GAAG,CAAC,EAAE,6CAA6C,CAAC,CAAC;IAC5E,KAAK,MAAM,CAAC,IAAI,SAAS,EAAE,CAAC;QAC3B,MAAM,CAAC,OAAO,CAAC,KAAK,QAAQ,IAAI,MAAM,CAAC,QAAQ,CAAC,CAAC,CAAC,IAAI,CAAC,IAAI,CAAC,IAAI,CAAC,IAAI,CAAC,EAAE,gCAAgC,CAAC,CAAC;IAC3G,CAAC;IACD,KAAK,MAAM,CAAC,IAAI,QAAQ;QAAE,MAAM,CAAC,CAAC,KAAK,CAAC,IAAI,CAAC,KAAK,CAAC,EAAE,6BAA6B,CAAC,CAAC;AAAA,CACpF;AAED,MAAM,UAAU,UAAU,CAAC,SAA4B,EAAE,QAA2B,EAAU;IAC7F,kBAAkB,CAAC,SAAS,EAAE,QAAQ,CAAC,CAAC;IACxC,IAAI,KAAK,GAAG,CAAC,CAAC;IACd,KAAK,MAAM,CAAC,CAAC,EAAE,CAAC,CAAC,IAAI,SAAS,CAAC,OAAO,EAAE;QAAE,KAAK,IAAI,CAAC,CAAC,GAAG,QAAQ,CAAC,CAAC,CAAE,CAAC,IAAI,CAAC,CAAC;IAC3E,OAAO,KAAK,GAAG,SAAS,CAAC,MAAM,CAAC;AAAA,CAChC;AAED,MAAM,UAAU,eAAe,CAC9B,SAA4B,EAC5B,QAA2B,EAC3B,IAAI,GAAW,aAAa,EACnB;IACT,kBAAkB,CAAC,SAAS,EAAE,QAAQ,CAAC,CAAC;IACxC,MAAM,CAAC,OAAO,IAAI,KAAK,QAAQ,IAAI,IAAI,GAAG,CAAC,IAAI,IAAI,GAAG,GAAG,EAAE,yBAAyB,CAAC,CAAC;IACtF,IAAI,KAAK,GAAG,CAAC,CAAC;IACd,KAAK,MAAM,CAAC,CAAC,EAAE,GAAG,CAAC,IAAI,SAAS,CAAC,OAAO,EAAE,EAAE,CAAC;QAC5C,MAAM,CAAC,GAAG,IAAI,CAAC,GAAG,CAAC,CAAC,GAAG,IAAI,EAAE,IAAI,CAAC,GAAG,CAAC,IAAI,EAAE,GAAG,CAAC,CAAC,CAAC;QAClD,KAAK,IAAI,QAAQ,CAAC,CAAC,CAAC,KAAK,CAAC,CAAC,CAAC,CAAC,IAAI,CAAC,GAAG,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,IAAI,CAAC,GAAG,CAAC,CAAC,GAAG,CAAC,CAAC,CAAC;IAC5D,CAAC;IACD,OAAO,KAAK,GAAG,SAAS,CAAC,MAAM,CAAC;AAAA,CAChC;AAgBD;;;;;GAKG;AACH,MAAM,UAAU,wBAAwB,CACvC,SAA4B,EAC5B,QAA2B,EAC3B,QAAgB,EACI;IACpB,kBAAkB,CAAC,SAAS,EAAE,QAAQ,CAAC,CAAC;IACxC,MAAM,CAAC,MAAM,CAAC,aAAa,CAAC,QAAQ,CAAC,IAAI,QAAQ,GAAG,CAAC,EAAE,qCAAqC,CAAC,CAAC;IAC9F,MAAM,IAAI,GAAG,KAAK,CAAC,IAAI,CAAC,EAAE,MAAM,EAAE,QAAQ,EAAE,EAAE,GAAG,EAAE,CAAC,CAAC,EAAE,KAAK,EAAE,CAAC,EAAE,OAAO,EAAE,CAAC,EAAE,QAAQ,EAAE,CAAC,EAAE,CAAC,CAAC,CAAC;IAC7F,KAAK,MAAM,CAAC,CAAC,EAAE,CAAC,CAAC,IAAI,SAAS,CAAC,OAAO,EAAE,EAAE,CAAC;QAC1C,MAAM,KAAK,GAAG,IAAI,CAAC,GAAG,CAAC,QAAQ,GAAG,CAAC,EAAE,IAAI,CAAC,KAAK,CAAC,CAAC,GAAG,QAAQ,CAAC,CAAC,CAAC;QAC/D,MAAM,GAAG,GAAG,IAAI,CAAC,KAAK,CAAE,CAAC;QACzB,GAAG,CAAC,KAAK,IAAI,CAAC,CAAC;QACf,GAAG,CAAC,OAAO,IAAI,QAAQ,CAAC,CAAC,CAAE,CAAC;QAC5B,GAAG,CAAC,QAAQ,IAAI,CAAC,CAAC;IACnB,CAAC;IACD,MAAM,OAAO,GAAG,SAAS,CAAC,MAAM,CAAC;IACjC,IAAI,KAAK,GAAG,CAAC,CAAC;IACd,MAAM,IAAI,GAAG,IAAI,CAAC,GAAG,CAAC,CAAC,GAAG,EAAE,KAAK,EAAE,EAAE,CAAC;QACrC,MAAM,QAAQ,GAAG,GAAG,CAAC,KAAK,GAAG,CAAC,CAAC,CAAC,CAAC,GAAG,CAAC,OAAO,GAAG,GAAG,CAAC,KAAK,CAAC,CAAC,CAAC,SAAS,CAAC;QACrE,MAAM,UAAU,GAAG,GAAG,CAAC,KAAK,GAAG,CAAC,CAAC,CAAC,CAAC,GAAG,CAAC,QAAQ,GAAG,GAAG,CAAC,KAAK,CAAC,CAAC,CAAC,SAAS,CAAC;QACxE,IAAI,QAAQ,KAAK,SAAS,IAAI,UAAU,KAAK,SAAS,EAAE,CAAC;YACxD,KAAK,IAAI,CAAC,GAAG,CAAC,KAAK,GAAG,OAAO,CAAC,GAAG,IAAI,CAAC,GAAG,CAAC,QAAQ,GAAG,UAAU,CAAC,CAAC;QAClE,CAAC;QACD,OAAO,EAAE,KAAK,EAAE,KAAK,GAAG,QAAQ,EAAE,KAAK,EAAE,CAAC,KAAK,GAAG,CAAC,CAAC,GAAG,QAAQ,EAAE,KAAK,EAAE,GAAG,CAAC,KAAK,EAAE,QAAQ,EAAE,UAAU,EAAE,CAAC;IAAA,CAC1G,CAAC,CAAC;IACH,OAAO,EAAE,KAAK,EAAE,IAAI,EAAE,OAAO,EAAE,CAAC;AAAA,CAChC;AASD;;;;;GAKG;AACH,MAAM,UAAU,kBAAkB,CACjC,MAAyB,EACzB,WAA8B,EAC9B,SAAiB,EACjB,WAA+B,EACJ;IAC3B,kBAAkB,CAAC,MAAM,EAAE,WAAW,CAAC,CAAC;IACxC,MAAM,CAAC,WAAW,CAAC,MAAM,KAAK,MAAM,CAAC,MAAM,EAAE,qCAAqC,CAAC,CAAC;IACpF,MAAM,CACL,OAAO,SAAS,KAAK,QAAQ,IAAI,MAAM,CAAC,QAAQ,CAAC,SAAS,CAAC,IAAI,SAAS,IAAI,CAAC,IAAI,SAAS,IAAI,CAAC,EAC/F,4BAA4B,CAC5B,CAAC;IACF,IAAI,QAAQ,GAAG,CAAC,CAAC;IACjB,IAAI,MAAM,GAAG,CAAC,CAAC;IACf,KAAK,MAAM,CAAC,CAAC,EAAE,KAAK,CAAC,IAAI,MAAM,CAAC,OAAO,EAAE,EAAE,CAAC;QAC3C,IAAI,KAAK,GAAG,SAAS,IAAI,WAAW,CAAC,CAAC,CAAC,KAAK,IAAI;YAAE,SAAS;QAC3D,QAAQ,IAAI,CAAC,CAAC;QACd,IAAI,WAAW,CAAC,CAAC,CAAC,KAAK,CAAC;YAAE,MAAM,IAAI,CAAC,CAAC;IACvC,CAAC;IACD,OAAO;QACN,QAAQ;QACR,QAAQ,EAAE,QAAQ,GAAG,MAAM,CAAC,MAAM;QAClC,IAAI,EAAE,QAAQ,GAAG,CAAC,CAAC,CAAC,CAAC,MAAM,GAAG,QAAQ,CAAC,CAAC,CAAC,SAAS;KAClD,CAAC;AAAA,CACF","sourcesContent":["/**\n * Probability calibration and selective execution (Jev audit algorithm A3).\n *\n * A provider's `confidence` is a statistic over its own option distribution,\n * not a calibrated probability of any event we care about. Before it can gate\n * execution, the predicted event has to be named — \"the chosen candidate was\n * appropriate\" is a different event from \"the driver executed it\" and from\n * \"the task succeeded\" — and the score has to be measured on held-out data.\n *\n * These are the measurement primitives, deliberately small: distribution\n * features, a post-hoc temperature rule, scoring rules, and a selective gate\n * that always reports coverage next to risk. None of them makes a score\n * trustworthy; they make it measurable.\n */\n\nimport { ensure } from \"./validation.ts\";\n\nconst SUM_TOLERANCE = 1e-9;\n/** Floor for numeric zeros in temperature scaling; recorded so results are reproducible. */\nexport const TEMPERATURE_EPSILON = 1e-12;\n/** Clip for log loss so one confident miss does not return Infinity. */\nexport const LOG_LOSS_CLIP = 1e-15;\n\nexport interface DistributionFeatures {\n\treadonly maxProbability: number;\n\t/** p(1) − p(2). Undefined for a single candidate: there is nothing to compare. */\n\treadonly margin: number | undefined;\n\t/** Undefined for a single candidate, where entropy carries no information. */\n\treadonly normalizedEntropy: number | undefined;\n\treadonly candidateCount: number;\n\treadonly logCandidateCount: number;\n}\n\nfunction assertDistribution(probabilities: readonly number[]): void {\n\tensure(Array.isArray(probabilities) && probabilities.length > 0, \"distribution must be non-empty\");\n\tlet total = 0;\n\tfor (const p of probabilities) {\n\t\tensure(typeof p === \"number\" && Number.isFinite(p) && p >= 0 && p <= 1, \"each probability must be in [0,1]\");\n\t\ttotal += p;\n\t}\n\tensure(Math.abs(total - 1) <= SUM_TOLERANCE, \"distribution must sum to one\");\n}\n\n/**\n * Summary features of a candidate distribution.\n *\n * A single candidate can show probability 1 while telling you nothing about\n * whether that candidate is correct, so its margin and entropy are undefined\n * rather than maximally confident.\n */\nexport function distributionFeatures(probabilities: readonly number[]): DistributionFeatures {\n\tassertDistribution(probabilities);\n\tconst sorted = [...probabilities].sort((a, b) => b - a);\n\tconst count = probabilities.length;\n\tconst maxProbability = sorted[0] ?? 0;\n\tif (count === 1) {\n\t\treturn {\n\t\t\tmaxProbability,\n\t\t\tmargin: undefined,\n\t\t\tnormalizedEntropy: undefined,\n\t\t\tcandidateCount: 1,\n\t\t\tlogCandidateCount: 0,\n\t\t};\n\t}\n\t// 0·log0 is defined as 0; skipping the term is exactly that definition.\n\tlet entropy = 0;\n\tfor (const p of probabilities) if (p > 0) entropy -= p * Math.log(p);\n\treturn {\n\t\tmaxProbability,\n\t\tmargin: maxProbability - (sorted[1] ?? 0),\n\t\tnormalizedEntropy: entropy / Math.log(count),\n\t\tcandidateCount: count,\n\t\tlogCandidateCount: Math.log(count),\n\t};\n}\n\n/**\n * Post-hoc temperature rule: p̃(a) ∝ max(p(a), ε)^(1/T).\n *\n * T is fitted on a calibration split, never on the data being scored. This\n * does not assume the input came from a softmax; whether it helps has to be\n * measured separately.\n */\nexport function temperatureScale(\n\tprobabilities: readonly number[],\n\ttemperature: number,\n\tepsilon: number = TEMPERATURE_EPSILON,\n): number[] {\n\tassertDistribution(probabilities);\n\tensure(\n\t\ttypeof temperature === \"number\" && Number.isFinite(temperature) && temperature > 0,\n\t\t\"temperature must be positive\",\n\t);\n\tensure(typeof epsilon === \"number\" && epsilon > 0 && epsilon < 1, \"epsilon must be in (0,1)\");\n\tconst raised = probabilities.map((p) => Math.max(p, epsilon) ** (1 / temperature));\n\tconst total = raised.reduce((sum, value) => sum + value, 0);\n\tensure(total > 0 && Number.isFinite(total), \"temperature scaling produced a degenerate distribution\");\n\treturn raised.map((value) => value / total);\n}\n\nfunction assertBinaryPaired(forecasts: readonly number[], outcomes: readonly number[]): void {\n\tensure(Array.isArray(forecasts) && Array.isArray(outcomes), \"forecasts and outcomes must be arrays\");\n\tensure(forecasts.length === outcomes.length, \"forecasts and outcomes must align\");\n\tensure(forecasts.length > 0, \"at least one paired observation is required\");\n\tfor (const q of forecasts) {\n\t\tensure(typeof q === \"number\" && Number.isFinite(q) && q >= 0 && q <= 1, \"each forecast must be in [0,1]\");\n\t}\n\tfor (const y of outcomes) ensure(y === 0 || y === 1, \"each outcome must be 0 or 1\");\n}\n\nexport function brierScore(forecasts: readonly number[], outcomes: readonly number[]): number {\n\tassertBinaryPaired(forecasts, outcomes);\n\tlet total = 0;\n\tfor (const [i, q] of forecasts.entries()) total += (q - outcomes[i]!) ** 2;\n\treturn total / forecasts.length;\n}\n\nexport function negativeLogLoss(\n\tforecasts: readonly number[],\n\toutcomes: readonly number[],\n\tclip: number = LOG_LOSS_CLIP,\n): number {\n\tassertBinaryPaired(forecasts, outcomes);\n\tensure(typeof clip === \"number\" && clip > 0 && clip < 0.5, \"clip must be in (0,0.5)\");\n\tlet total = 0;\n\tfor (const [i, raw] of forecasts.entries()) {\n\t\tconst q = Math.min(1 - clip, Math.max(clip, raw));\n\t\ttotal -= outcomes[i] === 1 ? Math.log(q) : Math.log(1 - q);\n\t}\n\treturn total / forecasts.length;\n}\n\nexport interface CalibrationBin {\n\treadonly lower: number;\n\treadonly upper: number;\n\treadonly count: number;\n\treadonly accuracy: number | undefined;\n\treadonly confidence: number | undefined;\n}\n\nexport interface CalibrationReport {\n\treadonly error: number;\n\treadonly bins: readonly CalibrationBin[];\n\treadonly samples: number;\n}\n\n/**\n * Expected calibration error with its bin occupancy.\n *\n * ECE is sensitive to binning, so the bins and their counts are returned with\n * it: a near-zero error over three samples is not evidence of calibration.\n */\nexport function expectedCalibrationError(\n\tforecasts: readonly number[],\n\toutcomes: readonly number[],\n\tbinCount: number,\n): CalibrationReport {\n\tassertBinaryPaired(forecasts, outcomes);\n\tensure(Number.isSafeInteger(binCount) && binCount > 0, \"binCount must be a positive integer\");\n\tconst sums = Array.from({ length: binCount }, () => ({ count: 0, outcome: 0, forecast: 0 }));\n\tfor (const [i, q] of forecasts.entries()) {\n\t\tconst index = Math.min(binCount - 1, Math.floor(q * binCount));\n\t\tconst bin = sums[index]!;\n\t\tbin.count += 1;\n\t\tbin.outcome += outcomes[i]!;\n\t\tbin.forecast += q;\n\t}\n\tconst samples = forecasts.length;\n\tlet error = 0;\n\tconst bins = sums.map((bin, index) => {\n\t\tconst accuracy = bin.count > 0 ? bin.outcome / bin.count : undefined;\n\t\tconst confidence = bin.count > 0 ? bin.forecast / bin.count : undefined;\n\t\tif (accuracy !== undefined && confidence !== undefined) {\n\t\t\terror += (bin.count / samples) * Math.abs(accuracy - confidence);\n\t\t}\n\t\treturn { lower: index / binCount, upper: (index + 1) / binCount, count: bin.count, accuracy, confidence };\n\t});\n\treturn { error, bins, samples };\n}\n\nexport interface SelectiveExecutionReport {\n\treadonly admitted: number;\n\treadonly coverage: number;\n\t/** Undefined when nothing was admitted: that is unmeasurable, not risk-free. */\n\treadonly risk: number | undefined;\n}\n\n/**\n * Selective execution gate: a(τ) = 1[q̂ ≥ τ] · 1[G = 1].\n *\n * Risk and coverage are returned together because refusing everything drives\n * measured risk to zero while delivering nothing.\n */\nexport function selectiveExecution(\n\tscores: readonly number[],\n\tappropriate: readonly number[],\n\tthreshold: number,\n\tpolicyGates: readonly boolean[],\n): SelectiveExecutionReport {\n\tassertBinaryPaired(scores, appropriate);\n\tensure(policyGates.length === scores.length, \"policy gates must align with scores\");\n\tensure(\n\t\ttypeof threshold === \"number\" && Number.isFinite(threshold) && threshold >= 0 && threshold <= 1,\n\t\t\"threshold must be in [0,1]\",\n\t);\n\tlet admitted = 0;\n\tlet errors = 0;\n\tfor (const [i, score] of scores.entries()) {\n\t\tif (score < threshold || policyGates[i] !== true) continue;\n\t\tadmitted += 1;\n\t\tif (appropriate[i] === 0) errors += 1;\n\t}\n\treturn {\n\t\tadmitted,\n\t\tcoverage: admitted / scores.length,\n\t\trisk: admitted > 0 ? errors / admitted : undefined,\n\t};\n}\n"]}
@@ -0,0 +1,62 @@
1
+ export interface ConditionKey {
2
+ readonly modelRevision: string;
3
+ readonly skillHashes: readonly string[];
4
+ readonly toolchain: string;
5
+ readonly taskBand: string;
6
+ readonly checkKind: string;
7
+ readonly checkHealth: "healthy" | "degraded" | "unverified";
8
+ readonly budgetBand: string;
9
+ readonly policyVersion: string;
10
+ }
11
+ export type BucketState = "insufficient-data" | "active" | "demoted";
12
+ export interface CalibrationBucket {
13
+ readonly key: ConditionKey;
14
+ readonly keyHash: string;
15
+ readonly priorAlpha: number;
16
+ readonly priorBeta: number;
17
+ successes: number;
18
+ failures: number;
19
+ drift: number;
20
+ state: BucketState;
21
+ readonly demotedReason?: string;
22
+ }
23
+ export interface CalibrationStore {
24
+ readonly buckets: Readonly<Record<string, CalibrationBucket>>;
25
+ readonly minSamples: number;
26
+ readonly referenceMean: number;
27
+ readonly slack: number;
28
+ readonly threshold: number;
29
+ }
30
+ export declare function conditionKeyHash(key: ConditionKey): string;
31
+ export declare function createCalibrationStore(options: {
32
+ readonly minSamples: number;
33
+ readonly priorAlpha: number;
34
+ readonly priorBeta: number;
35
+ readonly referenceMean: number;
36
+ readonly slack: number;
37
+ readonly threshold: number;
38
+ }): CalibrationStore;
39
+ /**
40
+ * Record one independent task outcome. Never count 100 runs of one task as
41
+ * 100 independent tasks; the caller supplies one pre-registered binary outcome
42
+ * per task. Environment failures belong in delivery metrics, not here.
43
+ */
44
+ export declare function recordOutcome(store: CalibrationStore, key: ConditionKey, outcome: "success" | "failure", loss: number): CalibrationStore;
45
+ /**
46
+ * Demote every bucket whose condition changed underneath it: new model
47
+ * revision, new check definition, or new major toolchain version. Explicit
48
+ * condition changes never wait for statistical drift detection.
49
+ */
50
+ export declare function demoteOnConditionChange(store: CalibrationStore, change: {
51
+ modelRevision?: string;
52
+ checkKind?: string;
53
+ toolchain?: string;
54
+ policyVersion?: string;
55
+ }): CalibrationStore;
56
+ /** Posterior mean success rate for one bucket, or `insufficient-data`. */
57
+ export declare function conditionalSuccessRate(store: CalibrationStore, key: ConditionKey): {
58
+ state: BucketState;
59
+ rate?: number;
60
+ samples: number;
61
+ };
62
+ //# sourceMappingURL=calibration.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"calibration.d.ts","sourceRoot":"","sources":["../../src/metacognition/calibration.ts"],"names":[],"mappings":"AAcA,MAAM,WAAW,YAAY;IAC5B,QAAQ,CAAC,aAAa,EAAE,MAAM,CAAC;IAC/B,QAAQ,CAAC,WAAW,EAAE,SAAS,MAAM,EAAE,CAAC;IACxC,QAAQ,CAAC,SAAS,EAAE,MAAM,CAAC;IAC3B,QAAQ,CAAC,QAAQ,EAAE,MAAM,CAAC;IAC1B,QAAQ,CAAC,SAAS,EAAE,MAAM,CAAC;IAC3B,QAAQ,CAAC,WAAW,EAAE,SAAS,GAAG,UAAU,GAAG,YAAY,CAAC;IAC5D,QAAQ,CAAC,UAAU,EAAE,MAAM,CAAC;IAC5B,QAAQ,CAAC,aAAa,EAAE,MAAM,CAAC;CAC/B;AACD,MAAM,MAAM,WAAW,GAAG,mBAAmB,GAAG,QAAQ,GAAG,SAAS,CAAC;AACrE,MAAM,WAAW,iBAAiB;IACjC,QAAQ,CAAC,GAAG,EAAE,YAAY,CAAC;IAC3B,QAAQ,CAAC,OAAO,EAAE,MAAM,CAAC;IACzB,QAAQ,CAAC,UAAU,EAAE,MAAM,CAAC;IAC5B,QAAQ,CAAC,SAAS,EAAE,MAAM,CAAC;IAC3B,SAAS,EAAE,MAAM,CAAC;IAClB,QAAQ,EAAE,MAAM,CAAC;IACjB,KAAK,EAAE,MAAM,CAAC;IACd,KAAK,EAAE,WAAW,CAAC;IACnB,QAAQ,CAAC,aAAa,CAAC,EAAE,MAAM,CAAC;CAChC;AACD,MAAM,WAAW,gBAAgB;IAChC,QAAQ,CAAC,OAAO,EAAE,QAAQ,CAAC,MAAM,CAAC,MAAM,EAAE,iBAAiB,CAAC,CAAC,CAAC;IAC9D,QAAQ,CAAC,UAAU,EAAE,MAAM,CAAC;IAC5B,QAAQ,CAAC,aAAa,EAAE,MAAM,CAAC;IAC/B,QAAQ,CAAC,KAAK,EAAE,MAAM,CAAC;IACvB,QAAQ,CAAC,SAAS,EAAE,MAAM,CAAC;CAC3B;AAED,wBAAgB,gBAAgB,CAAC,GAAG,EAAE,YAAY,GAAG,MAAM,CAkB1D;AAED,wBAAgB,sBAAsB,CAAC,OAAO,EAAE;IAC/C,QAAQ,CAAC,UAAU,EAAE,MAAM,CAAC;IAC5B,QAAQ,CAAC,UAAU,EAAE,MAAM,CAAC;IAC5B,QAAQ,CAAC,SAAS,EAAE,MAAM,CAAC;IAC3B,QAAQ,CAAC,aAAa,EAAE,MAAM,CAAC;IAC/B,QAAQ,CAAC,KAAK,EAAE,MAAM,CAAC;IACvB,QAAQ,CAAC,SAAS,EAAE,MAAM,CAAC;CAC3B,GAAG,gBAAgB,CAenB;AAkBD;;;;GAIG;AACH,wBAAgB,aAAa,CAC5B,KAAK,EAAE,gBAAgB,EACvB,GAAG,EAAE,YAAY,EACjB,OAAO,EAAE,SAAS,GAAG,SAAS,EAC9B,IAAI,EAAE,MAAM,GACV,gBAAgB,CAoBlB;AAED;;;;GAIG;AACH,wBAAgB,uBAAuB,CACtC,KAAK,EAAE,gBAAgB,EACvB,MAAM,EAAE;IAAE,aAAa,CAAC,EAAE,MAAM,CAAC;IAAC,SAAS,CAAC,EAAE,MAAM,CAAC;IAAC,SAAS,CAAC,EAAE,MAAM,CAAC;IAAC,aAAa,CAAC,EAAE,MAAM,CAAA;CAAE,GAChG,gBAAgB,CAclB;AAED,0EAA0E;AAC1E,wBAAgB,sBAAsB,CACrC,KAAK,EAAE,gBAAgB,EACvB,GAAG,EAAE,YAAY,GACf;IAAE,KAAK,EAAE,WAAW,CAAC;IAAC,IAAI,CAAC,EAAE,MAAM,CAAC;IAAC,OAAO,EAAE,MAAM,CAAA;CAAE,CAUxD","sourcesContent":["/**\n * Algorithms E and G — conditional capability/calibration records and\n * drift-triggered demotion. Spec §8, §10.3.\n *\n * Records are per condition bucket: model revision + loaded skill content\n * fingerprints + toolchain version + task/effect/difficulty band + check\n * type/health + budget band + selection policy version. A bucket with too few\n * samples stays `insufficient-data`; a new provider revision, check\n * definition, or major library version demotes the bucket immediately,\n * without waiting for statistical detection.\n */\nimport { betaPosteriorMean, driftStatistic } from \"./decision.ts\";\nimport { canonical, ensure, finite, integer, lexical, member, text } from \"./validation.ts\";\n\nexport interface ConditionKey {\n\treadonly modelRevision: string;\n\treadonly skillHashes: readonly string[];\n\treadonly toolchain: string;\n\treadonly taskBand: string;\n\treadonly checkKind: string;\n\treadonly checkHealth: \"healthy\" | \"degraded\" | \"unverified\";\n\treadonly budgetBand: string;\n\treadonly policyVersion: string;\n}\nexport type BucketState = \"insufficient-data\" | \"active\" | \"demoted\";\nexport interface CalibrationBucket {\n\treadonly key: ConditionKey;\n\treadonly keyHash: string;\n\treadonly priorAlpha: number;\n\treadonly priorBeta: number;\n\tsuccesses: number;\n\tfailures: number;\n\tdrift: number;\n\tstate: BucketState;\n\treadonly demotedReason?: string;\n}\nexport interface CalibrationStore {\n\treadonly buckets: Readonly<Record<string, CalibrationBucket>>;\n\treadonly minSamples: number;\n\treadonly referenceMean: number;\n\treadonly slack: number;\n\treadonly threshold: number;\n}\n\nexport function conditionKeyHash(key: ConditionKey): string {\n\ttext(key.modelRevision, \"modelRevision\", 128);\n\ttext(key.toolchain, \"toolchain\", 128);\n\ttext(key.taskBand, \"taskBand\", 128);\n\ttext(key.checkKind, \"checkKind\", 128);\n\tmember(key.checkHealth, [\"healthy\", \"degraded\", \"unverified\"], \"checkHealth\");\n\ttext(key.budgetBand, \"budgetBand\", 128);\n\ttext(key.policyVersion, \"policyVersion\", 64);\n\treturn canonical([\n\t\tkey.modelRevision,\n\t\t[...key.skillHashes].sort(lexical),\n\t\tkey.toolchain,\n\t\tkey.taskBand,\n\t\tkey.checkKind,\n\t\tkey.checkHealth,\n\t\tkey.budgetBand,\n\t\tkey.policyVersion,\n\t]);\n}\n\nexport function createCalibrationStore(options: {\n\treadonly minSamples: number;\n\treadonly priorAlpha: number;\n\treadonly priorBeta: number;\n\treadonly referenceMean: number;\n\treadonly slack: number;\n\treadonly threshold: number;\n}): CalibrationStore {\n\tinteger(options.minSamples, \"minSamples\", 100_000);\n\tfinite(options.priorAlpha, \"priorAlpha\", 1e9);\n\tfinite(options.priorBeta, \"priorBeta\", 1e9);\n\tensure(options.priorAlpha > 0 && options.priorBeta > 0, \"beta prior must be positive\");\n\tfinite(options.referenceMean, \"referenceMean\");\n\tfinite(options.slack, \"slack\");\n\tfinite(options.threshold, \"threshold\");\n\treturn {\n\t\tbuckets: {},\n\t\tminSamples: options.minSamples,\n\t\treferenceMean: options.referenceMean,\n\t\tslack: options.slack,\n\t\tthreshold: options.threshold,\n\t};\n}\n\nfunction bucketFor(store: CalibrationStore, key: ConditionKey): CalibrationBucket {\n\tconst keyHash = conditionKeyHash(key);\n\treturn (\n\t\tstore.buckets[keyHash] ?? {\n\t\t\tkey,\n\t\t\tkeyHash,\n\t\t\tpriorAlpha: 1,\n\t\t\tpriorBeta: 1,\n\t\t\tsuccesses: 0,\n\t\t\tfailures: 0,\n\t\t\tdrift: 0,\n\t\t\tstate: \"insufficient-data\",\n\t\t}\n\t);\n}\n\n/**\n * Record one independent task outcome. Never count 100 runs of one task as\n * 100 independent tasks; the caller supplies one pre-registered binary outcome\n * per task. Environment failures belong in delivery metrics, not here.\n */\nexport function recordOutcome(\n\tstore: CalibrationStore,\n\tkey: ConditionKey,\n\toutcome: \"success\" | \"failure\",\n\tloss: number,\n): CalibrationStore {\n\tmember(outcome, [\"success\", \"failure\"], \"outcome\");\n\tfinite(loss, \"loss\");\n\tconst bucket = bucketFor(store, key);\n\tif (bucket.state === \"demoted\") return store;\n\tconst next: CalibrationBucket = {\n\t\t...bucket,\n\t\tsuccesses: bucket.successes + (outcome === \"success\" ? 1 : 0),\n\t\tfailures: bucket.failures + (outcome === \"failure\" ? 1 : 0),\n\t\tdrift: driftStatistic(bucket.drift, loss, store.referenceMean, store.slack),\n\t\tstate: bucket.successes + bucket.failures + 1 >= store.minSamples ? \"active\" : \"insufficient-data\",\n\t};\n\tconst demoted = next.drift > store.threshold;\n\treturn {\n\t\t...store,\n\t\tbuckets: {\n\t\t\t...store.buckets,\n\t\t\t[next.keyHash]: demoted ? { ...next, state: \"demoted\", demotedReason: \"drift-threshold\" } : next,\n\t\t},\n\t};\n}\n\n/**\n * Demote every bucket whose condition changed underneath it: new model\n * revision, new check definition, or new major toolchain version. Explicit\n * condition changes never wait for statistical drift detection.\n */\nexport function demoteOnConditionChange(\n\tstore: CalibrationStore,\n\tchange: { modelRevision?: string; checkKind?: string; toolchain?: string; policyVersion?: string },\n): CalibrationStore {\n\tconst buckets: Record<string, CalibrationBucket> = {};\n\tfor (const [hash, bucket] of Object.entries(store.buckets)) {\n\t\tconst changed =\n\t\t\t(change.modelRevision !== undefined && bucket.key.modelRevision !== change.modelRevision) ||\n\t\t\t(change.checkKind !== undefined && bucket.key.checkKind !== change.checkKind) ||\n\t\t\t(change.toolchain !== undefined && bucket.key.toolchain !== change.toolchain) ||\n\t\t\t(change.policyVersion !== undefined && bucket.key.policyVersion !== change.policyVersion);\n\t\tbuckets[hash] =\n\t\t\tchanged && bucket.state !== \"demoted\"\n\t\t\t\t? { ...bucket, state: \"demoted\", demotedReason: \"condition-change\" }\n\t\t\t\t: bucket;\n\t}\n\treturn { ...store, buckets };\n}\n\n/** Posterior mean success rate for one bucket, or `insufficient-data`. */\nexport function conditionalSuccessRate(\n\tstore: CalibrationStore,\n\tkey: ConditionKey,\n): { state: BucketState; rate?: number; samples: number } {\n\tconst bucket = store.buckets[conditionKeyHash(key)];\n\tif (!bucket) return { state: \"insufficient-data\", samples: 0 };\n\tconst samples = bucket.successes + bucket.failures;\n\tif (bucket.state !== \"active\") return { state: bucket.state, samples };\n\treturn {\n\t\tstate: \"active\",\n\t\tsamples,\n\t\trate: betaPosteriorMean(bucket.priorAlpha, bucket.priorBeta, bucket.successes, bucket.failures),\n\t};\n}\n"]}
@@ -0,0 +1,122 @@
1
+ /**
2
+ * Algorithms E and G — conditional capability/calibration records and
3
+ * drift-triggered demotion. Spec §8, §10.3.
4
+ *
5
+ * Records are per condition bucket: model revision + loaded skill content
6
+ * fingerprints + toolchain version + task/effect/difficulty band + check
7
+ * type/health + budget band + selection policy version. A bucket with too few
8
+ * samples stays `insufficient-data`; a new provider revision, check
9
+ * definition, or major library version demotes the bucket immediately,
10
+ * without waiting for statistical detection.
11
+ */
12
+ import { betaPosteriorMean, driftStatistic } from "./decision.js";
13
+ import { canonical, ensure, finite, integer, lexical, member, text } from "./validation.js";
14
+ export function conditionKeyHash(key) {
15
+ text(key.modelRevision, "modelRevision", 128);
16
+ text(key.toolchain, "toolchain", 128);
17
+ text(key.taskBand, "taskBand", 128);
18
+ text(key.checkKind, "checkKind", 128);
19
+ member(key.checkHealth, ["healthy", "degraded", "unverified"], "checkHealth");
20
+ text(key.budgetBand, "budgetBand", 128);
21
+ text(key.policyVersion, "policyVersion", 64);
22
+ return canonical([
23
+ key.modelRevision,
24
+ [...key.skillHashes].sort(lexical),
25
+ key.toolchain,
26
+ key.taskBand,
27
+ key.checkKind,
28
+ key.checkHealth,
29
+ key.budgetBand,
30
+ key.policyVersion,
31
+ ]);
32
+ }
33
+ export function createCalibrationStore(options) {
34
+ integer(options.minSamples, "minSamples", 100_000);
35
+ finite(options.priorAlpha, "priorAlpha", 1e9);
36
+ finite(options.priorBeta, "priorBeta", 1e9);
37
+ ensure(options.priorAlpha > 0 && options.priorBeta > 0, "beta prior must be positive");
38
+ finite(options.referenceMean, "referenceMean");
39
+ finite(options.slack, "slack");
40
+ finite(options.threshold, "threshold");
41
+ return {
42
+ buckets: {},
43
+ minSamples: options.minSamples,
44
+ referenceMean: options.referenceMean,
45
+ slack: options.slack,
46
+ threshold: options.threshold,
47
+ };
48
+ }
49
+ function bucketFor(store, key) {
50
+ const keyHash = conditionKeyHash(key);
51
+ return (store.buckets[keyHash] ?? {
52
+ key,
53
+ keyHash,
54
+ priorAlpha: 1,
55
+ priorBeta: 1,
56
+ successes: 0,
57
+ failures: 0,
58
+ drift: 0,
59
+ state: "insufficient-data",
60
+ });
61
+ }
62
+ /**
63
+ * Record one independent task outcome. Never count 100 runs of one task as
64
+ * 100 independent tasks; the caller supplies one pre-registered binary outcome
65
+ * per task. Environment failures belong in delivery metrics, not here.
66
+ */
67
+ export function recordOutcome(store, key, outcome, loss) {
68
+ member(outcome, ["success", "failure"], "outcome");
69
+ finite(loss, "loss");
70
+ const bucket = bucketFor(store, key);
71
+ if (bucket.state === "demoted")
72
+ return store;
73
+ const next = {
74
+ ...bucket,
75
+ successes: bucket.successes + (outcome === "success" ? 1 : 0),
76
+ failures: bucket.failures + (outcome === "failure" ? 1 : 0),
77
+ drift: driftStatistic(bucket.drift, loss, store.referenceMean, store.slack),
78
+ state: bucket.successes + bucket.failures + 1 >= store.minSamples ? "active" : "insufficient-data",
79
+ };
80
+ const demoted = next.drift > store.threshold;
81
+ return {
82
+ ...store,
83
+ buckets: {
84
+ ...store.buckets,
85
+ [next.keyHash]: demoted ? { ...next, state: "demoted", demotedReason: "drift-threshold" } : next,
86
+ },
87
+ };
88
+ }
89
+ /**
90
+ * Demote every bucket whose condition changed underneath it: new model
91
+ * revision, new check definition, or new major toolchain version. Explicit
92
+ * condition changes never wait for statistical drift detection.
93
+ */
94
+ export function demoteOnConditionChange(store, change) {
95
+ const buckets = {};
96
+ for (const [hash, bucket] of Object.entries(store.buckets)) {
97
+ const changed = (change.modelRevision !== undefined && bucket.key.modelRevision !== change.modelRevision) ||
98
+ (change.checkKind !== undefined && bucket.key.checkKind !== change.checkKind) ||
99
+ (change.toolchain !== undefined && bucket.key.toolchain !== change.toolchain) ||
100
+ (change.policyVersion !== undefined && bucket.key.policyVersion !== change.policyVersion);
101
+ buckets[hash] =
102
+ changed && bucket.state !== "demoted"
103
+ ? { ...bucket, state: "demoted", demotedReason: "condition-change" }
104
+ : bucket;
105
+ }
106
+ return { ...store, buckets };
107
+ }
108
+ /** Posterior mean success rate for one bucket, or `insufficient-data`. */
109
+ export function conditionalSuccessRate(store, key) {
110
+ const bucket = store.buckets[conditionKeyHash(key)];
111
+ if (!bucket)
112
+ return { state: "insufficient-data", samples: 0 };
113
+ const samples = bucket.successes + bucket.failures;
114
+ if (bucket.state !== "active")
115
+ return { state: bucket.state, samples };
116
+ return {
117
+ state: "active",
118
+ samples,
119
+ rate: betaPosteriorMean(bucket.priorAlpha, bucket.priorBeta, bucket.successes, bucket.failures),
120
+ };
121
+ }
122
+ //# sourceMappingURL=calibration.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"calibration.js","sourceRoot":"","sources":["../../src/metacognition/calibration.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;GAUG;AACH,OAAO,EAAE,iBAAiB,EAAE,cAAc,EAAE,MAAM,eAAe,CAAC;AAClE,OAAO,EAAE,SAAS,EAAE,MAAM,EAAE,MAAM,EAAE,OAAO,EAAE,OAAO,EAAE,MAAM,EAAE,IAAI,EAAE,MAAM,iBAAiB,CAAC;AAgC5F,MAAM,UAAU,gBAAgB,CAAC,GAAiB,EAAU;IAC3D,IAAI,CAAC,GAAG,CAAC,aAAa,EAAE,eAAe,EAAE,GAAG,CAAC,CAAC;IAC9C,IAAI,CAAC,GAAG,CAAC,SAAS,EAAE,WAAW,EAAE,GAAG,CAAC,CAAC;IACtC,IAAI,CAAC,GAAG,CAAC,QAAQ,EAAE,UAAU,EAAE,GAAG,CAAC,CAAC;IACpC,IAAI,CAAC,GAAG,CAAC,SAAS,EAAE,WAAW,EAAE,GAAG,CAAC,CAAC;IACtC,MAAM,CAAC,GAAG,CAAC,WAAW,EAAE,CAAC,SAAS,EAAE,UAAU,EAAE,YAAY,CAAC,EAAE,aAAa,CAAC,CAAC;IAC9E,IAAI,CAAC,GAAG,CAAC,UAAU,EAAE,YAAY,EAAE,GAAG,CAAC,CAAC;IACxC,IAAI,CAAC,GAAG,CAAC,aAAa,EAAE,eAAe,EAAE,EAAE,CAAC,CAAC;IAC7C,OAAO,SAAS,CAAC;QAChB,GAAG,CAAC,aAAa;QACjB,CAAC,GAAG,GAAG,CAAC,WAAW,CAAC,CAAC,IAAI,CAAC,OAAO,CAAC;QAClC,GAAG,CAAC,SAAS;QACb,GAAG,CAAC,QAAQ;QACZ,GAAG,CAAC,SAAS;QACb,GAAG,CAAC,WAAW;QACf,GAAG,CAAC,UAAU;QACd,GAAG,CAAC,aAAa;KACjB,CAAC,CAAC;AAAA,CACH;AAED,MAAM,UAAU,sBAAsB,CAAC,OAOtC,EAAoB;IACpB,OAAO,CAAC,OAAO,CAAC,UAAU,EAAE,YAAY,EAAE,OAAO,CAAC,CAAC;IACnD,MAAM,CAAC,OAAO,CAAC,UAAU,EAAE,YAAY,EAAE,GAAG,CAAC,CAAC;IAC9C,MAAM,CAAC,OAAO,CAAC,SAAS,EAAE,WAAW,EAAE,GAAG,CAAC,CAAC;IAC5C,MAAM,CAAC,OAAO,CAAC,UAAU,GAAG,CAAC,IAAI,OAAO,CAAC,SAAS,GAAG,CAAC,EAAE,6BAA6B,CAAC,CAAC;IACvF,MAAM,CAAC,OAAO,CAAC,aAAa,EAAE,eAAe,CAAC,CAAC;IAC/C,MAAM,CAAC,OAAO,CAAC,KAAK,EAAE,OAAO,CAAC,CAAC;IAC/B,MAAM,CAAC,OAAO,CAAC,SAAS,EAAE,WAAW,CAAC,CAAC;IACvC,OAAO;QACN,OAAO,EAAE,EAAE;QACX,UAAU,EAAE,OAAO,CAAC,UAAU;QAC9B,aAAa,EAAE,OAAO,CAAC,aAAa;QACpC,KAAK,EAAE,OAAO,CAAC,KAAK;QACpB,SAAS,EAAE,OAAO,CAAC,SAAS;KAC5B,CAAC;AAAA,CACF;AAED,SAAS,SAAS,CAAC,KAAuB,EAAE,GAAiB,EAAqB;IACjF,MAAM,OAAO,GAAG,gBAAgB,CAAC,GAAG,CAAC,CAAC;IACtC,OAAO,CACN,KAAK,CAAC,OAAO,CAAC,OAAO,CAAC,IAAI;QACzB,GAAG;QACH,OAAO;QACP,UAAU,EAAE,CAAC;QACb,SAAS,EAAE,CAAC;QACZ,SAAS,EAAE,CAAC;QACZ,QAAQ,EAAE,CAAC;QACX,KAAK,EAAE,CAAC;QACR,KAAK,EAAE,mBAAmB;KAC1B,CACD,CAAC;AAAA,CACF;AAED;;;;GAIG;AACH,MAAM,UAAU,aAAa,CAC5B,KAAuB,EACvB,GAAiB,EACjB,OAA8B,EAC9B,IAAY,EACO;IACnB,MAAM,CAAC,OAAO,EAAE,CAAC,SAAS,EAAE,SAAS,CAAC,EAAE,SAAS,CAAC,CAAC;IACnD,MAAM,CAAC,IAAI,EAAE,MAAM,CAAC,CAAC;IACrB,MAAM,MAAM,GAAG,SAAS,CAAC,KAAK,EAAE,GAAG,CAAC,CAAC;IACrC,IAAI,MAAM,CAAC,KAAK,KAAK,SAAS;QAAE,OAAO,KAAK,CAAC;IAC7C,MAAM,IAAI,GAAsB;QAC/B,GAAG,MAAM;QACT,SAAS,EAAE,MAAM,CAAC,SAAS,GAAG,CAAC,OAAO,KAAK,SAAS,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC;QAC7D,QAAQ,EAAE,MAAM,CAAC,QAAQ,GAAG,CAAC,OAAO,KAAK,SAAS,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC;QAC3D,KAAK,EAAE,cAAc,CAAC,MAAM,CAAC,KAAK,EAAE,IAAI,EAAE,KAAK,CAAC,aAAa,EAAE,KAAK,CAAC,KAAK,CAAC;QAC3E,KAAK,EAAE,MAAM,CAAC,SAAS,GAAG,MAAM,CAAC,QAAQ,GAAG,CAAC,IAAI,KAAK,CAAC,UAAU,CAAC,CAAC,CAAC,QAAQ,CAAC,CAAC,CAAC,mBAAmB;KAClG,CAAC;IACF,MAAM,OAAO,GAAG,IAAI,CAAC,KAAK,GAAG,KAAK,CAAC,SAAS,CAAC;IAC7C,OAAO;QACN,GAAG,KAAK;QACR,OAAO,EAAE;YACR,GAAG,KAAK,CAAC,OAAO;YAChB,CAAC,IAAI,CAAC,OAAO,CAAC,EAAE,OAAO,CAAC,CAAC,CAAC,EAAE,GAAG,IAAI,EAAE,KAAK,EAAE,SAAS,EAAE,aAAa,EAAE,iBAAiB,EAAE,CAAC,CAAC,CAAC,IAAI;SAChG;KACD,CAAC;AAAA,CACF;AAED;;;;GAIG;AACH,MAAM,UAAU,uBAAuB,CACtC,KAAuB,EACvB,MAAkG,EAC/E;IACnB,MAAM,OAAO,GAAsC,EAAE,CAAC;IACtD,KAAK,MAAM,CAAC,IAAI,EAAE,MAAM,CAAC,IAAI,MAAM,CAAC,OAAO,CAAC,KAAK,CAAC,OAAO,CAAC,EAAE,CAAC;QAC5D,MAAM,OAAO,GACZ,CAAC,MAAM,CAAC,aAAa,KAAK,SAAS,IAAI,MAAM,CAAC,GAAG,CAAC,aAAa,KAAK,MAAM,CAAC,aAAa,CAAC;YACzF,CAAC,MAAM,CAAC,SAAS,KAAK,SAAS,IAAI,MAAM,CAAC,GAAG,CAAC,SAAS,KAAK,MAAM,CAAC,SAAS,CAAC;YAC7E,CAAC,MAAM,CAAC,SAAS,KAAK,SAAS,IAAI,MAAM,CAAC,GAAG,CAAC,SAAS,KAAK,MAAM,CAAC,SAAS,CAAC;YAC7E,CAAC,MAAM,CAAC,aAAa,KAAK,SAAS,IAAI,MAAM,CAAC,GAAG,CAAC,aAAa,KAAK,MAAM,CAAC,aAAa,CAAC,CAAC;QAC3F,OAAO,CAAC,IAAI,CAAC;YACZ,OAAO,IAAI,MAAM,CAAC,KAAK,KAAK,SAAS;gBACpC,CAAC,CAAC,EAAE,GAAG,MAAM,EAAE,KAAK,EAAE,SAAS,EAAE,aAAa,EAAE,kBAAkB,EAAE;gBACpE,CAAC,CAAC,MAAM,CAAC;IACZ,CAAC;IACD,OAAO,EAAE,GAAG,KAAK,EAAE,OAAO,EAAE,CAAC;AAAA,CAC7B;AAED,0EAA0E;AAC1E,MAAM,UAAU,sBAAsB,CACrC,KAAuB,EACvB,GAAiB,EACwC;IACzD,MAAM,MAAM,GAAG,KAAK,CAAC,OAAO,CAAC,gBAAgB,CAAC,GAAG,CAAC,CAAC,CAAC;IACpD,IAAI,CAAC,MAAM;QAAE,OAAO,EAAE,KAAK,EAAE,mBAAmB,EAAE,OAAO,EAAE,CAAC,EAAE,CAAC;IAC/D,MAAM,OAAO,GAAG,MAAM,CAAC,SAAS,GAAG,MAAM,CAAC,QAAQ,CAAC;IACnD,IAAI,MAAM,CAAC,KAAK,KAAK,QAAQ;QAAE,OAAO,EAAE,KAAK,EAAE,MAAM,CAAC,KAAK,EAAE,OAAO,EAAE,CAAC;IACvE,OAAO;QACN,KAAK,EAAE,QAAQ;QACf,OAAO;QACP,IAAI,EAAE,iBAAiB,CAAC,MAAM,CAAC,UAAU,EAAE,MAAM,CAAC,SAAS,EAAE,MAAM,CAAC,SAAS,EAAE,MAAM,CAAC,QAAQ,CAAC;KAC/F,CAAC;AAAA,CACF","sourcesContent":["/**\n * Algorithms E and G — conditional capability/calibration records and\n * drift-triggered demotion. Spec §8, §10.3.\n *\n * Records are per condition bucket: model revision + loaded skill content\n * fingerprints + toolchain version + task/effect/difficulty band + check\n * type/health + budget band + selection policy version. A bucket with too few\n * samples stays `insufficient-data`; a new provider revision, check\n * definition, or major library version demotes the bucket immediately,\n * without waiting for statistical detection.\n */\nimport { betaPosteriorMean, driftStatistic } from \"./decision.ts\";\nimport { canonical, ensure, finite, integer, lexical, member, text } from \"./validation.ts\";\n\nexport interface ConditionKey {\n\treadonly modelRevision: string;\n\treadonly skillHashes: readonly string[];\n\treadonly toolchain: string;\n\treadonly taskBand: string;\n\treadonly checkKind: string;\n\treadonly checkHealth: \"healthy\" | \"degraded\" | \"unverified\";\n\treadonly budgetBand: string;\n\treadonly policyVersion: string;\n}\nexport type BucketState = \"insufficient-data\" | \"active\" | \"demoted\";\nexport interface CalibrationBucket {\n\treadonly key: ConditionKey;\n\treadonly keyHash: string;\n\treadonly priorAlpha: number;\n\treadonly priorBeta: number;\n\tsuccesses: number;\n\tfailures: number;\n\tdrift: number;\n\tstate: BucketState;\n\treadonly demotedReason?: string;\n}\nexport interface CalibrationStore {\n\treadonly buckets: Readonly<Record<string, CalibrationBucket>>;\n\treadonly minSamples: number;\n\treadonly referenceMean: number;\n\treadonly slack: number;\n\treadonly threshold: number;\n}\n\nexport function conditionKeyHash(key: ConditionKey): string {\n\ttext(key.modelRevision, \"modelRevision\", 128);\n\ttext(key.toolchain, \"toolchain\", 128);\n\ttext(key.taskBand, \"taskBand\", 128);\n\ttext(key.checkKind, \"checkKind\", 128);\n\tmember(key.checkHealth, [\"healthy\", \"degraded\", \"unverified\"], \"checkHealth\");\n\ttext(key.budgetBand, \"budgetBand\", 128);\n\ttext(key.policyVersion, \"policyVersion\", 64);\n\treturn canonical([\n\t\tkey.modelRevision,\n\t\t[...key.skillHashes].sort(lexical),\n\t\tkey.toolchain,\n\t\tkey.taskBand,\n\t\tkey.checkKind,\n\t\tkey.checkHealth,\n\t\tkey.budgetBand,\n\t\tkey.policyVersion,\n\t]);\n}\n\nexport function createCalibrationStore(options: {\n\treadonly minSamples: number;\n\treadonly priorAlpha: number;\n\treadonly priorBeta: number;\n\treadonly referenceMean: number;\n\treadonly slack: number;\n\treadonly threshold: number;\n}): CalibrationStore {\n\tinteger(options.minSamples, \"minSamples\", 100_000);\n\tfinite(options.priorAlpha, \"priorAlpha\", 1e9);\n\tfinite(options.priorBeta, \"priorBeta\", 1e9);\n\tensure(options.priorAlpha > 0 && options.priorBeta > 0, \"beta prior must be positive\");\n\tfinite(options.referenceMean, \"referenceMean\");\n\tfinite(options.slack, \"slack\");\n\tfinite(options.threshold, \"threshold\");\n\treturn {\n\t\tbuckets: {},\n\t\tminSamples: options.minSamples,\n\t\treferenceMean: options.referenceMean,\n\t\tslack: options.slack,\n\t\tthreshold: options.threshold,\n\t};\n}\n\nfunction bucketFor(store: CalibrationStore, key: ConditionKey): CalibrationBucket {\n\tconst keyHash = conditionKeyHash(key);\n\treturn (\n\t\tstore.buckets[keyHash] ?? {\n\t\t\tkey,\n\t\t\tkeyHash,\n\t\t\tpriorAlpha: 1,\n\t\t\tpriorBeta: 1,\n\t\t\tsuccesses: 0,\n\t\t\tfailures: 0,\n\t\t\tdrift: 0,\n\t\t\tstate: \"insufficient-data\",\n\t\t}\n\t);\n}\n\n/**\n * Record one independent task outcome. Never count 100 runs of one task as\n * 100 independent tasks; the caller supplies one pre-registered binary outcome\n * per task. Environment failures belong in delivery metrics, not here.\n */\nexport function recordOutcome(\n\tstore: CalibrationStore,\n\tkey: ConditionKey,\n\toutcome: \"success\" | \"failure\",\n\tloss: number,\n): CalibrationStore {\n\tmember(outcome, [\"success\", \"failure\"], \"outcome\");\n\tfinite(loss, \"loss\");\n\tconst bucket = bucketFor(store, key);\n\tif (bucket.state === \"demoted\") return store;\n\tconst next: CalibrationBucket = {\n\t\t...bucket,\n\t\tsuccesses: bucket.successes + (outcome === \"success\" ? 1 : 0),\n\t\tfailures: bucket.failures + (outcome === \"failure\" ? 1 : 0),\n\t\tdrift: driftStatistic(bucket.drift, loss, store.referenceMean, store.slack),\n\t\tstate: bucket.successes + bucket.failures + 1 >= store.minSamples ? \"active\" : \"insufficient-data\",\n\t};\n\tconst demoted = next.drift > store.threshold;\n\treturn {\n\t\t...store,\n\t\tbuckets: {\n\t\t\t...store.buckets,\n\t\t\t[next.keyHash]: demoted ? { ...next, state: \"demoted\", demotedReason: \"drift-threshold\" } : next,\n\t\t},\n\t};\n}\n\n/**\n * Demote every bucket whose condition changed underneath it: new model\n * revision, new check definition, or new major toolchain version. Explicit\n * condition changes never wait for statistical drift detection.\n */\nexport function demoteOnConditionChange(\n\tstore: CalibrationStore,\n\tchange: { modelRevision?: string; checkKind?: string; toolchain?: string; policyVersion?: string },\n): CalibrationStore {\n\tconst buckets: Record<string, CalibrationBucket> = {};\n\tfor (const [hash, bucket] of Object.entries(store.buckets)) {\n\t\tconst changed =\n\t\t\t(change.modelRevision !== undefined && bucket.key.modelRevision !== change.modelRevision) ||\n\t\t\t(change.checkKind !== undefined && bucket.key.checkKind !== change.checkKind) ||\n\t\t\t(change.toolchain !== undefined && bucket.key.toolchain !== change.toolchain) ||\n\t\t\t(change.policyVersion !== undefined && bucket.key.policyVersion !== change.policyVersion);\n\t\tbuckets[hash] =\n\t\t\tchanged && bucket.state !== \"demoted\"\n\t\t\t\t? { ...bucket, state: \"demoted\", demotedReason: \"condition-change\" }\n\t\t\t\t: bucket;\n\t}\n\treturn { ...store, buckets };\n}\n\n/** Posterior mean success rate for one bucket, or `insufficient-data`. */\nexport function conditionalSuccessRate(\n\tstore: CalibrationStore,\n\tkey: ConditionKey,\n): { state: BucketState; rate?: number; samples: number } {\n\tconst bucket = store.buckets[conditionKeyHash(key)];\n\tif (!bucket) return { state: \"insufficient-data\", samples: 0 };\n\tconst samples = bucket.successes + bucket.failures;\n\tif (bucket.state !== \"active\") return { state: bucket.state, samples };\n\treturn {\n\t\tstate: \"active\",\n\t\tsamples,\n\t\trate: betaPosteriorMean(bucket.priorAlpha, bucket.priorBeta, bucket.successes, bucket.failures),\n\t};\n}\n"]}
@@ -0,0 +1,60 @@
1
+ /**
2
+ * §9.4 checkpoint ordering: refresh obligations from host observations,
3
+ * evaluate knowledge gaps, classify prediction mismatches, evaluate verifier
4
+ * blind spots and runner health, then select one bounded action.
5
+ *
6
+ * The host calls this at a safe decision boundary — never mid-stream, and
7
+ * never as a path for model reasoning to remove a `required` obligation.
8
+ */
9
+ import type { ChangeAtom, ObligationRule } from "./obligations.ts";
10
+ import { type ActionConstraints, type MetaAction, type MetaActionKind } from "./policy.ts";
11
+ import type { MetaState } from "./state.ts";
12
+ import { type FinishState } from "./state.ts";
13
+ import { type MutantRecord, type VerifierEvaluation } from "./verifier.ts";
14
+ export interface CheckpointInput {
15
+ readonly state: MetaState;
16
+ readonly atoms?: readonly ChangeAtom[];
17
+ readonly rules?: readonly ObligationRule[];
18
+ readonly nowMs: number;
19
+ readonly constraints: ActionConstraints;
20
+ readonly mutants?: readonly MutantRecord[];
21
+ readonly receipt?: {
22
+ receiptId: string;
23
+ candidateHash: string;
24
+ scope: string;
25
+ } | null;
26
+ readonly options?: {
27
+ readonly maxRepetitions?: number;
28
+ readonly switchCost?: number;
29
+ readonly estimatedSwitchGain?: number;
30
+ readonly voiCandidates?: readonly {
31
+ kind: MetaActionKind;
32
+ voi: number;
33
+ }[];
34
+ readonly progress?: {
35
+ readonly goalScope: string;
36
+ readonly candidateFamily: string;
37
+ readonly errorClass: string;
38
+ readonly approach: string;
39
+ readonly checkMethod: string;
40
+ readonly sourceFamily: string;
41
+ };
42
+ };
43
+ }
44
+ export interface CheckpointResult {
45
+ readonly state: MetaState;
46
+ readonly action: MetaAction;
47
+ readonly finish: FinishState;
48
+ readonly newObligations: number;
49
+ readonly newPredictions: number;
50
+ readonly mismatches: number;
51
+ readonly verifierEvaluations: readonly VerifierEvaluation[];
52
+ }
53
+ /**
54
+ * One checkpoint evaluation. Refreshes obligations from observed atoms when
55
+ * provided, resolves registered predictions against fresh observations when
56
+ * provided, evaluates verifier mutants when provided, then selects the next
57
+ * action under the §14 priority table.
58
+ */
59
+ export declare function checkpoint(input: CheckpointInput): CheckpointResult;
60
+ //# sourceMappingURL=checkpoint.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"checkpoint.d.ts","sourceRoot":"","sources":["../../src/metacognition/checkpoint.ts"],"names":[],"mappings":"AAAA;;;;;;;GAOG;AACH,OAAO,KAAK,EAAE,UAAU,EAAE,cAAc,EAAE,MAAM,kBAAkB,CAAC;AAEnE,OAAO,EAAE,KAAK,iBAAiB,EAAE,KAAK,UAAU,EAAE,KAAK,cAAc,EAAgB,MAAM,aAAa,CAAC;AACzG,OAAO,KAAK,EAAE,SAAS,EAAE,MAAM,YAAY,CAAC;AAC5C,OAAO,EAAE,KAAK,WAAW,EAAkC,MAAM,YAAY,CAAC;AAE9E,OAAO,EAAoB,KAAK,YAAY,EAAE,KAAK,kBAAkB,EAAE,MAAM,eAAe,CAAC;AAE7F,MAAM,WAAW,eAAe;IAC/B,QAAQ,CAAC,KAAK,EAAE,SAAS,CAAC;IAC1B,QAAQ,CAAC,KAAK,CAAC,EAAE,SAAS,UAAU,EAAE,CAAC;IACvC,QAAQ,CAAC,KAAK,CAAC,EAAE,SAAS,cAAc,EAAE,CAAC;IAC3C,QAAQ,CAAC,KAAK,EAAE,MAAM,CAAC;IACvB,QAAQ,CAAC,WAAW,EAAE,iBAAiB,CAAC;IACxC,QAAQ,CAAC,OAAO,CAAC,EAAE,SAAS,YAAY,EAAE,CAAC;IAC3C,QAAQ,CAAC,OAAO,CAAC,EAAE;QAAE,SAAS,EAAE,MAAM,CAAC;QAAC,aAAa,EAAE,MAAM,CAAC;QAAC,KAAK,EAAE,MAAM,CAAA;KAAE,GAAG,IAAI,CAAC;IACtF,QAAQ,CAAC,OAAO,CAAC,EAAE;QAClB,QAAQ,CAAC,cAAc,CAAC,EAAE,MAAM,CAAC;QACjC,QAAQ,CAAC,UAAU,CAAC,EAAE,MAAM,CAAC;QAC7B,QAAQ,CAAC,mBAAmB,CAAC,EAAE,MAAM,CAAC;QACtC,QAAQ,CAAC,aAAa,CAAC,EAAE,SAAS;YAAE,IAAI,EAAE,cAAc,CAAC;YAAC,GAAG,EAAE,MAAM,CAAA;SAAE,EAAE,CAAC;QAC1E,QAAQ,CAAC,QAAQ,CAAC,EAAE;YACnB,QAAQ,CAAC,SAAS,EAAE,MAAM,CAAC;YAC3B,QAAQ,CAAC,eAAe,EAAE,MAAM,CAAC;YACjC,QAAQ,CAAC,UAAU,EAAE,MAAM,CAAC;YAC5B,QAAQ,CAAC,QAAQ,EAAE,MAAM,CAAC;YAC1B,QAAQ,CAAC,WAAW,EAAE,MAAM,CAAC;YAC7B,QAAQ,CAAC,YAAY,EAAE,MAAM,CAAC;SAC9B,CAAC;KACF,CAAC;CACF;AACD,MAAM,WAAW,gBAAgB;IAChC,QAAQ,CAAC,KAAK,EAAE,SAAS,CAAC;IAC1B,QAAQ,CAAC,MAAM,EAAE,UAAU,CAAC;IAC5B,QAAQ,CAAC,MAAM,EAAE,WAAW,CAAC;IAC7B,QAAQ,CAAC,cAAc,EAAE,MAAM,CAAC;IAChC,QAAQ,CAAC,cAAc,EAAE,MAAM,CAAC;IAChC,QAAQ,CAAC,UAAU,EAAE,MAAM,CAAC;IAC5B,QAAQ,CAAC,mBAAmB,EAAE,SAAS,kBAAkB,EAAE,CAAC;CAC5D;AAED;;;;;GAKG;AACH,wBAAgB,UAAU,CAAC,KAAK,EAAE,eAAe,GAAG,gBAAgB,CA4CnE","sourcesContent":["/**\n * §9.4 checkpoint ordering: refresh obligations from host observations,\n * evaluate knowledge gaps, classify prediction mismatches, evaluate verifier\n * blind spots and runner health, then select one bounded action.\n *\n * The host calls this at a safe decision boundary — never mid-stream, and\n * never as a path for model reasoning to remove a `required` obligation.\n */\nimport type { ChangeAtom, ObligationRule } from \"./obligations.ts\";\nimport { instantiateObligations } from \"./obligations.ts\";\nimport { type ActionConstraints, type MetaAction, type MetaActionKind, selectAction } from \"./policy.ts\";\nimport type { MetaState } from \"./state.ts\";\nimport { type FinishState, finishState, validateMetaState } from \"./state.ts\";\nimport { integer } from \"./validation.ts\";\nimport { evaluateVerifier, type MutantRecord, type VerifierEvaluation } from \"./verifier.ts\";\n\nexport interface CheckpointInput {\n\treadonly state: MetaState;\n\treadonly atoms?: readonly ChangeAtom[];\n\treadonly rules?: readonly ObligationRule[];\n\treadonly nowMs: number;\n\treadonly constraints: ActionConstraints;\n\treadonly mutants?: readonly MutantRecord[];\n\treadonly receipt?: { receiptId: string; candidateHash: string; scope: string } | null;\n\treadonly options?: {\n\t\treadonly maxRepetitions?: number;\n\t\treadonly switchCost?: number;\n\t\treadonly estimatedSwitchGain?: number;\n\t\treadonly voiCandidates?: readonly { kind: MetaActionKind; voi: number }[];\n\t\treadonly progress?: {\n\t\t\treadonly goalScope: string;\n\t\t\treadonly candidateFamily: string;\n\t\t\treadonly errorClass: string;\n\t\t\treadonly approach: string;\n\t\t\treadonly checkMethod: string;\n\t\t\treadonly sourceFamily: string;\n\t\t};\n\t};\n}\nexport interface CheckpointResult {\n\treadonly state: MetaState;\n\treadonly action: MetaAction;\n\treadonly finish: FinishState;\n\treadonly newObligations: number;\n\treadonly newPredictions: number;\n\treadonly mismatches: number;\n\treadonly verifierEvaluations: readonly VerifierEvaluation[];\n}\n\n/**\n * One checkpoint evaluation. Refreshes obligations from observed atoms when\n * provided, resolves registered predictions against fresh observations when\n * provided, evaluates verifier mutants when provided, then selects the next\n * action under the §14 priority table.\n */\nexport function checkpoint(input: CheckpointInput): CheckpointResult {\n\tinteger(input.nowMs, \"nowMs\");\n\tvalidateMetaState(input.state);\n\tlet state = input.state;\n\tlet newObligations = 0;\n\tif (input.atoms && input.rules) {\n\t\tconst report = instantiateObligations(input.atoms, input.rules, input.nowMs);\n\t\tconst merged = [...state.obligations, ...report.required, ...report.candidates];\n\t\tnewObligations = report.required.length + report.candidates.length;\n\t\tstate = { ...state, obligations: merged };\n\t}\n\tlet mismatches = 0;\n\tfor (const p of state.predictions) {\n\t\tif (p.status === \"resolved\" && p.mismatch !== undefined && p.mismatch !== \"none\") mismatches += 1;\n\t}\n\tconst verifierEvaluations: VerifierEvaluation[] = [...state.verifier.evaluations];\n\tif (input.mutants) {\n\t\tfor (const obligation of state.obligations) {\n\t\t\tconst scoped = input.mutants.filter((m) => m.obligationId === obligation.id);\n\t\t\tif (scoped.length === 0) continue;\n\t\t\tverifierEvaluations.push(\n\t\t\t\tevaluateVerifier(obligation.id, input.mutants, state.verifier.runnerHealth === \"healthy\"),\n\t\t\t);\n\t\t}\n\t\tstate = { ...state, verifier: { ...state.verifier, evaluations: verifierEvaluations } };\n\t}\n\tconst action = selectAction(state, input.constraints, {\n\t\treceipt: input.receipt ?? null,\n\t\tmaxRepetitions: input.options?.maxRepetitions,\n\t\tswitchCost: input.options?.switchCost,\n\t\testimatedSwitchGain: input.options?.estimatedSwitchGain,\n\t\tvoiCandidates: input.options?.voiCandidates,\n\t\tprogress: input.options?.progress,\n\t});\n\tconst finish = finishState(state, input.receipt ?? null);\n\treturn {\n\t\tstate: { ...state, checkpointCount: state.checkpointCount + 1 },\n\t\taction,\n\t\tfinish,\n\t\tnewObligations,\n\t\tnewPredictions: 0,\n\t\tmismatches,\n\t\tverifierEvaluations,\n\t};\n}\n"]}