@adaptic/utils 0.0.1014 → 0.0.1016

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (152) hide show
  1. package/dist/index.cjs +2823 -25
  2. package/dist/index.cjs.map +1 -1
  3. package/dist/index.mjs +2783 -25
  4. package/dist/index.mjs.map +1 -1
  5. package/dist/types/__tests__/llm/client/support/rejections.d.ts +21 -0
  6. package/dist/types/__tests__/llm/client/support/rejections.d.ts.map +1 -0
  7. package/dist/types/__tests__/llm/client/support/routes.d.ts +67 -0
  8. package/dist/types/__tests__/llm/client/support/routes.d.ts.map +1 -0
  9. package/dist/types/__tests__/llm/client/support/streams.d.ts +74 -0
  10. package/dist/types/__tests__/llm/client/support/streams.d.ts.map +1 -0
  11. package/dist/types/__tests__/llm/client/support/transports.d.ts +105 -0
  12. package/dist/types/__tests__/llm/client/support/transports.d.ts.map +1 -0
  13. package/dist/types/index.d.ts +1 -0
  14. package/dist/types/index.d.ts.map +1 -1
  15. package/dist/types/llm/alias-client.d.ts +64 -0
  16. package/dist/types/llm/alias-client.d.ts.map +1 -0
  17. package/dist/types/llm/circuit-breaker.d.ts +124 -0
  18. package/dist/types/llm/circuit-breaker.d.ts.map +1 -0
  19. package/dist/types/llm/eval/comparators.d.ts +127 -0
  20. package/dist/types/llm/eval/comparators.d.ts.map +1 -0
  21. package/dist/types/llm/eval/coverage.d.ts +44 -0
  22. package/dist/types/llm/eval/coverage.d.ts.map +1 -0
  23. package/dist/types/llm/eval/golden-set.d.ts +47 -0
  24. package/dist/types/llm/eval/golden-set.d.ts.map +1 -0
  25. package/dist/types/llm/eval/index.d.ts +25 -0
  26. package/dist/types/llm/eval/index.d.ts.map +1 -0
  27. package/dist/types/llm/eval/json-shape.d.ts +74 -0
  28. package/dist/types/llm/eval/json-shape.d.ts.map +1 -0
  29. package/dist/types/llm/eval/judge.d.ts +131 -0
  30. package/dist/types/llm/eval/judge.d.ts.map +1 -0
  31. package/dist/types/llm/eval/metrics.d.ts +51 -0
  32. package/dist/types/llm/eval/metrics.d.ts.map +1 -0
  33. package/dist/types/llm/eval/run.d.ts +97 -0
  34. package/dist/types/llm/eval/run.d.ts.map +1 -0
  35. package/dist/types/llm/eval/types.d.ts +242 -0
  36. package/dist/types/llm/eval/types.d.ts.map +1 -0
  37. package/dist/types/llm/fallback-chain.d.ts +97 -0
  38. package/dist/types/llm/fallback-chain.d.ts.map +1 -0
  39. package/dist/types/llm/index.d.ts +30 -0
  40. package/dist/types/llm/index.d.ts.map +1 -0
  41. package/dist/types/llm/param-matrix.d.ts +65 -0
  42. package/dist/types/llm/param-matrix.d.ts.map +1 -0
  43. package/dist/types/llm/rate-guard.d.ts +119 -0
  44. package/dist/types/llm/rate-guard.d.ts.map +1 -0
  45. package/dist/types/llm/route-table.d.ts +155 -0
  46. package/dist/types/llm/route-table.d.ts.map +1 -0
  47. package/dist/types/llm/schema-retry.d.ts +80 -0
  48. package/dist/types/llm/schema-retry.d.ts.map +1 -0
  49. package/dist/types/llm/streaming.d.ts +93 -0
  50. package/dist/types/llm/streaming.d.ts.map +1 -0
  51. package/dist/types/llm/transports/direct.d.ts +92 -0
  52. package/dist/types/llm/transports/direct.d.ts.map +1 -0
  53. package/dist/types/llm/transports/gateway.d.ts +73 -0
  54. package/dist/types/llm/transports/gateway.d.ts.map +1 -0
  55. package/dist/types/llm/types.d.ts +292 -0
  56. package/dist/types/llm/types.d.ts.map +1 -0
  57. package/dist/types/schemas/alpaca-schemas.d.ts +6 -6
  58. package/package.json +2 -2
  59. package/dist/types/__tests__/alpaca-broker-error-preservation.test.d.ts +0 -2
  60. package/dist/types/__tests__/alpaca-broker-error-preservation.test.d.ts.map +0 -1
  61. package/dist/types/__tests__/alpaca-client-order-id.test.d.ts +0 -2
  62. package/dist/types/__tests__/alpaca-client-order-id.test.d.ts.map +0 -1
  63. package/dist/types/__tests__/alpaca-functions.test.d.ts +0 -2
  64. package/dist/types/__tests__/alpaca-functions.test.d.ts.map +0 -1
  65. package/dist/types/__tests__/alpaca-market-data-retry.test.d.ts +0 -2
  66. package/dist/types/__tests__/alpaca-market-data-retry.test.d.ts.map +0 -1
  67. package/dist/types/__tests__/alpaca-order-idempotency.test.d.ts +0 -2
  68. package/dist/types/__tests__/alpaca-order-idempotency.test.d.ts.map +0 -1
  69. package/dist/types/__tests__/alpaca-trading-api.test.d.ts +0 -2
  70. package/dist/types/__tests__/alpaca-trading-api.test.d.ts.map +0 -1
  71. package/dist/types/__tests__/api-endpoints.test.d.ts +0 -2
  72. package/dist/types/__tests__/api-endpoints.test.d.ts.map +0 -1
  73. package/dist/types/__tests__/asset-allocation.test.d.ts +0 -2
  74. package/dist/types/__tests__/asset-allocation.test.d.ts.map +0 -1
  75. package/dist/types/__tests__/atr.test.d.ts +0 -2
  76. package/dist/types/__tests__/atr.test.d.ts.map +0 -1
  77. package/dist/types/__tests__/auth-validator.test.d.ts +0 -2
  78. package/dist/types/__tests__/auth-validator.test.d.ts.map +0 -1
  79. package/dist/types/__tests__/broker-factory.test.d.ts +0 -2
  80. package/dist/types/__tests__/broker-factory.test.d.ts.map +0 -1
  81. package/dist/types/__tests__/broker-types.test.d.ts +0 -2
  82. package/dist/types/__tests__/broker-types.test.d.ts.map +0 -1
  83. package/dist/types/__tests__/cache.test.d.ts +0 -2
  84. package/dist/types/__tests__/cache.test.d.ts.map +0 -1
  85. package/dist/types/__tests__/errors.test.d.ts +0 -2
  86. package/dist/types/__tests__/errors.test.d.ts.map +0 -1
  87. package/dist/types/__tests__/financial-regression.test.d.ts +0 -2
  88. package/dist/types/__tests__/financial-regression.test.d.ts.map +0 -1
  89. package/dist/types/__tests__/format-tools.test.d.ts +0 -2
  90. package/dist/types/__tests__/format-tools.test.d.ts.map +0 -1
  91. package/dist/types/__tests__/http-keep-alive.test.d.ts +0 -2
  92. package/dist/types/__tests__/http-keep-alive.test.d.ts.map +0 -1
  93. package/dist/types/__tests__/http-timeout.test.d.ts +0 -2
  94. package/dist/types/__tests__/http-timeout.test.d.ts.map +0 -1
  95. package/dist/types/__tests__/index.test.d.ts +0 -2
  96. package/dist/types/__tests__/index.test.d.ts.map +0 -1
  97. package/dist/types/__tests__/legacy-auth.test.d.ts +0 -2
  98. package/dist/types/__tests__/legacy-auth.test.d.ts.map +0 -1
  99. package/dist/types/__tests__/logger.test.d.ts +0 -2
  100. package/dist/types/__tests__/logger.test.d.ts.map +0 -1
  101. package/dist/types/__tests__/logging.test.d.ts +0 -2
  102. package/dist/types/__tests__/logging.test.d.ts.map +0 -1
  103. package/dist/types/__tests__/market-time.test.d.ts +0 -2
  104. package/dist/types/__tests__/market-time.test.d.ts.map +0 -1
  105. package/dist/types/__tests__/massive.test.d.ts +0 -2
  106. package/dist/types/__tests__/massive.test.d.ts.map +0 -1
  107. package/dist/types/__tests__/metrics-calcs-direction.test.d.ts +0 -2
  108. package/dist/types/__tests__/metrics-calcs-direction.test.d.ts.map +0 -1
  109. package/dist/types/__tests__/misc-utils.test.d.ts +0 -2
  110. package/dist/types/__tests__/misc-utils.test.d.ts.map +0 -1
  111. package/dist/types/__tests__/paginator.test.d.ts +0 -2
  112. package/dist/types/__tests__/paginator.test.d.ts.map +0 -1
  113. package/dist/types/__tests__/performance-metrics-fees.test.d.ts +0 -2
  114. package/dist/types/__tests__/performance-metrics-fees.test.d.ts.map +0 -1
  115. package/dist/types/__tests__/performance-metrics.test.d.ts +0 -2
  116. package/dist/types/__tests__/performance-metrics.test.d.ts.map +0 -1
  117. package/dist/types/__tests__/price-utils-fees.test.d.ts +0 -2
  118. package/dist/types/__tests__/price-utils-fees.test.d.ts.map +0 -1
  119. package/dist/types/__tests__/price-utils.test.d.ts +0 -2
  120. package/dist/types/__tests__/price-utils.test.d.ts.map +0 -1
  121. package/dist/types/__tests__/property-based-financial.test.d.ts +0 -2
  122. package/dist/types/__tests__/property-based-financial.test.d.ts.map +0 -1
  123. package/dist/types/__tests__/protective-order-sides.test.d.ts +0 -2
  124. package/dist/types/__tests__/protective-order-sides.test.d.ts.map +0 -1
  125. package/dist/types/__tests__/rate-limiter.test.d.ts +0 -2
  126. package/dist/types/__tests__/rate-limiter.test.d.ts.map +0 -1
  127. package/dist/types/__tests__/retry-classification.test.d.ts +0 -2
  128. package/dist/types/__tests__/retry-classification.test.d.ts.map +0 -1
  129. package/dist/types/__tests__/retry.test.d.ts +0 -2
  130. package/dist/types/__tests__/retry.test.d.ts.map +0 -1
  131. package/dist/types/__tests__/risk-free-rate.test.d.ts +0 -2
  132. package/dist/types/__tests__/risk-free-rate.test.d.ts.map +0 -1
  133. package/dist/types/__tests__/risk-metrics.test.d.ts +0 -2
  134. package/dist/types/__tests__/risk-metrics.test.d.ts.map +0 -1
  135. package/dist/types/__tests__/schema-validation.test.d.ts +0 -2
  136. package/dist/types/__tests__/schema-validation.test.d.ts.map +0 -1
  137. package/dist/types/__tests__/stampede-load-timeout.test.d.ts +0 -2
  138. package/dist/types/__tests__/stampede-load-timeout.test.d.ts.map +0 -1
  139. package/dist/types/__tests__/strategy-metrics.test.d.ts +0 -2
  140. package/dist/types/__tests__/strategy-metrics.test.d.ts.map +0 -1
  141. package/dist/types/__tests__/technical-analysis-totality.test.d.ts +0 -2
  142. package/dist/types/__tests__/technical-analysis-totality.test.d.ts.map +0 -1
  143. package/dist/types/__tests__/technical-analysis.test.d.ts +0 -2
  144. package/dist/types/__tests__/technical-analysis.test.d.ts.map +0 -1
  145. package/dist/types/__tests__/time-utils.test.d.ts +0 -2
  146. package/dist/types/__tests__/time-utils.test.d.ts.map +0 -1
  147. package/dist/types/__tests__/trading-policy-schemas.test.d.ts +0 -2
  148. package/dist/types/__tests__/trading-policy-schemas.test.d.ts.map +0 -1
  149. package/dist/types/__tests__/trailing-stops-portfolio.test.d.ts +0 -2
  150. package/dist/types/__tests__/trailing-stops-portfolio.test.d.ts.map +0 -1
  151. package/dist/types/__tests__/volatility.test.d.ts +0 -2
  152. package/dist/types/__tests__/volatility.test.d.ts.map +0 -1
@@ -0,0 +1 @@
1
+ {"version":3,"file":"comparators.d.ts","sourceRoot":"","sources":["../../../../src/llm/eval/comparators.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;;GAoBG;AAEH,OAAO,KAAK,EAAE,aAAa,EAAE,QAAQ,EAAE,OAAO,EAAE,MAAM,SAAS,CAAC;AAEhE;;;;;;GAMG;AACH,eAAO,MAAM,sBAAsB,IAAI,CAAC;AAExC;;;;;;;GAOG;AACH,eAAO,MAAM,wBAAwB,OAAO,CAAC;AAE7C,8FAA8F;AAC9F,eAAO,MAAM,cAAc,MAAM,CAAC;AAiBlC,qDAAqD;AACrD,MAAM,WAAW,eAAe;IAC9B,oFAAoF;IACpF,QAAQ,CAAC,SAAS,EAAE,MAAM,GAAG,SAAS,CAAC;IACvC,wFAAwF;IACxF,QAAQ,CAAC,SAAS,EAAE,MAAM,GAAG,SAAS,CAAC;IACvC,4CAA4C;IAC5C,QAAQ,CAAC,CAAC,EAAE,MAAM,CAAC;IACnB,6CAA6C;IAC7C,QAAQ,CAAC,IAAI,EAAE,MAAM,CAAC;CACvB;AAED,yDAAyD;AACzD,MAAM,MAAM,UAAU,GAAG,CAAC,KAAK,EAAE,eAAe,KAAK,OAAO,CAAC;AAkF7D;;;;;;;;GAQG;AACH,eAAO,MAAM,sBAAsB,EAAE,UACsB,CAAC;AAE5D;;;;;;;;;GASG;AACH,eAAO,MAAM,oBAAoB,EAAE,UACmB,CAAC;AAEvD;;;;;;;;;GASG;AACH,eAAO,MAAM,iBAAiB,EAAE,UA8B/B,CAAC;AAEF;;;;;;;;;GASG;AACH,eAAO,MAAM,iBAAiB,EAAE,UA8B/B,CAAC;AAEF;;;;;;;;;;GAUG;AACH,eAAO,MAAM,iBAAiB,EAAE,UA2B/B,CAAC;AAEF;;;;;;GAMG;AACH,eAAO,MAAM,WAAW,EAAE,QAAQ,CAAC,MAAM,CAAC,aAAa,EAAE,UAAU,CAAC,CAMnE,CAAC;AAEF;;;;;;;GAOG;AACH,eAAO,MAAM,oBAAoB,EAAE,QAAQ,CAAC,MAAM,CAAC,QAAQ,EAAE,SAAS,aAAa,EAAE,CAAC,CAKrF,CAAC"}
@@ -0,0 +1,44 @@
1
+ /**
2
+ * Coverage of the harness itself.
3
+ *
4
+ * A gate is only worth its CI minutes if every assertion it claims to enforce
5
+ * has been demonstrated to fire in both directions. "Implemented" is not the
6
+ * bar — a comparator that always returns `pass` is implemented. The bar is
7
+ * EXECUTED EVIDENCE: for each assertion, a fixture that this run graded as
8
+ * passing and a fixture that this run graded as failing.
9
+ *
10
+ * The route table is the other half of the question. An alias whose eval gate
11
+ * maps to no assertion bundle is an alias nothing grades, and it would sail
12
+ * through a migration unmeasured, so the coverage check reads the table rather
13
+ * than a list someone maintained by hand.
14
+ *
15
+ * @module llm/eval/coverage
16
+ */
17
+ import type { LlmRouteTable } from "../types";
18
+ import type { EvalAssertion, EvalStatus } from "./types";
19
+ /** Assertions a run actually demonstrated, in each direction. */
20
+ export interface ObservedAssertions {
21
+ /** Assertions this run graded as passing on at least one fixture. */
22
+ readonly green: readonly EvalAssertion[];
23
+ /** Assertions this run graded as failing on at least one fixture. */
24
+ readonly red: readonly EvalAssertion[];
25
+ }
26
+ /** The outcome of the coverage check. */
27
+ export interface CoverageReport {
28
+ /** Assertions required by at least one eval gate in the route table. */
29
+ readonly required: readonly EvalAssertion[];
30
+ /** Everything missing; empty is the only passing state. */
31
+ readonly problems: readonly string[];
32
+ /** PASSED only when nothing is missing. */
33
+ readonly status: EvalStatus;
34
+ }
35
+ /**
36
+ * Check that every assertion the route table requires is implemented and was
37
+ * demonstrated in both directions by this run.
38
+ *
39
+ * @param observed Assertions this run graded as passing and as failing.
40
+ * @param table The route table to read; defaults to the canonical one.
41
+ * @returns The coverage report.
42
+ */
43
+ export declare function computeCoverage(observed: ObservedAssertions, table?: LlmRouteTable): CoverageReport;
44
+ //# sourceMappingURL=coverage.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"coverage.d.ts","sourceRoot":"","sources":["../../../../src/llm/eval/coverage.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;GAeG;AAGH,OAAO,KAAK,EAAE,aAAa,EAAE,MAAM,UAAU,CAAC;AAG9C,OAAO,KAAK,EAAE,aAAa,EAAY,UAAU,EAAE,MAAM,SAAS,CAAC;AAEnE,iEAAiE;AACjE,MAAM,WAAW,kBAAkB;IACjC,qEAAqE;IACrE,QAAQ,CAAC,KAAK,EAAE,SAAS,aAAa,EAAE,CAAC;IACzC,qEAAqE;IACrE,QAAQ,CAAC,GAAG,EAAE,SAAS,aAAa,EAAE,CAAC;CACxC;AAED,yCAAyC;AACzC,MAAM,WAAW,cAAc;IAC7B,wEAAwE;IACxE,QAAQ,CAAC,QAAQ,EAAE,SAAS,aAAa,EAAE,CAAC;IAC5C,2DAA2D;IAC3D,QAAQ,CAAC,QAAQ,EAAE,SAAS,MAAM,EAAE,CAAC;IACrC,2CAA2C;IAC3C,QAAQ,CAAC,MAAM,EAAE,UAAU,CAAC;CAC7B;AA4BD;;;;;;;GAOG;AACH,wBAAgB,eAAe,CAC7B,QAAQ,EAAE,kBAAkB,EAC5B,KAAK,GAAE,aAA0B,GAChC,cAAc,CAwChB"}
@@ -0,0 +1,47 @@
1
+ /**
2
+ * Parsing and structural validation of golden sets and candidate runs.
3
+ *
4
+ * A golden set is evidence, and evidence that is not validated on the way in is
5
+ * evidence nobody can rely on later. Every field is checked here, at the edge,
6
+ * so that by the time a comparator sees a number the only remaining question is
7
+ * the one the comparator exists to answer.
8
+ *
9
+ * Validation is strict in one particular direction: a set that declares an
10
+ * assertion must carry everything that assertion needs. A set asserting
11
+ * `schema-valid` without a response shape, or `tool-call` without a tool
12
+ * contract, is rejected rather than quietly graded on the assertions it does
13
+ * support — a gate that drops the check it cannot perform is a gate that gets
14
+ * weaker exactly where the evidence is thinnest.
15
+ *
16
+ * @module llm/eval/golden-set
17
+ */
18
+ import type { CandidateRun, GoldenSet } from "./types";
19
+ /** Thrown when a golden set or candidate run is not well formed. */
20
+ export declare class GoldenSetError extends Error {
21
+ /** Where the malformed document came from, so the message is actionable. */
22
+ readonly origin: string;
23
+ /**
24
+ * @param origin Where the document came from.
25
+ * @param detail What is wrong with it.
26
+ */
27
+ constructor(origin: string, detail: string);
28
+ }
29
+ /**
30
+ * Parse and validate a golden set.
31
+ *
32
+ * @param raw The parsed JSON document.
33
+ * @param origin Where it came from, quoted in any error.
34
+ * @returns The validated golden set.
35
+ * @throws {GoldenSetError} When the document is not a valid golden set.
36
+ */
37
+ export declare function parseGoldenSet(raw: unknown, origin: string): GoldenSet;
38
+ /**
39
+ * Parse and validate a candidate run.
40
+ *
41
+ * @param raw The parsed JSON document.
42
+ * @param origin Where it came from, quoted in any error.
43
+ * @returns The validated candidate run.
44
+ * @throws {GoldenSetError} When the document is not a valid candidate run.
45
+ */
46
+ export declare function parseCandidateRun(raw: unknown, origin: string): CandidateRun;
47
+ //# sourceMappingURL=golden-set.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"golden-set.d.ts","sourceRoot":"","sources":["../../../../src/llm/eval/golden-set.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;GAgBG;AAKH,OAAO,KAAK,EACV,YAAY,EAIZ,SAAS,EAOV,MAAM,SAAS,CAAC;AAEjB,oEAAoE;AACpE,qBAAa,cAAe,SAAQ,KAAK;IACvC,4EAA4E;IAC5E,SAAgB,MAAM,EAAE,MAAM,CAAC;IAE/B;;;OAGG;gBACgB,MAAM,EAAE,MAAM,EAAE,MAAM,EAAE,MAAM;CAKlD;AA8QD;;;;;;;GAOG;AACH,wBAAgB,cAAc,CAAC,GAAG,EAAE,OAAO,EAAE,MAAM,EAAE,MAAM,GAAG,SAAS,CAyJtE;AAED;;;;;;;GAOG;AACH,wBAAgB,iBAAiB,CAAC,GAAG,EAAE,OAAO,EAAE,MAAM,EAAE,MAAM,GAAG,YAAY,CAa5E"}
@@ -0,0 +1,25 @@
1
+ /**
2
+ * Public surface of the LLM migration eval harness.
3
+ *
4
+ * The harness grades a candidate model against a recorded incumbent baseline on
5
+ * captured golden sets, and renders every verdict by code from recorded data —
6
+ * PD-6 forbids agent judgement from standing in for a gate, and an exported
7
+ * surface with no "just tell me if this looks fine" entry point is what makes
8
+ * that structural rather than aspirational.
9
+ *
10
+ * @module llm/eval
11
+ */
12
+ export { COMPARATORS, EVAL_GATE_ASSERTIONS, JUDGE_TOLERANCE_FRACTION, MATCH_TOLERANCE_POINTS, RATE_TO_POINTS, compareJudgeScore, compareLatencyP95, compareMatchScore, compareSchemaValidRate, compareValidCallRate, } from "./comparators";
13
+ export type { Comparator, ComparatorInput } from "./comparators";
14
+ export { computeCoverage } from "./coverage";
15
+ export type { CoverageReport, ObservedAssertions } from "./coverage";
16
+ export { GoldenSetError, parseCandidateRun, parseGoldenSet } from "./golden-set";
17
+ export { F1_SCALE_POINTS, fieldF1, jsonEquals, leafFields, p95, satisfiesShape } from "./json-shape";
18
+ export { JUDGE_SCORE_MAX, JUDGE_SCORE_MIN, JudgeNotPinnedError, PINNED_JUDGE, assertPinnedJudge, scoreCandidateRun, scoreWithPinnedJudge, } from "./judge";
19
+ export type { JudgeRequest, PinnedJudgeIdentity, ResolvedJudge } from "./judge";
20
+ export { MAX_STRUCTURED_RETRIES, deriveMetrics, isValidToolCall } from "./metrics";
21
+ export type { MetricSample } from "./metrics";
22
+ export { BASELINE_ATTESTATION_TOLERANCE, evaluateRun, evaluateSet, formatRunReport, formatVerdict, } from "./run";
23
+ export type { EvalPair, EvaluateOptions } from "./run";
24
+ export type { BaselineMetrics, CandidateRun, EvalAssertion, EvalGate, EvalStatus, GoldenCase, GoldenSet, IncumbentBaseline, JsonShape, MatchMetric, MetricUnit, RecordedResult, RecordedToolCall, RunReport, SetReport, ToolExpectation, Verdict, } from "./types";
25
+ //# sourceMappingURL=index.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"index.d.ts","sourceRoot":"","sources":["../../../../src/llm/eval/index.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;GAUG;AAEH,OAAO,EACL,WAAW,EACX,oBAAoB,EACpB,wBAAwB,EACxB,sBAAsB,EACtB,cAAc,EACd,iBAAiB,EACjB,iBAAiB,EACjB,iBAAiB,EACjB,sBAAsB,EACtB,oBAAoB,GACrB,MAAM,eAAe,CAAC;AACvB,YAAY,EAAE,UAAU,EAAE,eAAe,EAAE,MAAM,eAAe,CAAC;AAEjE,OAAO,EAAE,eAAe,EAAE,MAAM,YAAY,CAAC;AAC7C,YAAY,EAAE,cAAc,EAAE,kBAAkB,EAAE,MAAM,YAAY,CAAC;AAErE,OAAO,EAAE,cAAc,EAAE,iBAAiB,EAAE,cAAc,EAAE,MAAM,cAAc,CAAC;AAEjF,OAAO,EAAE,eAAe,EAAE,OAAO,EAAE,UAAU,EAAE,UAAU,EAAE,GAAG,EAAE,cAAc,EAAE,MAAM,cAAc,CAAC;AAErG,OAAO,EACL,eAAe,EACf,eAAe,EACf,mBAAmB,EACnB,YAAY,EACZ,iBAAiB,EACjB,iBAAiB,EACjB,oBAAoB,GACrB,MAAM,SAAS,CAAC;AACjB,YAAY,EAAE,YAAY,EAAE,mBAAmB,EAAE,aAAa,EAAE,MAAM,SAAS,CAAC;AAEhF,OAAO,EAAE,sBAAsB,EAAE,aAAa,EAAE,eAAe,EAAE,MAAM,WAAW,CAAC;AACnF,YAAY,EAAE,YAAY,EAAE,MAAM,WAAW,CAAC;AAE9C,OAAO,EACL,8BAA8B,EAC9B,WAAW,EACX,WAAW,EACX,eAAe,EACf,aAAa,GACd,MAAM,OAAO,CAAC;AACf,YAAY,EAAE,QAAQ,EAAE,eAAe,EAAE,MAAM,OAAO,CAAC;AAEvD,YAAY,EACV,eAAe,EACf,YAAY,EACZ,aAAa,EACb,QAAQ,EACR,UAAU,EACV,UAAU,EACV,SAAS,EACT,iBAAiB,EACjB,SAAS,EACT,WAAW,EACX,UAAU,EACV,cAAc,EACd,gBAAgB,EAChB,SAAS,EACT,SAAS,EACT,eAAe,EACf,OAAO,GACR,MAAM,SAAS,CAAC"}
@@ -0,0 +1,74 @@
1
+ /**
2
+ * Structural comparison primitives the comparators are built from.
3
+ *
4
+ * These are separated from the comparators so that "did this answer parse" and
5
+ * "was this answer right" stay distinct measurements. Conflating them hides the
6
+ * most useful diagnostic in a model swap: a candidate can be perfectly
7
+ * well-formed and consistently wrong, or consistently right and badly
8
+ * formatted, and those two have opposite remedies.
9
+ *
10
+ * The validator honours a documented subset of JSON Schema and adds no
11
+ * dependency. An unrecognised keyword is ignored rather than treated as
12
+ * satisfied, so the subset can only ever be stricter than a caller expects,
13
+ * never more permissive.
14
+ *
15
+ * @module llm/eval/json-shape
16
+ */
17
+ import type { JsonShape } from "./types";
18
+ /** Points on the F1 scale, so a tolerance expressed "in points" has a fixed meaning. */
19
+ export declare const F1_SCALE_POINTS = 100;
20
+ /**
21
+ * Whether a value satisfies a shape.
22
+ *
23
+ * @param value The value to check.
24
+ * @param shape The shape it must satisfy.
25
+ * @returns Whether the value satisfies the shape.
26
+ */
27
+ export declare function satisfiesShape(value: unknown, shape: JsonShape): boolean;
28
+ /**
29
+ * Structural equality over JSON values.
30
+ *
31
+ * Order-sensitive for arrays and order-insensitive for object keys, which is
32
+ * what JSON itself means: two objects with the same entries are the same
33
+ * answer, while a reordered list is a different one.
34
+ *
35
+ * @param left The first value.
36
+ * @param right The second value.
37
+ * @returns Whether the two are structurally equal.
38
+ */
39
+ export declare function jsonEquals(left: unknown, right: unknown): boolean;
40
+ /**
41
+ * Flatten a JSON value into dotted leaf paths paired with their serialised values.
42
+ *
43
+ * Field-level F1 needs a set of comparable atoms, and a leaf path is the
44
+ * natural one for extraction output: it credits a candidate for the fields it
45
+ * got right instead of scoring the whole record all-or-nothing, which is the
46
+ * difference between a metric that can move by one point and one that can only
47
+ * move by whole cases.
48
+ *
49
+ * @param value The value to flatten.
50
+ * @param prefix Path prefix used by the recursion.
51
+ * @returns Leaf paths mapped to their serialised values.
52
+ */
53
+ export declare function leafFields(value: unknown, prefix?: string): Map<string, string>;
54
+ /**
55
+ * Field-level F1 between a produced value and the reference, in points.
56
+ *
57
+ * @param produced What the model returned.
58
+ * @param expected The reference answer.
59
+ * @returns F1 on a 0-100 point scale; 0 when nothing matched.
60
+ */
61
+ export declare function fieldF1(produced: unknown, expected: unknown): number;
62
+ /**
63
+ * The p95 of a sample by nearest-rank.
64
+ *
65
+ * Nearest-rank rather than an interpolating estimator because a latency gate
66
+ * must name a value the system actually produced; an interpolated p95 is a
67
+ * number no request ever took, and a gate is easier to trust when its threshold
68
+ * is an observation.
69
+ *
70
+ * @param values The sample.
71
+ * @returns The p95 value, or `null` for an empty sample.
72
+ */
73
+ export declare function p95(values: readonly number[]): number | null;
74
+ //# sourceMappingURL=json-shape.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"json-shape.d.ts","sourceRoot":"","sources":["../../../../src/llm/eval/json-shape.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;GAeG;AAEH,OAAO,KAAK,EAAE,SAAS,EAAE,MAAM,SAAS,CAAC;AAEzC,wFAAwF;AACxF,eAAO,MAAM,eAAe,MAAM,CAAC;AAEnC;;;;;;GAMG;AACH,wBAAgB,cAAc,CAAC,KAAK,EAAE,OAAO,EAAE,KAAK,EAAE,SAAS,GAAG,OAAO,CA6BxE;AA4BD;;;;;;;;;;GAUG;AACH,wBAAgB,UAAU,CAAC,IAAI,EAAE,OAAO,EAAE,KAAK,EAAE,OAAO,GAAG,OAAO,CA6BjE;AAED;;;;;;;;;;;;GAYG;AACH,wBAAgB,UAAU,CAAC,KAAK,EAAE,OAAO,EAAE,MAAM,SAAK,GAAG,GAAG,CAAC,MAAM,EAAE,MAAM,CAAC,CAqB3E;AAED;;;;;;GAMG;AACH,wBAAgB,OAAO,CAAC,QAAQ,EAAE,OAAO,EAAE,QAAQ,EAAE,OAAO,GAAG,MAAM,CAepE;AAED;;;;;;;;;;GAUG;AACH,wBAAgB,GAAG,CAAC,MAAM,EAAE,SAAS,MAAM,EAAE,GAAG,MAAM,GAAG,IAAI,CAQ5D"}
@@ -0,0 +1,131 @@
1
+ /**
2
+ * The pinned judge, and the guard that keeps it pinned.
3
+ *
4
+ * PD-6 states that the judge model is pinned and never auto-swapped. That is a
5
+ * stronger requirement than "configured": a judge that could move would make
6
+ * every gate it decided incomparable with the gates decided before the move,
7
+ * because the scores either side of it were produced against different
8
+ * standards. A migration programme whose bar drifts under it cannot demonstrate
9
+ * anything.
10
+ *
11
+ * The pin is therefore recorded in TWO places that must agree — this module and
12
+ * the route table — and a judged assertion refuses to run when they disagree.
13
+ * The duplication is deliberate. A pin that lived only in the table would be
14
+ * satisfied by any model the table happened to name, so changing the table
15
+ * would silently change the standard; requiring both to move makes a judge swap
16
+ * an explicit, reviewable act in the same change that re-baselines the sets.
17
+ *
18
+ * @module llm/eval/judge
19
+ */
20
+ import type { LlmAlias, LlmRouteTable } from "../types";
21
+ import type { CandidateRun, GoldenSet } from "./types";
22
+ /** Lowest score the judge rubric may return. */
23
+ export declare const JUDGE_SCORE_MIN = 0;
24
+ /** Highest score the judge rubric may return, matching the point scale the comparators use. */
25
+ export declare const JUDGE_SCORE_MAX = 100;
26
+ /** The identity a judged assertion requires the route table to resolve to. */
27
+ export interface PinnedJudgeIdentity {
28
+ /** The only alias a judged assertion may be served by. */
29
+ readonly alias: LlmAlias;
30
+ /** The provider the judge route must name. */
31
+ readonly provider: string;
32
+ /** The exact model id the judge route must name. */
33
+ readonly modelId: string;
34
+ }
35
+ /**
36
+ * The pinned judge identity.
37
+ *
38
+ * This is the one place in the codebase where naming a vendor model string is
39
+ * the requirement rather than a violation of it: PD-5 forbids model strings in
40
+ * APPLICATION code so that routing stays in config, while PD-6 requires the
41
+ * judge to be pinned so that grading stays comparable. Pinning by anything
42
+ * looser — a family, a provider, "whatever the table says" — would not pin
43
+ * anything.
44
+ */
45
+ export declare const PINNED_JUDGE: PinnedJudgeIdentity;
46
+ /**
47
+ * Thrown when a judged assertion is asked to run against a judge that is not
48
+ * the pinned one.
49
+ *
50
+ * A refusal rather than a warning or a downgraded score: a judged gate decided
51
+ * by an unknown judge looks exactly like a judged gate decided by the right
52
+ * one, so the only safe outcome is to produce no verdict at all.
53
+ */
54
+ export declare class JudgeNotPinnedError extends Error {
55
+ /** The alias the caller tried to judge through. */
56
+ readonly alias: string;
57
+ /**
58
+ * @param alias The alias the caller tried to judge through.
59
+ * @param detail What specifically failed the pin check.
60
+ */
61
+ constructor(alias: string, detail: string);
62
+ }
63
+ /** The judge identity a route table actually resolves to. */
64
+ export interface ResolvedJudge {
65
+ /** The alias that was verified. */
66
+ readonly alias: LlmAlias;
67
+ /** The provider the table names for it. */
68
+ readonly provider: string;
69
+ /** The model id the table names for it. */
70
+ readonly modelId: string;
71
+ }
72
+ /**
73
+ * Verify that an alias is the pinned judge, or refuse.
74
+ *
75
+ * @param alias The alias a judged assertion would be served by.
76
+ * @param table The route table to verify against; defaults to the canonical one.
77
+ * @returns The resolved judge identity when every pin condition holds.
78
+ * @throws {JudgeNotPinnedError} When any pin condition fails.
79
+ */
80
+ export declare function assertPinnedJudge(alias: LlmAlias, table?: LlmRouteTable): ResolvedJudge;
81
+ /** One case put to the judge. */
82
+ export interface JudgeRequest {
83
+ /** The grading rubric, stated on the golden set rather than improvised per call. */
84
+ readonly rubric: string;
85
+ /** The original input the answer responds to. */
86
+ readonly input: string;
87
+ /** The reference answer. */
88
+ readonly reference: string;
89
+ /** The answer being graded. */
90
+ readonly answer: string;
91
+ }
92
+ /**
93
+ * Score one answer with the pinned judge.
94
+ *
95
+ * The pin is verified before the call, not after, so a mis-pinned judge costs
96
+ * nothing and produces nothing rather than producing a score that would have to
97
+ * be retracted.
98
+ *
99
+ * @param request The case to grade.
100
+ * @param options Overrides used by the harness's own pinning check.
101
+ * @param options.alias The alias to judge through; only the pinned one is accepted.
102
+ * @param options.table The route table to verify the pin against.
103
+ * @returns The score, on the same 0-100 point scale the comparators use.
104
+ * @throws {JudgeNotPinnedError} When the judge is not the pinned one.
105
+ */
106
+ export declare function scoreWithPinnedJudge(request: JudgeRequest, options?: {
107
+ alias?: LlmAlias;
108
+ table?: LlmRouteTable;
109
+ }): Promise<number>;
110
+ /**
111
+ * Fill in any missing judge scores on a candidate run, using the pinned judge.
112
+ *
113
+ * The pin is verified once before the first call rather than per case, so a
114
+ * mis-pinned judge cannot grade part of a set before it is caught — a half-
115
+ * graded set is worse than an ungraded one, because its mean looks like a
116
+ * measurement.
117
+ *
118
+ * @param set The golden set being graded, which supplies the reference answers.
119
+ * @param candidate The candidate run whose answers need scoring.
120
+ * @param rubric The grading rubric, stated on the set rather than improvised.
121
+ * @param options Overrides used by the harness's own pinning check.
122
+ * @param options.alias The alias to judge through; only the pinned one is accepted.
123
+ * @param options.table The route table to verify the pin against.
124
+ * @returns A candidate run with a judge score on every case.
125
+ * @throws {JudgeNotPinnedError} When the judge is not the pinned one.
126
+ */
127
+ export declare function scoreCandidateRun(set: GoldenSet, candidate: CandidateRun, rubric: string, options?: {
128
+ alias?: LlmAlias;
129
+ table?: LlmRouteTable;
130
+ }): Promise<CandidateRun>;
131
+ //# sourceMappingURL=judge.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"judge.d.ts","sourceRoot":"","sources":["../../../../src/llm/eval/judge.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;GAkBG;AAIH,OAAO,KAAK,EACV,QAAQ,EACR,aAAa,EAEd,MAAM,UAAU,CAAC;AAClB,OAAO,KAAK,EAAE,YAAY,EAAE,SAAS,EAAkB,MAAM,SAAS,CAAC;AAEvE,gDAAgD;AAChD,eAAO,MAAM,eAAe,IAAI,CAAC;AAEjC,+FAA+F;AAC/F,eAAO,MAAM,eAAe,MAAM,CAAC;AAEnC,8EAA8E;AAC9E,MAAM,WAAW,mBAAmB;IAClC,0DAA0D;IAC1D,QAAQ,CAAC,KAAK,EAAE,QAAQ,CAAC;IACzB,8CAA8C;IAC9C,QAAQ,CAAC,QAAQ,EAAE,MAAM,CAAC;IAC1B,oDAAoD;IACpD,QAAQ,CAAC,OAAO,EAAE,MAAM,CAAC;CAC1B;AAED;;;;;;;;;GASG;AACH,eAAO,MAAM,YAAY,EAAE,mBAI1B,CAAC;AAEF;;;;;;;GAOG;AACH,qBAAa,mBAAoB,SAAQ,KAAK;IAC5C,mDAAmD;IACnD,SAAgB,KAAK,EAAE,MAAM,CAAC;IAE9B;;;OAGG;gBACgB,KAAK,EAAE,MAAM,EAAE,MAAM,EAAE,MAAM;CASjD;AAED,6DAA6D;AAC7D,MAAM,WAAW,aAAa;IAC5B,mCAAmC;IACnC,QAAQ,CAAC,KAAK,EAAE,QAAQ,CAAC;IACzB,2CAA2C;IAC3C,QAAQ,CAAC,QAAQ,EAAE,MAAM,CAAC;IAC1B,2CAA2C;IAC3C,QAAQ,CAAC,OAAO,EAAE,MAAM,CAAC;CAC1B;AAED;;;;;;;GAOG;AACH,wBAAgB,iBAAiB,CAC/B,KAAK,EAAE,QAAQ,EACf,KAAK,GAAE,aAA0B,GAChC,aAAa,CAiDf;AAED,iCAAiC;AACjC,MAAM,WAAW,YAAY;IAC3B,oFAAoF;IACpF,QAAQ,CAAC,MAAM,EAAE,MAAM,CAAC;IACxB,iDAAiD;IACjD,QAAQ,CAAC,KAAK,EAAE,MAAM,CAAC;IACvB,4BAA4B;IAC5B,QAAQ,CAAC,SAAS,EAAE,MAAM,CAAC;IAC3B,+BAA+B;IAC/B,QAAQ,CAAC,MAAM,EAAE,MAAM,CAAC;CACzB;AAyBD;;;;;;;;;;;;;GAaG;AACH,wBAAsB,oBAAoB,CACxC,OAAO,EAAE,YAAY,EACrB,OAAO,GAAE;IAAE,KAAK,CAAC,EAAE,QAAQ,CAAC;IAAC,KAAK,CAAC,EAAE,aAAa,CAAA;CAAO,GACxD,OAAO,CAAC,MAAM,CAAC,CAqBjB;AAED;;;;;;;;;;;;;;;;GAgBG;AACH,wBAAsB,iBAAiB,CACrC,GAAG,EAAE,SAAS,EACd,SAAS,EAAE,YAAY,EACvB,MAAM,EAAE,MAAM,EACd,OAAO,GAAE;IAAE,KAAK,CAAC,EAAE,QAAQ,CAAC;IAAC,KAAK,CAAC,EAAE,aAAa,CAAA;CAAO,GACxD,OAAO,CAAC,YAAY,CAAC,CA0BvB"}
@@ -0,0 +1,51 @@
1
+ /**
2
+ * Derivation of eval metrics from recorded model answers.
3
+ *
4
+ * Both sides of every comparison are measured HERE, by the same code, from
5
+ * records of the same shape. That symmetry is the point: an incumbent baseline
6
+ * computed by one routine and a candidate score computed by another would
7
+ * differ by the routines as much as by the models, and the gate would be
8
+ * measuring its own implementation.
9
+ *
10
+ * A metric that cannot be derived from the records present is returned as
11
+ * `undefined` rather than as a zero or a default. Zero is a measurement; absence
12
+ * is not, and the comparators depend on being able to tell them apart in order
13
+ * to render `indeterminate` instead of a verdict they have not earned.
14
+ *
15
+ * @module llm/eval/metrics
16
+ */
17
+ import type { BaselineMetrics, GoldenCase, GoldenSet, RecordedResult, ToolExpectation } from "./types";
18
+ /**
19
+ * Structured retries a tool-call site may spend before the answer stops
20
+ * counting as a valid call.
21
+ *
22
+ * Section 6 permits exactly one: the retry exists to recover a malformed call
23
+ * from a model that otherwise understood the request, and a site that needs
24
+ * more than one has not produced a valid call — it has been coached into one,
25
+ * which is a different capability from the one being measured.
26
+ */
27
+ export declare const MAX_STRUCTURED_RETRIES = 1;
28
+ /** One graded case paired with the answer being scored. */
29
+ export interface MetricSample {
30
+ /** The golden case. */
31
+ readonly goldenCase: GoldenCase;
32
+ /** The answer under measurement — incumbent or candidate. */
33
+ readonly result: RecordedResult;
34
+ }
35
+ /**
36
+ * Whether one recorded answer is a valid tool call under the site's contract.
37
+ *
38
+ * @param result The recorded answer.
39
+ * @param expectation The tool contract the site requires.
40
+ * @returns Whether the answer is a valid call within the permitted retry budget.
41
+ */
42
+ export declare function isValidToolCall(result: RecordedResult, expectation: ToolExpectation): boolean;
43
+ /**
44
+ * Derive every metric a set's assertions need, from one side's answers.
45
+ *
46
+ * @param set The golden set, which fixes what is measured and how.
47
+ * @param samples Cases paired with the answers being scored.
48
+ * @returns The derived metrics; a metric the records cannot support is absent.
49
+ */
50
+ export declare function deriveMetrics(set: GoldenSet, samples: readonly MetricSample[]): BaselineMetrics;
51
+ //# sourceMappingURL=metrics.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"metrics.d.ts","sourceRoot":"","sources":["../../../../src/llm/eval/metrics.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;GAeG;AAGH,OAAO,KAAK,EACV,eAAe,EACf,UAAU,EACV,SAAS,EACT,cAAc,EACd,eAAe,EAChB,MAAM,SAAS,CAAC;AAEjB;;;;;;;;GAQG;AACH,eAAO,MAAM,sBAAsB,IAAI,CAAC;AAExC,2DAA2D;AAC3D,MAAM,WAAW,YAAY;IAC3B,uBAAuB;IACvB,QAAQ,CAAC,UAAU,EAAE,UAAU,CAAC;IAChC,6DAA6D;IAC7D,QAAQ,CAAC,MAAM,EAAE,cAAc,CAAC;CACjC;AAED;;;;;;GAMG;AACH,wBAAgB,eAAe,CAC7B,MAAM,EAAE,cAAc,EACtB,WAAW,EAAE,eAAe,GAC3B,OAAO,CAwBT;AAuBD;;;;;;GAMG;AACH,wBAAgB,aAAa,CAC3B,GAAG,EAAE,SAAS,EACd,OAAO,EAAE,SAAS,YAAY,EAAE,GAC/B,eAAe,CAmDjB"}
@@ -0,0 +1,97 @@
1
+ /**
2
+ * Grading one candidate run against one golden set, and a whole evaluation
3
+ * against many.
4
+ *
5
+ * The orchestration here holds two invariants that the comparators cannot hold
6
+ * on their own.
7
+ *
8
+ * The bar is the ATTESTED baseline, not a number recomputed at grading time —
9
+ * so the bar a candidate clears is the same bar every earlier candidate cleared,
10
+ * and a re-baselining is a visible edit to the set rather than a silent
11
+ * consequence of running the harness again. The recomputation still happens,
12
+ * as an integrity check: when the attested baseline and the records it claims
13
+ * to summarise disagree, the set renders `indeterminate` instead of grading
14
+ * against a bar that has been moved by hand.
15
+ *
16
+ * And an incomplete candidate run is never graded. Scoring the subset of cases
17
+ * that happened to produce an answer silently reweights the set toward whatever
18
+ * the candidate found easy, which is the most flattering possible error a model
19
+ * comparison can make.
20
+ *
21
+ * @module llm/eval/run
22
+ */
23
+ import type { LlmAlias, LlmRouteTable } from "../types";
24
+ import type { CandidateRun, GoldenSet, RunReport, SetReport, Verdict } from "./types";
25
+ /**
26
+ * Absolute agreement required between an attested baseline metric and the same
27
+ * metric recomputed from the set's own records.
28
+ *
29
+ * Tight enough that only floating-point representation error fits inside it: the
30
+ * check exists to catch a baseline edited independently of its evidence, and a
31
+ * generous tolerance would let exactly that through.
32
+ */
33
+ export declare const BASELINE_ATTESTATION_TOLERANCE = 0.000001;
34
+ /**
35
+ * Overrides for grading, used only by the harness's own pinning check.
36
+ *
37
+ * Production grading passes none of these: the judge is the pinned one and the
38
+ * route table is the canonical one, and an override that a caller could reach in
39
+ * normal use would be a way around PD-6 rather than a way to verify it.
40
+ */
41
+ export interface EvaluateOptions {
42
+ /** The alias judged assertions are served by. Only the pinned judge is accepted. */
43
+ readonly judgeAlias?: LlmAlias;
44
+ /** The route table the pin is verified against. */
45
+ readonly routeTable?: LlmRouteTable;
46
+ }
47
+ /** One golden set paired with the candidate answers to grade against it. */
48
+ export interface EvalPair {
49
+ /** The golden set. */
50
+ readonly set: GoldenSet;
51
+ /** The candidate run answering it. */
52
+ readonly candidate: CandidateRun;
53
+ }
54
+ /**
55
+ * Grade one candidate run against one golden set.
56
+ *
57
+ * A set that asserts `judge` verifies the judge pin BEFORE it grades, even
58
+ * though the scores it grades were recorded earlier. The scores are only
59
+ * meaningful as the output of a specific, unchanged instrument, so a set whose
60
+ * judge has moved since capture has no valid scores to grade — and refusing is
61
+ * the only outcome that says so.
62
+ *
63
+ * @param set The golden set.
64
+ * @param candidate The candidate's answers.
65
+ * @param options Overrides used by the harness's own pinning check.
66
+ * @returns The set's report.
67
+ * @throws {GoldenSetError} When the candidate run answers a different set.
68
+ * @throws {JudgeNotPinnedError} When a judged set's judge is not the pinned one.
69
+ */
70
+ export declare function evaluateSet(set: GoldenSet, candidate: CandidateRun, options?: EvaluateOptions): SetReport;
71
+ /**
72
+ * Grade a whole evaluation.
73
+ *
74
+ * An evaluation with no sets is FAILED, not PASSED. A gate that reports green
75
+ * when it graded nothing is indistinguishable from a gate that is working, and
76
+ * is the exact failure this harness exists to prevent.
77
+ *
78
+ * @param pairs The sets and the candidate runs answering them.
79
+ * @param options Overrides used by the harness's own pinning check.
80
+ * @returns The run report.
81
+ */
82
+ export declare function evaluateRun(pairs: readonly EvalPair[], options?: EvaluateOptions): RunReport;
83
+ /**
84
+ * Render one verdict as a single log line.
85
+ *
86
+ * @param verdict The verdict.
87
+ * @returns A line naming the assertion, the outcome and the numbers behind it.
88
+ */
89
+ export declare function formatVerdict(verdict: Verdict): string;
90
+ /**
91
+ * Render a run report as lines for a CI log.
92
+ *
93
+ * @param runReport The report.
94
+ * @returns The lines, in display order.
95
+ */
96
+ export declare function formatRunReport(runReport: RunReport): string[];
97
+ //# sourceMappingURL=run.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"run.d.ts","sourceRoot":"","sources":["../../../../src/llm/eval/run.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;;;GAqBG;AAEH,OAAO,KAAK,EAAE,QAAQ,EAAE,aAAa,EAAE,MAAM,UAAU,CAAC;AAMxD,OAAO,KAAK,EAEV,YAAY,EAGZ,SAAS,EACT,SAAS,EACT,SAAS,EACT,OAAO,EACR,MAAM,SAAS,CAAC;AAEjB;;;;;;;GAOG;AACH,eAAO,MAAM,8BAA8B,WAAO,CAAC;AAKnD;;;;;;GAMG;AACH,MAAM,WAAW,eAAe;IAC9B,oFAAoF;IACpF,QAAQ,CAAC,UAAU,CAAC,EAAE,QAAQ,CAAC;IAC/B,mDAAmD;IACnD,QAAQ,CAAC,UAAU,CAAC,EAAE,aAAa,CAAC;CACrC;AAED,4EAA4E;AAC5E,MAAM,WAAW,QAAQ;IACvB,sBAAsB;IACtB,QAAQ,CAAC,GAAG,EAAE,SAAS,CAAC;IACxB,sCAAsC;IACtC,QAAQ,CAAC,SAAS,EAAE,YAAY,CAAC;CAClC;AAmID;;;;;;;;;;;;;;;GAeG;AACH,wBAAgB,WAAW,CACzB,GAAG,EAAE,SAAS,EACd,SAAS,EAAE,YAAY,EACvB,OAAO,GAAE,eAAoB,GAC5B,SAAS,CAsDX;AAsBD;;;;;;;;;;GAUG;AACH,wBAAgB,WAAW,CACzB,KAAK,EAAE,SAAS,QAAQ,EAAE,EAC1B,OAAO,GAAE,eAAoB,GAC5B,SAAS,CAiBX;AAED;;;;;GAKG;AACH,wBAAgB,aAAa,CAAC,OAAO,EAAE,OAAO,GAAG,MAAM,CAQtD;AAED;;;;;GAKG;AACH,wBAAgB,eAAe,CAAC,SAAS,EAAE,SAAS,GAAG,MAAM,EAAE,CAU9D"}