@adaptic/utils 0.0.1014 → 0.0.1016
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/index.cjs +2823 -25
- package/dist/index.cjs.map +1 -1
- package/dist/index.mjs +2783 -25
- package/dist/index.mjs.map +1 -1
- package/dist/types/__tests__/llm/client/support/rejections.d.ts +21 -0
- package/dist/types/__tests__/llm/client/support/rejections.d.ts.map +1 -0
- package/dist/types/__tests__/llm/client/support/routes.d.ts +67 -0
- package/dist/types/__tests__/llm/client/support/routes.d.ts.map +1 -0
- package/dist/types/__tests__/llm/client/support/streams.d.ts +74 -0
- package/dist/types/__tests__/llm/client/support/streams.d.ts.map +1 -0
- package/dist/types/__tests__/llm/client/support/transports.d.ts +105 -0
- package/dist/types/__tests__/llm/client/support/transports.d.ts.map +1 -0
- package/dist/types/index.d.ts +1 -0
- package/dist/types/index.d.ts.map +1 -1
- package/dist/types/llm/alias-client.d.ts +64 -0
- package/dist/types/llm/alias-client.d.ts.map +1 -0
- package/dist/types/llm/circuit-breaker.d.ts +124 -0
- package/dist/types/llm/circuit-breaker.d.ts.map +1 -0
- package/dist/types/llm/eval/comparators.d.ts +127 -0
- package/dist/types/llm/eval/comparators.d.ts.map +1 -0
- package/dist/types/llm/eval/coverage.d.ts +44 -0
- package/dist/types/llm/eval/coverage.d.ts.map +1 -0
- package/dist/types/llm/eval/golden-set.d.ts +47 -0
- package/dist/types/llm/eval/golden-set.d.ts.map +1 -0
- package/dist/types/llm/eval/index.d.ts +25 -0
- package/dist/types/llm/eval/index.d.ts.map +1 -0
- package/dist/types/llm/eval/json-shape.d.ts +74 -0
- package/dist/types/llm/eval/json-shape.d.ts.map +1 -0
- package/dist/types/llm/eval/judge.d.ts +131 -0
- package/dist/types/llm/eval/judge.d.ts.map +1 -0
- package/dist/types/llm/eval/metrics.d.ts +51 -0
- package/dist/types/llm/eval/metrics.d.ts.map +1 -0
- package/dist/types/llm/eval/run.d.ts +97 -0
- package/dist/types/llm/eval/run.d.ts.map +1 -0
- package/dist/types/llm/eval/types.d.ts +242 -0
- package/dist/types/llm/eval/types.d.ts.map +1 -0
- package/dist/types/llm/fallback-chain.d.ts +97 -0
- package/dist/types/llm/fallback-chain.d.ts.map +1 -0
- package/dist/types/llm/index.d.ts +30 -0
- package/dist/types/llm/index.d.ts.map +1 -0
- package/dist/types/llm/param-matrix.d.ts +65 -0
- package/dist/types/llm/param-matrix.d.ts.map +1 -0
- package/dist/types/llm/rate-guard.d.ts +119 -0
- package/dist/types/llm/rate-guard.d.ts.map +1 -0
- package/dist/types/llm/route-table.d.ts +155 -0
- package/dist/types/llm/route-table.d.ts.map +1 -0
- package/dist/types/llm/schema-retry.d.ts +80 -0
- package/dist/types/llm/schema-retry.d.ts.map +1 -0
- package/dist/types/llm/streaming.d.ts +93 -0
- package/dist/types/llm/streaming.d.ts.map +1 -0
- package/dist/types/llm/transports/direct.d.ts +92 -0
- package/dist/types/llm/transports/direct.d.ts.map +1 -0
- package/dist/types/llm/transports/gateway.d.ts +73 -0
- package/dist/types/llm/transports/gateway.d.ts.map +1 -0
- package/dist/types/llm/types.d.ts +292 -0
- package/dist/types/llm/types.d.ts.map +1 -0
- package/dist/types/schemas/alpaca-schemas.d.ts +6 -6
- package/package.json +2 -2
- package/dist/types/__tests__/alpaca-broker-error-preservation.test.d.ts +0 -2
- package/dist/types/__tests__/alpaca-broker-error-preservation.test.d.ts.map +0 -1
- package/dist/types/__tests__/alpaca-client-order-id.test.d.ts +0 -2
- package/dist/types/__tests__/alpaca-client-order-id.test.d.ts.map +0 -1
- package/dist/types/__tests__/alpaca-functions.test.d.ts +0 -2
- package/dist/types/__tests__/alpaca-functions.test.d.ts.map +0 -1
- package/dist/types/__tests__/alpaca-market-data-retry.test.d.ts +0 -2
- package/dist/types/__tests__/alpaca-market-data-retry.test.d.ts.map +0 -1
- package/dist/types/__tests__/alpaca-order-idempotency.test.d.ts +0 -2
- package/dist/types/__tests__/alpaca-order-idempotency.test.d.ts.map +0 -1
- package/dist/types/__tests__/alpaca-trading-api.test.d.ts +0 -2
- package/dist/types/__tests__/alpaca-trading-api.test.d.ts.map +0 -1
- package/dist/types/__tests__/api-endpoints.test.d.ts +0 -2
- package/dist/types/__tests__/api-endpoints.test.d.ts.map +0 -1
- package/dist/types/__tests__/asset-allocation.test.d.ts +0 -2
- package/dist/types/__tests__/asset-allocation.test.d.ts.map +0 -1
- package/dist/types/__tests__/atr.test.d.ts +0 -2
- package/dist/types/__tests__/atr.test.d.ts.map +0 -1
- package/dist/types/__tests__/auth-validator.test.d.ts +0 -2
- package/dist/types/__tests__/auth-validator.test.d.ts.map +0 -1
- package/dist/types/__tests__/broker-factory.test.d.ts +0 -2
- package/dist/types/__tests__/broker-factory.test.d.ts.map +0 -1
- package/dist/types/__tests__/broker-types.test.d.ts +0 -2
- package/dist/types/__tests__/broker-types.test.d.ts.map +0 -1
- package/dist/types/__tests__/cache.test.d.ts +0 -2
- package/dist/types/__tests__/cache.test.d.ts.map +0 -1
- package/dist/types/__tests__/errors.test.d.ts +0 -2
- package/dist/types/__tests__/errors.test.d.ts.map +0 -1
- package/dist/types/__tests__/financial-regression.test.d.ts +0 -2
- package/dist/types/__tests__/financial-regression.test.d.ts.map +0 -1
- package/dist/types/__tests__/format-tools.test.d.ts +0 -2
- package/dist/types/__tests__/format-tools.test.d.ts.map +0 -1
- package/dist/types/__tests__/http-keep-alive.test.d.ts +0 -2
- package/dist/types/__tests__/http-keep-alive.test.d.ts.map +0 -1
- package/dist/types/__tests__/http-timeout.test.d.ts +0 -2
- package/dist/types/__tests__/http-timeout.test.d.ts.map +0 -1
- package/dist/types/__tests__/index.test.d.ts +0 -2
- package/dist/types/__tests__/index.test.d.ts.map +0 -1
- package/dist/types/__tests__/legacy-auth.test.d.ts +0 -2
- package/dist/types/__tests__/legacy-auth.test.d.ts.map +0 -1
- package/dist/types/__tests__/logger.test.d.ts +0 -2
- package/dist/types/__tests__/logger.test.d.ts.map +0 -1
- package/dist/types/__tests__/logging.test.d.ts +0 -2
- package/dist/types/__tests__/logging.test.d.ts.map +0 -1
- package/dist/types/__tests__/market-time.test.d.ts +0 -2
- package/dist/types/__tests__/market-time.test.d.ts.map +0 -1
- package/dist/types/__tests__/massive.test.d.ts +0 -2
- package/dist/types/__tests__/massive.test.d.ts.map +0 -1
- package/dist/types/__tests__/metrics-calcs-direction.test.d.ts +0 -2
- package/dist/types/__tests__/metrics-calcs-direction.test.d.ts.map +0 -1
- package/dist/types/__tests__/misc-utils.test.d.ts +0 -2
- package/dist/types/__tests__/misc-utils.test.d.ts.map +0 -1
- package/dist/types/__tests__/paginator.test.d.ts +0 -2
- package/dist/types/__tests__/paginator.test.d.ts.map +0 -1
- package/dist/types/__tests__/performance-metrics-fees.test.d.ts +0 -2
- package/dist/types/__tests__/performance-metrics-fees.test.d.ts.map +0 -1
- package/dist/types/__tests__/performance-metrics.test.d.ts +0 -2
- package/dist/types/__tests__/performance-metrics.test.d.ts.map +0 -1
- package/dist/types/__tests__/price-utils-fees.test.d.ts +0 -2
- package/dist/types/__tests__/price-utils-fees.test.d.ts.map +0 -1
- package/dist/types/__tests__/price-utils.test.d.ts +0 -2
- package/dist/types/__tests__/price-utils.test.d.ts.map +0 -1
- package/dist/types/__tests__/property-based-financial.test.d.ts +0 -2
- package/dist/types/__tests__/property-based-financial.test.d.ts.map +0 -1
- package/dist/types/__tests__/protective-order-sides.test.d.ts +0 -2
- package/dist/types/__tests__/protective-order-sides.test.d.ts.map +0 -1
- package/dist/types/__tests__/rate-limiter.test.d.ts +0 -2
- package/dist/types/__tests__/rate-limiter.test.d.ts.map +0 -1
- package/dist/types/__tests__/retry-classification.test.d.ts +0 -2
- package/dist/types/__tests__/retry-classification.test.d.ts.map +0 -1
- package/dist/types/__tests__/retry.test.d.ts +0 -2
- package/dist/types/__tests__/retry.test.d.ts.map +0 -1
- package/dist/types/__tests__/risk-free-rate.test.d.ts +0 -2
- package/dist/types/__tests__/risk-free-rate.test.d.ts.map +0 -1
- package/dist/types/__tests__/risk-metrics.test.d.ts +0 -2
- package/dist/types/__tests__/risk-metrics.test.d.ts.map +0 -1
- package/dist/types/__tests__/schema-validation.test.d.ts +0 -2
- package/dist/types/__tests__/schema-validation.test.d.ts.map +0 -1
- package/dist/types/__tests__/stampede-load-timeout.test.d.ts +0 -2
- package/dist/types/__tests__/stampede-load-timeout.test.d.ts.map +0 -1
- package/dist/types/__tests__/strategy-metrics.test.d.ts +0 -2
- package/dist/types/__tests__/strategy-metrics.test.d.ts.map +0 -1
- package/dist/types/__tests__/technical-analysis-totality.test.d.ts +0 -2
- package/dist/types/__tests__/technical-analysis-totality.test.d.ts.map +0 -1
- package/dist/types/__tests__/technical-analysis.test.d.ts +0 -2
- package/dist/types/__tests__/technical-analysis.test.d.ts.map +0 -1
- package/dist/types/__tests__/time-utils.test.d.ts +0 -2
- package/dist/types/__tests__/time-utils.test.d.ts.map +0 -1
- package/dist/types/__tests__/trading-policy-schemas.test.d.ts +0 -2
- package/dist/types/__tests__/trading-policy-schemas.test.d.ts.map +0 -1
- package/dist/types/__tests__/trailing-stops-portfolio.test.d.ts +0 -2
- package/dist/types/__tests__/trailing-stops-portfolio.test.d.ts.map +0 -1
- package/dist/types/__tests__/volatility.test.d.ts +0 -2
- package/dist/types/__tests__/volatility.test.d.ts.map +0 -1
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"comparators.d.ts","sourceRoot":"","sources":["../../../../src/llm/eval/comparators.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;;GAoBG;AAEH,OAAO,KAAK,EAAE,aAAa,EAAE,QAAQ,EAAE,OAAO,EAAE,MAAM,SAAS,CAAC;AAEhE;;;;;;GAMG;AACH,eAAO,MAAM,sBAAsB,IAAI,CAAC;AAExC;;;;;;;GAOG;AACH,eAAO,MAAM,wBAAwB,OAAO,CAAC;AAE7C,8FAA8F;AAC9F,eAAO,MAAM,cAAc,MAAM,CAAC;AAiBlC,qDAAqD;AACrD,MAAM,WAAW,eAAe;IAC9B,oFAAoF;IACpF,QAAQ,CAAC,SAAS,EAAE,MAAM,GAAG,SAAS,CAAC;IACvC,wFAAwF;IACxF,QAAQ,CAAC,SAAS,EAAE,MAAM,GAAG,SAAS,CAAC;IACvC,4CAA4C;IAC5C,QAAQ,CAAC,CAAC,EAAE,MAAM,CAAC;IACnB,6CAA6C;IAC7C,QAAQ,CAAC,IAAI,EAAE,MAAM,CAAC;CACvB;AAED,yDAAyD;AACzD,MAAM,MAAM,UAAU,GAAG,CAAC,KAAK,EAAE,eAAe,KAAK,OAAO,CAAC;AAkF7D;;;;;;;;GAQG;AACH,eAAO,MAAM,sBAAsB,EAAE,UACsB,CAAC;AAE5D;;;;;;;;;GASG;AACH,eAAO,MAAM,oBAAoB,EAAE,UACmB,CAAC;AAEvD;;;;;;;;;GASG;AACH,eAAO,MAAM,iBAAiB,EAAE,UA8B/B,CAAC;AAEF;;;;;;;;;GASG;AACH,eAAO,MAAM,iBAAiB,EAAE,UA8B/B,CAAC;AAEF;;;;;;;;;;GAUG;AACH,eAAO,MAAM,iBAAiB,EAAE,UA2B/B,CAAC;AAEF;;;;;;GAMG;AACH,eAAO,MAAM,WAAW,EAAE,QAAQ,CAAC,MAAM,CAAC,aAAa,EAAE,UAAU,CAAC,CAMnE,CAAC;AAEF;;;;;;;GAOG;AACH,eAAO,MAAM,oBAAoB,EAAE,QAAQ,CAAC,MAAM,CAAC,QAAQ,EAAE,SAAS,aAAa,EAAE,CAAC,CAKrF,CAAC"}
|
|
@@ -0,0 +1,44 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Coverage of the harness itself.
|
|
3
|
+
*
|
|
4
|
+
* A gate is only worth its CI minutes if every assertion it claims to enforce
|
|
5
|
+
* has been demonstrated to fire in both directions. "Implemented" is not the
|
|
6
|
+
* bar — a comparator that always returns `pass` is implemented. The bar is
|
|
7
|
+
* EXECUTED EVIDENCE: for each assertion, a fixture that this run graded as
|
|
8
|
+
* passing and a fixture that this run graded as failing.
|
|
9
|
+
*
|
|
10
|
+
* The route table is the other half of the question. An alias whose eval gate
|
|
11
|
+
* maps to no assertion bundle is an alias nothing grades, and it would sail
|
|
12
|
+
* through a migration unmeasured, so the coverage check reads the table rather
|
|
13
|
+
* than a list someone maintained by hand.
|
|
14
|
+
*
|
|
15
|
+
* @module llm/eval/coverage
|
|
16
|
+
*/
|
|
17
|
+
import type { LlmRouteTable } from "../types";
|
|
18
|
+
import type { EvalAssertion, EvalStatus } from "./types";
|
|
19
|
+
/** Assertions a run actually demonstrated, in each direction. */
|
|
20
|
+
export interface ObservedAssertions {
|
|
21
|
+
/** Assertions this run graded as passing on at least one fixture. */
|
|
22
|
+
readonly green: readonly EvalAssertion[];
|
|
23
|
+
/** Assertions this run graded as failing on at least one fixture. */
|
|
24
|
+
readonly red: readonly EvalAssertion[];
|
|
25
|
+
}
|
|
26
|
+
/** The outcome of the coverage check. */
|
|
27
|
+
export interface CoverageReport {
|
|
28
|
+
/** Assertions required by at least one eval gate in the route table. */
|
|
29
|
+
readonly required: readonly EvalAssertion[];
|
|
30
|
+
/** Everything missing; empty is the only passing state. */
|
|
31
|
+
readonly problems: readonly string[];
|
|
32
|
+
/** PASSED only when nothing is missing. */
|
|
33
|
+
readonly status: EvalStatus;
|
|
34
|
+
}
|
|
35
|
+
/**
|
|
36
|
+
* Check that every assertion the route table requires is implemented and was
|
|
37
|
+
* demonstrated in both directions by this run.
|
|
38
|
+
*
|
|
39
|
+
* @param observed Assertions this run graded as passing and as failing.
|
|
40
|
+
* @param table The route table to read; defaults to the canonical one.
|
|
41
|
+
* @returns The coverage report.
|
|
42
|
+
*/
|
|
43
|
+
export declare function computeCoverage(observed: ObservedAssertions, table?: LlmRouteTable): CoverageReport;
|
|
44
|
+
//# sourceMappingURL=coverage.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"coverage.d.ts","sourceRoot":"","sources":["../../../../src/llm/eval/coverage.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;GAeG;AAGH,OAAO,KAAK,EAAE,aAAa,EAAE,MAAM,UAAU,CAAC;AAG9C,OAAO,KAAK,EAAE,aAAa,EAAY,UAAU,EAAE,MAAM,SAAS,CAAC;AAEnE,iEAAiE;AACjE,MAAM,WAAW,kBAAkB;IACjC,qEAAqE;IACrE,QAAQ,CAAC,KAAK,EAAE,SAAS,aAAa,EAAE,CAAC;IACzC,qEAAqE;IACrE,QAAQ,CAAC,GAAG,EAAE,SAAS,aAAa,EAAE,CAAC;CACxC;AAED,yCAAyC;AACzC,MAAM,WAAW,cAAc;IAC7B,wEAAwE;IACxE,QAAQ,CAAC,QAAQ,EAAE,SAAS,aAAa,EAAE,CAAC;IAC5C,2DAA2D;IAC3D,QAAQ,CAAC,QAAQ,EAAE,SAAS,MAAM,EAAE,CAAC;IACrC,2CAA2C;IAC3C,QAAQ,CAAC,MAAM,EAAE,UAAU,CAAC;CAC7B;AA4BD;;;;;;;GAOG;AACH,wBAAgB,eAAe,CAC7B,QAAQ,EAAE,kBAAkB,EAC5B,KAAK,GAAE,aAA0B,GAChC,cAAc,CAwChB"}
|
|
@@ -0,0 +1,47 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Parsing and structural validation of golden sets and candidate runs.
|
|
3
|
+
*
|
|
4
|
+
* A golden set is evidence, and evidence that is not validated on the way in is
|
|
5
|
+
* evidence nobody can rely on later. Every field is checked here, at the edge,
|
|
6
|
+
* so that by the time a comparator sees a number the only remaining question is
|
|
7
|
+
* the one the comparator exists to answer.
|
|
8
|
+
*
|
|
9
|
+
* Validation is strict in one particular direction: a set that declares an
|
|
10
|
+
* assertion must carry everything that assertion needs. A set asserting
|
|
11
|
+
* `schema-valid` without a response shape, or `tool-call` without a tool
|
|
12
|
+
* contract, is rejected rather than quietly graded on the assertions it does
|
|
13
|
+
* support — a gate that drops the check it cannot perform is a gate that gets
|
|
14
|
+
* weaker exactly where the evidence is thinnest.
|
|
15
|
+
*
|
|
16
|
+
* @module llm/eval/golden-set
|
|
17
|
+
*/
|
|
18
|
+
import type { CandidateRun, GoldenSet } from "./types";
|
|
19
|
+
/** Thrown when a golden set or candidate run is not well formed. */
|
|
20
|
+
export declare class GoldenSetError extends Error {
|
|
21
|
+
/** Where the malformed document came from, so the message is actionable. */
|
|
22
|
+
readonly origin: string;
|
|
23
|
+
/**
|
|
24
|
+
* @param origin Where the document came from.
|
|
25
|
+
* @param detail What is wrong with it.
|
|
26
|
+
*/
|
|
27
|
+
constructor(origin: string, detail: string);
|
|
28
|
+
}
|
|
29
|
+
/**
|
|
30
|
+
* Parse and validate a golden set.
|
|
31
|
+
*
|
|
32
|
+
* @param raw The parsed JSON document.
|
|
33
|
+
* @param origin Where it came from, quoted in any error.
|
|
34
|
+
* @returns The validated golden set.
|
|
35
|
+
* @throws {GoldenSetError} When the document is not a valid golden set.
|
|
36
|
+
*/
|
|
37
|
+
export declare function parseGoldenSet(raw: unknown, origin: string): GoldenSet;
|
|
38
|
+
/**
|
|
39
|
+
* Parse and validate a candidate run.
|
|
40
|
+
*
|
|
41
|
+
* @param raw The parsed JSON document.
|
|
42
|
+
* @param origin Where it came from, quoted in any error.
|
|
43
|
+
* @returns The validated candidate run.
|
|
44
|
+
* @throws {GoldenSetError} When the document is not a valid candidate run.
|
|
45
|
+
*/
|
|
46
|
+
export declare function parseCandidateRun(raw: unknown, origin: string): CandidateRun;
|
|
47
|
+
//# sourceMappingURL=golden-set.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"golden-set.d.ts","sourceRoot":"","sources":["../../../../src/llm/eval/golden-set.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;GAgBG;AAKH,OAAO,KAAK,EACV,YAAY,EAIZ,SAAS,EAOV,MAAM,SAAS,CAAC;AAEjB,oEAAoE;AACpE,qBAAa,cAAe,SAAQ,KAAK;IACvC,4EAA4E;IAC5E,SAAgB,MAAM,EAAE,MAAM,CAAC;IAE/B;;;OAGG;gBACgB,MAAM,EAAE,MAAM,EAAE,MAAM,EAAE,MAAM;CAKlD;AA8QD;;;;;;;GAOG;AACH,wBAAgB,cAAc,CAAC,GAAG,EAAE,OAAO,EAAE,MAAM,EAAE,MAAM,GAAG,SAAS,CAyJtE;AAED;;;;;;;GAOG;AACH,wBAAgB,iBAAiB,CAAC,GAAG,EAAE,OAAO,EAAE,MAAM,EAAE,MAAM,GAAG,YAAY,CAa5E"}
|
|
@@ -0,0 +1,25 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Public surface of the LLM migration eval harness.
|
|
3
|
+
*
|
|
4
|
+
* The harness grades a candidate model against a recorded incumbent baseline on
|
|
5
|
+
* captured golden sets, and renders every verdict by code from recorded data —
|
|
6
|
+
* PD-6 forbids agent judgement from standing in for a gate, and an exported
|
|
7
|
+
* surface with no "just tell me if this looks fine" entry point is what makes
|
|
8
|
+
* that structural rather than aspirational.
|
|
9
|
+
*
|
|
10
|
+
* @module llm/eval
|
|
11
|
+
*/
|
|
12
|
+
export { COMPARATORS, EVAL_GATE_ASSERTIONS, JUDGE_TOLERANCE_FRACTION, MATCH_TOLERANCE_POINTS, RATE_TO_POINTS, compareJudgeScore, compareLatencyP95, compareMatchScore, compareSchemaValidRate, compareValidCallRate, } from "./comparators";
|
|
13
|
+
export type { Comparator, ComparatorInput } from "./comparators";
|
|
14
|
+
export { computeCoverage } from "./coverage";
|
|
15
|
+
export type { CoverageReport, ObservedAssertions } from "./coverage";
|
|
16
|
+
export { GoldenSetError, parseCandidateRun, parseGoldenSet } from "./golden-set";
|
|
17
|
+
export { F1_SCALE_POINTS, fieldF1, jsonEquals, leafFields, p95, satisfiesShape } from "./json-shape";
|
|
18
|
+
export { JUDGE_SCORE_MAX, JUDGE_SCORE_MIN, JudgeNotPinnedError, PINNED_JUDGE, assertPinnedJudge, scoreCandidateRun, scoreWithPinnedJudge, } from "./judge";
|
|
19
|
+
export type { JudgeRequest, PinnedJudgeIdentity, ResolvedJudge } from "./judge";
|
|
20
|
+
export { MAX_STRUCTURED_RETRIES, deriveMetrics, isValidToolCall } from "./metrics";
|
|
21
|
+
export type { MetricSample } from "./metrics";
|
|
22
|
+
export { BASELINE_ATTESTATION_TOLERANCE, evaluateRun, evaluateSet, formatRunReport, formatVerdict, } from "./run";
|
|
23
|
+
export type { EvalPair, EvaluateOptions } from "./run";
|
|
24
|
+
export type { BaselineMetrics, CandidateRun, EvalAssertion, EvalGate, EvalStatus, GoldenCase, GoldenSet, IncumbentBaseline, JsonShape, MatchMetric, MetricUnit, RecordedResult, RecordedToolCall, RunReport, SetReport, ToolExpectation, Verdict, } from "./types";
|
|
25
|
+
//# sourceMappingURL=index.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"index.d.ts","sourceRoot":"","sources":["../../../../src/llm/eval/index.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;GAUG;AAEH,OAAO,EACL,WAAW,EACX,oBAAoB,EACpB,wBAAwB,EACxB,sBAAsB,EACtB,cAAc,EACd,iBAAiB,EACjB,iBAAiB,EACjB,iBAAiB,EACjB,sBAAsB,EACtB,oBAAoB,GACrB,MAAM,eAAe,CAAC;AACvB,YAAY,EAAE,UAAU,EAAE,eAAe,EAAE,MAAM,eAAe,CAAC;AAEjE,OAAO,EAAE,eAAe,EAAE,MAAM,YAAY,CAAC;AAC7C,YAAY,EAAE,cAAc,EAAE,kBAAkB,EAAE,MAAM,YAAY,CAAC;AAErE,OAAO,EAAE,cAAc,EAAE,iBAAiB,EAAE,cAAc,EAAE,MAAM,cAAc,CAAC;AAEjF,OAAO,EAAE,eAAe,EAAE,OAAO,EAAE,UAAU,EAAE,UAAU,EAAE,GAAG,EAAE,cAAc,EAAE,MAAM,cAAc,CAAC;AAErG,OAAO,EACL,eAAe,EACf,eAAe,EACf,mBAAmB,EACnB,YAAY,EACZ,iBAAiB,EACjB,iBAAiB,EACjB,oBAAoB,GACrB,MAAM,SAAS,CAAC;AACjB,YAAY,EAAE,YAAY,EAAE,mBAAmB,EAAE,aAAa,EAAE,MAAM,SAAS,CAAC;AAEhF,OAAO,EAAE,sBAAsB,EAAE,aAAa,EAAE,eAAe,EAAE,MAAM,WAAW,CAAC;AACnF,YAAY,EAAE,YAAY,EAAE,MAAM,WAAW,CAAC;AAE9C,OAAO,EACL,8BAA8B,EAC9B,WAAW,EACX,WAAW,EACX,eAAe,EACf,aAAa,GACd,MAAM,OAAO,CAAC;AACf,YAAY,EAAE,QAAQ,EAAE,eAAe,EAAE,MAAM,OAAO,CAAC;AAEvD,YAAY,EACV,eAAe,EACf,YAAY,EACZ,aAAa,EACb,QAAQ,EACR,UAAU,EACV,UAAU,EACV,SAAS,EACT,iBAAiB,EACjB,SAAS,EACT,WAAW,EACX,UAAU,EACV,cAAc,EACd,gBAAgB,EAChB,SAAS,EACT,SAAS,EACT,eAAe,EACf,OAAO,GACR,MAAM,SAAS,CAAC"}
|
|
@@ -0,0 +1,74 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Structural comparison primitives the comparators are built from.
|
|
3
|
+
*
|
|
4
|
+
* These are separated from the comparators so that "did this answer parse" and
|
|
5
|
+
* "was this answer right" stay distinct measurements. Conflating them hides the
|
|
6
|
+
* most useful diagnostic in a model swap: a candidate can be perfectly
|
|
7
|
+
* well-formed and consistently wrong, or consistently right and badly
|
|
8
|
+
* formatted, and those two have opposite remedies.
|
|
9
|
+
*
|
|
10
|
+
* The validator honours a documented subset of JSON Schema and adds no
|
|
11
|
+
* dependency. An unrecognised keyword is ignored rather than treated as
|
|
12
|
+
* satisfied, so the subset can only ever be stricter than a caller expects,
|
|
13
|
+
* never more permissive.
|
|
14
|
+
*
|
|
15
|
+
* @module llm/eval/json-shape
|
|
16
|
+
*/
|
|
17
|
+
import type { JsonShape } from "./types";
|
|
18
|
+
/** Points on the F1 scale, so a tolerance expressed "in points" has a fixed meaning. */
|
|
19
|
+
export declare const F1_SCALE_POINTS = 100;
|
|
20
|
+
/**
|
|
21
|
+
* Whether a value satisfies a shape.
|
|
22
|
+
*
|
|
23
|
+
* @param value The value to check.
|
|
24
|
+
* @param shape The shape it must satisfy.
|
|
25
|
+
* @returns Whether the value satisfies the shape.
|
|
26
|
+
*/
|
|
27
|
+
export declare function satisfiesShape(value: unknown, shape: JsonShape): boolean;
|
|
28
|
+
/**
|
|
29
|
+
* Structural equality over JSON values.
|
|
30
|
+
*
|
|
31
|
+
* Order-sensitive for arrays and order-insensitive for object keys, which is
|
|
32
|
+
* what JSON itself means: two objects with the same entries are the same
|
|
33
|
+
* answer, while a reordered list is a different one.
|
|
34
|
+
*
|
|
35
|
+
* @param left The first value.
|
|
36
|
+
* @param right The second value.
|
|
37
|
+
* @returns Whether the two are structurally equal.
|
|
38
|
+
*/
|
|
39
|
+
export declare function jsonEquals(left: unknown, right: unknown): boolean;
|
|
40
|
+
/**
|
|
41
|
+
* Flatten a JSON value into dotted leaf paths paired with their serialised values.
|
|
42
|
+
*
|
|
43
|
+
* Field-level F1 needs a set of comparable atoms, and a leaf path is the
|
|
44
|
+
* natural one for extraction output: it credits a candidate for the fields it
|
|
45
|
+
* got right instead of scoring the whole record all-or-nothing, which is the
|
|
46
|
+
* difference between a metric that can move by one point and one that can only
|
|
47
|
+
* move by whole cases.
|
|
48
|
+
*
|
|
49
|
+
* @param value The value to flatten.
|
|
50
|
+
* @param prefix Path prefix used by the recursion.
|
|
51
|
+
* @returns Leaf paths mapped to their serialised values.
|
|
52
|
+
*/
|
|
53
|
+
export declare function leafFields(value: unknown, prefix?: string): Map<string, string>;
|
|
54
|
+
/**
|
|
55
|
+
* Field-level F1 between a produced value and the reference, in points.
|
|
56
|
+
*
|
|
57
|
+
* @param produced What the model returned.
|
|
58
|
+
* @param expected The reference answer.
|
|
59
|
+
* @returns F1 on a 0-100 point scale; 0 when nothing matched.
|
|
60
|
+
*/
|
|
61
|
+
export declare function fieldF1(produced: unknown, expected: unknown): number;
|
|
62
|
+
/**
|
|
63
|
+
* The p95 of a sample by nearest-rank.
|
|
64
|
+
*
|
|
65
|
+
* Nearest-rank rather than an interpolating estimator because a latency gate
|
|
66
|
+
* must name a value the system actually produced; an interpolated p95 is a
|
|
67
|
+
* number no request ever took, and a gate is easier to trust when its threshold
|
|
68
|
+
* is an observation.
|
|
69
|
+
*
|
|
70
|
+
* @param values The sample.
|
|
71
|
+
* @returns The p95 value, or `null` for an empty sample.
|
|
72
|
+
*/
|
|
73
|
+
export declare function p95(values: readonly number[]): number | null;
|
|
74
|
+
//# sourceMappingURL=json-shape.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"json-shape.d.ts","sourceRoot":"","sources":["../../../../src/llm/eval/json-shape.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;GAeG;AAEH,OAAO,KAAK,EAAE,SAAS,EAAE,MAAM,SAAS,CAAC;AAEzC,wFAAwF;AACxF,eAAO,MAAM,eAAe,MAAM,CAAC;AAEnC;;;;;;GAMG;AACH,wBAAgB,cAAc,CAAC,KAAK,EAAE,OAAO,EAAE,KAAK,EAAE,SAAS,GAAG,OAAO,CA6BxE;AA4BD;;;;;;;;;;GAUG;AACH,wBAAgB,UAAU,CAAC,IAAI,EAAE,OAAO,EAAE,KAAK,EAAE,OAAO,GAAG,OAAO,CA6BjE;AAED;;;;;;;;;;;;GAYG;AACH,wBAAgB,UAAU,CAAC,KAAK,EAAE,OAAO,EAAE,MAAM,SAAK,GAAG,GAAG,CAAC,MAAM,EAAE,MAAM,CAAC,CAqB3E;AAED;;;;;;GAMG;AACH,wBAAgB,OAAO,CAAC,QAAQ,EAAE,OAAO,EAAE,QAAQ,EAAE,OAAO,GAAG,MAAM,CAepE;AAED;;;;;;;;;;GAUG;AACH,wBAAgB,GAAG,CAAC,MAAM,EAAE,SAAS,MAAM,EAAE,GAAG,MAAM,GAAG,IAAI,CAQ5D"}
|
|
@@ -0,0 +1,131 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The pinned judge, and the guard that keeps it pinned.
|
|
3
|
+
*
|
|
4
|
+
* PD-6 states that the judge model is pinned and never auto-swapped. That is a
|
|
5
|
+
* stronger requirement than "configured": a judge that could move would make
|
|
6
|
+
* every gate it decided incomparable with the gates decided before the move,
|
|
7
|
+
* because the scores either side of it were produced against different
|
|
8
|
+
* standards. A migration programme whose bar drifts under it cannot demonstrate
|
|
9
|
+
* anything.
|
|
10
|
+
*
|
|
11
|
+
* The pin is therefore recorded in TWO places that must agree — this module and
|
|
12
|
+
* the route table — and a judged assertion refuses to run when they disagree.
|
|
13
|
+
* The duplication is deliberate. A pin that lived only in the table would be
|
|
14
|
+
* satisfied by any model the table happened to name, so changing the table
|
|
15
|
+
* would silently change the standard; requiring both to move makes a judge swap
|
|
16
|
+
* an explicit, reviewable act in the same change that re-baselines the sets.
|
|
17
|
+
*
|
|
18
|
+
* @module llm/eval/judge
|
|
19
|
+
*/
|
|
20
|
+
import type { LlmAlias, LlmRouteTable } from "../types";
|
|
21
|
+
import type { CandidateRun, GoldenSet } from "./types";
|
|
22
|
+
/** Lowest score the judge rubric may return. */
|
|
23
|
+
export declare const JUDGE_SCORE_MIN = 0;
|
|
24
|
+
/** Highest score the judge rubric may return, matching the point scale the comparators use. */
|
|
25
|
+
export declare const JUDGE_SCORE_MAX = 100;
|
|
26
|
+
/** The identity a judged assertion requires the route table to resolve to. */
|
|
27
|
+
export interface PinnedJudgeIdentity {
|
|
28
|
+
/** The only alias a judged assertion may be served by. */
|
|
29
|
+
readonly alias: LlmAlias;
|
|
30
|
+
/** The provider the judge route must name. */
|
|
31
|
+
readonly provider: string;
|
|
32
|
+
/** The exact model id the judge route must name. */
|
|
33
|
+
readonly modelId: string;
|
|
34
|
+
}
|
|
35
|
+
/**
|
|
36
|
+
* The pinned judge identity.
|
|
37
|
+
*
|
|
38
|
+
* This is the one place in the codebase where naming a vendor model string is
|
|
39
|
+
* the requirement rather than a violation of it: PD-5 forbids model strings in
|
|
40
|
+
* APPLICATION code so that routing stays in config, while PD-6 requires the
|
|
41
|
+
* judge to be pinned so that grading stays comparable. Pinning by anything
|
|
42
|
+
* looser — a family, a provider, "whatever the table says" — would not pin
|
|
43
|
+
* anything.
|
|
44
|
+
*/
|
|
45
|
+
export declare const PINNED_JUDGE: PinnedJudgeIdentity;
|
|
46
|
+
/**
|
|
47
|
+
* Thrown when a judged assertion is asked to run against a judge that is not
|
|
48
|
+
* the pinned one.
|
|
49
|
+
*
|
|
50
|
+
* A refusal rather than a warning or a downgraded score: a judged gate decided
|
|
51
|
+
* by an unknown judge looks exactly like a judged gate decided by the right
|
|
52
|
+
* one, so the only safe outcome is to produce no verdict at all.
|
|
53
|
+
*/
|
|
54
|
+
export declare class JudgeNotPinnedError extends Error {
|
|
55
|
+
/** The alias the caller tried to judge through. */
|
|
56
|
+
readonly alias: string;
|
|
57
|
+
/**
|
|
58
|
+
* @param alias The alias the caller tried to judge through.
|
|
59
|
+
* @param detail What specifically failed the pin check.
|
|
60
|
+
*/
|
|
61
|
+
constructor(alias: string, detail: string);
|
|
62
|
+
}
|
|
63
|
+
/** The judge identity a route table actually resolves to. */
|
|
64
|
+
export interface ResolvedJudge {
|
|
65
|
+
/** The alias that was verified. */
|
|
66
|
+
readonly alias: LlmAlias;
|
|
67
|
+
/** The provider the table names for it. */
|
|
68
|
+
readonly provider: string;
|
|
69
|
+
/** The model id the table names for it. */
|
|
70
|
+
readonly modelId: string;
|
|
71
|
+
}
|
|
72
|
+
/**
|
|
73
|
+
* Verify that an alias is the pinned judge, or refuse.
|
|
74
|
+
*
|
|
75
|
+
* @param alias The alias a judged assertion would be served by.
|
|
76
|
+
* @param table The route table to verify against; defaults to the canonical one.
|
|
77
|
+
* @returns The resolved judge identity when every pin condition holds.
|
|
78
|
+
* @throws {JudgeNotPinnedError} When any pin condition fails.
|
|
79
|
+
*/
|
|
80
|
+
export declare function assertPinnedJudge(alias: LlmAlias, table?: LlmRouteTable): ResolvedJudge;
|
|
81
|
+
/** One case put to the judge. */
|
|
82
|
+
export interface JudgeRequest {
|
|
83
|
+
/** The grading rubric, stated on the golden set rather than improvised per call. */
|
|
84
|
+
readonly rubric: string;
|
|
85
|
+
/** The original input the answer responds to. */
|
|
86
|
+
readonly input: string;
|
|
87
|
+
/** The reference answer. */
|
|
88
|
+
readonly reference: string;
|
|
89
|
+
/** The answer being graded. */
|
|
90
|
+
readonly answer: string;
|
|
91
|
+
}
|
|
92
|
+
/**
|
|
93
|
+
* Score one answer with the pinned judge.
|
|
94
|
+
*
|
|
95
|
+
* The pin is verified before the call, not after, so a mis-pinned judge costs
|
|
96
|
+
* nothing and produces nothing rather than producing a score that would have to
|
|
97
|
+
* be retracted.
|
|
98
|
+
*
|
|
99
|
+
* @param request The case to grade.
|
|
100
|
+
* @param options Overrides used by the harness's own pinning check.
|
|
101
|
+
* @param options.alias The alias to judge through; only the pinned one is accepted.
|
|
102
|
+
* @param options.table The route table to verify the pin against.
|
|
103
|
+
* @returns The score, on the same 0-100 point scale the comparators use.
|
|
104
|
+
* @throws {JudgeNotPinnedError} When the judge is not the pinned one.
|
|
105
|
+
*/
|
|
106
|
+
export declare function scoreWithPinnedJudge(request: JudgeRequest, options?: {
|
|
107
|
+
alias?: LlmAlias;
|
|
108
|
+
table?: LlmRouteTable;
|
|
109
|
+
}): Promise<number>;
|
|
110
|
+
/**
|
|
111
|
+
* Fill in any missing judge scores on a candidate run, using the pinned judge.
|
|
112
|
+
*
|
|
113
|
+
* The pin is verified once before the first call rather than per case, so a
|
|
114
|
+
* mis-pinned judge cannot grade part of a set before it is caught — a half-
|
|
115
|
+
* graded set is worse than an ungraded one, because its mean looks like a
|
|
116
|
+
* measurement.
|
|
117
|
+
*
|
|
118
|
+
* @param set The golden set being graded, which supplies the reference answers.
|
|
119
|
+
* @param candidate The candidate run whose answers need scoring.
|
|
120
|
+
* @param rubric The grading rubric, stated on the set rather than improvised.
|
|
121
|
+
* @param options Overrides used by the harness's own pinning check.
|
|
122
|
+
* @param options.alias The alias to judge through; only the pinned one is accepted.
|
|
123
|
+
* @param options.table The route table to verify the pin against.
|
|
124
|
+
* @returns A candidate run with a judge score on every case.
|
|
125
|
+
* @throws {JudgeNotPinnedError} When the judge is not the pinned one.
|
|
126
|
+
*/
|
|
127
|
+
export declare function scoreCandidateRun(set: GoldenSet, candidate: CandidateRun, rubric: string, options?: {
|
|
128
|
+
alias?: LlmAlias;
|
|
129
|
+
table?: LlmRouteTable;
|
|
130
|
+
}): Promise<CandidateRun>;
|
|
131
|
+
//# sourceMappingURL=judge.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"judge.d.ts","sourceRoot":"","sources":["../../../../src/llm/eval/judge.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;GAkBG;AAIH,OAAO,KAAK,EACV,QAAQ,EACR,aAAa,EAEd,MAAM,UAAU,CAAC;AAClB,OAAO,KAAK,EAAE,YAAY,EAAE,SAAS,EAAkB,MAAM,SAAS,CAAC;AAEvE,gDAAgD;AAChD,eAAO,MAAM,eAAe,IAAI,CAAC;AAEjC,+FAA+F;AAC/F,eAAO,MAAM,eAAe,MAAM,CAAC;AAEnC,8EAA8E;AAC9E,MAAM,WAAW,mBAAmB;IAClC,0DAA0D;IAC1D,QAAQ,CAAC,KAAK,EAAE,QAAQ,CAAC;IACzB,8CAA8C;IAC9C,QAAQ,CAAC,QAAQ,EAAE,MAAM,CAAC;IAC1B,oDAAoD;IACpD,QAAQ,CAAC,OAAO,EAAE,MAAM,CAAC;CAC1B;AAED;;;;;;;;;GASG;AACH,eAAO,MAAM,YAAY,EAAE,mBAI1B,CAAC;AAEF;;;;;;;GAOG;AACH,qBAAa,mBAAoB,SAAQ,KAAK;IAC5C,mDAAmD;IACnD,SAAgB,KAAK,EAAE,MAAM,CAAC;IAE9B;;;OAGG;gBACgB,KAAK,EAAE,MAAM,EAAE,MAAM,EAAE,MAAM;CASjD;AAED,6DAA6D;AAC7D,MAAM,WAAW,aAAa;IAC5B,mCAAmC;IACnC,QAAQ,CAAC,KAAK,EAAE,QAAQ,CAAC;IACzB,2CAA2C;IAC3C,QAAQ,CAAC,QAAQ,EAAE,MAAM,CAAC;IAC1B,2CAA2C;IAC3C,QAAQ,CAAC,OAAO,EAAE,MAAM,CAAC;CAC1B;AAED;;;;;;;GAOG;AACH,wBAAgB,iBAAiB,CAC/B,KAAK,EAAE,QAAQ,EACf,KAAK,GAAE,aAA0B,GAChC,aAAa,CAiDf;AAED,iCAAiC;AACjC,MAAM,WAAW,YAAY;IAC3B,oFAAoF;IACpF,QAAQ,CAAC,MAAM,EAAE,MAAM,CAAC;IACxB,iDAAiD;IACjD,QAAQ,CAAC,KAAK,EAAE,MAAM,CAAC;IACvB,4BAA4B;IAC5B,QAAQ,CAAC,SAAS,EAAE,MAAM,CAAC;IAC3B,+BAA+B;IAC/B,QAAQ,CAAC,MAAM,EAAE,MAAM,CAAC;CACzB;AAyBD;;;;;;;;;;;;;GAaG;AACH,wBAAsB,oBAAoB,CACxC,OAAO,EAAE,YAAY,EACrB,OAAO,GAAE;IAAE,KAAK,CAAC,EAAE,QAAQ,CAAC;IAAC,KAAK,CAAC,EAAE,aAAa,CAAA;CAAO,GACxD,OAAO,CAAC,MAAM,CAAC,CAqBjB;AAED;;;;;;;;;;;;;;;;GAgBG;AACH,wBAAsB,iBAAiB,CACrC,GAAG,EAAE,SAAS,EACd,SAAS,EAAE,YAAY,EACvB,MAAM,EAAE,MAAM,EACd,OAAO,GAAE;IAAE,KAAK,CAAC,EAAE,QAAQ,CAAC;IAAC,KAAK,CAAC,EAAE,aAAa,CAAA;CAAO,GACxD,OAAO,CAAC,YAAY,CAAC,CA0BvB"}
|
|
@@ -0,0 +1,51 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Derivation of eval metrics from recorded model answers.
|
|
3
|
+
*
|
|
4
|
+
* Both sides of every comparison are measured HERE, by the same code, from
|
|
5
|
+
* records of the same shape. That symmetry is the point: an incumbent baseline
|
|
6
|
+
* computed by one routine and a candidate score computed by another would
|
|
7
|
+
* differ by the routines as much as by the models, and the gate would be
|
|
8
|
+
* measuring its own implementation.
|
|
9
|
+
*
|
|
10
|
+
* A metric that cannot be derived from the records present is returned as
|
|
11
|
+
* `undefined` rather than as a zero or a default. Zero is a measurement; absence
|
|
12
|
+
* is not, and the comparators depend on being able to tell them apart in order
|
|
13
|
+
* to render `indeterminate` instead of a verdict they have not earned.
|
|
14
|
+
*
|
|
15
|
+
* @module llm/eval/metrics
|
|
16
|
+
*/
|
|
17
|
+
import type { BaselineMetrics, GoldenCase, GoldenSet, RecordedResult, ToolExpectation } from "./types";
|
|
18
|
+
/**
|
|
19
|
+
* Structured retries a tool-call site may spend before the answer stops
|
|
20
|
+
* counting as a valid call.
|
|
21
|
+
*
|
|
22
|
+
* Section 6 permits exactly one: the retry exists to recover a malformed call
|
|
23
|
+
* from a model that otherwise understood the request, and a site that needs
|
|
24
|
+
* more than one has not produced a valid call — it has been coached into one,
|
|
25
|
+
* which is a different capability from the one being measured.
|
|
26
|
+
*/
|
|
27
|
+
export declare const MAX_STRUCTURED_RETRIES = 1;
|
|
28
|
+
/** One graded case paired with the answer being scored. */
|
|
29
|
+
export interface MetricSample {
|
|
30
|
+
/** The golden case. */
|
|
31
|
+
readonly goldenCase: GoldenCase;
|
|
32
|
+
/** The answer under measurement — incumbent or candidate. */
|
|
33
|
+
readonly result: RecordedResult;
|
|
34
|
+
}
|
|
35
|
+
/**
|
|
36
|
+
* Whether one recorded answer is a valid tool call under the site's contract.
|
|
37
|
+
*
|
|
38
|
+
* @param result The recorded answer.
|
|
39
|
+
* @param expectation The tool contract the site requires.
|
|
40
|
+
* @returns Whether the answer is a valid call within the permitted retry budget.
|
|
41
|
+
*/
|
|
42
|
+
export declare function isValidToolCall(result: RecordedResult, expectation: ToolExpectation): boolean;
|
|
43
|
+
/**
|
|
44
|
+
* Derive every metric a set's assertions need, from one side's answers.
|
|
45
|
+
*
|
|
46
|
+
* @param set The golden set, which fixes what is measured and how.
|
|
47
|
+
* @param samples Cases paired with the answers being scored.
|
|
48
|
+
* @returns The derived metrics; a metric the records cannot support is absent.
|
|
49
|
+
*/
|
|
50
|
+
export declare function deriveMetrics(set: GoldenSet, samples: readonly MetricSample[]): BaselineMetrics;
|
|
51
|
+
//# sourceMappingURL=metrics.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"metrics.d.ts","sourceRoot":"","sources":["../../../../src/llm/eval/metrics.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;GAeG;AAGH,OAAO,KAAK,EACV,eAAe,EACf,UAAU,EACV,SAAS,EACT,cAAc,EACd,eAAe,EAChB,MAAM,SAAS,CAAC;AAEjB;;;;;;;;GAQG;AACH,eAAO,MAAM,sBAAsB,IAAI,CAAC;AAExC,2DAA2D;AAC3D,MAAM,WAAW,YAAY;IAC3B,uBAAuB;IACvB,QAAQ,CAAC,UAAU,EAAE,UAAU,CAAC;IAChC,6DAA6D;IAC7D,QAAQ,CAAC,MAAM,EAAE,cAAc,CAAC;CACjC;AAED;;;;;;GAMG;AACH,wBAAgB,eAAe,CAC7B,MAAM,EAAE,cAAc,EACtB,WAAW,EAAE,eAAe,GAC3B,OAAO,CAwBT;AAuBD;;;;;;GAMG;AACH,wBAAgB,aAAa,CAC3B,GAAG,EAAE,SAAS,EACd,OAAO,EAAE,SAAS,YAAY,EAAE,GAC/B,eAAe,CAmDjB"}
|
|
@@ -0,0 +1,97 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Grading one candidate run against one golden set, and a whole evaluation
|
|
3
|
+
* against many.
|
|
4
|
+
*
|
|
5
|
+
* The orchestration here holds two invariants that the comparators cannot hold
|
|
6
|
+
* on their own.
|
|
7
|
+
*
|
|
8
|
+
* The bar is the ATTESTED baseline, not a number recomputed at grading time —
|
|
9
|
+
* so the bar a candidate clears is the same bar every earlier candidate cleared,
|
|
10
|
+
* and a re-baselining is a visible edit to the set rather than a silent
|
|
11
|
+
* consequence of running the harness again. The recomputation still happens,
|
|
12
|
+
* as an integrity check: when the attested baseline and the records it claims
|
|
13
|
+
* to summarise disagree, the set renders `indeterminate` instead of grading
|
|
14
|
+
* against a bar that has been moved by hand.
|
|
15
|
+
*
|
|
16
|
+
* And an incomplete candidate run is never graded. Scoring the subset of cases
|
|
17
|
+
* that happened to produce an answer silently reweights the set toward whatever
|
|
18
|
+
* the candidate found easy, which is the most flattering possible error a model
|
|
19
|
+
* comparison can make.
|
|
20
|
+
*
|
|
21
|
+
* @module llm/eval/run
|
|
22
|
+
*/
|
|
23
|
+
import type { LlmAlias, LlmRouteTable } from "../types";
|
|
24
|
+
import type { CandidateRun, GoldenSet, RunReport, SetReport, Verdict } from "./types";
|
|
25
|
+
/**
|
|
26
|
+
* Absolute agreement required between an attested baseline metric and the same
|
|
27
|
+
* metric recomputed from the set's own records.
|
|
28
|
+
*
|
|
29
|
+
* Tight enough that only floating-point representation error fits inside it: the
|
|
30
|
+
* check exists to catch a baseline edited independently of its evidence, and a
|
|
31
|
+
* generous tolerance would let exactly that through.
|
|
32
|
+
*/
|
|
33
|
+
export declare const BASELINE_ATTESTATION_TOLERANCE = 0.000001;
|
|
34
|
+
/**
|
|
35
|
+
* Overrides for grading, used only by the harness's own pinning check.
|
|
36
|
+
*
|
|
37
|
+
* Production grading passes none of these: the judge is the pinned one and the
|
|
38
|
+
* route table is the canonical one, and an override that a caller could reach in
|
|
39
|
+
* normal use would be a way around PD-6 rather than a way to verify it.
|
|
40
|
+
*/
|
|
41
|
+
export interface EvaluateOptions {
|
|
42
|
+
/** The alias judged assertions are served by. Only the pinned judge is accepted. */
|
|
43
|
+
readonly judgeAlias?: LlmAlias;
|
|
44
|
+
/** The route table the pin is verified against. */
|
|
45
|
+
readonly routeTable?: LlmRouteTable;
|
|
46
|
+
}
|
|
47
|
+
/** One golden set paired with the candidate answers to grade against it. */
|
|
48
|
+
export interface EvalPair {
|
|
49
|
+
/** The golden set. */
|
|
50
|
+
readonly set: GoldenSet;
|
|
51
|
+
/** The candidate run answering it. */
|
|
52
|
+
readonly candidate: CandidateRun;
|
|
53
|
+
}
|
|
54
|
+
/**
|
|
55
|
+
* Grade one candidate run against one golden set.
|
|
56
|
+
*
|
|
57
|
+
* A set that asserts `judge` verifies the judge pin BEFORE it grades, even
|
|
58
|
+
* though the scores it grades were recorded earlier. The scores are only
|
|
59
|
+
* meaningful as the output of a specific, unchanged instrument, so a set whose
|
|
60
|
+
* judge has moved since capture has no valid scores to grade — and refusing is
|
|
61
|
+
* the only outcome that says so.
|
|
62
|
+
*
|
|
63
|
+
* @param set The golden set.
|
|
64
|
+
* @param candidate The candidate's answers.
|
|
65
|
+
* @param options Overrides used by the harness's own pinning check.
|
|
66
|
+
* @returns The set's report.
|
|
67
|
+
* @throws {GoldenSetError} When the candidate run answers a different set.
|
|
68
|
+
* @throws {JudgeNotPinnedError} When a judged set's judge is not the pinned one.
|
|
69
|
+
*/
|
|
70
|
+
export declare function evaluateSet(set: GoldenSet, candidate: CandidateRun, options?: EvaluateOptions): SetReport;
|
|
71
|
+
/**
|
|
72
|
+
* Grade a whole evaluation.
|
|
73
|
+
*
|
|
74
|
+
* An evaluation with no sets is FAILED, not PASSED. A gate that reports green
|
|
75
|
+
* when it graded nothing is indistinguishable from a gate that is working, and
|
|
76
|
+
* is the exact failure this harness exists to prevent.
|
|
77
|
+
*
|
|
78
|
+
* @param pairs The sets and the candidate runs answering them.
|
|
79
|
+
* @param options Overrides used by the harness's own pinning check.
|
|
80
|
+
* @returns The run report.
|
|
81
|
+
*/
|
|
82
|
+
export declare function evaluateRun(pairs: readonly EvalPair[], options?: EvaluateOptions): RunReport;
|
|
83
|
+
/**
|
|
84
|
+
* Render one verdict as a single log line.
|
|
85
|
+
*
|
|
86
|
+
* @param verdict The verdict.
|
|
87
|
+
* @returns A line naming the assertion, the outcome and the numbers behind it.
|
|
88
|
+
*/
|
|
89
|
+
export declare function formatVerdict(verdict: Verdict): string;
|
|
90
|
+
/**
|
|
91
|
+
* Render a run report as lines for a CI log.
|
|
92
|
+
*
|
|
93
|
+
* @param runReport The report.
|
|
94
|
+
* @returns The lines, in display order.
|
|
95
|
+
*/
|
|
96
|
+
export declare function formatRunReport(runReport: RunReport): string[];
|
|
97
|
+
//# sourceMappingURL=run.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"run.d.ts","sourceRoot":"","sources":["../../../../src/llm/eval/run.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;;;GAqBG;AAEH,OAAO,KAAK,EAAE,QAAQ,EAAE,aAAa,EAAE,MAAM,UAAU,CAAC;AAMxD,OAAO,KAAK,EAEV,YAAY,EAGZ,SAAS,EACT,SAAS,EACT,SAAS,EACT,OAAO,EACR,MAAM,SAAS,CAAC;AAEjB;;;;;;;GAOG;AACH,eAAO,MAAM,8BAA8B,WAAO,CAAC;AAKnD;;;;;;GAMG;AACH,MAAM,WAAW,eAAe;IAC9B,oFAAoF;IACpF,QAAQ,CAAC,UAAU,CAAC,EAAE,QAAQ,CAAC;IAC/B,mDAAmD;IACnD,QAAQ,CAAC,UAAU,CAAC,EAAE,aAAa,CAAC;CACrC;AAED,4EAA4E;AAC5E,MAAM,WAAW,QAAQ;IACvB,sBAAsB;IACtB,QAAQ,CAAC,GAAG,EAAE,SAAS,CAAC;IACxB,sCAAsC;IACtC,QAAQ,CAAC,SAAS,EAAE,YAAY,CAAC;CAClC;AAmID;;;;;;;;;;;;;;;GAeG;AACH,wBAAgB,WAAW,CACzB,GAAG,EAAE,SAAS,EACd,SAAS,EAAE,YAAY,EACvB,OAAO,GAAE,eAAoB,GAC5B,SAAS,CAsDX;AAsBD;;;;;;;;;;GAUG;AACH,wBAAgB,WAAW,CACzB,KAAK,EAAE,SAAS,QAAQ,EAAE,EAC1B,OAAO,GAAE,eAAoB,GAC5B,SAAS,CAiBX;AAED;;;;;GAKG;AACH,wBAAgB,aAAa,CAAC,OAAO,EAAE,OAAO,GAAG,MAAM,CAQtD;AAED;;;;;GAKG;AACH,wBAAgB,eAAe,CAAC,SAAS,EAAE,SAAS,GAAG,MAAM,EAAE,CAU9D"}
|