@tangle-network/agent-eval 0.93.0 → 0.94.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/adapters/otel.d.ts +4 -4
- package/dist/analyst/index.d.ts +10 -10
- package/dist/analyst/index.js +6 -6
- package/dist/{analyze-runs-DwCEkpO_.d.ts → analyze-runs-B6Ljo_dI.d.ts} +2 -2
- package/dist/{baseline-DE36-Np7.d.ts → baseline-Bbid3WoO.d.ts} +1 -1
- package/dist/belief-state/index.d.ts +2 -2
- package/dist/belief-state/index.js +1 -1
- package/dist/builder-eval/index.d.ts +3 -3
- package/dist/builder-eval/index.js +2 -2
- package/dist/{calibration-Cpr3WaX3.d.ts → calibration-BPmzuVPk.d.ts} +1 -1
- package/dist/campaign/index.d.ts +13 -13
- package/dist/campaign/index.js +8 -8
- package/dist/{chunk-TWS7AZEY.js → chunk-2K6UUZ7P.js} +2 -2
- package/dist/{chunk-Y47J2LJ3.js → chunk-2KNZHH3P.js} +5 -5
- package/dist/{chunk-LMZQ2Z4U.js → chunk-56GC6VYO.js} +126 -1
- package/dist/chunk-56GC6VYO.js.map +1 -0
- package/dist/{chunk-QG2OVF2D.js → chunk-CTBHKLEU.js} +99 -30
- package/dist/chunk-CTBHKLEU.js.map +1 -0
- package/dist/{chunk-CVVHBFGN.js → chunk-CWNP4DV4.js} +53 -5
- package/dist/chunk-CWNP4DV4.js.map +1 -0
- package/dist/{chunk-QS3RBQPI.js → chunk-D5KBVULY.js} +2 -2
- package/dist/{chunk-ZFIBGEOL.js → chunk-EGPMSBEZ.js} +4 -4
- package/dist/{chunk-GSH6QNNS.js → chunk-KW53MSA5.js} +101 -78
- package/dist/chunk-KW53MSA5.js.map +1 -0
- package/dist/{chunk-UMMZHCPB.js → chunk-MIFZUPEK.js} +7 -5
- package/dist/chunk-MIFZUPEK.js.map +1 -0
- package/dist/{chunk-XY4DDNEG.js → chunk-MPQWFX6Y.js} +2 -2
- package/dist/{chunk-L3JOU6XM.js → chunk-NACM5RS7.js} +3 -3
- package/dist/{chunk-D3V5B42D.js → chunk-Q5LIB7BC.js} +7 -4
- package/dist/{chunk-D3V5B42D.js.map → chunk-Q5LIB7BC.js.map} +1 -1
- package/dist/{chunk-QAY5UIJO.js → chunk-QMUEXQJS.js} +99 -57
- package/dist/chunk-QMUEXQJS.js.map +1 -0
- package/dist/{chunk-4FBZZIYD.js → chunk-QTMYW64F.js} +2 -2
- package/dist/{chunk-6SOJM3VR.js → chunk-SD2YFWQQ.js} +4 -4
- package/dist/{chunk-UHMJT4T7.js → chunk-TBDR6PAI.js} +2 -2
- package/dist/{chunk-47X6LRCE.js → chunk-UFSG7ACU.js} +10 -7
- package/dist/chunk-UFSG7ACU.js.map +1 -0
- package/dist/{chunk-CY6U5S3X.js → chunk-URWVKRCS.js} +2 -2
- package/dist/{chunk-T375SUOZ.js → chunk-ZOPLCJSY.js} +86 -31
- package/dist/chunk-ZOPLCJSY.js.map +1 -0
- package/dist/cli.js +2 -2
- package/dist/contract/index.d.ts +17 -17
- package/dist/contract/index.js +11 -11
- package/dist/{control-_Qb7skHX.d.ts → control-D6qwHXIR.d.ts} +4 -4
- package/dist/{control-runtime-DuFBYg7A.d.ts → control-runtime-Acf9CGhw.d.ts} +2 -2
- package/dist/control.d.ts +5 -5
- package/dist/{corpus-BoR-041R.d.ts → corpus-B8A4BDR3.d.ts} +1 -1
- package/dist/{counterfactual-Dwibr5IW.d.ts → counterfactual-DlOz8PBx.d.ts} +3 -3
- package/dist/{default-registry-zoGHUQEH.d.ts → default-registry-6dhErQbs.d.ts} +2 -2
- package/dist/diagnose.d.ts +9 -9
- package/dist/diagnose.js +1 -1
- package/dist/{emitter-DEZwY14K.d.ts → emitter-C2rqGH_l.d.ts} +1 -1
- package/dist/{failure-cluster-CL7IVgkJ.d.ts → failure-cluster-DH9Flgcf.d.ts} +1 -1
- package/dist/{feedback-trajectory-D9OVLrg9.d.ts → feedback-trajectory-BxY0cKfs.d.ts} +1 -1
- package/dist/governance/index.d.ts +2 -2
- package/dist/{harness-optimizer-EnEnQPsr.d.ts → harness-optimizer-mOl9XX_O.d.ts} +1 -1
- package/dist/hosted/index.d.ts +4 -4
- package/dist/index.d.ts +68 -46
- package/dist/index.js +261 -87
- package/dist/index.js.map +1 -1
- package/dist/{insight-report-BBwvOh6x.d.ts → insight-report-DWl3z9tl.d.ts} +1 -1
- package/dist/{integrity-VJ9A7aST.d.ts → integrity-D2t12mMw.d.ts} +1 -1
- package/dist/{kind-factory-5b7xXXOr.d.ts → kind-factory-0BhLSI27.d.ts} +1 -1
- package/dist/knowledge/index.d.ts +3 -3
- package/dist/{llm-client-BeEcAokY.d.ts → llm-client-Bj7g0rqu.d.ts} +31 -0
- package/dist/meta-eval/index.d.ts +4 -4
- package/dist/meta-eval/index.js +1 -1
- package/dist/openapi.json +1 -1
- package/dist/pipelines/index.d.ts +6 -6
- package/dist/pipelines/index.js +40 -15
- package/dist/pipelines/index.js.map +1 -1
- package/dist/prm/index.d.ts +11 -7
- package/dist/prm/index.js +55 -12
- package/dist/prm/index.js.map +1 -1
- package/dist/{provenance-LnqRT0sS.d.ts → provenance-P-bCL2Fo.d.ts} +3 -3
- package/dist/{query-CqTxMwDw.d.ts → query-B7GGjRox.d.ts} +2 -1
- package/dist/{red-team-BXHil6c8.d.ts → red-team-BWdoyleI.d.ts} +1 -1
- package/dist/{release-report-euXIV_Sk.d.ts → release-report-BEbWmVYj.d.ts} +1 -1
- package/dist/reporting.d.ts +6 -6
- package/dist/reporting.js +3 -3
- package/dist/{researcher-DE6Gpnb4.d.ts → researcher-B0C2_fVO.d.ts} +5 -5
- package/dist/rl.d.ts +10 -10
- package/dist/rl.js +4 -4
- package/dist/{rubric-BOfxn4ja.d.ts → rubric-Cc6UHvUb.d.ts} +2 -2
- package/dist/{run-campaign-RDGAM5KJ.js → run-campaign-7WNXMDSN.js} +3 -3
- package/dist/{run-critic-BAIjX99r.d.ts → run-critic-CmMf05uV.d.ts} +1 -1
- package/dist/{run-improvement-loop-5z_l5zDz.d.ts → run-improvement-loop-DBahB8Ax.d.ts} +1 -1
- package/dist/{semantic-concept-judge-Dn8Z6KEG.d.ts → semantic-concept-judge-B9MgmBnM.d.ts} +3 -3
- package/dist/{statistics-C7PozGrZ.d.ts → statistics-CCJpTGOS.d.ts} +117 -2
- package/dist/{store-CKUAgsJz.d.ts → store-BcFXE6LG.d.ts} +16 -1
- package/dist/{summary-report-DGmUucwQ.d.ts → summary-report-BDOFevaT.d.ts} +1 -1
- package/dist/{test-graded-scenario-BdVaPyHT.d.ts → test-graded-scenario-DeODGLra.d.ts} +21 -2
- package/dist/traces.d.ts +19 -6
- package/dist/traces.js +4 -4
- package/dist/{trajectory-GEdXJCL5.d.ts → trajectory-2TkpSEVh.d.ts} +1 -1
- package/dist/{types-Croy5h7V.d.ts → types-C7DGg5ex.d.ts} +2 -0
- package/dist/{types-2VVIL04s.d.ts → types-Ce17tDlG.d.ts} +2 -2
- package/dist/wire/index.d.ts +4 -4
- package/dist/wire/index.js +2 -2
- package/dist/workflow/index.d.ts +13 -13
- package/dist/workflow/index.js +1 -1
- package/package.json +1 -1
- package/dist/chunk-47X6LRCE.js.map +0 -1
- package/dist/chunk-CVVHBFGN.js.map +0 -1
- package/dist/chunk-GSH6QNNS.js.map +0 -1
- package/dist/chunk-LMZQ2Z4U.js.map +0 -1
- package/dist/chunk-QAY5UIJO.js.map +0 -1
- package/dist/chunk-QG2OVF2D.js.map +0 -1
- package/dist/chunk-T375SUOZ.js.map +0 -1
- package/dist/chunk-UMMZHCPB.js.map +0 -1
- /package/dist/{chunk-TWS7AZEY.js.map → chunk-2K6UUZ7P.js.map} +0 -0
- /package/dist/{chunk-Y47J2LJ3.js.map → chunk-2KNZHH3P.js.map} +0 -0
- /package/dist/{chunk-QS3RBQPI.js.map → chunk-D5KBVULY.js.map} +0 -0
- /package/dist/{chunk-ZFIBGEOL.js.map → chunk-EGPMSBEZ.js.map} +0 -0
- /package/dist/{chunk-XY4DDNEG.js.map → chunk-MPQWFX6Y.js.map} +0 -0
- /package/dist/{chunk-L3JOU6XM.js.map → chunk-NACM5RS7.js.map} +0 -0
- /package/dist/{chunk-4FBZZIYD.js.map → chunk-QTMYW64F.js.map} +0 -0
- /package/dist/{chunk-6SOJM3VR.js.map → chunk-SD2YFWQQ.js.map} +0 -0
- /package/dist/{chunk-UHMJT4T7.js.map → chunk-TBDR6PAI.js.map} +0 -0
- /package/dist/{chunk-CY6U5S3X.js.map → chunk-URWVKRCS.js.map} +0 -0
- /package/dist/{run-campaign-RDGAM5KJ.js.map → run-campaign-7WNXMDSN.js.map} +0 -0
package/dist/prm/index.js.map
CHANGED
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"sources":["../../src/prm/builtin-rubrics.ts","../../src/prm/inference.ts","../../src/prm/rubric.ts"],"sourcesContent":["/**\n * Built-in reference rubrics. Consumers combine these with domain\n * rubrics. All are deterministic, rule-based — cheap to run + easy\n * to unit-test. LLM-based rubrics are trivially authored by\n * following the StepRubric contract.\n */\n\nimport type { LlmSpan, ToolSpan } from '../trace/schema'\nimport type { StepRubric } from './rubric'\n\n/** Penalize very short or very long assistant outputs. */\nexport function outputLengthRubric(\n args: { minChars?: number; maxChars?: number; weight?: number } = {},\n): StepRubric {\n const min = args.minChars ?? 20\n const max = args.maxChars ?? 8000\n return {\n id: 'output-length',\n kinds: ['llm'],\n weight: args.weight ?? 0.5,\n async grade({ step }) {\n const llm = step.span as LlmSpan\n const len = (llm.output ?? '').length\n if (len === 0) return { score: 0, rationale: 'empty output' }\n if (len < min)\n return { score: Math.max(0, len / min), rationale: `below min (${len} < ${min})` }\n if (len > max)\n return {\n score: Math.max(0, 1 - (len - max) / max),\n rationale: `above max (${len} > ${max})`,\n }\n return { score: 1, rationale: `${len} chars in bounds` }\n },\n }\n}\n\n/** Reward tool calls that succeeded (status='ok') with an informative result. */\nexport function toolSuccessRubric(args: { weight?: number } = {}): StepRubric {\n return {\n id: 'tool-success',\n kinds: ['tool'],\n weight: args.weight ?? 1,\n async grade({ step }) {\n const tool = step.span as ToolSpan\n if (tool.status === 'error')\n return { score: 0, rationale: `error: ${tool.error ?? 'unknown'}` }\n const r = tool.result\n if (r === null || r === undefined) return { score: 0.3, rationale: 'empty result' }\n const asText = typeof r === 'string' ? r : JSON.stringify(r)\n if (asText.length < 4) return { score: 0.5, rationale: 'tiny result' }\n return { score: 1, rationale: `${tool.toolName} ok` }\n },\n }\n}\n\n/** Penalize tool calls that duplicate a prior call with identical args. */\nexport function toolNonRedundantRubric(args: { weight?: number } = {}): StepRubric {\n const weight = args.weight ?? 0.5\n return {\n id: 'tool-non-redundant',\n kinds: ['tool'],\n weight,\n async grade({ step, prior }) {\n const tool = step.span as ToolSpan\n const priorMatches = prior.filter((p) => {\n if (p.span.kind !== 'tool') return false\n const pt = p.span as ToolSpan\n return (\n pt.toolName === tool.toolName && stableStringify(pt.args) === stableStringify(tool.args)\n )\n })\n if (priorMatches.length === 0) return { score: 1, rationale: 'novel call' }\n return {\n score: Math.max(0, 1 - priorMatches.length * 0.5),\n rationale: `${priorMatches.length} duplicate(s)`,\n }\n },\n }\n}\n\n/** Penalize LLM outputs that contain common refusal markers when a refusal\n * is NOT expected (caller inverts weight for scenarios where refusal IS expected). */\nexport function nonRefusalRubric(args: { markers?: RegExp[]; weight?: number } = {}): StepRubric {\n const weight = args.weight ?? 1\n const markers = args.markers ?? [\n /\\bi\\s+(?:can(?:not|'t)|won't|will\\s+not)\\b/i,\n /\\b(?:as\\s+an?\\s+)?ai\\b.*?\\b(?:can't|cannot)\\b/i,\n ]\n return {\n id: 'non-refusal',\n kinds: ['llm'],\n weight,\n async grade({ step }) {\n const llm = step.span as LlmSpan\n const out = llm.output ?? ''\n const refused = markers.some((re) => re.test(out))\n return refused\n ? { score: 0, rationale: 'refusal marker present' }\n : { score: 1, rationale: 'no refusal' }\n },\n }\n}\n\n/** Reward outputs that invoke the next-step tool the trajectory actually uses\n * (i.e. the LLM span announced \"I will call X\" and the following tool span IS X). */\nexport function toolIntentAlignmentRubric(args: { weight?: number } = {}): StepRubric {\n return {\n id: 'tool-intent-alignment',\n kinds: ['llm'],\n weight: args.weight ?? 0.5,\n async grade({ step, next }) {\n const llm = step.span as LlmSpan\n const nextTool = next.find((s) => s.span.kind === 'tool')\n if (!nextTool) return null\n const toolName = (nextTool.span as ToolSpan).toolName\n const out = (llm.output ?? '').toLowerCase()\n const mentioned = out.includes(toolName.toLowerCase())\n return mentioned\n ? { score: 1, rationale: `mentioned \"${toolName}\" before calling it` }\n : { score: 0.5, rationale: `called \"${toolName}\" without announcing it` }\n },\n }\n}\n\nfunction stableStringify(value: unknown): string {\n if (value === null || typeof value !== 'object') return JSON.stringify(value)\n if (Array.isArray(value)) return `[${value.map(stableStringify).join(',')}]`\n const keys = Object.keys(value as Record<string, unknown>).sort()\n return `{${keys.map((k) => `${JSON.stringify(k)}:${stableStringify((value as Record<string, unknown>)[k])}`).join(',')}}`\n}\n","/**\n * Inference-time PRM scoring — pick the best of N candidate trajectories\n * using a trained reward model (or a rule-based PRM as a proxy).\n *\n * The canonical Best-of-N pattern: generate N completions, score each\n * with a PRM, pick the winner. Here the scoring loop is framework-agnostic\n * — supply a TraceStore + PrmGrader + N run IDs → get ranking + winner.\n */\n\nimport type { TraceStore } from '../trace/store'\nimport type { PrmGradedTrace, PrmGrader } from './rubric'\n\nexport interface BestOfNResult {\n winner: PrmGradedTrace\n ranked: PrmGradedTrace[]\n /** Standard deviation of aggregate scores — small = candidates were homogenous. */\n stdDev: number\n}\n\nexport async function prmBestOfN(\n store: TraceStore,\n grader: PrmGrader,\n runIds: string[],\n): Promise<BestOfNResult> {\n if (runIds.length === 0) throw new Error('prmBestOfN: at least 1 candidate required')\n const graded = await Promise.all(runIds.map((id) => grader.grade(store, id)))\n const ranked = [...graded].sort((a, b) => b.aggregateScore - a.aggregateScore)\n const mean = graded.reduce((a, g) => a + g.aggregateScore, 0) / graded.length\n const variance = graded.reduce((a, g) => a + (g.aggregateScore - mean) ** 2, 0) / graded.length\n return { winner: ranked[0]!, ranked, stdDev: Math.sqrt(variance) }\n}\n\n/**\n * Weighted vote across multiple graders — use when you want a PRM ensemble\n * (e.g. rule-based + LLM-based + trained model). Each grader produces its\n * own ranking; we aggregate via rank-sum (Borda count) so no single grader\n * dominates via a different score scale.\n */\nexport async function prmEnsembleBestOfN(\n store: TraceStore,\n graders: PrmGrader[],\n runIds: string[],\n): Promise<BestOfNResult> {\n if (graders.length === 0) throw new Error('prmEnsembleBestOfN: at least 1 grader')\n const perGrader = await Promise.all(\n graders.map(async (g) => {\n const graded = await Promise.all(runIds.map((id) => g.grade(store, id)))\n return graded.sort((a, b) => b.aggregateScore - a.aggregateScore)\n }),\n )\n // Borda: rank-sum across graders.\n const bordaScores = new Map<string, number>()\n for (const ranking of perGrader) {\n ranking.forEach((g, rank) => {\n bordaScores.set(g.runId, (bordaScores.get(g.runId) ?? 0) + (ranking.length - rank))\n })\n }\n // Return a synthesized ranking using the first grader's graded traces\n // ordered by Borda score. aggregateScore field kept for UX.\n const canonical = perGrader[0]!\n const byRun = new Map(canonical.map((g) => [g.runId, g]))\n const ranked = [...byRun.values()].sort(\n (a, b) => (bordaScores.get(b.runId) ?? 0) - (bordaScores.get(a.runId) ?? 0),\n )\n const mean = ranked.reduce((a, g) => a + g.aggregateScore, 0) / ranked.length\n const variance = ranked.reduce((a, g) => a + (g.aggregateScore - mean) ** 2, 0) / ranked.length\n return { winner: ranked[0]!, ranked, stdDev: Math.sqrt(variance) }\n}\n","/**\n * Process Reward Modeling — per-step rubric grading.\n *\n * A StepRubric inspects one span and returns a score + rationale.\n * PrmGrader applies an array of rubrics to every LLM span in a\n * trajectory (consumers can broaden to tool/retrieval spans via the\n * `kind` filter on each rubric).\n *\n * Why this matters: outcome-only eval (did the final artifact work?)\n * gives sparse reward — most agent turns are unattributable. PRMs\n * densify the signal so optimizers and RL fine-tuning can assign\n * credit per turn.\n */\n\nimport { TraceEmitter } from '../trace/emitter'\nimport type { JudgeSpan, Span } from '../trace/schema'\nimport type { TraceStore } from '../trace/store'\nimport { buildTrajectory, type Trajectory, type TrajectoryStep } from '../trajectory'\n\nexport interface StepContext {\n trajectory: Trajectory\n step: TrajectoryStep\n /** Steps preceding `step` in trajectory order. */\n prior: TrajectoryStep[]\n /** Steps following `step`. */\n next: TrajectoryStep[]\n}\n\nexport interface StepRubric {\n id: string\n /** Only grade spans of these kinds (default: all). */\n kinds?: Array<Span['kind']>\n /** Weight in the aggregate score. Default 1. */\n weight?: number\n /** Returns score in 0..1 + optional rationale/evidence. Return `null` to\n * skip grading (rubric doesn't apply to this step). */\n grade: (\n ctx: StepContext,\n ) => Promise<{ score: number; rationale?: string; evidence?: string } | null>\n}\n\nexport interface GradedStep {\n spanId: string\n rubricId: string\n score: number\n weight: number\n rationale?: string\n evidence?: string\n}\n\nexport interface PrmGradedTrace {\n runId: string\n steps: GradedStep[]\n /** Weighted mean of all graded steps; 0..1. */\n aggregateScore: number\n /** Number of spans graded — useful for sanity-checking coverage. */\n gradedCount: number\n /** Number of spans in the trajectory that no rubric matched. */\n ungradedCount: number\n}\n\nexport class PrmGrader {\n constructor(private rubrics: StepRubric[]) {\n if (rubrics.length === 0) throw new Error('PrmGrader: at least 1 rubric required')\n }\n\n /**\n * Grade every eligible span in a run. Emits a JudgeVerdict span for each\n * (rubric × span) verdict so the result is visible to downstream pipelines\n * (judgeAgreementView, etc.) — PRM is just \"a judge that runs per span.\"\n */\n async grade(store: TraceStore, runId: string): Promise<PrmGradedTrace> {\n const trajectory = await buildTrajectory(store, runId)\n const emitter = new TraceEmitter(store, { runId })\n const steps: GradedStep[] = []\n let ungraded = 0\n for (let i = 0; i < trajectory.steps.length; i++) {\n const step = trajectory.steps[i]!\n const ctx: StepContext = {\n trajectory,\n step,\n prior: trajectory.steps.slice(0, i),\n next: trajectory.steps.slice(i + 1),\n }\n let gradedThis = false\n for (const rubric of this.rubrics) {\n if (rubric.kinds && !rubric.kinds.includes(step.span.kind)) continue\n const verdict = await rubric.grade(ctx)\n if (verdict === null) continue\n const weight = rubric.weight ?? 1\n steps.push({\n spanId: step.span.spanId,\n rubricId: rubric.id,\n score: verdict.score,\n weight,\n rationale: verdict.rationale,\n evidence: verdict.evidence,\n })\n gradedThis = true\n // Persist the verdict as a JudgeSpan so the query pipelines see it\n await emitter.recordJudge({\n judgeId: `prm:${rubric.id}`,\n targetSpanId: step.span.spanId,\n dimension: 'step_quality',\n score: verdict.score,\n rationale: verdict.rationale,\n evidence: verdict.evidence,\n name: `prm:${rubric.id}`,\n })\n }\n if (!gradedThis) ungraded++\n }\n\n const totalWeight = steps.reduce((a, s) => a + s.weight, 0)\n const aggregateScore =\n totalWeight === 0 ? 0 : steps.reduce((a, s) => a + s.score * s.weight, 0) / totalWeight\n\n return { runId, steps, aggregateScore, gradedCount: steps.length, ungradedCount: ungraded }\n }\n}\n\n/** Helper: reads JudgeVerdict spans that PRM emitted so downstream pipelines\n * can distinguish PRM verdicts from human or top-level LLM judges. */\nexport function isPrmVerdict(verdict: JudgeSpan): boolean {\n return verdict.judgeId.startsWith('prm:')\n}\n"],"mappings":";;;;;;;;;;;;;;AAWO,SAAS,mBACd,OAAkE,CAAC,GACvD;AACZ,QAAM,MAAM,KAAK,YAAY;AAC7B,QAAM,MAAM,KAAK,YAAY;AAC7B,SAAO;AAAA,IACL,IAAI;AAAA,IACJ,OAAO,CAAC,KAAK;AAAA,IACb,QAAQ,KAAK,UAAU;AAAA,IACvB,MAAM,MAAM,EAAE,KAAK,GAAG;AACpB,YAAM,MAAM,KAAK;AACjB,YAAM,OAAO,IAAI,UAAU,IAAI;AAC/B,UAAI,QAAQ,EAAG,QAAO,EAAE,OAAO,GAAG,WAAW,eAAe;AAC5D,UAAI,MAAM;AACR,eAAO,EAAE,OAAO,KAAK,IAAI,GAAG,MAAM,GAAG,GAAG,WAAW,cAAc,GAAG,MAAM,GAAG,IAAI;AACnF,UAAI,MAAM;AACR,eAAO;AAAA,UACL,OAAO,KAAK,IAAI,GAAG,KAAK,MAAM,OAAO,GAAG;AAAA,UACxC,WAAW,cAAc,GAAG,MAAM,GAAG;AAAA,QACvC;AACF,aAAO,EAAE,OAAO,GAAG,WAAW,GAAG,GAAG,mBAAmB;AAAA,IACzD;AAAA,EACF;AACF;AAGO,SAAS,kBAAkB,OAA4B,CAAC,GAAe;AAC5E,SAAO;AAAA,IACL,IAAI;AAAA,IACJ,OAAO,CAAC,MAAM;AAAA,IACd,QAAQ,KAAK,UAAU;AAAA,IACvB,MAAM,MAAM,EAAE,KAAK,GAAG;AACpB,YAAM,OAAO,KAAK;AAClB,UAAI,KAAK,WAAW;AAClB,eAAO,EAAE,OAAO,GAAG,WAAW,UAAU,KAAK,SAAS,SAAS,GAAG;AACpE,YAAM,IAAI,KAAK;AACf,UAAI,MAAM,QAAQ,MAAM,OAAW,QAAO,EAAE,OAAO,KAAK,WAAW,eAAe;AAClF,YAAM,SAAS,OAAO,MAAM,WAAW,IAAI,KAAK,UAAU,CAAC;AAC3D,UAAI,OAAO,SAAS,EAAG,QAAO,EAAE,OAAO,KAAK,WAAW,cAAc;AACrE,aAAO,EAAE,OAAO,GAAG,WAAW,GAAG,KAAK,QAAQ,MAAM;AAAA,IACtD;AAAA,EACF;AACF;AAGO,SAAS,uBAAuB,OAA4B,CAAC,GAAe;AACjF,QAAM,SAAS,KAAK,UAAU;AAC9B,SAAO;AAAA,IACL,IAAI;AAAA,IACJ,OAAO,CAAC,MAAM;AAAA,IACd;AAAA,IACA,MAAM,MAAM,EAAE,MAAM,MAAM,GAAG;AAC3B,YAAM,OAAO,KAAK;AAClB,YAAM,eAAe,MAAM,OAAO,CAAC,MAAM;AACvC,YAAI,EAAE,KAAK,SAAS,OAAQ,QAAO;AACnC,cAAM,KAAK,EAAE;AACb,eACE,GAAG,aAAa,KAAK,YAAY,gBAAgB,GAAG,IAAI,MAAM,gBAAgB,KAAK,IAAI;AAAA,MAE3F,CAAC;AACD,UAAI,aAAa,WAAW,EAAG,QAAO,EAAE,OAAO,GAAG,WAAW,aAAa;AAC1E,aAAO;AAAA,QACL,OAAO,KAAK,IAAI,GAAG,IAAI,aAAa,SAAS,GAAG;AAAA,QAChD,WAAW,GAAG,aAAa,MAAM;AAAA,MACnC;AAAA,IACF;AAAA,EACF;AACF;AAIO,SAAS,iBAAiB,OAAgD,CAAC,GAAe;AAC/F,QAAM,SAAS,KAAK,UAAU;AAC9B,QAAM,UAAU,KAAK,WAAW;AAAA,IAC9B;AAAA,IACA;AAAA,EACF;AACA,SAAO;AAAA,IACL,IAAI;AAAA,IACJ,OAAO,CAAC,KAAK;AAAA,IACb;AAAA,IACA,MAAM,MAAM,EAAE,KAAK,GAAG;AACpB,YAAM,MAAM,KAAK;AACjB,YAAM,MAAM,IAAI,UAAU;AAC1B,YAAM,UAAU,QAAQ,KAAK,CAAC,OAAO,GAAG,KAAK,GAAG,CAAC;AACjD,aAAO,UACH,EAAE,OAAO,GAAG,WAAW,yBAAyB,IAChD,EAAE,OAAO,GAAG,WAAW,aAAa;AAAA,IAC1C;AAAA,EACF;AACF;AAIO,SAAS,0BAA0B,OAA4B,CAAC,GAAe;AACpF,SAAO;AAAA,IACL,IAAI;AAAA,IACJ,OAAO,CAAC,KAAK;AAAA,IACb,QAAQ,KAAK,UAAU;AAAA,IACvB,MAAM,MAAM,EAAE,MAAM,KAAK,GAAG;AAC1B,YAAM,MAAM,KAAK;AACjB,YAAM,WAAW,KAAK,KAAK,CAAC,MAAM,EAAE,KAAK,SAAS,MAAM;AACxD,UAAI,CAAC,SAAU,QAAO;AACtB,YAAM,WAAY,SAAS,KAAkB;AAC7C,YAAM,OAAO,IAAI,UAAU,IAAI,YAAY;AAC3C,YAAM,YAAY,IAAI,SAAS,SAAS,YAAY,CAAC;AACrD,aAAO,YACH,EAAE,OAAO,GAAG,WAAW,cAAc,QAAQ,sBAAsB,IACnE,EAAE,OAAO,KAAK,WAAW,WAAW,QAAQ,0BAA0B;AAAA,IAC5E;AAAA,EACF;AACF;AAEA,SAAS,gBAAgB,OAAwB;AAC/C,MAAI,UAAU,QAAQ,OAAO,UAAU,SAAU,QAAO,KAAK,UAAU,KAAK;AAC5E,MAAI,MAAM,QAAQ,KAAK,EAAG,QAAO,IAAI,MAAM,IAAI,eAAe,EAAE,KAAK,GAAG,CAAC;AACzE,QAAM,OAAO,OAAO,KAAK,KAAgC,EAAE,KAAK;AAChE,SAAO,IAAI,KAAK,IAAI,CAAC,MAAM,GAAG,KAAK,UAAU,CAAC,CAAC,IAAI,gBAAiB,MAAkC,CAAC,CAAC,CAAC,EAAE,EAAE,KAAK,GAAG,CAAC;AACxH;;;AC9GA,eAAsB,WACpB,OACA,QACA,QACwB;AACxB,MAAI,OAAO,WAAW,EAAG,OAAM,IAAI,MAAM,2CAA2C;AACpF,QAAM,SAAS,MAAM,QAAQ,IAAI,OAAO,IAAI,CAAC,OAAO,OAAO,MAAM,OAAO,EAAE,CAAC,CAAC;AAC5E,QAAM,SAAS,CAAC,GAAG,MAAM,EAAE,KAAK,CAAC,GAAG,MAAM,EAAE,iBAAiB,EAAE,cAAc;AAC7E,QAAM,OAAO,OAAO,OAAO,CAAC,GAAG,MAAM,IAAI,EAAE,gBAAgB,CAAC,IAAI,OAAO;AACvE,QAAM,WAAW,OAAO,OAAO,CAAC,GAAG,MAAM,KAAK,EAAE,iBAAiB,SAAS,GAAG,CAAC,IAAI,OAAO;AACzF,SAAO,EAAE,QAAQ,OAAO,CAAC,GAAI,QAAQ,QAAQ,KAAK,KAAK,QAAQ,EAAE;AACnE;AAQA,eAAsB,mBACpB,OACA,SACA,QACwB;AACxB,MAAI,QAAQ,WAAW,EAAG,OAAM,IAAI,MAAM,uCAAuC;AACjF,QAAM,YAAY,MAAM,QAAQ;AAAA,IAC9B,QAAQ,IAAI,OAAO,MAAM;AACvB,YAAM,SAAS,MAAM,QAAQ,IAAI,OAAO,IAAI,CAAC,OAAO,EAAE,MAAM,OAAO,EAAE,CAAC,CAAC;AACvE,aAAO,OAAO,KAAK,CAAC,GAAG,MAAM,EAAE,iBAAiB,EAAE,cAAc;AAAA,IAClE,CAAC;AAAA,EACH;AAEA,QAAM,cAAc,oBAAI,IAAoB;AAC5C,aAAW,WAAW,WAAW;AAC/B,YAAQ,QAAQ,CAAC,GAAG,SAAS;AAC3B,kBAAY,IAAI,EAAE,QAAQ,YAAY,IAAI,EAAE,KAAK,KAAK,MAAM,QAAQ,SAAS,KAAK;AAAA,IACpF,CAAC;AAAA,EACH;AAGA,QAAM,YAAY,UAAU,CAAC;AAC7B,QAAM,QAAQ,IAAI,IAAI,UAAU,IAAI,CAAC,MAAM,CAAC,EAAE,OAAO,CAAC,CAAC,CAAC;AACxD,QAAM,SAAS,CAAC,GAAG,MAAM,OAAO,CAAC,EAAE;AAAA,IACjC,CAAC,GAAG,OAAO,YAAY,IAAI,EAAE,KAAK,KAAK,MAAM,YAAY,IAAI,EAAE,KAAK,KAAK;AAAA,EAC3E;AACA,QAAM,OAAO,OAAO,OAAO,CAAC,GAAG,MAAM,IAAI,EAAE,gBAAgB,CAAC,IAAI,OAAO;AACvE,QAAM,WAAW,OAAO,OAAO,CAAC,GAAG,MAAM,KAAK,EAAE,iBAAiB,SAAS,GAAG,CAAC,IAAI,OAAO;AACzF,SAAO,EAAE,QAAQ,OAAO,CAAC,GAAI,QAAQ,QAAQ,KAAK,KAAK,QAAQ,EAAE;AACnE;;;ACNO,IAAM,YAAN,MAAgB;AAAA,EACrB,YAAoB,SAAuB;AAAvB;AAClB,QAAI,QAAQ,WAAW,EAAG,OAAM,IAAI,MAAM,uCAAuC;AAAA,EACnF;AAAA,EAFoB;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA,EASpB,MAAM,MAAM,OAAmB,OAAwC;AACrE,UAAM,aAAa,MAAM,gBAAgB,OAAO,KAAK;AACrD,UAAM,UAAU,IAAI,aAAa,OAAO,EAAE,MAAM,CAAC;AACjD,UAAM,QAAsB,CAAC;AAC7B,QAAI,WAAW;AACf,aAAS,IAAI,GAAG,IAAI,WAAW,MAAM,QAAQ,KAAK;AAChD,YAAM,OAAO,WAAW,MAAM,CAAC;AAC/B,YAAM,MAAmB;AAAA,QACvB;AAAA,QACA;AAAA,QACA,OAAO,WAAW,MAAM,MAAM,GAAG,CAAC;AAAA,QAClC,MAAM,WAAW,MAAM,MAAM,IAAI,CAAC;AAAA,MACpC;AACA,UAAI,aAAa;AACjB,iBAAW,UAAU,KAAK,SAAS;AACjC,YAAI,OAAO,SAAS,CAAC,OAAO,MAAM,SAAS,KAAK,KAAK,IAAI,EAAG;AAC5D,cAAM,UAAU,MAAM,OAAO,MAAM,GAAG;AACtC,YAAI,YAAY,KAAM;AACtB,cAAM,SAAS,OAAO,UAAU;AAChC,cAAM,KAAK;AAAA,UACT,QAAQ,KAAK,KAAK;AAAA,UAClB,UAAU,OAAO;AAAA,UACjB,OAAO,QAAQ;AAAA,UACf;AAAA,UACA,WAAW,QAAQ;AAAA,UACnB,UAAU,QAAQ;AAAA,QACpB,CAAC;AACD,qBAAa;AAEb,cAAM,QAAQ,YAAY;AAAA,UACxB,SAAS,OAAO,OAAO,EAAE;AAAA,UACzB,cAAc,KAAK,KAAK;AAAA,UACxB,WAAW;AAAA,UACX,OAAO,QAAQ;AAAA,UACf,WAAW,QAAQ;AAAA,UACnB,UAAU,QAAQ;AAAA,UAClB,MAAM,OAAO,OAAO,EAAE;AAAA,QACxB,CAAC;AAAA,MACH;AACA,UAAI,CAAC,WAAY;AAAA,IACnB;AAEA,UAAM,cAAc,MAAM,OAAO,CAAC,GAAG,MAAM,IAAI,EAAE,QAAQ,CAAC;AAC1D,UAAM,iBACJ,gBAAgB,IAAI,IAAI,MAAM,OAAO,CAAC,GAAG,MAAM,IAAI,EAAE,QAAQ,EAAE,QAAQ,CAAC,IAAI;AAE9E,WAAO,EAAE,OAAO,OAAO,gBAAgB,aAAa,MAAM,QAAQ,eAAe,SAAS;AAAA,EAC5F;AACF;AAIO,SAAS,aAAa,SAA6B;AACxD,SAAO,QAAQ,QAAQ,WAAW,MAAM;AAC1C;","names":[]}
|
|
1
|
+
{"version":3,"sources":["../../src/prm/builtin-rubrics.ts","../../src/prm/inference.ts","../../src/prm/rubric.ts"],"sourcesContent":["/**\n * Built-in reference rubrics. Consumers combine these with domain\n * rubrics. All are deterministic, rule-based — cheap to run + easy\n * to unit-test. LLM-based rubrics are trivially authored by\n * following the StepRubric contract.\n */\n\nimport type { LlmSpan, ToolSpan } from '../trace/schema'\nimport type { StepRubric } from './rubric'\n\n/** Penalize very short or very long assistant outputs. */\nexport function outputLengthRubric(\n args: { minChars?: number; maxChars?: number; weight?: number } = {},\n): StepRubric {\n const min = args.minChars ?? 20\n const max = args.maxChars ?? 8000\n return {\n id: 'output-length',\n kinds: ['llm'],\n weight: args.weight ?? 0.5,\n async grade({ step }) {\n const llm = step.span as LlmSpan\n const len = (llm.output ?? '').length\n if (len === 0) return { score: 0, rationale: 'empty output' }\n if (len < min)\n return { score: Math.max(0, len / min), rationale: `below min (${len} < ${min})` }\n if (len > max)\n return {\n score: Math.max(0, 1 - (len - max) / max),\n rationale: `above max (${len} > ${max})`,\n }\n return { score: 1, rationale: `${len} chars in bounds` }\n },\n }\n}\n\n/** Reward tool calls that succeeded (status='ok') with an informative result. */\nexport function toolSuccessRubric(args: { weight?: number } = {}): StepRubric {\n return {\n id: 'tool-success',\n kinds: ['tool'],\n weight: args.weight ?? 1,\n async grade({ step }) {\n const tool = step.span as ToolSpan\n if (tool.status === 'error')\n return { score: 0, rationale: `error: ${tool.error ?? 'unknown'}` }\n const r = tool.result\n if (r === null || r === undefined) return { score: 0.3, rationale: 'empty result' }\n const asText = typeof r === 'string' ? r : JSON.stringify(r)\n if (asText.length < 4) return { score: 0.5, rationale: 'tiny result' }\n return { score: 1, rationale: `${tool.toolName} ok` }\n },\n }\n}\n\n/** Penalize tool calls that duplicate a prior call with identical args. */\nexport function toolNonRedundantRubric(args: { weight?: number } = {}): StepRubric {\n const weight = args.weight ?? 0.5\n return {\n id: 'tool-non-redundant',\n kinds: ['tool'],\n weight,\n async grade({ step, prior }) {\n const tool = step.span as ToolSpan\n const priorMatches = prior.filter((p) => {\n if (p.span.kind !== 'tool') return false\n const pt = p.span as ToolSpan\n return (\n pt.toolName === tool.toolName && stableStringify(pt.args) === stableStringify(tool.args)\n )\n })\n if (priorMatches.length === 0) return { score: 1, rationale: 'novel call' }\n return {\n score: Math.max(0, 1 - priorMatches.length * 0.5),\n rationale: `${priorMatches.length} duplicate(s)`,\n }\n },\n }\n}\n\n/** Penalize LLM outputs that contain common refusal markers when a refusal\n * is NOT expected (caller inverts weight for scenarios where refusal IS expected). */\nexport function nonRefusalRubric(args: { markers?: RegExp[]; weight?: number } = {}): StepRubric {\n const weight = args.weight ?? 1\n const markers = args.markers ?? [\n /\\bi\\s+(?:can(?:not|'t)|won't|will\\s+not)\\b/i,\n /\\b(?:as\\s+an?\\s+)?ai\\b.*?\\b(?:can't|cannot)\\b/i,\n ]\n return {\n id: 'non-refusal',\n kinds: ['llm'],\n weight,\n async grade({ step }) {\n const llm = step.span as LlmSpan\n const out = llm.output ?? ''\n const refused = markers.some((re) => re.test(out))\n return refused\n ? { score: 0, rationale: 'refusal marker present' }\n : { score: 1, rationale: 'no refusal' }\n },\n }\n}\n\n/** Reward outputs that invoke the next-step tool the trajectory actually uses\n * (i.e. the LLM span announced \"I will call X\" and the following tool span IS X). */\nexport function toolIntentAlignmentRubric(args: { weight?: number } = {}): StepRubric {\n return {\n id: 'tool-intent-alignment',\n kinds: ['llm'],\n weight: args.weight ?? 0.5,\n async grade({ step, next }) {\n const llm = step.span as LlmSpan\n const nextTool = next.find((s) => s.span.kind === 'tool')\n if (!nextTool) return null\n const toolName = (nextTool.span as ToolSpan).toolName\n const out = (llm.output ?? '').toLowerCase()\n const mentioned = out.includes(toolName.toLowerCase())\n return mentioned\n ? { score: 1, rationale: `mentioned \"${toolName}\" before calling it` }\n : { score: 0.5, rationale: `called \"${toolName}\" without announcing it` }\n },\n }\n}\n\nfunction stableStringify(value: unknown): string {\n if (value === null || typeof value !== 'object') return JSON.stringify(value)\n if (Array.isArray(value)) return `[${value.map(stableStringify).join(',')}]`\n const keys = Object.keys(value as Record<string, unknown>).sort()\n return `{${keys.map((k) => `${JSON.stringify(k)}:${stableStringify((value as Record<string, unknown>)[k])}`).join(',')}}`\n}\n","/**\n * Inference-time PRM scoring — pick the best of N candidate trajectories\n * using a trained reward model (or a rule-based PRM as a proxy).\n *\n * The canonical Best-of-N pattern: generate N completions, score each\n * with a PRM, pick the winner. Here the scoring loop is framework-agnostic\n * — supply a TraceStore + PrmGrader + N run IDs → get ranking + winner.\n */\n\nimport type { TraceStore } from '../trace/store'\nimport type { PrmGradedTrace, PrmGrader } from './rubric'\n\nexport interface BestOfNResult {\n winner: PrmGradedTrace\n ranked: PrmGradedTrace[]\n /** Standard deviation of aggregate scores — small = candidates were homogenous. */\n stdDev: number\n}\n\n/** Default max concurrent grader calls. Bounds LLM fan-out so a wide ensemble\n * doesn't trigger a provider rate-limit storm. */\nconst DEFAULT_PRM_CONCURRENCY = 4\n\nexport interface PrmBestOfNOptions {\n /** Max concurrent `grader.grade` calls. Default 4. */\n concurrency?: number\n}\n\nexport async function prmBestOfN(\n store: TraceStore,\n grader: PrmGrader,\n runIds: string[],\n options: PrmBestOfNOptions = {},\n): Promise<BestOfNResult> {\n if (runIds.length === 0) throw new Error('prmBestOfN: at least 1 candidate required')\n const concurrency = Math.max(1, options.concurrency ?? DEFAULT_PRM_CONCURRENCY)\n const graded: PrmGradedTrace[] = new Array(runIds.length)\n let cursor = 0\n async function worker(): Promise<void> {\n while (true) {\n const i = cursor++\n if (i >= runIds.length) return\n graded[i] = await grader.grade(store, runIds[i]!)\n }\n }\n await Promise.all(Array.from({ length: Math.min(concurrency, runIds.length) }, () => worker()))\n const ranked = [...graded].sort((a, b) => b.aggregateScore - a.aggregateScore)\n const mean = graded.reduce((a, g) => a + g.aggregateScore, 0) / graded.length\n const variance = graded.reduce((a, g) => a + (g.aggregateScore - mean) ** 2, 0) / graded.length\n return { winner: ranked[0]!, ranked, stdDev: Math.sqrt(variance) }\n}\n\n/**\n * Weighted vote across multiple graders — use when you want a PRM ensemble\n * (e.g. rule-based + LLM-based + trained model). Each grader produces its\n * own ranking; we aggregate via rank-sum (Borda count) so no single grader\n * dominates via a different score scale.\n */\nexport async function prmEnsembleBestOfN(\n store: TraceStore,\n graders: PrmGrader[],\n runIds: string[],\n options: PrmBestOfNOptions = {},\n): Promise<BestOfNResult> {\n if (graders.length === 0) throw new Error('prmEnsembleBestOfN: at least 1 grader')\n if (runIds.length === 0) throw new Error('prmEnsembleBestOfN: at least 1 candidate required')\n const concurrency = Math.max(1, options.concurrency ?? DEFAULT_PRM_CONCURRENCY)\n\n // Flatten the (grader, runId) product and run it through a single bounded\n // pool. Nested unbounded fan-out (graders × runIds) would launch every LLM\n // call at once. allSettled isolates failures: one grader (or one runId)\n // failing doesn't void the whole ensemble.\n type Job = { graderIdx: number; runId: string }\n const jobs: Job[] = []\n for (let gi = 0; gi < graders.length; gi++) {\n for (const runId of runIds) jobs.push({ graderIdx: gi, runId })\n }\n const settled: PromiseSettledResult<PrmGradedTrace>[] = new Array(jobs.length)\n let cursor = 0\n async function worker(): Promise<void> {\n while (true) {\n const i = cursor++\n if (i >= jobs.length) return\n const job = jobs[i]!\n settled[i] = await graders[job.graderIdx]!.grade(store, job.runId)\n .then((value): PromiseSettledResult<PrmGradedTrace> => ({ status: 'fulfilled', value }))\n .catch((reason): PromiseSettledResult<PrmGradedTrace> => ({ status: 'rejected', reason }))\n }\n }\n await Promise.all(Array.from({ length: Math.min(concurrency, jobs.length) }, () => worker()))\n\n // Regroup fulfilled results into per-grader rankings. A grader contributes\n // only the candidates it successfully graded; a grader that graded nothing\n // is dropped from the vote rather than skewing it with phantom zeros.\n const perGrader: PrmGradedTrace[][] = graders.map(() => [])\n const failures: { graderIdx: number; runId: string; error: string }[] = []\n for (let i = 0; i < jobs.length; i++) {\n const job = jobs[i]!\n const result = settled[i]!\n if (result.status === 'fulfilled') perGrader[job.graderIdx]!.push(result.value)\n else {\n const error = result.reason instanceof Error ? result.reason.message : String(result.reason)\n failures.push({ graderIdx: job.graderIdx, runId: job.runId, error })\n }\n }\n const survivingGraders = perGrader.filter((ranking) => ranking.length > 0)\n if (survivingGraders.length === 0) {\n throw new Error(\n `prmEnsembleBestOfN: every grader failed on every candidate (${failures.length} call(s)). First error: ${failures[0]?.error ?? 'unknown'}`,\n )\n }\n for (const ranking of survivingGraders)\n ranking.sort((a, b) => b.aggregateScore - a.aggregateScore)\n\n // Borda: rank-sum across surviving graders.\n const bordaScores = new Map<string, number>()\n for (const ranking of survivingGraders) {\n ranking.forEach((g, rank) => {\n bordaScores.set(g.runId, (bordaScores.get(g.runId) ?? 0) + (ranking.length - rank))\n })\n }\n // Synthesize a ranking from the union of every successfully-graded trace,\n // ordered by Borda score. aggregateScore field kept for UX. Using the union\n // (not just the first grader) keeps a candidate that one grader dropped but\n // another graded.\n const byRun = new Map<string, PrmGradedTrace>()\n for (const ranking of survivingGraders) {\n for (const g of ranking) if (!byRun.has(g.runId)) byRun.set(g.runId, g)\n }\n const ranked = [...byRun.values()].sort(\n (a, b) => (bordaScores.get(b.runId) ?? 0) - (bordaScores.get(a.runId) ?? 0),\n )\n const mean = ranked.reduce((a, g) => a + g.aggregateScore, 0) / ranked.length\n const variance = ranked.reduce((a, g) => a + (g.aggregateScore - mean) ** 2, 0) / ranked.length\n return { winner: ranked[0]!, ranked, stdDev: Math.sqrt(variance) }\n}\n","/**\n * Process Reward Modeling — per-step rubric grading.\n *\n * A StepRubric inspects one span and returns a score + rationale.\n * PrmGrader applies an array of rubrics to every LLM span in a\n * trajectory (consumers can broaden to tool/retrieval spans via the\n * `kind` filter on each rubric).\n *\n * Why this matters: outcome-only eval (did the final artifact work?)\n * gives sparse reward — most agent turns are unattributable. PRMs\n * densify the signal so optimizers and RL fine-tuning can assign\n * credit per turn.\n */\n\nimport { TraceEmitter } from '../trace/emitter'\nimport type { JudgeSpan, Span } from '../trace/schema'\nimport type { TraceStore } from '../trace/store'\nimport { buildTrajectory, type Trajectory, type TrajectoryStep } from '../trajectory'\n\nexport interface StepContext {\n trajectory: Trajectory\n step: TrajectoryStep\n /** Steps preceding `step` in trajectory order. */\n prior: TrajectoryStep[]\n /** Steps following `step`. */\n next: TrajectoryStep[]\n}\n\nexport interface StepRubric {\n id: string\n /** Only grade spans of these kinds (default: all). */\n kinds?: Array<Span['kind']>\n /** Weight in the aggregate score. Default 1. */\n weight?: number\n /** Returns score in 0..1 + optional rationale/evidence. Return `null` to\n * skip grading (rubric doesn't apply to this step). */\n grade: (\n ctx: StepContext,\n ) => Promise<{ score: number; rationale?: string; evidence?: string } | null>\n}\n\nexport interface GradedStep {\n spanId: string\n rubricId: string\n score: number\n weight: number\n rationale?: string\n evidence?: string\n}\n\nexport interface PrmGradedTrace {\n runId: string\n steps: GradedStep[]\n /** Weighted mean of all graded steps; 0..1. */\n aggregateScore: number\n /** Number of spans graded — useful for sanity-checking coverage. */\n gradedCount: number\n /** Number of spans in the trajectory that no rubric matched. */\n ungradedCount: number\n}\n\nexport class PrmGrader {\n constructor(private rubrics: StepRubric[]) {\n if (rubrics.length === 0) throw new Error('PrmGrader: at least 1 rubric required')\n }\n\n /**\n * Grade every eligible span in a run. Emits a JudgeVerdict span for each\n * (rubric × span) verdict so the result is visible to downstream pipelines\n * (judgeAgreementView, etc.) — PRM is just \"a judge that runs per span.\"\n */\n async grade(store: TraceStore, runId: string): Promise<PrmGradedTrace> {\n const trajectory = await buildTrajectory(store, runId)\n const emitter = new TraceEmitter(store, { runId })\n const steps: GradedStep[] = []\n let ungraded = 0\n for (let i = 0; i < trajectory.steps.length; i++) {\n const step = trajectory.steps[i]!\n const ctx: StepContext = {\n trajectory,\n step,\n prior: trajectory.steps.slice(0, i),\n next: trajectory.steps.slice(i + 1),\n }\n let gradedThis = false\n for (const rubric of this.rubrics) {\n if (rubric.kinds && !rubric.kinds.includes(step.span.kind)) continue\n const verdict = await rubric.grade(ctx)\n if (verdict === null) continue\n const weight = rubric.weight ?? 1\n steps.push({\n spanId: step.span.spanId,\n rubricId: rubric.id,\n score: verdict.score,\n weight,\n rationale: verdict.rationale,\n evidence: verdict.evidence,\n })\n gradedThis = true\n // Persist the verdict as a JudgeSpan so the query pipelines see it\n await emitter.recordJudge({\n judgeId: `prm:${rubric.id}`,\n targetSpanId: step.span.spanId,\n dimension: 'step_quality',\n score: verdict.score,\n rationale: verdict.rationale,\n evidence: verdict.evidence,\n name: `prm:${rubric.id}`,\n })\n }\n if (!gradedThis) ungraded++\n }\n\n const totalWeight = steps.reduce((a, s) => a + s.weight, 0)\n const aggregateScore =\n totalWeight === 0 ? 0 : steps.reduce((a, s) => a + s.score * s.weight, 0) / totalWeight\n\n return { runId, steps, aggregateScore, gradedCount: steps.length, ungradedCount: ungraded }\n }\n}\n\n/** Helper: reads JudgeVerdict spans that PRM emitted so downstream pipelines\n * can distinguish PRM verdicts from human or top-level LLM judges. */\nexport function isPrmVerdict(verdict: JudgeSpan): boolean {\n return verdict.judgeId.startsWith('prm:')\n}\n"],"mappings":";;;;;;;;;;;;;;AAWO,SAAS,mBACd,OAAkE,CAAC,GACvD;AACZ,QAAM,MAAM,KAAK,YAAY;AAC7B,QAAM,MAAM,KAAK,YAAY;AAC7B,SAAO;AAAA,IACL,IAAI;AAAA,IACJ,OAAO,CAAC,KAAK;AAAA,IACb,QAAQ,KAAK,UAAU;AAAA,IACvB,MAAM,MAAM,EAAE,KAAK,GAAG;AACpB,YAAM,MAAM,KAAK;AACjB,YAAM,OAAO,IAAI,UAAU,IAAI;AAC/B,UAAI,QAAQ,EAAG,QAAO,EAAE,OAAO,GAAG,WAAW,eAAe;AAC5D,UAAI,MAAM;AACR,eAAO,EAAE,OAAO,KAAK,IAAI,GAAG,MAAM,GAAG,GAAG,WAAW,cAAc,GAAG,MAAM,GAAG,IAAI;AACnF,UAAI,MAAM;AACR,eAAO;AAAA,UACL,OAAO,KAAK,IAAI,GAAG,KAAK,MAAM,OAAO,GAAG;AAAA,UACxC,WAAW,cAAc,GAAG,MAAM,GAAG;AAAA,QACvC;AACF,aAAO,EAAE,OAAO,GAAG,WAAW,GAAG,GAAG,mBAAmB;AAAA,IACzD;AAAA,EACF;AACF;AAGO,SAAS,kBAAkB,OAA4B,CAAC,GAAe;AAC5E,SAAO;AAAA,IACL,IAAI;AAAA,IACJ,OAAO,CAAC,MAAM;AAAA,IACd,QAAQ,KAAK,UAAU;AAAA,IACvB,MAAM,MAAM,EAAE,KAAK,GAAG;AACpB,YAAM,OAAO,KAAK;AAClB,UAAI,KAAK,WAAW;AAClB,eAAO,EAAE,OAAO,GAAG,WAAW,UAAU,KAAK,SAAS,SAAS,GAAG;AACpE,YAAM,IAAI,KAAK;AACf,UAAI,MAAM,QAAQ,MAAM,OAAW,QAAO,EAAE,OAAO,KAAK,WAAW,eAAe;AAClF,YAAM,SAAS,OAAO,MAAM,WAAW,IAAI,KAAK,UAAU,CAAC;AAC3D,UAAI,OAAO,SAAS,EAAG,QAAO,EAAE,OAAO,KAAK,WAAW,cAAc;AACrE,aAAO,EAAE,OAAO,GAAG,WAAW,GAAG,KAAK,QAAQ,MAAM;AAAA,IACtD;AAAA,EACF;AACF;AAGO,SAAS,uBAAuB,OAA4B,CAAC,GAAe;AACjF,QAAM,SAAS,KAAK,UAAU;AAC9B,SAAO;AAAA,IACL,IAAI;AAAA,IACJ,OAAO,CAAC,MAAM;AAAA,IACd;AAAA,IACA,MAAM,MAAM,EAAE,MAAM,MAAM,GAAG;AAC3B,YAAM,OAAO,KAAK;AAClB,YAAM,eAAe,MAAM,OAAO,CAAC,MAAM;AACvC,YAAI,EAAE,KAAK,SAAS,OAAQ,QAAO;AACnC,cAAM,KAAK,EAAE;AACb,eACE,GAAG,aAAa,KAAK,YAAY,gBAAgB,GAAG,IAAI,MAAM,gBAAgB,KAAK,IAAI;AAAA,MAE3F,CAAC;AACD,UAAI,aAAa,WAAW,EAAG,QAAO,EAAE,OAAO,GAAG,WAAW,aAAa;AAC1E,aAAO;AAAA,QACL,OAAO,KAAK,IAAI,GAAG,IAAI,aAAa,SAAS,GAAG;AAAA,QAChD,WAAW,GAAG,aAAa,MAAM;AAAA,MACnC;AAAA,IACF;AAAA,EACF;AACF;AAIO,SAAS,iBAAiB,OAAgD,CAAC,GAAe;AAC/F,QAAM,SAAS,KAAK,UAAU;AAC9B,QAAM,UAAU,KAAK,WAAW;AAAA,IAC9B;AAAA,IACA;AAAA,EACF;AACA,SAAO;AAAA,IACL,IAAI;AAAA,IACJ,OAAO,CAAC,KAAK;AAAA,IACb;AAAA,IACA,MAAM,MAAM,EAAE,KAAK,GAAG;AACpB,YAAM,MAAM,KAAK;AACjB,YAAM,MAAM,IAAI,UAAU;AAC1B,YAAM,UAAU,QAAQ,KAAK,CAAC,OAAO,GAAG,KAAK,GAAG,CAAC;AACjD,aAAO,UACH,EAAE,OAAO,GAAG,WAAW,yBAAyB,IAChD,EAAE,OAAO,GAAG,WAAW,aAAa;AAAA,IAC1C;AAAA,EACF;AACF;AAIO,SAAS,0BAA0B,OAA4B,CAAC,GAAe;AACpF,SAAO;AAAA,IACL,IAAI;AAAA,IACJ,OAAO,CAAC,KAAK;AAAA,IACb,QAAQ,KAAK,UAAU;AAAA,IACvB,MAAM,MAAM,EAAE,MAAM,KAAK,GAAG;AAC1B,YAAM,MAAM,KAAK;AACjB,YAAM,WAAW,KAAK,KAAK,CAAC,MAAM,EAAE,KAAK,SAAS,MAAM;AACxD,UAAI,CAAC,SAAU,QAAO;AACtB,YAAM,WAAY,SAAS,KAAkB;AAC7C,YAAM,OAAO,IAAI,UAAU,IAAI,YAAY;AAC3C,YAAM,YAAY,IAAI,SAAS,SAAS,YAAY,CAAC;AACrD,aAAO,YACH,EAAE,OAAO,GAAG,WAAW,cAAc,QAAQ,sBAAsB,IACnE,EAAE,OAAO,KAAK,WAAW,WAAW,QAAQ,0BAA0B;AAAA,IAC5E;AAAA,EACF;AACF;AAEA,SAAS,gBAAgB,OAAwB;AAC/C,MAAI,UAAU,QAAQ,OAAO,UAAU,SAAU,QAAO,KAAK,UAAU,KAAK;AAC5E,MAAI,MAAM,QAAQ,KAAK,EAAG,QAAO,IAAI,MAAM,IAAI,eAAe,EAAE,KAAK,GAAG,CAAC;AACzE,QAAM,OAAO,OAAO,KAAK,KAAgC,EAAE,KAAK;AAChE,SAAO,IAAI,KAAK,IAAI,CAAC,MAAM,GAAG,KAAK,UAAU,CAAC,CAAC,IAAI,gBAAiB,MAAkC,CAAC,CAAC,CAAC,EAAE,EAAE,KAAK,GAAG,CAAC;AACxH;;;AC5GA,IAAM,0BAA0B;AAOhC,eAAsB,WACpB,OACA,QACA,QACA,UAA6B,CAAC,GACN;AACxB,MAAI,OAAO,WAAW,EAAG,OAAM,IAAI,MAAM,2CAA2C;AACpF,QAAM,cAAc,KAAK,IAAI,GAAG,QAAQ,eAAe,uBAAuB;AAC9E,QAAM,SAA2B,IAAI,MAAM,OAAO,MAAM;AACxD,MAAI,SAAS;AACb,iBAAe,SAAwB;AACrC,WAAO,MAAM;AACX,YAAM,IAAI;AACV,UAAI,KAAK,OAAO,OAAQ;AACxB,aAAO,CAAC,IAAI,MAAM,OAAO,MAAM,OAAO,OAAO,CAAC,CAAE;AAAA,IAClD;AAAA,EACF;AACA,QAAM,QAAQ,IAAI,MAAM,KAAK,EAAE,QAAQ,KAAK,IAAI,aAAa,OAAO,MAAM,EAAE,GAAG,MAAM,OAAO,CAAC,CAAC;AAC9F,QAAM,SAAS,CAAC,GAAG,MAAM,EAAE,KAAK,CAAC,GAAG,MAAM,EAAE,iBAAiB,EAAE,cAAc;AAC7E,QAAM,OAAO,OAAO,OAAO,CAAC,GAAG,MAAM,IAAI,EAAE,gBAAgB,CAAC,IAAI,OAAO;AACvE,QAAM,WAAW,OAAO,OAAO,CAAC,GAAG,MAAM,KAAK,EAAE,iBAAiB,SAAS,GAAG,CAAC,IAAI,OAAO;AACzF,SAAO,EAAE,QAAQ,OAAO,CAAC,GAAI,QAAQ,QAAQ,KAAK,KAAK,QAAQ,EAAE;AACnE;AAQA,eAAsB,mBACpB,OACA,SACA,QACA,UAA6B,CAAC,GACN;AACxB,MAAI,QAAQ,WAAW,EAAG,OAAM,IAAI,MAAM,uCAAuC;AACjF,MAAI,OAAO,WAAW,EAAG,OAAM,IAAI,MAAM,mDAAmD;AAC5F,QAAM,cAAc,KAAK,IAAI,GAAG,QAAQ,eAAe,uBAAuB;AAO9E,QAAM,OAAc,CAAC;AACrB,WAAS,KAAK,GAAG,KAAK,QAAQ,QAAQ,MAAM;AAC1C,eAAW,SAAS,OAAQ,MAAK,KAAK,EAAE,WAAW,IAAI,MAAM,CAAC;AAAA,EAChE;AACA,QAAM,UAAkD,IAAI,MAAM,KAAK,MAAM;AAC7E,MAAI,SAAS;AACb,iBAAe,SAAwB;AACrC,WAAO,MAAM;AACX,YAAM,IAAI;AACV,UAAI,KAAK,KAAK,OAAQ;AACtB,YAAM,MAAM,KAAK,CAAC;AAClB,cAAQ,CAAC,IAAI,MAAM,QAAQ,IAAI,SAAS,EAAG,MAAM,OAAO,IAAI,KAAK,EAC9D,KAAK,CAAC,WAAiD,EAAE,QAAQ,aAAa,MAAM,EAAE,EACtF,MAAM,CAAC,YAAkD,EAAE,QAAQ,YAAY,OAAO,EAAE;AAAA,IAC7F;AAAA,EACF;AACA,QAAM,QAAQ,IAAI,MAAM,KAAK,EAAE,QAAQ,KAAK,IAAI,aAAa,KAAK,MAAM,EAAE,GAAG,MAAM,OAAO,CAAC,CAAC;AAK5F,QAAM,YAAgC,QAAQ,IAAI,MAAM,CAAC,CAAC;AAC1D,QAAM,WAAkE,CAAC;AACzE,WAAS,IAAI,GAAG,IAAI,KAAK,QAAQ,KAAK;AACpC,UAAM,MAAM,KAAK,CAAC;AAClB,UAAM,SAAS,QAAQ,CAAC;AACxB,QAAI,OAAO,WAAW,YAAa,WAAU,IAAI,SAAS,EAAG,KAAK,OAAO,KAAK;AAAA,SACzE;AACH,YAAM,QAAQ,OAAO,kBAAkB,QAAQ,OAAO,OAAO,UAAU,OAAO,OAAO,MAAM;AAC3F,eAAS,KAAK,EAAE,WAAW,IAAI,WAAW,OAAO,IAAI,OAAO,MAAM,CAAC;AAAA,IACrE;AAAA,EACF;AACA,QAAM,mBAAmB,UAAU,OAAO,CAAC,YAAY,QAAQ,SAAS,CAAC;AACzE,MAAI,iBAAiB,WAAW,GAAG;AACjC,UAAM,IAAI;AAAA,MACR,+DAA+D,SAAS,MAAM,2BAA2B,SAAS,CAAC,GAAG,SAAS,SAAS;AAAA,IAC1I;AAAA,EACF;AACA,aAAW,WAAW;AACpB,YAAQ,KAAK,CAAC,GAAG,MAAM,EAAE,iBAAiB,EAAE,cAAc;AAG5D,QAAM,cAAc,oBAAI,IAAoB;AAC5C,aAAW,WAAW,kBAAkB;AACtC,YAAQ,QAAQ,CAAC,GAAG,SAAS;AAC3B,kBAAY,IAAI,EAAE,QAAQ,YAAY,IAAI,EAAE,KAAK,KAAK,MAAM,QAAQ,SAAS,KAAK;AAAA,IACpF,CAAC;AAAA,EACH;AAKA,QAAM,QAAQ,oBAAI,IAA4B;AAC9C,aAAW,WAAW,kBAAkB;AACtC,eAAW,KAAK,QAAS,KAAI,CAAC,MAAM,IAAI,EAAE,KAAK,EAAG,OAAM,IAAI,EAAE,OAAO,CAAC;AAAA,EACxE;AACA,QAAM,SAAS,CAAC,GAAG,MAAM,OAAO,CAAC,EAAE;AAAA,IACjC,CAAC,GAAG,OAAO,YAAY,IAAI,EAAE,KAAK,KAAK,MAAM,YAAY,IAAI,EAAE,KAAK,KAAK;AAAA,EAC3E;AACA,QAAM,OAAO,OAAO,OAAO,CAAC,GAAG,MAAM,IAAI,EAAE,gBAAgB,CAAC,IAAI,OAAO;AACvE,QAAM,WAAW,OAAO,OAAO,CAAC,GAAG,MAAM,KAAK,EAAE,iBAAiB,SAAS,GAAG,CAAC,IAAI,OAAO;AACzF,SAAO,EAAE,QAAQ,OAAO,CAAC,GAAI,QAAQ,QAAQ,KAAK,KAAK,QAAQ,EAAE;AACnE;;;AC1EO,IAAM,YAAN,MAAgB;AAAA,EACrB,YAAoB,SAAuB;AAAvB;AAClB,QAAI,QAAQ,WAAW,EAAG,OAAM,IAAI,MAAM,uCAAuC;AAAA,EACnF;AAAA,EAFoB;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA,EASpB,MAAM,MAAM,OAAmB,OAAwC;AACrE,UAAM,aAAa,MAAM,gBAAgB,OAAO,KAAK;AACrD,UAAM,UAAU,IAAI,aAAa,OAAO,EAAE,MAAM,CAAC;AACjD,UAAM,QAAsB,CAAC;AAC7B,QAAI,WAAW;AACf,aAAS,IAAI,GAAG,IAAI,WAAW,MAAM,QAAQ,KAAK;AAChD,YAAM,OAAO,WAAW,MAAM,CAAC;AAC/B,YAAM,MAAmB;AAAA,QACvB;AAAA,QACA;AAAA,QACA,OAAO,WAAW,MAAM,MAAM,GAAG,CAAC;AAAA,QAClC,MAAM,WAAW,MAAM,MAAM,IAAI,CAAC;AAAA,MACpC;AACA,UAAI,aAAa;AACjB,iBAAW,UAAU,KAAK,SAAS;AACjC,YAAI,OAAO,SAAS,CAAC,OAAO,MAAM,SAAS,KAAK,KAAK,IAAI,EAAG;AAC5D,cAAM,UAAU,MAAM,OAAO,MAAM,GAAG;AACtC,YAAI,YAAY,KAAM;AACtB,cAAM,SAAS,OAAO,UAAU;AAChC,cAAM,KAAK;AAAA,UACT,QAAQ,KAAK,KAAK;AAAA,UAClB,UAAU,OAAO;AAAA,UACjB,OAAO,QAAQ;AAAA,UACf;AAAA,UACA,WAAW,QAAQ;AAAA,UACnB,UAAU,QAAQ;AAAA,QACpB,CAAC;AACD,qBAAa;AAEb,cAAM,QAAQ,YAAY;AAAA,UACxB,SAAS,OAAO,OAAO,EAAE;AAAA,UACzB,cAAc,KAAK,KAAK;AAAA,UACxB,WAAW;AAAA,UACX,OAAO,QAAQ;AAAA,UACf,WAAW,QAAQ;AAAA,UACnB,UAAU,QAAQ;AAAA,UAClB,MAAM,OAAO,OAAO,EAAE;AAAA,QACxB,CAAC;AAAA,MACH;AACA,UAAI,CAAC,WAAY;AAAA,IACnB;AAEA,UAAM,cAAc,MAAM,OAAO,CAAC,GAAG,MAAM,IAAI,EAAE,QAAQ,CAAC;AAC1D,UAAM,iBACJ,gBAAgB,IAAI,IAAI,MAAM,OAAO,CAAC,GAAG,MAAM,IAAI,EAAE,QAAQ,EAAE,QAAQ,CAAC,IAAI;AAE9E,WAAO,EAAE,OAAO,OAAO,gBAAgB,aAAa,MAAM,QAAQ,eAAe,SAAS;AAAA,EAC5F;AACF;AAIO,SAAS,aAAa,SAA6B;AACxD,SAAO,QAAQ,QAAQ,WAAW,MAAM;AAC1C;","names":[]}
|
|
@@ -1,9 +1,9 @@
|
|
|
1
1
|
import { o as Mutator, I as ImprovementDriver, S as Scenario, G as Gate, k as GateResult, j as GateContext, g as CampaignResult, M as MutableSurface, c as GateDecision } from './types-BU-7W85F.js';
|
|
2
|
-
import { R as RedTeamCase } from './red-team-
|
|
2
|
+
import { R as RedTeamCase } from './red-team-BWdoyleI.js';
|
|
3
3
|
import { R as RunRecord } from './run-record-e7vj1uZQ.js';
|
|
4
4
|
import { D as Direction } from './pareto-E-pembql.js';
|
|
5
|
-
import { a as PairedBootstrapResult } from './statistics-
|
|
6
|
-
import { b as RunCampaignOptions, C as CampaignStorage } from './run-improvement-loop-
|
|
5
|
+
import { a as PairedBootstrapResult } from './statistics-CCJpTGOS.js';
|
|
6
|
+
import { b as RunCampaignOptions, C as CampaignStorage } from './run-improvement-loop-DBahB8Ax.js';
|
|
7
7
|
import { HostedClient, TraceSpanEvent } from './hosted/index.js';
|
|
8
8
|
|
|
9
9
|
/**
|
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
import { L as LlmSpan, J as JudgeSpan, R as Run, F as FailureClass, T as ToolSpan } from './schema-m0gsnbt3.js';
|
|
2
|
-
import { T as TraceStore } from './store-
|
|
2
|
+
import { T as TraceStore } from './store-BcFXE6LG.js';
|
|
3
3
|
|
|
4
4
|
/**
|
|
5
5
|
* Typed query helpers over TraceStore.
|
|
@@ -23,6 +23,7 @@ declare function aggregateLlm(spans: LlmSpan[]): {
|
|
|
23
23
|
inputTokens: number;
|
|
24
24
|
outputTokens: number;
|
|
25
25
|
cachedTokens: number;
|
|
26
|
+
reasoningTokens: number;
|
|
26
27
|
costUsd: number;
|
|
27
28
|
};
|
|
28
29
|
/** Pick the outcome's failure class when present, else derive 'success' from run status. */
|
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
import { D as DatasetSplit, c as DatasetManifest, a as DatasetScenario } from './dataset-BbGkaN2I.js';
|
|
2
|
-
import { m as GateDecision } from './summary-report-
|
|
2
|
+
import { m as GateDecision } from './summary-report-BDOFevaT.js';
|
|
3
3
|
import { R as RunRecord, a as RunSplitTag } from './run-record-e7vj1uZQ.js';
|
|
4
4
|
|
|
5
5
|
/**
|
package/dist/reporting.d.ts
CHANGED
|
@@ -1,15 +1,15 @@
|
|
|
1
1
|
export { R as RubricOutcomePair, a as RubricPredictiveValidityInput, b as RubricPredictiveValidityReport, c as RubricRanking, r as rubricPredictiveValidity } from './rubric-predictive-validity-Cy_W-hWZ.js';
|
|
2
|
-
export { B as BootstrapOptions, a as BootstrapResult, J as JudgeReplayGateArgs, R as ReleaseConfidenceAxis, b as ReleaseConfidenceAxisName, c as ReleaseConfidenceInput, d as ReleaseConfidenceIssue, e as ReleaseConfidenceMetrics, f as ReleaseConfidenceScorecard, g as ReleaseConfidenceStatus, h as ReleaseConfidenceThresholds, i as ReleaseTraceEvidence, j as RenderReleaseReportOptions, V as Verdict, k as assertReleaseConfidence, l as bootstrapCi, m as evaluateReleaseConfidence, n as judgeReplayGate, r as renderReleaseReport } from './release-report-
|
|
2
|
+
export { B as BootstrapOptions, a as BootstrapResult, J as JudgeReplayGateArgs, R as ReleaseConfidenceAxis, b as ReleaseConfidenceAxisName, c as ReleaseConfidenceInput, d as ReleaseConfidenceIssue, e as ReleaseConfidenceMetrics, f as ReleaseConfidenceScorecard, g as ReleaseConfidenceStatus, h as ReleaseConfidenceThresholds, i as ReleaseTraceEvidence, j as RenderReleaseReportOptions, V as Verdict, k as assertReleaseConfidence, l as bootstrapCi, m as evaluateReleaseConfidence, n as judgeReplayGate, r as renderReleaseReport } from './release-report-BEbWmVYj.js';
|
|
3
3
|
export { I as InterimReleaseConfidence, a as InterimReleaseConfidenceInput, P as PairedEvalueOptions, b as PairedEvalueSequence, c as PairedEvalueStep, S as SequentialDecision, e as evaluateInterimReleaseConfidence, p as pairedEvalueSequence } from './sequential-5iSVfzl2.js';
|
|
4
|
-
export { P as PairedBootstrapOptions, a as PairedBootstrapResult, b as benjaminiHochberg, p as pairedBootstrap, w as wilcoxonSignedRank } from './statistics-
|
|
5
|
-
export { G as GainDistributionBin, a as GainDistributionFigureSpec, b as GainDistributionOptions, P as ParetoFigureSpec, c as ParetoPoint, R as RESEARCH_REPORT_HARD_PAIR_FLOOR, d as ResearchReport, e as ResearchReportCandidate, f as ResearchReportDecision, g as ResearchReportMethodology, h as ResearchReportOptions, i as ResearchReportRecommendation, S as SummaryTable, j as SummaryTableOptions, k as SummaryTableRow, l as gainHistogram, p as paretoChart, r as researchReport, s as summaryTable } from './summary-report-
|
|
4
|
+
export { P as PairedBootstrapOptions, a as PairedBootstrapResult, b as benjaminiHochberg, p as pairedBootstrap, w as wilcoxonSignedRank } from './statistics-CCJpTGOS.js';
|
|
5
|
+
export { G as GainDistributionBin, a as GainDistributionFigureSpec, b as GainDistributionOptions, P as ParetoFigureSpec, c as ParetoPoint, R as RESEARCH_REPORT_HARD_PAIR_FLOOR, d as ResearchReport, e as ResearchReportCandidate, f as ResearchReportDecision, g as ResearchReportMethodology, h as ResearchReportOptions, i as ResearchReportRecommendation, S as SummaryTable, j as SummaryTableOptions, k as SummaryTableRow, l as gainHistogram, p as paretoChart, r as researchReport, s as summaryTable } from './summary-report-BDOFevaT.js';
|
|
6
6
|
import './run-record-e7vj1uZQ.js';
|
|
7
7
|
import './errors-CzMUYo7b.js';
|
|
8
8
|
import './schema-m0gsnbt3.js';
|
|
9
9
|
import './outcome-store-rnXLEqSn.js';
|
|
10
10
|
import './dataset-BbGkaN2I.js';
|
|
11
11
|
import './judge-calibration-0p2QcWNE.js';
|
|
12
|
-
import './types-
|
|
12
|
+
import './types-C7DGg5ex.js';
|
|
13
13
|
import '@tangle-network/tcloud';
|
|
14
|
-
import './failure-cluster-
|
|
15
|
-
import './store-
|
|
14
|
+
import './failure-cluster-DH9Flgcf.js';
|
|
15
|
+
import './store-BcFXE6LG.js';
|
package/dist/reporting.js
CHANGED
|
@@ -4,7 +4,7 @@ import {
|
|
|
4
4
|
evaluateReleaseConfidence,
|
|
5
5
|
judgeReplayGate,
|
|
6
6
|
renderReleaseReport
|
|
7
|
-
} from "./chunk-
|
|
7
|
+
} from "./chunk-QTMYW64F.js";
|
|
8
8
|
import {
|
|
9
9
|
rubricPredictiveValidity
|
|
10
10
|
} from "./chunk-YRZ4M5GS.js";
|
|
@@ -18,12 +18,12 @@ import {
|
|
|
18
18
|
paretoChart,
|
|
19
19
|
researchReport,
|
|
20
20
|
summaryTable
|
|
21
|
-
} from "./chunk-
|
|
21
|
+
} from "./chunk-URWVKRCS.js";
|
|
22
22
|
import {
|
|
23
23
|
benjaminiHochberg,
|
|
24
24
|
pairedBootstrap,
|
|
25
25
|
wilcoxonSignedRank
|
|
26
|
-
} from "./chunk-
|
|
26
|
+
} from "./chunk-56GC6VYO.js";
|
|
27
27
|
import "./chunk-VSMTAMNK.js";
|
|
28
28
|
import "./chunk-3BFEG2F6.js";
|
|
29
29
|
import "./chunk-PZ5AY32C.js";
|
|
@@ -1,11 +1,11 @@
|
|
|
1
1
|
import { a as RunSplitTag, b as RunTokenUsage, c as RunJudgeMetadata, J as JudgeScoresRecord, A as AgentProfileCell, d as AgentProfileCellInput, R as RunRecord } from './run-record-e7vj1uZQ.js';
|
|
2
|
-
import { L as LlmClientOptions, a as LlmRouteRequirements } from './llm-client-
|
|
3
|
-
import { h as ResearchReportOptions, d as ResearchReport, m as GateDecision } from './summary-report-
|
|
4
|
-
import { T as TraceEmitter, R as RunCompleteHook } from './emitter-
|
|
5
|
-
import { R as RunIntegrityExpectations, a as RunIntegrityReport } from './integrity-
|
|
2
|
+
import { L as LlmClientOptions, a as LlmRouteRequirements } from './llm-client-Bj7g0rqu.js';
|
|
3
|
+
import { h as ResearchReportOptions, d as ResearchReport, m as GateDecision } from './summary-report-BDOFevaT.js';
|
|
4
|
+
import { T as TraceEmitter, R as RunCompleteHook } from './emitter-C2rqGH_l.js';
|
|
5
|
+
import { R as RunIntegrityExpectations, a as RunIntegrityReport } from './integrity-D2t12mMw.js';
|
|
6
6
|
import { R as RawProviderSink } from './raw-provider-sink-C46HDghv.js';
|
|
7
7
|
import { F as FailureClass } from './schema-m0gsnbt3.js';
|
|
8
|
-
import { T as TraceStore } from './store-
|
|
8
|
+
import { T as TraceStore } from './store-BcFXE6LG.js';
|
|
9
9
|
|
|
10
10
|
/**
|
|
11
11
|
* EvalCampaign — opinionated matrix runner that wires the four
|
package/dist/rl.d.ts
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
export { O as OffPolicyEstimate, a as OffPolicyOptions, b as OffPolicyTrajectory, d as doublyRobust, i as inverseProbabilityWeighting, o as offPolicyEstimateAll, s as selfNormalizedImportanceWeighting } from './off-policy-DiwuKKg7.js';
|
|
2
|
-
import { P as PreferenceExtractionReport, D as DpoExportRow, G as GrpoExportRow, S as SftExportRow, E as ExtractPreferencesOptions, a as DpoLookups, b as GrpoLookups, c as SftLookups } from './corpus-
|
|
3
|
-
export { d as CorpusAppendResult, C as CorpusRecord, e as DatasetFormat, f as ExtractStepRewardsOptions, H as HarvestOptions, g as PreferenceStrategy, h as PreferenceTriple, i as PrmExportRow, j as PrmLookups, k as PrmTrainingTriple, R as RewardKind, l as RewardStats, m as RlDatasetBundle, n as RlDatasetConfig, o as RlDatasetManifest, p as RlDatasetStats, q as RunwiseStepSummary, r as StepReward, s as StepRewardJsonlRow, t as StepScorer, u as appendToCorpus, v as buildDatasetFromCorpus, w as buildRlDataset, x as datasheetToMarkdown, y as extractPreferences, z as extractStepRewards, A as prmTrainingPairs, B as readCorpus, F as runwiseStepRewardSummary, I as stepRewardsToJsonl, J as toAnthropicFormat, K as toDpoJsonl, L as toDpoRows, M as toGrpoJsonl, N as toGrpoRows, O as toPrmJsonl, Q as toPrmRows, T as toSftJsonl, U as toSftRows, V as toTRLFormat } from './corpus-
|
|
2
|
+
import { P as PreferenceExtractionReport, D as DpoExportRow, G as GrpoExportRow, S as SftExportRow, E as ExtractPreferencesOptions, a as DpoLookups, b as GrpoLookups, c as SftLookups } from './corpus-B8A4BDR3.js';
|
|
3
|
+
export { d as CorpusAppendResult, C as CorpusRecord, e as DatasetFormat, f as ExtractStepRewardsOptions, H as HarvestOptions, g as PreferenceStrategy, h as PreferenceTriple, i as PrmExportRow, j as PrmLookups, k as PrmTrainingTriple, R as RewardKind, l as RewardStats, m as RlDatasetBundle, n as RlDatasetConfig, o as RlDatasetManifest, p as RlDatasetStats, q as RunwiseStepSummary, r as StepReward, s as StepRewardJsonlRow, t as StepScorer, u as appendToCorpus, v as buildDatasetFromCorpus, w as buildRlDataset, x as datasheetToMarkdown, y as extractPreferences, z as extractStepRewards, A as prmTrainingPairs, B as readCorpus, F as runwiseStepRewardSummary, I as stepRewardsToJsonl, J as toAnthropicFormat, K as toDpoJsonl, L as toDpoRows, M as toGrpoJsonl, N as toGrpoRows, O as toPrmJsonl, Q as toPrmRows, T as toSftJsonl, U as toSftRows, V as toTRLFormat } from './corpus-B8A4BDR3.js';
|
|
4
4
|
import { g as CampaignResult } from './types-BU-7W85F.js';
|
|
5
5
|
import { a as RunSplitTag, R as RunRecord } from './run-record-e7vj1uZQ.js';
|
|
6
6
|
import { a as VerificationReport } from './multi-layer-verifier-DUZXrPDA.js';
|
|
@@ -8,19 +8,19 @@ export { A as AdversarialMutation, a as AdversarialScenario, b as AdversarialSea
|
|
|
8
8
|
import { b as OutcomeStore } from './outcome-store-rnXLEqSn.js';
|
|
9
9
|
export { D as DeploymentOutcome, F as FileSystemOutcomeStore, a as FileSystemOutcomeStoreOptions, I as InMemoryOutcomeStore } from './outcome-store-rnXLEqSn.js';
|
|
10
10
|
import { b as RubricPredictiveValidityReport } from './rubric-predictive-validity-Cy_W-hWZ.js';
|
|
11
|
-
import { R as Researcher, F as FailureMode, S as SteeringChange, E as ExperimentPlan, a as ExperimentResult, b as EvalCampaignResult, c as EvalCampaignOptions } from './researcher-
|
|
12
|
-
export { r as runEvalCampaign } from './researcher-
|
|
11
|
+
import { R as Researcher, F as FailureMode, S as SteeringChange, E as ExperimentPlan, a as ExperimentResult, b as EvalCampaignResult, c as EvalCampaignOptions } from './researcher-B0C2_fVO.js';
|
|
12
|
+
export { r as runEvalCampaign } from './researcher-B0C2_fVO.js';
|
|
13
13
|
import { I as InterimReleaseConfidence } from './sequential-5iSVfzl2.js';
|
|
14
14
|
import './schema-m0gsnbt3.js';
|
|
15
|
-
import './store-
|
|
15
|
+
import './store-BcFXE6LG.js';
|
|
16
16
|
import './errors-CzMUYo7b.js';
|
|
17
17
|
import './verdict-C9MlYujm.js';
|
|
18
|
-
import './llm-client-
|
|
18
|
+
import './llm-client-Bj7g0rqu.js';
|
|
19
19
|
import './raw-provider-sink-C46HDghv.js';
|
|
20
|
-
import './summary-report-
|
|
21
|
-
import './failure-cluster-
|
|
22
|
-
import './emitter-
|
|
23
|
-
import './integrity-
|
|
20
|
+
import './summary-report-BDOFevaT.js';
|
|
21
|
+
import './failure-cluster-DH9Flgcf.js';
|
|
22
|
+
import './emitter-C2rqGH_l.js';
|
|
23
|
+
import './integrity-D2t12mMw.js';
|
|
24
24
|
|
|
25
25
|
/**
|
|
26
26
|
* Test-time compute scaling curves.
|
package/dist/rl.js
CHANGED
|
@@ -16,19 +16,19 @@ import {
|
|
|
16
16
|
} from "./chunk-3RF76KTD.js";
|
|
17
17
|
import {
|
|
18
18
|
runEvalCampaign
|
|
19
|
-
} from "./chunk-
|
|
20
|
-
import "./chunk-
|
|
19
|
+
} from "./chunk-KW53MSA5.js";
|
|
20
|
+
import "./chunk-CWNP4DV4.js";
|
|
21
21
|
import {
|
|
22
22
|
rubricPredictiveValidity
|
|
23
23
|
} from "./chunk-YRZ4M5GS.js";
|
|
24
24
|
import {
|
|
25
25
|
evaluateInterimReleaseConfidence
|
|
26
26
|
} from "./chunk-MAZ26DC7.js";
|
|
27
|
-
import "./chunk-
|
|
27
|
+
import "./chunk-URWVKRCS.js";
|
|
28
28
|
import {
|
|
29
29
|
benjaminiHochberg,
|
|
30
30
|
wilcoxonSignedRank
|
|
31
|
-
} from "./chunk-
|
|
31
|
+
} from "./chunk-56GC6VYO.js";
|
|
32
32
|
import {
|
|
33
33
|
observationsFromRunRecords,
|
|
34
34
|
thompsonCurriculum,
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
import { S as Span, J as JudgeSpan } from './schema-m0gsnbt3.js';
|
|
2
|
-
import { T as TraceStore } from './store-
|
|
3
|
-
import { T as Trajectory, a as TrajectoryStep } from './trajectory-
|
|
2
|
+
import { T as TraceStore } from './store-BcFXE6LG.js';
|
|
3
|
+
import { T as Trajectory, a as TrajectoryStep } from './trajectory-2TkpSEVh.js';
|
|
4
4
|
|
|
5
5
|
/**
|
|
6
6
|
* Process Reward Modeling — per-step rubric grading.
|
|
@@ -1,10 +1,10 @@
|
|
|
1
1
|
import {
|
|
2
2
|
runCampaign
|
|
3
|
-
} from "./chunk-
|
|
4
|
-
import "./chunk-
|
|
3
|
+
} from "./chunk-2K6UUZ7P.js";
|
|
4
|
+
import "./chunk-56GC6VYO.js";
|
|
5
5
|
import "./chunk-3BFEG2F6.js";
|
|
6
6
|
import "./chunk-PZ5AY32C.js";
|
|
7
7
|
export {
|
|
8
8
|
runCampaign
|
|
9
9
|
};
|
|
10
|
-
//# sourceMappingURL=run-campaign-
|
|
10
|
+
//# sourceMappingURL=run-campaign-7WNXMDSN.js.map
|
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
import { R as Run, S as Span, a as TraceEvent, A as Artifact, B as BudgetLedgerEntry } from './schema-m0gsnbt3.js';
|
|
2
|
-
import { T as TraceStore } from './store-
|
|
2
|
+
import { T as TraceStore } from './store-BcFXE6LG.js';
|
|
3
3
|
|
|
4
4
|
interface RunScore {
|
|
5
5
|
success: number;
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import { L as LlmClientOptions } from './llm-client-
|
|
1
|
+
import { L as LlmClientOptions } from './llm-client-Bj7g0rqu.js';
|
|
2
2
|
import { I as ImprovementDriver, S as Scenario, g as CampaignResult, k as GateResult, D as DispatchFn, a as JudgeConfig, L as LabeledScenarioStore, h as CampaignTraceWriter, m as GenerationRecord, M as MutableSurface, P as ParetoParent, G as Gate } from './types-BU-7W85F.js';
|
|
3
3
|
|
|
4
4
|
/**
|
|
@@ -1,9 +1,9 @@
|
|
|
1
1
|
import { AxAIService } from '@ax-llm/ax';
|
|
2
2
|
import { z } from 'zod';
|
|
3
|
-
import { c as AnalystFinding, A as Analyst, a as AnalystContext } from './types-
|
|
4
|
-
import { T as TraceAnalystKindSpec } from './kind-factory-
|
|
3
|
+
import { c as AnalystFinding, A as Analyst, a as AnalystContext } from './types-Ce17tDlG.js';
|
|
4
|
+
import { T as TraceAnalystKindSpec } from './kind-factory-0BhLSI27.js';
|
|
5
5
|
import { a as TraceAnalystSpan } from './store-C1YxJDEK.js';
|
|
6
|
-
import { L as LlmClientOptions } from './llm-client-
|
|
6
|
+
import { L as LlmClientOptions } from './llm-client-Bj7g0rqu.js';
|
|
7
7
|
import { S as Severity } from './multi-layer-verifier-DUZXrPDA.js';
|
|
8
8
|
|
|
9
9
|
interface CreateAnalystAiConfig {
|
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
import { d as ContinuousAgreementOptions, C as ContinuousAgreement } from './judge-calibration-0p2QcWNE.js';
|
|
2
|
-
import { J as JudgeScore } from './types-
|
|
2
|
+
import { J as JudgeScore } from './types-C7DGg5ex.js';
|
|
3
3
|
|
|
4
4
|
/** Identity: dimensions already follow "higher = better" by prompt convention
|
|
5
5
|
* (inverted dims like hallucination are scored 10 = best at the source). */
|
|
@@ -199,6 +199,39 @@ declare function pairedMde(opts: {
|
|
|
199
199
|
power?: number;
|
|
200
200
|
twoSided?: boolean;
|
|
201
201
|
}): number;
|
|
202
|
+
/**
|
|
203
|
+
* Number of paired observations needed for a McNemar test to reach a target
|
|
204
|
+
* power — the pre-registration companion to {@link mcnemar}. Parametrised by the
|
|
205
|
+
* expected discordant-cell probabilities `p10` (P[treatment wins on a pair]) and
|
|
206
|
+
* `p01` (P[control wins]); concordant pairs carry no information, so the count
|
|
207
|
+
* is driven entirely by the discordant rate. Lachin's (1992) asymptotic normal
|
|
208
|
+
* approximation: with discordant rate `pDisc = p10 + p01` and marginal effect
|
|
209
|
+
* `δ = p10 − p01`,
|
|
210
|
+
* n = ( z_{1-α/2}·√pDisc + z_{1-β}·√(pDisc − δ²) )² / δ².
|
|
211
|
+
* Returns Infinity when there is no effect (p10 === p01). Asymptotic — at the
|
|
212
|
+
* tiny discordant counts where the exact {@link mcnemar} differs from the normal
|
|
213
|
+
* approximation, treat the result as a lower bound and prefer the discordant-pair
|
|
214
|
+
* floor.
|
|
215
|
+
*/
|
|
216
|
+
declare function mcnemarRequiredN(opts: {
|
|
217
|
+
p10: number;
|
|
218
|
+
p01: number;
|
|
219
|
+
alpha?: number;
|
|
220
|
+
power?: number;
|
|
221
|
+
twoSided?: boolean;
|
|
222
|
+
}): number;
|
|
223
|
+
/**
|
|
224
|
+
* Power of a McNemar test at a given number of paired observations, the inverse
|
|
225
|
+
* of {@link mcnemarRequiredN} (same Lachin asymptotic model, same parameters).
|
|
226
|
+
* Returns a value in [0, 1]; equals `alpha` when there is no effect.
|
|
227
|
+
*/
|
|
228
|
+
declare function mcnemarPower(opts: {
|
|
229
|
+
p10: number;
|
|
230
|
+
p01: number;
|
|
231
|
+
nPairs: number;
|
|
232
|
+
alpha?: number;
|
|
233
|
+
twoSided?: boolean;
|
|
234
|
+
}): number;
|
|
202
235
|
/** Bonferroni adjustment: multiply every p-value by the test count, clamp at 1. */
|
|
203
236
|
declare function bonferroni(pValues: number[], alpha?: number): {
|
|
204
237
|
adjusted: number[];
|
|
@@ -245,6 +278,88 @@ interface PairedBootstrapOptions {
|
|
|
245
278
|
* gain is real at the confidence level. Throws on unequal sample sizes.
|
|
246
279
|
*/
|
|
247
280
|
declare function pairedBootstrap(before: number[], after: number[], opts?: PairedBootstrapOptions): PairedBootstrapResult;
|
|
281
|
+
/** A binomial proportion estimate with a confidence interval. */
|
|
282
|
+
interface ProportionInterval {
|
|
283
|
+
/** Point estimate successes / n (0 when n = 0). */
|
|
284
|
+
estimate: number;
|
|
285
|
+
/** Lower bound, clamped to [0, 1]. */
|
|
286
|
+
lower: number;
|
|
287
|
+
/** Upper bound, clamped to [0, 1]. */
|
|
288
|
+
upper: number;
|
|
289
|
+
}
|
|
290
|
+
/**
|
|
291
|
+
* Wilson score interval for a binomial proportion. Correct at small n and near
|
|
292
|
+
* 0/1, where the normal (Wald) approximation produces bounds outside [0, 1] and
|
|
293
|
+
* understates coverage. Use this for any pass-rate / hit-rate / realness-rate
|
|
294
|
+
* CI — the continuous `confidenceInterval` assumes the wrong distribution for a
|
|
295
|
+
* proportion. `n = 0 ⇒ {0, 0, 0}`.
|
|
296
|
+
*/
|
|
297
|
+
declare function wilson(successes: number, n: number, confidence?: number): ProportionInterval;
|
|
298
|
+
/** Result of a McNemar paired-binary significance test. */
|
|
299
|
+
interface McNemarResult {
|
|
300
|
+
/** Total paired observations. */
|
|
301
|
+
n: number;
|
|
302
|
+
/** Discordant pairs (b + c) — the only ones that carry signal. */
|
|
303
|
+
nDiscordant: number;
|
|
304
|
+
/** Pairs where treatment succeeded and control failed ("newly correct"). */
|
|
305
|
+
b: number;
|
|
306
|
+
/** Pairs where control succeeded and treatment failed ("newly wrong"). */
|
|
307
|
+
c: number;
|
|
308
|
+
/** Continuity-corrected chi-square statistic (reference; exact p drives the call). */
|
|
309
|
+
statistic: number;
|
|
310
|
+
/** Two-sided p-value. Exact (binomial sign test on discordant pairs). */
|
|
311
|
+
pValue: number;
|
|
312
|
+
}
|
|
313
|
+
/**
|
|
314
|
+
* McNemar's test for paired binary outcomes — the correct significance test for
|
|
315
|
+
* "does treatment change the success rate vs control on the SAME items". Only
|
|
316
|
+
* discordant pairs (one arm right, the other wrong) carry information; concordant
|
|
317
|
+
* pairs are uninformative, so a paired t-test / two-proportion z-test on the raw
|
|
318
|
+
* rates is wrong here. The p-value is exact: under H0 the b "treatment-wins" are
|
|
319
|
+
* Binomial(b + c, 0.5), so the two-sided p is the doubled binomial tail — correct
|
|
320
|
+
* at the small discordant counts typical of eval runs (no continuity-corrected
|
|
321
|
+
* chi-square approximation needed, though it is returned as `statistic` for
|
|
322
|
+
* reference). Inputs are paired 0/1 (or boolean) arrays, control first to match
|
|
323
|
+
* the module's (before, after) convention. Throws on unequal lengths.
|
|
324
|
+
*/
|
|
325
|
+
declare function mcnemar(control: ArrayLike<number | boolean>, treatment: ArrayLike<number | boolean>): McNemarResult;
|
|
326
|
+
/** A paired binary effect size (treatment rate − control rate) with a CI. */
|
|
327
|
+
interface RiskDifferenceResult {
|
|
328
|
+
/** Total paired observations. */
|
|
329
|
+
n: number;
|
|
330
|
+
/** Discordant pairs: treatment-win count. */
|
|
331
|
+
b: number;
|
|
332
|
+
/** Discordant pairs: control-win count. */
|
|
333
|
+
c: number;
|
|
334
|
+
/** Paired risk difference p(treatment) − p(control) = (b − c) / n. */
|
|
335
|
+
riskDifference: number;
|
|
336
|
+
/** Lower bound of the CI, clamped to [-1, 1]. */
|
|
337
|
+
lower: number;
|
|
338
|
+
/** Upper bound of the CI, clamped to [-1, 1]. */
|
|
339
|
+
upper: number;
|
|
340
|
+
/** Confidence level used. */
|
|
341
|
+
confidence: number;
|
|
342
|
+
}
|
|
343
|
+
/**
|
|
344
|
+
* Paired risk difference (the effect-size companion to {@link mcnemar}): the
|
|
345
|
+
* change in success rate p(treatment) − p(control) on matched items, which for
|
|
346
|
+
* paired binary data equals (b − c) / n. The CI uses the paired variance from
|
|
347
|
+
* the discordant counts, not the independent-samples formula (which overstates
|
|
348
|
+
* the interval by ignoring the pairing). Inputs are paired 0/1 (or boolean)
|
|
349
|
+
* arrays, control first. Throws on unequal lengths.
|
|
350
|
+
*/
|
|
351
|
+
declare function pairedRiskDifference(control: ArrayLike<number | boolean>, treatment: ArrayLike<number | boolean>, confidence?: number): RiskDifferenceResult;
|
|
352
|
+
/**
|
|
353
|
+
* Unbiased pass@k for code generation (Chen et al. 2021, "Evaluating Large
|
|
354
|
+
* Language Models Trained on Code"). Given `n` independent samples for one
|
|
355
|
+
* problem of which `c` pass, the probability that at least one of a random k of
|
|
356
|
+
* them passes is 1 − C(n−c, k) / C(n, k). Estimating pass@k as "did any of the
|
|
357
|
+
* first k pass" is biased high at small n; this is the variance-reduced estimator
|
|
358
|
+
* averaged implicitly over all k-subsets. Average the per-problem values across
|
|
359
|
+
* the suite for the corpus pass@k. Computed in the numerically stable product
|
|
360
|
+
* form. Requires 1 ≤ k ≤ n and 0 ≤ c ≤ n.
|
|
361
|
+
*/
|
|
362
|
+
declare function passAtK(n: number, c: number, k: number): number;
|
|
248
363
|
interface EProcessOptions {
|
|
249
364
|
/** Type-I error budget. The process decides when wealth ≥ 1/alpha
|
|
250
365
|
* (Ville's inequality). Default 0.05. */
|
|
@@ -315,4 +430,4 @@ declare function eProcess(opts?: EProcessOptions): EProcess;
|
|
|
315
430
|
* gate verdicts is non-reproducible by construction). */
|
|
316
431
|
declare function mulberry32(seed: number): () => number;
|
|
317
432
|
|
|
318
|
-
export {
|
|
433
|
+
export { mulberry32 as A, normalizeScores as B, type CorpusAgreementReport as C, pairedMde as D, type EProcessState as E, pairedRiskDifference as F, pairedTTest as G, partialCredit as H, passAtK as I, requiredSampleSize as J, weightedComposite as K, weightedMean as L, type McNemarResult as M, wilson as N, type PairedBootstrapOptions as P, type RiskDifferenceResult as R, type WeightedCompositeInput as W, type PairedBootstrapResult as a, benjaminiHochberg as b, type CliffsMagnitude as c, type CorpusAgreementOptions as d, type CorpusAgreementPerDimension as e, type CorpusScoreRecord as f, type EProcess as g, type EProcessOptions as h, type EProcessStep as i, type ProportionInterval as j, type WeightedCompositeResult as k, bonferroni as l, cliffsDelta as m, cohensD as n, confidenceInterval as o, pairedBootstrap as p, corpusInterRaterAgreement as q, corpusInterRaterAgreementFromJudgeScores as r, eProcess as s, interRaterReliability as t, interpretCliffs as u, mannWhitneyU as v, wilcoxonSignedRank as w, mcnemar as x, mcnemarPower as y, mcnemarRequiredN as z };
|
|
@@ -77,12 +77,27 @@ declare class FileSystemTraceStore implements TraceStore {
|
|
|
77
77
|
private maxBytes;
|
|
78
78
|
/** Lazy in-memory index for queries — populated on first read. */
|
|
79
79
|
private index?;
|
|
80
|
-
|
|
80
|
+
/** Memoized index build — concurrent first reads share one build, and an
|
|
81
|
+
* append racing an in-flight load awaits this so its row isn't lost. */
|
|
82
|
+
private indexPromise?;
|
|
83
|
+
/** Strictly-increasing rollover stamp. Date.now() alone collides when two
|
|
84
|
+
* rollovers land in the same millisecond, overwriting a rolled file. */
|
|
85
|
+
private lastRolloverStamp;
|
|
86
|
+
/**
|
|
87
|
+
* Per-file append serialization. stat → conditional-rename → appendFile is a
|
|
88
|
+
* read-modify-write on the active file; without a lock two concurrent appends
|
|
89
|
+
* to the same `name` can both pass the size check (exceeding maxBytes) or one
|
|
90
|
+
* can append to a file the other just renamed away. Each entry chains the
|
|
91
|
+
* next append behind the prior one for that file.
|
|
92
|
+
*/
|
|
93
|
+
private appendLocks;
|
|
81
94
|
constructor(options: FileSystemTraceStoreOptions);
|
|
82
95
|
private ensureDir;
|
|
83
96
|
private append;
|
|
97
|
+
private appendLocked;
|
|
84
98
|
private insertInto;
|
|
85
99
|
private load;
|
|
100
|
+
private buildIndex;
|
|
86
101
|
appendRun(run: Run): Promise<void>;
|
|
87
102
|
updateRun(runId: string, patch: Partial<Run>): Promise<void>;
|
|
88
103
|
appendSpan(span: Span): Promise<void>;
|
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
import { R as RunRecord } from './run-record-e7vj1uZQ.js';
|
|
2
|
-
import { F as FailureClusterReport } from './failure-cluster-
|
|
2
|
+
import { F as FailureClusterReport } from './failure-cluster-DH9Flgcf.js';
|
|
3
3
|
|
|
4
4
|
/**
|
|
5
5
|
* HeldOutGate — first-class held-out paired-delta promotion gate.
|
|
@@ -1,6 +1,6 @@
|
|
|
1
|
-
import { T as TraceEmitter } from './emitter-
|
|
1
|
+
import { T as TraceEmitter } from './emitter-C2rqGH_l.js';
|
|
2
2
|
import { R as Run, F as FailureClass } from './schema-m0gsnbt3.js';
|
|
3
|
-
import { T as TraceStore } from './store-
|
|
3
|
+
import { T as TraceStore } from './store-BcFXE6LG.js';
|
|
4
4
|
|
|
5
5
|
/**
|
|
6
6
|
* SandboxHarness — executes a scenario in an isolated environment and
|
|
@@ -27,6 +27,12 @@ interface HarnessConfig {
|
|
|
27
27
|
cwd?: string;
|
|
28
28
|
/** Max wall-clock per phase in ms. Default 10 minutes. */
|
|
29
29
|
timeoutMs?: number;
|
|
30
|
+
/**
|
|
31
|
+
* Cap on captured stdout+stderr bytes per phase. A runaway process can
|
|
32
|
+
* otherwise grow the in-memory buffer without bound. Once hit, further
|
|
33
|
+
* output is dropped and `outputTruncated` is set. Default 16 MiB.
|
|
34
|
+
*/
|
|
35
|
+
maxOutputBytes?: number;
|
|
30
36
|
/** Image for the docker driver. */
|
|
31
37
|
image?: string;
|
|
32
38
|
/** Extra env vars (validated; shell-escaped). */
|
|
@@ -49,6 +55,19 @@ interface SandboxResult {
|
|
|
49
55
|
wallMs: number;
|
|
50
56
|
testsTotal?: number;
|
|
51
57
|
testsPassed?: number;
|
|
58
|
+
/**
|
|
59
|
+
* True when the process was killed because it exceeded `timeoutMs`. A
|
|
60
|
+
* SIGKILLed child can still close with exit code 0; callers MUST treat
|
|
61
|
+
* a timed-out phase as a hard failure regardless of `exitCode`, never
|
|
62
|
+
* as a pass. `undefined`/`false` means the process completed on its own.
|
|
63
|
+
*/
|
|
64
|
+
killedByTimeout?: boolean;
|
|
65
|
+
/**
|
|
66
|
+
* True when captured stdout/stderr hit `maxOutputBytes` and further
|
|
67
|
+
* output was dropped. The result is still returned (the process was
|
|
68
|
+
* not killed for this), but downstream parsers see truncated text.
|
|
69
|
+
*/
|
|
70
|
+
outputTruncated?: boolean;
|
|
52
71
|
}
|
|
53
72
|
interface SandboxDriver {
|
|
54
73
|
id: string;
|
package/dist/traces.d.ts
CHANGED
|
@@ -1,12 +1,12 @@
|
|
|
1
1
|
import { N as NotFoundError, R as ReplayError } from './errors-CzMUYo7b.js';
|
|
2
2
|
import { P as ProviderRedactor, R as RawProviderSink, d as RawProviderEvent } from './raw-provider-sink-C46HDghv.js';
|
|
3
3
|
export { F as FileSystemRawProviderSink, a as FileSystemRawProviderSinkOptions, I as InMemoryRawProviderSink, b as InMemoryRawProviderSinkOptions, N as NoopRawProviderSink, c as RawProviderDirection, e as RawProviderSinkFilter, f as defaultProviderRedactor, p as providerFromBaseUrl } from './raw-provider-sink-C46HDghv.js';
|
|
4
|
-
import { a as RunCompleteHookContext, R as RunCompleteHook } from './emitter-
|
|
5
|
-
export { S as SpanHandle, T as TraceEmitter, b as TraceEmitterOptions, l as llmSpanFromProvider } from './emitter-
|
|
6
|
-
export { b as RunIntegrityError, R as RunIntegrityExpectations, c as RunIntegrityIssue, d as RunIntegrityIssueCode, a as RunIntegrityReport, e as assertRunCaptured, t as throwIfRunIncomplete } from './integrity-
|
|
7
|
-
import { T as TraceStore } from './store-
|
|
8
|
-
export { E as EventFilter, F as FileSystemTraceStore, a as FileSystemTraceStoreOptions, I as InMemoryTraceStore, R as RunFilter, S as SpanFilter } from './store-
|
|
9
|
-
export { a as aggregateLlm, b as argHash, g as groupBy, j as judgeSpans, l as llmSpans, r as runFailureClass, c as runsForScenario, t as toolSpans } from './query-
|
|
4
|
+
import { a as RunCompleteHookContext, R as RunCompleteHook } from './emitter-C2rqGH_l.js';
|
|
5
|
+
export { S as SpanHandle, T as TraceEmitter, b as TraceEmitterOptions, l as llmSpanFromProvider } from './emitter-C2rqGH_l.js';
|
|
6
|
+
export { b as RunIntegrityError, R as RunIntegrityExpectations, c as RunIntegrityIssue, d as RunIntegrityIssueCode, a as RunIntegrityReport, e as assertRunCaptured, t as throwIfRunIncomplete } from './integrity-D2t12mMw.js';
|
|
7
|
+
import { T as TraceStore } from './store-BcFXE6LG.js';
|
|
8
|
+
export { E as EventFilter, F as FileSystemTraceStore, a as FileSystemTraceStoreOptions, I as InMemoryTraceStore, R as RunFilter, S as SpanFilter } from './store-BcFXE6LG.js';
|
|
9
|
+
export { a as aggregateLlm, b as argHash, g as groupBy, j as judgeSpans, l as llmSpans, r as runFailureClass, c as runsForScenario, t as toolSpans } from './query-B7GGjRox.js';
|
|
10
10
|
export { D as DEFAULT_REDACTION_RULES, b as REDACTION_VERSION, a as RedactionReport, R as RedactionRule, r as redactString, c as redactValue } from './redact-B40YG2M_.js';
|
|
11
11
|
import { R as Run } from './schema-m0gsnbt3.js';
|
|
12
12
|
export { A as Artifact, B as BudgetLedgerEntry, h as BudgetSpec, E as EventKind, i as FAILURE_CLASSES, F as FailureClass, G as GenericSpan, J as JudgeSpan, L as LlmSpan, M as Message, d as RetrievalSpan, g as RunLayer, b as RunOutcome, f as RunStatus, e as SandboxSpan, S as Span, j as SpanBase, c as SpanKind, k as SpanStatus, l as TRACE_SCHEMA_VERSION, T as ToolSpan, a as TraceEvent, m as isJudgeSpan, n as isLlmSpan, o as isRetrievalSpan, p as isSandboxSpan, q as isToolSpan } from './schema-m0gsnbt3.js';
|
|
@@ -474,6 +474,14 @@ interface OtlpFileTraceStoreOptions {
|
|
|
474
474
|
perCallByteCeiling?: number;
|
|
475
475
|
/** Override the per-match text budget. */
|
|
476
476
|
perMatchTextBudget?: number;
|
|
477
|
+
/**
|
|
478
|
+
* Hard ceiling on the trace file size in bytes. The store reads the
|
|
479
|
+
* whole file into one Buffer and indexes it in memory, so an
|
|
480
|
+
* unbounded file OOMs the process. Above this size the store fails
|
|
481
|
+
* loud with `TraceFileTooLargeError` instead of degrading silently.
|
|
482
|
+
* Default 256 MiB.
|
|
483
|
+
*/
|
|
484
|
+
maxFileBytes?: number;
|
|
477
485
|
}
|
|
478
486
|
declare class OtlpFileTraceStore implements TraceAnalysisStore {
|
|
479
487
|
private readonly path;
|
|
@@ -481,6 +489,7 @@ declare class OtlpFileTraceStore implements TraceAnalysisStore {
|
|
|
481
489
|
private readonly perAttributeSpanBudget;
|
|
482
490
|
private readonly perCallByteCeiling;
|
|
483
491
|
private readonly perMatchTextBudget;
|
|
492
|
+
private readonly maxFileBytes;
|
|
484
493
|
private indexPromise?;
|
|
485
494
|
/** Cached UTF-8 buffer of the file. We pin it once because every
|
|
486
495
|
* read needs slice access and re-reading on each call balloons the
|
|
@@ -518,6 +527,10 @@ declare class OtlpFileTraceStore implements TraceAnalysisStore {
|
|
|
518
527
|
* before the first agent call. */
|
|
519
528
|
ensureIndexed(): Promise<void>;
|
|
520
529
|
private buffer;
|
|
530
|
+
/** Stat-then-read so an oversized file fails loud BEFORE we allocate a
|
|
531
|
+
* multi-hundred-MB Buffer and OOM the process. A missing file surfaces
|
|
532
|
+
* as TraceFileMissingError; any other stat/read error propagates. */
|
|
533
|
+
private readGuarded;
|
|
521
534
|
private index;
|
|
522
535
|
private buildIndex;
|
|
523
536
|
private matchedTraces;
|