npm - @tangle-network/agent-eval - Versions diffs - 0.30.0 → 0.31.1 - Mend

@tangle-network/agent-eval 0.30.0 → 0.31.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.

Files changed (83) hide show

package/CHANGELOG.md CHANGED Viewed

@@ -1,5 +1,84 @@
 # Changelog
+## 0.31.1 — 2026-05-20
+### Republish of 0.31.0 — dist drift fix
+The `v0.31.0` tag's npm tarball shipped a stale `dist/` — `JudgeScoresRecord`
+was missing from `dist/index.d.ts` and the `recordOutcome.judgeScores`
+propagation never made it into `dist/index.js`, even though the source on
+the tagged commit had both. Consumers that bumped to `^0.31.0` got a
+typecheck failure on `RunOutcome.judgeScores` (since the type wasn't
+re-exported) and a silent drop on the wire (since the campaign runner
+didn't carry the field through).
+Cause: a build artifact picked up by the publish workflow predated the
+source merge. The retag forces a clean `pnpm build` and republish; this
+patch carries no source change beyond the version bump.
+Verified after this tag: `dist/index.d.ts` contains `JudgeScoresRecord`,
+`dist/index.js` propagates `outcome.judgeScores` end-to-end via
+`recordOutcome.judgeScores`, and a downstream `pnpm install
+@tangle-network/agent-eval@0.31.1` types-clean against the shape
+documented in 0.31.0.
+## 0.31.0 — 2026-05-20
+### `JudgeScoresRecord` on `RunRecord.outcome` — substrate-blessed ensemble shape
+Multi-judge consumers (forge-chat in agent-builder, and four sibling
+product agents on the same trajectory) compute per-judge per-dimension
+scores per cell, then collapse to a single composite for the gate. The
+substrate's `RunOutcome` only had a slot for the composite plus a free
+`raw: Record<string, number>` bag. Consumers were either dropping the
+breakdown on the floor or smuggling it through stringly-typed `raw`
+keys like `judge_kimi_helpfulness` — neither survives a corpus-IRR run
+(0.27.2's `corpusInterRaterAgreement` expects structured per-judge
+per-dim records, not parsed strings).
+This release ships the typed slot so every product agent speaks the
+same shape, and the inter-rater primitives consume it without a
+per-consumer adapter.
+### Added
+- **`JudgeScoresRecord`** (`src/run-record.ts`) — `perJudge[judgeId][dim]`
+  is the canonical store; `perDimMean` and `composite` are precomputed
+  projections so reporters and IRR primitives don't repeat the
+  aggregation; `failedJudges?: string[]` records dead-judge ids
+  explicitly (no inferring partial-failure from missing keys);
+  `notes?: string` carries panel prose.
+- **`RunOutcome.judgeScores?: JudgeScoresRecord`** — optional. Single-
+  judge or scalar-only runs leave it unset; ensemble runs populate it.
+- **`CampaignRunOutcome.judgeScores?: JudgeScoresRecord`** — runners
+  return it on the per-cell outcome; `runEvalCampaign` threads it onto
+  the resulting `RunRecord.outcome.judgeScores` without coercion.
+### Validator extended
+`validateRunRecord` validates `outcome.judgeScores` when present.
+Every `perJudge[judge][dim]` and every `perDimMean[dim]` and the
+`composite` must be finite numbers — the NaN-as-silent-zero bug class
+banned by `CLAUDE.md` cannot pass the boundary. `failedJudges` must be
+an array of non-empty strings; `notes` must be a string. Round-trip
+tested in `tests/run-record.test.ts`.
+### Fail-loud contract
+A judge that throws lands in `failedJudges` by id, not a silent zero
+in `perJudge`. The composite is computed over surviving judges only;
+the partial-failure signal is preserved through to the gate.
+`tests/eval-campaign.test.ts` covers the four shapes (full, partial,
+missing, with notes) plus an explicit fail-loud case where one judge
+throws and the run record carries `failedJudges: ['glm-5.1@...']`.
+### Consumer contract
+`tests/consumer-contract.test.ts` pins `JudgeScoresRecord` as a
+type-level export at the root entry. The 0.30.0 surface is preserved —
+the new field is additive on `RunOutcome` and the new type is a new
+export, so existing consumers stay green.
 ## 0.29.0 — 2026-05-19
 ### Analyst kinds + cross-run findings context

package/dist/{baseline-BwdCXUS8.d.ts → baseline-4R5deP0N.d.ts} RENAMED Viewed

@@ -1,4 +1,4 @@
-import { T as TraceStore } from './store-BP5be6s7.js';
+import { T as TraceStore } from './store-Db2Bv8Cf.js';
 /**
  * Tool-use metrics — derived purely from trace data.

package/dist/benchmarks/index.d.ts CHANGED Viewed

@@ -1,3 +1,3 @@
-export { B as BENCHMARK_SPLIT_SEED, a as BenchmarkAdapter, b as BenchmarkDatasetItem, c as BenchmarkEvaluation, d as deterministicSplit, e as routing } from '../index--fVrWDiR.js';
-import '../run-record-CqzahIbx.js';
-import '../errors-BZ9sTdz7.js';
+export { B as BENCHMARK_SPLIT_SEED, a as BenchmarkAdapter, b as BenchmarkDatasetItem, c as BenchmarkEvaluation, d as deterministicSplit, e as routing } from '../index-BTqhGHJT.js';
+import '../run-record-BfX5y68A.js';
+import '../errors-mje_cKOs.js';

package/dist/builder-eval/index.d.ts CHANGED Viewed

@@ -1,6 +1,6 @@
-import { S as SandboxDriver, H as HarnessConfig, a as SandboxHarnessResult, T as TestGradedScenario, b as TestGradedRunResult } from '../test-graded-scenario-BJ54PDan.js';
-import { T as TraceEmitter } from '../emitter-BqjeOvJh.js';
-import { T as TraceStore, R as Run } from '../store-BP5be6s7.js';
+import { S as SandboxDriver, H as HarnessConfig, a as SandboxHarnessResult, T as TestGradedScenario, b as TestGradedRunResult } from '../test-graded-scenario-B2kWEdh9.js';
+import { T as TraceEmitter } from '../emitter-DP_cSSiw.js';
+import { T as TraceStore, R as Run } from '../store-Db2Bv8Cf.js';
 /**
  * BuilderSession — ties a builder-of-builders workflow together.

package/dist/builder-eval/index.js CHANGED Viewed

@@ -1,7 +1,7 @@
 import {
   SandboxHarness,
   runTestGradedScenario
-} from "../chunk-QHF6EQKK.js";
+} from "../chunk-YTMXBHFM.js";
 import {
   judgeSpans
 } from "../chunk-47X6LRCE.js";
@@ -9,7 +9,7 @@ import "../chunk-5BKGXME7.js";
 import {
   TraceEmitter
 } from "../chunk-TVVP3ZZQ.js";
-import "../chunk-NG236HPC.js";
+import "../chunk-QYJT52YW.js";
 import "../chunk-PZ5AY32C.js";
 // src/builder-eval/builder-session.ts

package/dist/{chunk-R5UQJNKC.js → chunk-4L3WJXQJ.js} RENAMED Viewed

@@ -1,6 +1,6 @@
 import {
   ValidationError
-} from "./chunk-NG236HPC.js";
+} from "./chunk-QYJT52YW.js";
 // src/judge-calibration.ts
 function calibrateJudge(golden, candidate) {
@@ -719,4 +719,4 @@ export {
   corpusInterRaterAgreement,
   corpusInterRaterAgreementFromJudgeScores
 };
-//# sourceMappingURL=chunk-R5UQJNKC.js.map
+//# sourceMappingURL=chunk-4L3WJXQJ.js.map

package/dist/{chunk-SZSBQUIJ.js → chunk-B73G44OH.js} RENAMED Viewed

@@ -1,10 +1,10 @@
 import {
   validateRunRecord
-} from "./chunk-NLMNWKVM.js";
+} from "./chunk-ZN2CMQIW.js";
 import {
   pairedBootstrap,
   pairedWilcoxon
-} from "./chunk-5AKPEK5L.js";
+} from "./chunk-CXJOVDJR.js";
 // src/feedback-trajectory.ts
 var DEFAULT_SPLIT_POLICY = {
@@ -1409,4 +1409,4 @@ export {
   CallbackResearcher,
   NoopResearcher
 };
-//# sourceMappingURL=chunk-SZSBQUIJ.js.map
+//# sourceMappingURL=chunk-B73G44OH.js.map

package/dist/{chunk-5AKPEK5L.js → chunk-CXJOVDJR.js} RENAMED Viewed

@@ -2,7 +2,7 @@ import {
   cohensD,
   confidenceInterval,
   wilcoxonSignedRank
-} from "./chunk-R5UQJNKC.js";
+} from "./chunk-4L3WJXQJ.js";
 import {
   canonicalize,
   hashJson
@@ -1047,4 +1047,4 @@ export {
   RESEARCH_REPORT_HARD_PAIR_FLOOR,
   researchReport
 };
-//# sourceMappingURL=chunk-5AKPEK5L.js.map
+//# sourceMappingURL=chunk-CXJOVDJR.js.map

package/dist/{chunk-RUI6SIHY.js → chunk-DTEJNZYK.js} RENAMED Viewed

@@ -1,13 +1,13 @@
 import {
   assertLlmRoute
-} from "./chunk-4S4BM3QQ.js";
+} from "./chunk-M6RZ5LJN.js";
 import {
   researchReport
-} from "./chunk-5AKPEK5L.js";
+} from "./chunk-CXJOVDJR.js";
 import {
   RunIntegrityError,
   assertRunCaptured
-} from "./chunk-KTGTIOFD.js";
+} from "./chunk-UBPIXOC4.js";
 import {
   FileSystemRawProviderSink
 } from "./chunk-PC4UYEBM.js";
@@ -202,6 +202,7 @@ async function runEvalCampaign(opts) {
     };
     if (splitTag === "holdout") recordOutcome.holdoutScore = outcome.score;
     else recordOutcome.searchScore = outcome.score;
+    if (outcome.judgeScores !== void 0) recordOutcome.judgeScores = outcome.judgeScores;
     const record = {
       runId,
       experimentId: opts.campaignId,
@@ -284,4 +285,4 @@ function defaultRunId(params) {
 export {
   runEvalCampaign
 };
-//# sourceMappingURL=chunk-RUI6SIHY.js.map
+//# sourceMappingURL=chunk-DTEJNZYK.js.map

package/dist/chunk-DTEJNZYK.js.map ADDED Viewed

@@ -0,0 +1 @@

+ {"version":3,"sources":["../src/eval-campaign.ts"],"sourcesContent":["/**\n * EvalCampaign — opinionated matrix runner that wires the four\n * capture-integrity directives by construction.\n *\n * The canonical benchmark shape — matrix runner → for each\n * (variant, scenario, seed) → start a TraceEmitter → call LLMs → end the\n * run → analyze — has a bug class at the integration boundary: raw\n * events not captured, route silently wrong, integrity not asserted,\n * analyst never run. The directives in `SKILL.md § Capture integrity`\n * are the mitigations.\n *\n * `EvalCampaign` is the structural fix — consumers don't wire the\n * integrity surface themselves; the campaign owns it. Specifically:\n *\n * - calls `assertLlmRoute` once at preflight before any work runs\n * - constructs a per-run `TraceStore` and `RawProviderSink` via factories\n * - constructs the `TraceEmitter` with `onRunComplete: [analyst hook]`\n * - hands the runner an `LlmClientOptions` pre-wired with the sink and\n * trace context — the runner can't accidentally call an LLM without\n * capturing the raw HTTP envelope\n * - calls `assertRunCaptured` after every `endRun` and routes failures\n * through a configurable policy (`throw` / `mark_failed` / `log`)\n * - assembles per-run `RunRecord`s and runs `researchReport` at the end\n * so the campaign artifact is launch-decision-grade by default\n * - embeds the campaign fingerprint (a SHA-256 over the canonicalised\n * run set) and optional `preregistrationHash` in the report\n *\n * The runner contract is intentionally narrow: produce a `CampaignRunOutcome`\n * given a fully-wired `CampaignRunContext`. Everything orchestration-shaped\n * lives in the campaign. This is the inversion-of-control point — consumers\n * stop writing matrix runners and start writing scenario-runners.\n *\n * Out of scope for v1 (tracked in `docs/research-report-methodology.md`):\n *\n * - Distributed/cluster execution (concurrency is local async)\n * - Adaptive sampling / sequential interim looks\n * - Resume from partial state across crashes\n * - LLM-call retry beyond what `LlmClient` already does\n */\n\nimport { assertLlmRoute, type LlmClientOptions, type LlmRouteRequirements } from './llm-client'\nimport { canonicalize, hashJson } from './pre-registration'\nimport type {\n JudgeScoresRecord,\n RunJudgeMetadata,\n RunOutcome,\n RunRecord,\n RunSplitTag,\n RunTokenUsage,\n} from './run-record'\nimport { type ResearchReport, type ResearchReportOptions, researchReport } from './summary-report'\nimport type { RunCompleteHook } from './trace/emitter'\nimport { TraceEmitter } from './trace/emitter'\nimport {\n assertRunCaptured,\n RunIntegrityError,\n type RunIntegrityExpectations,\n type RunIntegrityReport,\n} from './trace/integrity'\nimport { FileSystemRawProviderSink, type RawProviderSink } from './trace/raw-provider-sink'\nimport type { TraceStore } from './trace/store'\n\n// ── Public types ─────────────────────────────────────────────────────────\n\nexport interface CampaignVariant<V> {\n id: string\n payload: V\n}\n\nexport interface CampaignScenario {\n scenarioId: string\n /** Free-form metadata propagated to runs and reports. */\n tags?: Record<string, string>\n}\n\nexport interface CampaignRunContext<V> {\n /** Stable run id. The campaign generates this; the runner does not. */\n runId: string\n /** Logical experiment id (campaignId by default; overridable per-run via opts). */\n experimentId: string\n variant: V\n variantId: string\n scenarioId: string\n scenarioTags: Record<string, string>\n seed: number\n splitTag: RunSplitTag\n /**\n * The TraceEmitter for this run, with `onRunComplete` hooks pre-wired\n * (analyst auto-execution if configured, plus integrity check). The\n * runner MUST call `emitter.startRun` before doing any work and either\n * `emitter.endRun` or `emitter.abortRun` before returning.\n */\n emitter: TraceEmitter\n store: TraceStore\n rawSink: RawProviderSink\n /**\n * Pre-wired LLM client options — `rawSink` and `traceContext` are populated\n * so any `callLlm(req, ctx.llmOpts)` automatically captures raw HTTP. The\n * runner can spread additional fields if needed.\n */\n llmOpts: LlmClientOptions\n}\n\nexport interface CampaignRunOutcome {\n /** Did the run pass? Mirrors `RunOutcome.pass` semantics. */\n pass: boolean\n /** Score for the run on its split. Maps to `searchScore` or `holdoutScore`. */\n score: number\n /** Mandatory cost in USD. Use 0 + raw.cost_unknown=1 only if truly unknown. */\n costUsd: number\n tokenUsage: RunTokenUsage\n /** Snapshot model id (e.g. `claude-sonnet-4-6@2025-04-15`). */\n model: string\n /** sha256 of the effective prompt sent to the model. */\n promptHash: string\n /** sha256 of the effective config (model, temperature, tools, judges, splits). */\n configHash: string\n /** Optional extra numeric metrics to land in `outcome.raw`. */\n raw?: Record<string, number>\n /** Optional failure-taxonomy tag if the run failed. */\n failureMode?: string\n /** Optional judge metadata when a judge was used. */\n judgeMetadata?: RunJudgeMetadata\n /**\n * Optional per-judge / per-dim breakdown for ensemble-judged runs.\n * Propagated to `outcome.judgeScores` on the resulting `RunRecord`.\n * Single-judge or scalar-only runs leave this unset.\n */\n judgeScores?: JudgeScoresRecord\n}\n\nexport type CampaignRunner<V> = (ctx: CampaignRunContext<V>) => Promise<CampaignRunOutcome>\n\nexport type CampaignIntegrityPolicy = 'throw' | 'mark_failed' | 'log'\n\nexport interface EvalCampaignOptions<V> {\n /**\n * Stable id for the campaign. Used as the default `experimentId` on\n * every run, and folded into the campaign fingerprint.\n */\n campaignId: string\n variants: CampaignVariant<V>[]\n scenarios: CampaignScenario[]\n /** Default `[0, 1, 2]`. */\n seeds?: number[]\n /** Default `'holdout'` — the split that anchors a launch decision. */\n splitTag?: RunSplitTag\n /** Git SHA the campaign is run against. Mandatory; `RunRecord` rejects unset. */\n commitSha: string\n /**\n * LLM client config. Augmented per-run with `rawSink` and `traceContext`\n * before being passed to the runner. The campaign asserts this config\n * matches `routeRequirements` once at preflight.\n */\n llmOpts: LlmClientOptions\n /**\n * Default `{ requireExplicitBaseUrl: true, requireAuth: true }` — fail\n * loud if the campaign would silently fall back to the public router or\n * run unauthenticated. Override with an empty object to disable.\n */\n routeRequirements?: LlmRouteRequirements\n /**\n * Per-run TraceStore factory. Common shape: a fresh store per run keyed\n * on `runId`. Implementations that share a store across the campaign\n * are valid — the campaign only writes through `emitter`.\n */\n storeFactory: (params: CampaignFactoryParams) => TraceStore\n /**\n * Per-run RawProviderSink factory. Defaults to `FileSystemRawProviderSink`\n * rooted at `${workDir}/raw-events/${runId}` if `workDir` is supplied;\n * otherwise required. Forensic capture is non-negotiable in a campaign\n * run — pass `NoopRawProviderSink` explicitly if you want to opt out.\n */\n rawSinkFactory?: (params: CampaignFactoryParams) => RawProviderSink\n /**\n * Filesystem root for default `rawSinkFactory`. Ignored if\n * `rawSinkFactory` is supplied.\n */\n workDir?: string\n /**\n * Extra `onRunComplete` hooks the campaign appends (after its own\n * integrity-check hook). Pass `traceAnalystOnRunComplete(...)` here.\n */\n onRunComplete?: RunCompleteHook[]\n /**\n * Per-run integrity expectations. Defaults to:\n * `{ llmSpansMin: 1, requireRawCoverageOfLlmSpans: true, requireOutcome: true }`.\n * Override (e.g. `{ llmSpansMin: 0 }`) for runs that don't call LLMs.\n */\n integrity?: RunIntegrityExpectations\n /** Behaviour when integrity fails. Default `'mark_failed'`. */\n onIntegrityFailure?: CampaignIntegrityPolicy\n /**\n * Per-run runner. Receives a fully-wired context; produces an outcome\n * the campaign converts into a `RunRecord`.\n */\n runner: CampaignRunner<V>\n /**\n * If set, the campaign computes `researchReport` at the end. `comparator`\n * is a `variantId`. Other fields are forwarded verbatim.\n */\n report?: { comparator?: string } & Omit<\n ResearchReportOptions,\n 'comparator' | 'preregistrationHash' | 'generatedAt'\n >\n /**\n * Hash of a signed `HypothesisManifest` (see `pre-registration.ts`).\n * Embedded in the campaign fingerprint and the research report.\n */\n preregistrationHash?: string\n /** Local concurrency. Default `1` (sequential). */\n concurrency?: number\n /**\n * Override the time source. Tests pass a mock to make wallMs deterministic.\n */\n now?: () => number\n /** Override the runId generator. Tests pin this. */\n runId?: (params: CampaignFactoryParams) => string\n}\n\nexport interface CampaignFactoryParams {\n campaignId: string\n runId: string\n variantId: string\n scenarioId: string\n seed: number\n}\n\nexport interface FailedRun {\n runId: string\n variantId: string\n scenarioId: string\n seed: number\n reason: string\n error?: string\n}\n\nexport interface EvalCampaignResult {\n campaignId: string\n /** SHA-256 over canonicalised `(variantIds, scenarioIds, seeds, comparator, splitTag, baseUrl, provider, preregistrationHash)`. */\n campaignFingerprint: string\n preregistrationHash: string | null\n /** Successful runs only. Failed runs land in `failedRuns`. */\n runs: RunRecord[]\n /** Integrity reports for every successful run. */\n integrityReports: RunIntegrityReport[]\n failedRuns: FailedRun[]\n /** Computed when `report` is set on options. */\n report?: ResearchReport\n startedAt: string\n endedAt: string\n}\n\n// ── Implementation ───────────────────────────────────────────────────────\n\nconst DEFAULT_INTEGRITY: RunIntegrityExpectations = {\n llmSpansMin: 1,\n requireRawCoverageOfLlmSpans: true,\n requireOutcome: true,\n}\n\nconst DEFAULT_ROUTE: LlmRouteRequirements = {\n requireExplicitBaseUrl: true,\n requireAuth: true,\n}\n\nexport async function runEvalCampaign<V>(\n opts: EvalCampaignOptions<V>,\n): Promise<EvalCampaignResult> {\n // ── Preflight ──────────────────────────────────────────────────────\n assertLlmRoute(opts.llmOpts, opts.routeRequirements ?? DEFAULT_ROUTE)\n\n if (opts.variants.length === 0) {\n throw new Error('runEvalCampaign: variants must be non-empty.')\n }\n if (opts.scenarios.length === 0) {\n throw new Error('runEvalCampaign: scenarios must be non-empty.')\n }\n const variantIds = new Set<string>()\n for (const v of opts.variants) {\n if (variantIds.has(v.id)) {\n throw new Error(`runEvalCampaign: duplicate variant id \"${v.id}\".`)\n }\n variantIds.add(v.id)\n }\n const scenarioIds = new Set<string>()\n for (const s of opts.scenarios) {\n if (scenarioIds.has(s.scenarioId)) {\n throw new Error(`runEvalCampaign: duplicate scenarioId \"${s.scenarioId}\".`)\n }\n scenarioIds.add(s.scenarioId)\n }\n if (opts.report?.comparator && !variantIds.has(opts.report.comparator)) {\n throw new Error(\n `runEvalCampaign: report.comparator \"${opts.report.comparator}\" is not a configured variantId.`,\n )\n }\n if (!opts.commitSha) {\n throw new Error('runEvalCampaign: commitSha is required (every RunRecord needs it).')\n }\n\n const seeds = opts.seeds ?? [0, 1, 2]\n const splitTag: RunSplitTag = opts.splitTag ?? 'holdout'\n const concurrency = Math.max(1, opts.concurrency ?? 1)\n const integrity = { ...DEFAULT_INTEGRITY, ...(opts.integrity ?? {}) }\n const onIntegrityFailure: CampaignIntegrityPolicy = opts.onIntegrityFailure ?? 'mark_failed'\n const now = opts.now ?? (() => Date.now())\n const baseUrl = (opts.llmOpts.baseUrl ?? '').replace(/\\/+$/, '')\n const provider = opts.llmOpts.provider ?? null\n const preregistrationHash = opts.preregistrationHash ?? null\n\n const rawSinkFactory = opts.rawSinkFactory ?? defaultRawSinkFactory(opts.workDir)\n\n // ── Fingerprint ────────────────────────────────────────────────────\n const campaignFingerprint = await hashJson(\n canonicalize({\n campaignId: opts.campaignId,\n variants: opts.variants.map((v) => v.id).sort(),\n scenarios: opts.scenarios.map((s) => s.scenarioId).sort(),\n seeds: [...seeds].sort((a, b) => a - b),\n splitTag,\n comparator: opts.report?.comparator ?? null,\n baseUrl,\n provider,\n preregistrationHash,\n }),\n )\n\n // ── Plan the matrix ────────────────────────────────────────────────\n type Cell = { variant: CampaignVariant<V>; scenario: CampaignScenario; seed: number }\n const cells: Cell[] = []\n for (const variant of opts.variants) {\n for (const scenario of opts.scenarios) {\n for (const seed of seeds) {\n cells.push({ variant, scenario, seed })\n }\n }\n }\n\n const startedAt = new Date(now()).toISOString()\n const runs: RunRecord[] = []\n const integrityReports: RunIntegrityReport[] = []\n const failedRuns: FailedRun[] = []\n\n // ── Execute (bounded-concurrency worker pool) ──────────────────────\n let cursor = 0\n async function worker(): Promise<void> {\n while (true) {\n const i = cursor++\n if (i >= cells.length) return\n const cell = cells[i]!\n try {\n const result = await runOneCell(cell)\n runs.push(result.record)\n integrityReports.push(result.integrity)\n } catch (err) {\n if (err instanceof CellExecutionError) {\n failedRuns.push(err.failed)\n if (err.integrity) integrityReports.push(err.integrity)\n } else {\n // Genuine bug — not a runner failure, not an integrity failure.\n // Surface it; don't silently mask.\n throw err\n }\n }\n }\n }\n\n async function runOneCell(\n cell: Cell,\n ): Promise<{ record: RunRecord; integrity: RunIntegrityReport }> {\n const runId = (opts.runId ?? defaultRunId)({\n campaignId: opts.campaignId,\n runId: '', // unused by default generator\n variantId: cell.variant.id,\n scenarioId: cell.scenario.scenarioId,\n seed: cell.seed,\n })\n const factoryParams: CampaignFactoryParams = {\n campaignId: opts.campaignId,\n runId,\n variantId: cell.variant.id,\n scenarioId: cell.scenario.scenarioId,\n seed: cell.seed,\n }\n const store = opts.storeFactory(factoryParams)\n const rawSink = rawSinkFactory(factoryParams)\n\n const emitter = new TraceEmitter(store, {\n runId,\n now: opts.now,\n onRunComplete: opts.onRunComplete,\n })\n\n const llmOpts: LlmClientOptions = {\n ...opts.llmOpts,\n rawSink,\n traceContext: { runId },\n }\n\n const ctx: CampaignRunContext<V> = {\n runId,\n experimentId: opts.campaignId,\n variant: cell.variant.payload,\n variantId: cell.variant.id,\n scenarioId: cell.scenario.scenarioId,\n scenarioTags: cell.scenario.tags ?? {},\n seed: cell.seed,\n splitTag,\n emitter,\n store,\n rawSink,\n llmOpts,\n }\n\n const wallStart = now()\n let outcome: CampaignRunOutcome\n try {\n outcome = await opts.runner(ctx)\n } catch (err) {\n const message = err instanceof Error ? err.message : String(err)\n // The runner threw mid-execution; give it a chance to have aborted.\n try {\n await emitter.abortRun(message)\n } catch {\n // Already aborted/ended; ignore.\n }\n throw new CellExecutionError({\n runId,\n variantId: cell.variant.id,\n scenarioId: cell.scenario.scenarioId,\n seed: cell.seed,\n reason: 'runner_threw',\n error: message,\n })\n }\n const wallMs = now() - wallStart\n\n const integrityReport = await assertRunCaptured(store, runId, { ...integrity, rawSink })\n if (!integrityReport.ok) {\n switch (onIntegrityFailure) {\n case 'throw':\n throw new RunIntegrityError(integrityReport)\n case 'mark_failed':\n throw new CellExecutionError(\n {\n runId,\n variantId: cell.variant.id,\n scenarioId: cell.scenario.scenarioId,\n seed: cell.seed,\n reason: 'integrity_failed',\n error: integrityReport.issues.map((i) => i.code).join(', '),\n },\n integrityReport,\n )\n case 'log':\n // Caller wants the run admitted with a flagged report; fall through.\n break\n }\n }\n\n const recordOutcome: RunOutcome = {\n raw: outcome.raw ?? {},\n }\n if (splitTag === 'holdout') recordOutcome.holdoutScore = outcome.score\n else recordOutcome.searchScore = outcome.score\n if (outcome.judgeScores !== undefined) recordOutcome.judgeScores = outcome.judgeScores\n\n const record: RunRecord = {\n runId,\n experimentId: opts.campaignId,\n candidateId: cell.variant.id,\n seed: cell.seed,\n model: outcome.model,\n promptHash: outcome.promptHash,\n configHash: outcome.configHash,\n commitSha: opts.commitSha,\n wallMs,\n costUsd: outcome.costUsd,\n tokenUsage: outcome.tokenUsage,\n judgeMetadata: outcome.judgeMetadata,\n outcome: recordOutcome,\n failureMode: outcome.failureMode,\n splitTag,\n scenarioId: cell.scenario.scenarioId,\n }\n return { record, integrity: integrityReport }\n }\n\n const workers = Array.from({ length: Math.min(concurrency, cells.length) }, () => worker())\n await Promise.all(workers)\n\n // ── Optional research report ───────────────────────────────────────\n let report: ResearchReport | undefined\n if (opts.report) {\n const reportOpts: ResearchReportOptions = {\n ...opts.report,\n comparator: opts.report.comparator,\n split: splitTag === 'dev' ? 'search' : splitTag,\n generatedAt: new Date(now()).toISOString(),\n preregistrationHash: preregistrationHash ?? undefined,\n }\n report = await researchReport(runs, reportOpts)\n }\n\n const endedAt = new Date(now()).toISOString()\n\n return {\n campaignId: opts.campaignId,\n campaignFingerprint,\n preregistrationHash,\n runs,\n integrityReports,\n failedRuns,\n report,\n startedAt,\n endedAt,\n }\n}\n\n// ── Internal ─────────────────────────────────────────────────────────────\n\nclass CellExecutionError extends Error {\n readonly failed: FailedRun\n readonly integrity?: RunIntegrityReport\n constructor(failed: FailedRun, integrity?: RunIntegrityReport) {\n super(`cell ${failed.variantId}/${failed.scenarioId}@${failed.seed} failed: ${failed.reason}`)\n this.failed = failed\n this.integrity = integrity\n }\n}\n\nfunction defaultRawSinkFactory(workDir: string | undefined) {\n return (params: CampaignFactoryParams): RawProviderSink => {\n if (!workDir) {\n throw new Error(\n 'runEvalCampaign: rawSinkFactory not supplied and workDir not set. Pass either to enable raw provider capture, or pass `new NoopRawProviderSink()` via rawSinkFactory to opt out explicitly.',\n )\n }\n return new FileSystemRawProviderSink({\n dir: `${workDir}/raw-events/${params.runId}`,\n })\n }\n}\n\nfunction defaultRunId(params: CampaignFactoryParams): string {\n // Stable across re-runs: fingerprint of (campaignId, variantId, scenarioId, seed).\n // Caller can override via opts.runId for non-deterministic IDs.\n const base = `${params.campaignId}::${params.variantId}::${params.scenarioId}::${params.seed}`\n // Lightweight hex: we don't need crypto-grade here, just stability + uniqueness.\n let h1 = 0x811c9dc5\n let h2 = 0x12345678\n for (let i = 0; i < base.length; i++) {\n const c = base.charCodeAt(i)\n h1 = Math.imul(h1 ^ c, 0x01000193) >>> 0\n h2 = Math.imul(h2 ^ c, 0x9e3779b1) >>> 0\n }\n return `run-${h1.toString(16).padStart(8, '0')}${h2.toString(16).padStart(8, '0')}`\n}\n"],"mappings":";;;;;;;;;;;;;;;;;;;;;;AA+PA,IAAM,oBAA8C;AAAA,EAClD,aAAa;AAAA,EACb,8BAA8B;AAAA,EAC9B,gBAAgB;AAClB;AAEA,IAAM,gBAAsC;AAAA,EAC1C,wBAAwB;AAAA,EACxB,aAAa;AACf;AAEA,eAAsB,gBACpB,MAC6B;AAE7B,iBAAe,KAAK,SAAS,KAAK,qBAAqB,aAAa;AAEpE,MAAI,KAAK,SAAS,WAAW,GAAG;AAC9B,UAAM,IAAI,MAAM,8CAA8C;AAAA,EAChE;AACA,MAAI,KAAK,UAAU,WAAW,GAAG;AAC/B,UAAM,IAAI,MAAM,+CAA+C;AAAA,EACjE;AACA,QAAM,aAAa,oBAAI,IAAY;AACnC,aAAW,KAAK,KAAK,UAAU;AAC7B,QAAI,WAAW,IAAI,EAAE,EAAE,GAAG;AACxB,YAAM,IAAI,MAAM,0CAA0C,EAAE,EAAE,IAAI;AAAA,IACpE;AACA,eAAW,IAAI,EAAE,EAAE;AAAA,EACrB;AACA,QAAM,cAAc,oBAAI,IAAY;AACpC,aAAW,KAAK,KAAK,WAAW;AAC9B,QAAI,YAAY,IAAI,EAAE,UAAU,GAAG;AACjC,YAAM,IAAI,MAAM,0CAA0C,EAAE,UAAU,IAAI;AAAA,IAC5E;AACA,gBAAY,IAAI,EAAE,UAAU;AAAA,EAC9B;AACA,MAAI,KAAK,QAAQ,cAAc,CAAC,WAAW,IAAI,KAAK,OAAO,UAAU,GAAG;AACtE,UAAM,IAAI;AAAA,MACR,uCAAuC,KAAK,OAAO,UAAU;AAAA,IAC/D;AAAA,EACF;AACA,MAAI,CAAC,KAAK,WAAW;AACnB,UAAM,IAAI,MAAM,oEAAoE;AAAA,EACtF;AAEA,QAAM,QAAQ,KAAK,SAAS,CAAC,GAAG,GAAG,CAAC;AACpC,QAAM,WAAwB,KAAK,YAAY;AAC/C,QAAM,cAAc,KAAK,IAAI,GAAG,KAAK,eAAe,CAAC;AACrD,QAAM,YAAY,EAAE,GAAG,mBAAmB,GAAI,KAAK,aAAa,CAAC,EAAG;AACpE,QAAM,qBAA8C,KAAK,sBAAsB;AAC/E,QAAM,MAAM,KAAK,QAAQ,MAAM,KAAK,IAAI;AACxC,QAAM,WAAW,KAAK,QAAQ,WAAW,IAAI,QAAQ,QAAQ,EAAE;AAC/D,QAAM,WAAW,KAAK,QAAQ,YAAY;AAC1C,QAAM,sBAAsB,KAAK,uBAAuB;AAExD,QAAM,iBAAiB,KAAK,kBAAkB,sBAAsB,KAAK,OAAO;AAGhF,QAAM,sBAAsB,MAAM;AAAA,IAChC,aAAa;AAAA,MACX,YAAY,KAAK;AAAA,MACjB,UAAU,KAAK,SAAS,IAAI,CAAC,MAAM,EAAE,EAAE,EAAE,KAAK;AAAA,MAC9C,WAAW,KAAK,UAAU,IAAI,CAAC,MAAM,EAAE,UAAU,EAAE,KAAK;AAAA,MACxD,OAAO,CAAC,GAAG,KAAK,EAAE,KAAK,CAAC,GAAG,MAAM,IAAI,CAAC;AAAA,MACtC;AAAA,MACA,YAAY,KAAK,QAAQ,cAAc;AAAA,MACvC;AAAA,MACA;AAAA,MACA;AAAA,IACF,CAAC;AAAA,EACH;AAIA,QAAM,QAAgB,CAAC;AACvB,aAAW,WAAW,KAAK,UAAU;AACnC,eAAW,YAAY,KAAK,WAAW;AACrC,iBAAW,QAAQ,OAAO;AACxB,cAAM,KAAK,EAAE,SAAS,UAAU,KAAK,CAAC;AAAA,MACxC;AAAA,IACF;AAAA,EACF;AAEA,QAAM,YAAY,IAAI,KAAK,IAAI,CAAC,EAAE,YAAY;AAC9C,QAAM,OAAoB,CAAC;AAC3B,QAAM,mBAAyC,CAAC;AAChD,QAAM,aAA0B,CAAC;AAGjC,MAAI,SAAS;AACb,iBAAe,SAAwB;AACrC,WAAO,MAAM;AACX,YAAM,IAAI;AACV,UAAI,KAAK,MAAM,OAAQ;AACvB,YAAM,OAAO,MAAM,CAAC;AACpB,UAAI;AACF,cAAM,SAAS,MAAM,WAAW,IAAI;AACpC,aAAK,KAAK,OAAO,MAAM;AACvB,yBAAiB,KAAK,OAAO,SAAS;AAAA,MACxC,SAAS,KAAK;AACZ,YAAI,eAAe,oBAAoB;AACrC,qBAAW,KAAK,IAAI,MAAM;AAC1B,cAAI,IAAI,UAAW,kBAAiB,KAAK,IAAI,SAAS;AAAA,QACxD,OAAO;AAGL,gBAAM;AAAA,QACR;AAAA,MACF;AAAA,IACF;AAAA,EACF;AAEA,iBAAe,WACb,MAC+D;AAC/D,UAAM,SAAS,KAAK,SAAS,cAAc;AAAA,MACzC,YAAY,KAAK;AAAA,MACjB,OAAO;AAAA;AAAA,MACP,WAAW,KAAK,QAAQ;AAAA,MACxB,YAAY,KAAK,SAAS;AAAA,MAC1B,MAAM,KAAK;AAAA,IACb,CAAC;AACD,UAAM,gBAAuC;AAAA,MAC3C,YAAY,KAAK;AAAA,MACjB;AAAA,MACA,WAAW,KAAK,QAAQ;AAAA,MACxB,YAAY,KAAK,SAAS;AAAA,MAC1B,MAAM,KAAK;AAAA,IACb;AACA,UAAM,QAAQ,KAAK,aAAa,aAAa;AAC7C,UAAM,UAAU,eAAe,aAAa;AAE5C,UAAM,UAAU,IAAI,aAAa,OAAO;AAAA,MACtC;AAAA,MACA,KAAK,KAAK;AAAA,MACV,eAAe,KAAK;AAAA,IACtB,CAAC;AAED,UAAM,UAA4B;AAAA,MAChC,GAAG,KAAK;AAAA,MACR;AAAA,MACA,cAAc,EAAE,MAAM;AAAA,IACxB;AAEA,UAAM,MAA6B;AAAA,MACjC;AAAA,MACA,cAAc,KAAK;AAAA,MACnB,SAAS,KAAK,QAAQ;AAAA,MACtB,WAAW,KAAK,QAAQ;AAAA,MACxB,YAAY,KAAK,SAAS;AAAA,MAC1B,cAAc,KAAK,SAAS,QAAQ,CAAC;AAAA,MACrC,MAAM,KAAK;AAAA,MACX;AAAA,MACA;AAAA,MACA;AAAA,MACA;AAAA,MACA;AAAA,IACF;AAEA,UAAM,YAAY,IAAI;AACtB,QAAI;AACJ,QAAI;AACF,gBAAU,MAAM,KAAK,OAAO,GAAG;AAAA,IACjC,SAAS,KAAK;AACZ,YAAM,UAAU,eAAe,QAAQ,IAAI,UAAU,OAAO,GAAG;AAE/D,UAAI;AACF,cAAM,QAAQ,SAAS,OAAO;AAAA,MAChC,QAAQ;AAAA,MAER;AACA,YAAM,IAAI,mBAAmB;AAAA,QAC3B;AAAA,QACA,WAAW,KAAK,QAAQ;AAAA,QACxB,YAAY,KAAK,SAAS;AAAA,QAC1B,MAAM,KAAK;AAAA,QACX,QAAQ;AAAA,QACR,OAAO;AAAA,MACT,CAAC;AAAA,IACH;AACA,UAAM,SAAS,IAAI,IAAI;AAEvB,UAAM,kBAAkB,MAAM,kBAAkB,OAAO,OAAO,EAAE,GAAG,WAAW,QAAQ,CAAC;AACvF,QAAI,CAAC,gBAAgB,IAAI;AACvB,cAAQ,oBAAoB;AAAA,QAC1B,KAAK;AACH,gBAAM,IAAI,kBAAkB,eAAe;AAAA,QAC7C,KAAK;AACH,gBAAM,IAAI;AAAA,YACR;AAAA,cACE;AAAA,cACA,WAAW,KAAK,QAAQ;AAAA,cACxB,YAAY,KAAK,SAAS;AAAA,cAC1B,MAAM,KAAK;AAAA,cACX,QAAQ;AAAA,cACR,OAAO,gBAAgB,OAAO,IAAI,CAAC,MAAM,EAAE,IAAI,EAAE,KAAK,IAAI;AAAA,YAC5D;AAAA,YACA;AAAA,UACF;AAAA,QACF,KAAK;AAEH;AAAA,MACJ;AAAA,IACF;AAEA,UAAM,gBAA4B;AAAA,MAChC,KAAK,QAAQ,OAAO,CAAC;AAAA,IACvB;AACA,QAAI,aAAa,UAAW,eAAc,eAAe,QAAQ;AAAA,QAC5D,eAAc,cAAc,QAAQ;AACzC,QAAI,QAAQ,gBAAgB,OAAW,eAAc,cAAc,QAAQ;AAE3E,UAAM,SAAoB;AAAA,MACxB;AAAA,MACA,cAAc,KAAK;AAAA,MACnB,aAAa,KAAK,QAAQ;AAAA,MAC1B,MAAM,KAAK;AAAA,MACX,OAAO,QAAQ;AAAA,MACf,YAAY,QAAQ;AAAA,MACpB,YAAY,QAAQ;AAAA,MACpB,WAAW,KAAK;AAAA,MAChB;AAAA,MACA,SAAS,QAAQ;AAAA,MACjB,YAAY,QAAQ;AAAA,MACpB,eAAe,QAAQ;AAAA,MACvB,SAAS;AAAA,MACT,aAAa,QAAQ;AAAA,MACrB;AAAA,MACA,YAAY,KAAK,SAAS;AAAA,IAC5B;AACA,WAAO,EAAE,QAAQ,WAAW,gBAAgB;AAAA,EAC9C;AAEA,QAAM,UAAU,MAAM,KAAK,EAAE,QAAQ,KAAK,IAAI,aAAa,MAAM,MAAM,EAAE,GAAG,MAAM,OAAO,CAAC;AAC1F,QAAM,QAAQ,IAAI,OAAO;AAGzB,MAAI;AACJ,MAAI,KAAK,QAAQ;AACf,UAAM,aAAoC;AAAA,MACxC,GAAG,KAAK;AAAA,MACR,YAAY,KAAK,OAAO;AAAA,MACxB,OAAO,aAAa,QAAQ,WAAW;AAAA,MACvC,aAAa,IAAI,KAAK,IAAI,CAAC,EAAE,YAAY;AAAA,MACzC,qBAAqB,uBAAuB;AAAA,IAC9C;AACA,aAAS,MAAM,eAAe,MAAM,UAAU;AAAA,EAChD;AAEA,QAAM,UAAU,IAAI,KAAK,IAAI,CAAC,EAAE,YAAY;AAE5C,SAAO;AAAA,IACL,YAAY,KAAK;AAAA,IACjB;AAAA,IACA;AAAA,IACA;AAAA,IACA;AAAA,IACA;AAAA,IACA;AAAA,IACA;AAAA,IACA;AAAA,EACF;AACF;AAIA,IAAM,qBAAN,cAAiC,MAAM;AAAA,EAC5B;AAAA,EACA;AAAA,EACT,YAAY,QAAmB,WAAgC;AAC7D,UAAM,QAAQ,OAAO,SAAS,IAAI,OAAO,UAAU,IAAI,OAAO,IAAI,YAAY,OAAO,MAAM,EAAE;AAC7F,SAAK,SAAS;AACd,SAAK,YAAY;AAAA,EACnB;AACF;AAEA,SAAS,sBAAsB,SAA6B;AAC1D,SAAO,CAAC,WAAmD;AACzD,QAAI,CAAC,SAAS;AACZ,YAAM,IAAI;AAAA,QACR;AAAA,MACF;AAAA,IACF;AACA,WAAO,IAAI,0BAA0B;AAAA,MACnC,KAAK,GAAG,OAAO,eAAe,OAAO,KAAK;AAAA,IAC5C,CAAC;AAAA,EACH;AACF;AAEA,SAAS,aAAa,QAAuC;AAG3D,QAAM,OAAO,GAAG,OAAO,UAAU,KAAK,OAAO,SAAS,KAAK,OAAO,UAAU,KAAK,OAAO,IAAI;AAE5F,MAAI,KAAK;AACT,MAAI,KAAK;AACT,WAAS,IAAI,GAAG,IAAI,KAAK,QAAQ,KAAK;AACpC,UAAM,IAAI,KAAK,WAAW,CAAC;AAC3B,SAAK,KAAK,KAAK,KAAK,GAAG,QAAU,MAAM;AACvC,SAAK,KAAK,KAAK,KAAK,GAAG,UAAU,MAAM;AAAA,EACzC;AACA,SAAO,OAAO,GAAG,SAAS,EAAE,EAAE,SAAS,GAAG,GAAG,CAAC,GAAG,GAAG,SAAS,EAAE,EAAE,SAAS,GAAG,GAAG,CAAC;AACnF;","names":[]}

package/dist/{chunk-K33INZHH.js → chunk-GVQT44CS.js} RENAMED Viewed

@@ -1,6 +1,6 @@
 import {
   cohensD
-} from "./chunk-R5UQJNKC.js";
+} from "./chunk-4L3WJXQJ.js";
 import {
   argHash,
   groupBy,
@@ -615,4 +615,4 @@ export {
   iqr,
   welchsTTest
 };
-//# sourceMappingURL=chunk-K33INZHH.js.map
+//# sourceMappingURL=chunk-GVQT44CS.js.map

package/dist/{chunk-UW4NOOZI.js → chunk-HIO4UIS5.js} RENAMED Viewed

@@ -5,7 +5,7 @@ import {
 import {
   NotFoundError,
   ReplayError
-} from "./chunk-NG236HPC.js";
+} from "./chunk-QYJT52YW.js";
 // src/trace-analyst/prompts.ts
 var TRACE_ANALYST_ACTOR_DESCRIPTION = `You answer questions about an OTLP-shaped JSONL trace dataset using the trace tools provided in the \`traces\` namespace.
@@ -986,6 +986,302 @@ function normalizeRecordArray(value) {
   );
 }
+// src/trace-analyst/hook.ts
+var DEFAULT_QUESTION = "Summarise what happened in this run. Surface any failure modes, surprising findings, or evidence that the run's verdict is wrong.";
+function traceAnalystOnRunComplete(opts) {
+  return async (ctx) => {
+    if (opts.shouldRun && !opts.shouldRun(ctx)) return;
+    const source = opts.analyze.source;
+    if (source === void 0) {
+      await ctx.store.appendEvent({
+        eventId: `analyst-skip-${ctx.runId}`,
+        runId: ctx.runId,
+        kind: "log",
+        timestamp: Date.now(),
+        payload: { source: "trace_analyst_hook", reason: "no source configured" }
+      });
+      return;
+    }
+    const result = await analyzeTraces({ question: opts.question ?? DEFAULT_QUESTION }, {
+      ...opts.analyze,
+      source
+    });
+    if (opts.save) await opts.save(result, ctx);
+    if (opts.gateOn && !opts.gateOn(result, ctx)) {
+      await ctx.store.appendEvent({
+        eventId: `analyst-gate-${ctx.runId}`,
+        runId: ctx.runId,
+        kind: "log",
+        timestamp: Date.now(),
+        payload: {
+          source: "trace_analyst_hook",
+          reason: "analyst_gate_failed",
+          findings: result.findings
+        }
+      });
+    }
+  };
+}
+// src/trace-analyst/insights.ts
+var DOMAIN_STOP_WORDS = /* @__PURE__ */ new Set([
+  "and",
+  "advanced",
+  "app",
+  "build",
+  "create",
+  "easy",
+  "expert",
+  "extreme",
+  "for",
+  "from",
+  "hard",
+  "implementation",
+  "integrate",
+  "medium",
+  "project",
+  "task",
+  "the",
+  "this",
+  "with",
+  "workflow"
+]);
+function tokenizeDomainWords(value) {
+  return [...value.matchAll(/[A-Za-z][A-Za-z0-9.+#-]{2,}/g)].map((match) => match[0].toLowerCase()).filter((word) => !DOMAIN_STOP_WORDS.has(word));
+}
+function inferDomainKeywords(suite) {
+  const suiteWords = new Set(tokenizeDomainWords(`${suite.name} ${suite.collectionId ?? ""}`));
+  const source = [
+    suite.name,
+    suite.collectionId ?? "",
+    ...suite.tasks.flatMap((task) => [
+      task.id,
+      task.name,
+      task.prompt ?? "",
+      task.difficulty ?? "",
+      ...task.tags ?? [],
+      ...task.gaps ?? []
+    ])
+  ].join(" ");
+  const counts = /* @__PURE__ */ new Map();
+  for (const word of tokenizeDomainWords(source)) counts.set(word, (counts.get(word) ?? 0) + 1);
+  return [...counts.entries()].filter(([word, count]) => count >= 2 || suiteWords.has(word)).sort((a, b) => b[1] - a[1] || a[0].localeCompare(b[0])).map(([word]) => word).slice(0, 18);
+}
+function domainEvidencePattern(keywords) {
+  const escaped = keywords.filter((keyword) => keyword.length >= 3).map((keyword) => keyword.replace(/[.*+?^${}()|[\]\\]/g, "\\$&"));
+  return escaped.length > 0 ? new RegExp(`(?<![A-Za-z0-9])(?:${escaped.join("|")})(?![A-Za-z0-9])`, "i") : /(?<![A-Za-z0-9])(?:sdk|api|css|dns|xml|provider|client|service|integration|webhook|transaction|auth|oauth|graphql|rest)(?![A-Za-z0-9])/i;
+}
+function describeTraceInsightScope(suite) {
+  const taskLabel = suite.tasks.length === 1 ? "1 implementation task" : `${suite.tasks.length} implementation tasks`;
+  const tags = /* @__PURE__ */ new Map();
+  for (const task of suite.tasks) {
+    for (const tag of task.tags ?? []) tags.set(tag, (tags.get(tag) ?? 0) + 1);
+  }
+  const topTags = [...tags.entries()].sort((a, b) => b[1] - a[1] || a[0].localeCompare(b[0])).slice(0, 8).map(([tag]) => tag);
+  if (topTags.length > 0) return `${taskLabel} across ${topTags.join(", ")}.`;
+  const difficulties = [
+    ...new Set(
+      suite.tasks.map((task) => task.difficulty).filter((value) => Boolean(value))
+    )
+  ].join(", ");
+  return `${taskLabel} across ${difficulties || "the selected benchmark scope"}.`;
+}
+function planTraceInsightQuestions(input) {
+  const hasFailures = input.suite.tasks.some((task) => task.outcome && task.outcome !== "satisfied");
+  const hasMultipleShots = input.suite.tasks.some(
+    (task) => (task.gaps ?? []).some((gap) => /shot|review|retry|continue/i.test(gap))
+  );
+  const questions = [
+    {
+      id: "execution-path",
+      question: "What did the worker actually do before the first meaningful implementation edit?",
+      why: "Separates grounded execution from polished but shallow output."
+    },
+    {
+      id: "research-grounding",
+      question: "Did the worker inspect docs, source, examples, or package references before committing to an implementation path?",
+      why: "Identifies whether failures came from weak retrieval, weak examples, or premature coding."
+    },
+    {
+      id: "domain-proof",
+      question: "Which tasks produced executable domain proof versus UI copy, placeholders, or inferred behavior?",
+      why: "Keeps product-quality claims tied to concrete evidence."
+    },
+    {
+      id: "root-cause",
+      question: "For each major failure cluster, is the likely root cause prompt/scaffold, docs/examples, SDK/API ergonomics, evaluator, runtime, or model behavior?",
+      why: "Turns trace observations into actionable ownership."
+    },
+    {
+      id: "evidence-quality",
+      question: "Which external-facing claims are directly supported by trace ids, span ids, verifier findings, reviewer notes, or generated code?",
+      why: "Prevents unsupported customer-report conclusions."
+    }
+  ];
+  if (hasMultipleShots) {
+    questions.push({
+      id: "reviewer-lift",
+      question: "Where did reviewer feedback improve score, stall, or regress across shots?",
+      why: "Shows whether the driver loop is learning or merely repeating work."
+    });
+  }
+  if (hasFailures) {
+    questions.push({
+      id: "optimization-targets",
+      question: "Which prompt, evaluator, scaffold, or workflow changes should feed the next GEPA/autoresearch optimization run?",
+      why: "Connects benchmark evidence to the optimization loop."
+    });
+  }
+  return questions;
+}
+function buildTraceInsightContext(input) {
+  return {
+    suite: input.suite,
+    scope: describeTraceInsightScope(input.suite),
+    keywords: inferDomainKeywords(input.suite),
+    questions: planTraceInsightQuestions(input),
+    panel: defaultTraceInsightPanel(),
+    findings: input.findings ?? [],
+    agent: input.agent ?? null,
+    totals: input.totals ?? null
+  };
+}
+function scoreTraceInsightReadiness(context) {
+  const failedTasks = context.suite.tasks.filter(
+    (task) => task.outcome && task.outcome !== "satisfied"
+  );
+  const findingTaskIds = new Set(context.findings.flatMap((finding) => finding.taskIds));
+  const failedTasksWithFindings = failedTasks.filter((task) => findingTaskIds.has(task.id));
+  const tasksWithGaps = context.suite.tasks.filter((task) => (task.gaps ?? []).length > 0);
+  const gates = [
+    {
+      id: "domain-context",
+      label: "Domain context inferred",
+      passed: context.keywords.length > 0,
+      severity: "high",
+      detail: context.keywords.length > 0 ? `${context.keywords.length} domain terms inferred: ${context.keywords.slice(0, 8).join(", ")}` : "No domain terms were inferred from suite, tasks, prompts, tags, or gaps."
+    },
+    {
+      id: "panel-coverage",
+      label: "Analyst panel planned",
+      passed: context.panel.length >= 4 && context.questions.length >= 5,
+      severity: "high",
+      detail: `${context.panel.length} panel roles and ${context.questions.length} investigation questions planned.`
+    },
+    {
+      id: "failure-coverage",
+      label: "Failures mapped to findings",
+      passed: failedTasks.length === 0 || failedTasksWithFindings.length / failedTasks.length >= 0.5,
+      severity: "critical",
+      detail: failedTasks.length === 0 ? "No failed tasks in suite." : `${failedTasksWithFindings.length}/${failedTasks.length} failed tasks appear in finding clusters.`
+    },
+    {
+      id: "gap-evidence",
+      label: "Task gaps captured",
+      passed: failedTasks.length === 0 || tasksWithGaps.length / failedTasks.length >= 0.5,
+      severity: "medium",
+      detail: `${tasksWithGaps.length} tasks include explicit evaluator or analyst gaps.`
+    }
+  ];
+  const penalty = gates.reduce((sum, gate) => {
+    if (gate.passed) return sum;
+    if (gate.severity === "critical") return sum + 35;
+    if (gate.severity === "high") return sum + 20;
+    if (gate.severity === "medium") return sum + 10;
+    return sum + 5;
+  }, 0);
+  const score = Math.max(0, Math.min(1, 1 - penalty / 100));
+  return {
+    score,
+    grade: score >= 0.9 ? "external-ready" : score >= 0.7 ? "internal-review" : "raw-analysis",
+    gates
+  };
+}
+function defaultTraceInsightPanel() {
+  return [
+    {
+      id: "trace-forensics",
+      name: "Trace Forensics",
+      responsibility: "Reconstruct what the worker did in order, including research, edits, reviewer interventions, verifier feedback, and stop reason."
+    },
+    {
+      id: "root-cause",
+      name: "Root Cause",
+      responsibility: "Map failures to prompt/scaffold, docs/examples, SDK/API/product ergonomics, evaluator, runtime, or model behavior."
+    },
+    {
+      id: "optimization",
+      name: "Optimization",
+      responsibility: "Identify prompt, reviewer, evaluator, scaffold, and GEPA/autoresearch changes that should be tested next."
+    },
+    {
+      id: "external-evidence",
+      name: "External Evidence",
+      responsibility: "Separate customer-safe claims from internal harness findings and reject conclusions without task, trace, span, code, reviewer, or verifier evidence."
+    }
+  ];
+}
+function buildTraceInsightPrompt(input) {
+  const context = buildTraceInsightContext(input);
+  const maxRepresentativeTraces = input.maxRepresentativeTraces ?? 6;
+  return `Analyze this benchmark run and produce evidence-backed trace intelligence.
+Audience:
+- internal AI/product leadership
+- possible customer-facing report for ${input.suite.name}
+Investigation plan:
+${context.questions.map((item, index) => `${index + 1}. ${item.question} (${item.why})`).join("\n")}
+Analyst panel:
+${context.panel.map((role) => `- ${role.name}: ${role.responsibility}`).join("\n")}
+If the task branches are independent, use subagents for the panel roles above and aggregate their findings. Do not run a panel role unless its answer will change the final report.
+Required output:
+1. Executive verdict: what this run proves and does not prove.
+2. The investigation questions you answered and the evidence used.
+3. Failure taxonomy: agent prompting, evaluator/harness, docs/examples, SDK/API/product integration, infra.
+4. Evidence-backed examples with trace ids/task ids and concrete verifier findings.
+5. Highest-ROI fixes for the benchmark harness, prompt/GEPA optimization, and customer-facing product/docs surface.
+6. What is safe for an external report versus what must stay internal.
+7. One rerun plan that would validate lift after optimization.
+Budget:
+- Inspect the dataset overview, the failure summary, and at most ${maxRepresentativeTraces} representative traces.
+- Prefer traces named in the failure summary over broad exploration.
+- Do not do exhaustive trace sweeps.
+- Return the final report as soon as the taxonomy and examples are supported.
+Run summary:
+${JSON.stringify(
+    {
+      suite: input.suite.name,
+      scope: context.scope,
+      inferredKeywords: context.keywords,
+      agent: context.agent,
+      totals: context.totals,
+      findings: context.findings.map((finding) => ({
+        kind: finding.kind,
+        severity: finding.severity,
+        taskCount: finding.taskIds.length,
+        proposedFixClass: finding.proposedFixClass
+      })),
+      failures: input.suite.tasks.filter((task) => task.outcome && task.outcome !== "satisfied").map((task) => ({
+        task: task.id,
+        difficulty: task.difficulty,
+        outcome: task.outcome,
+        score: task.score,
+        gaps: task.gaps ?? []
+      }))
+    },
+    null,
+    2
+  )}
+Use the trace tools. Do not invent facts. Cite task ids. Separate customer-facing claims from internal harness/model findings.`;
+}
 // src/trace/store.ts
 var InMemoryTraceStore = class {
   runs = /* @__PURE__ */ new Map();
@@ -1545,6 +1841,16 @@ export {
   buildTraceAnalystTools,
   traceAnalystFunctionGroup,
   analyzeTraces,
+  traceAnalystOnRunComplete,
+  tokenizeDomainWords,
+  inferDomainKeywords,
+  domainEvidencePattern,
+  describeTraceInsightScope,
+  planTraceInsightQuestions,
+  buildTraceInsightContext,
+  scoreTraceInsightReadiness,
+  defaultTraceInsightPanel,
+  buildTraceInsightPrompt,
   InMemoryTraceStore,
   FileSystemTraceStore,
   OTEL_AGENT_EVAL_SCOPE,
@@ -1558,4 +1864,4 @@ export {
   createReplayFetch,
   iterateRawCalls
 };
-//# sourceMappingURL=chunk-UW4NOOZI.js.map
+//# sourceMappingURL=chunk-HIO4UIS5.js.map