@tangle-network/agent-eval 0.174.0 → 0.176.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (128) hide show
  1. package/CHANGELOG.md +37 -0
  2. package/README.md +1 -1
  3. package/dist/adapters/http.d.ts +1 -1
  4. package/dist/agent-profile-cell-0gSi5ffD.js +374 -0
  5. package/dist/agent-profile-cell-0gSi5ffD.js.map +1 -0
  6. package/dist/analyst/index.d.ts +3 -3
  7. package/dist/analyst/index.js +4 -4
  8. package/dist/{benchmark-command-mZIlR-ra.js → benchmark-command-yPqjcZnC.js} +7 -7
  9. package/dist/{benchmark-command-mZIlR-ra.js.map → benchmark-command-yPqjcZnC.js.map} +1 -1
  10. package/dist/benchmarks/index.d.ts +1 -1
  11. package/dist/benchmarks/index.js +3 -3
  12. package/dist/campaign/index.d.ts +3 -3
  13. package/dist/campaign/index.js +9 -9
  14. package/dist/{campaign-BzMSCejE.js → campaign-85igdlgG.js} +12 -12
  15. package/dist/{campaign-BzMSCejE.js.map → campaign-85igdlgG.js.map} +1 -1
  16. package/dist/campaign-evidence-D8DBLqLI.js +2083 -0
  17. package/dist/campaign-evidence-D8DBLqLI.js.map +1 -0
  18. package/dist/{opencode-sqlite-eK6HW6dr.js → claude-jsonl-CxZZrDJ3.js} +9 -149
  19. package/dist/claude-jsonl-CxZZrDJ3.js.map +1 -0
  20. package/dist/cli.js +9 -2
  21. package/dist/cli.js.map +1 -1
  22. package/dist/contract/index.d.ts +4 -4
  23. package/dist/contract/index.js +9 -9
  24. package/dist/{default-registry-CrAp0pYq.js → default-registry-DBqVI4pq.js} +2 -2
  25. package/dist/{default-registry-CrAp0pYq.js.map → default-registry-DBqVI4pq.js.map} +1 -1
  26. package/dist/{define-agent-eval-V1jQyCDR.d.ts → define-agent-eval-CCbl8k2E.d.ts} +11 -4
  27. package/dist/define-agent-eval-CCbl8k2E.d.ts.map +1 -0
  28. package/dist/{define-agent-eval-ox5McL6e.js → define-agent-eval-DEMsu5eA.js} +54 -36
  29. package/dist/define-agent-eval-DEMsu5eA.js.map +1 -0
  30. package/dist/{dspy-rlm-engine-Caz2pl4L.js → dspy-rlm-engine-DqjER2sV.js} +2 -2
  31. package/dist/{dspy-rlm-engine-Caz2pl4L.js.map → dspy-rlm-engine-DqjER2sV.js.map} +1 -1
  32. package/dist/{eval-campaign-BeAjdhzC.js → eval-campaign-Cs-7MiCs.js} +4 -5
  33. package/dist/{eval-campaign-BeAjdhzC.js.map → eval-campaign-Cs-7MiCs.js.map} +1 -1
  34. package/dist/experiment/index.d.ts +3 -68
  35. package/dist/experiment/index.d.ts.map +1 -1
  36. package/dist/experiment/index.js +6 -128
  37. package/dist/experiment/index.js.map +1 -1
  38. package/dist/{attestation-XSUpbc4o.js → experiment-tracker-BKEumQug.js} +2 -96
  39. package/dist/experiment-tracker-BKEumQug.js.map +1 -0
  40. package/dist/{attestation-c1QvaBdX.d.ts → experiment-tracker-CNwqCZFD.d.ts} +2 -78
  41. package/dist/experiment-tracker-CNwqCZFD.d.ts.map +1 -0
  42. package/dist/{external-optimizer-process-CxnFL1hd.js → external-optimizer-process-Dlz8YxrT.js} +3 -3
  43. package/dist/{external-optimizer-process-CxnFL1hd.js.map → external-optimizer-process-Dlz8YxrT.js.map} +1 -1
  44. package/dist/{external-optimizer-subprocess-CQi27uEI.js → external-optimizer-subprocess-q3VzlGAO.js} +2 -2
  45. package/dist/{external-optimizer-subprocess-CQi27uEI.js.map → external-optimizer-subprocess-q3VzlGAO.js.map} +1 -1
  46. package/dist/{index-Bn-nlnSV.d.ts → index-BAAiSF3_.d.ts} +2 -2
  47. package/dist/{index-Bn-nlnSV.d.ts.map → index-BAAiSF3_.d.ts.map} +1 -1
  48. package/dist/{index-DKXuBPXf.d.ts → index-Bg6OT2Dd.d.ts} +23 -10
  49. package/dist/{index-DKXuBPXf.d.ts.map → index-Bg6OT2Dd.d.ts.map} +1 -1
  50. package/dist/{index-BTrx5s8m.d.ts → index-DBkcm_9H.d.ts} +4 -4
  51. package/dist/{index-BTrx5s8m.d.ts.map → index-DBkcm_9H.d.ts.map} +1 -1
  52. package/dist/{index-D-UdhAmg.d.ts → index-u0d1Jp4F.d.ts} +4 -2
  53. package/dist/{index-D-UdhAmg.d.ts.map → index-u0d1Jp4F.d.ts.map} +1 -1
  54. package/dist/index.d.ts +5 -5
  55. package/dist/index.js +15 -16
  56. package/dist/index.js.map +1 -1
  57. package/dist/{integrity-BWywb34E.js → integrity-DsHWCebQ.js} +11 -435
  58. package/dist/integrity-DsHWCebQ.js.map +1 -0
  59. package/dist/ledger-core/index.d.ts +2 -2
  60. package/dist/ledger-core/index.js +2 -2
  61. package/dist/{ledger-core-PIfjCbKn.js → ledger-core-Cs9f7385.js} +60 -47
  62. package/dist/{ledger-core-PIfjCbKn.js.map → ledger-core-Cs9f7385.js.map} +1 -1
  63. package/dist/{llm-judge-DmNaBrXB.js → llm-judge-DliimmRb.js} +994 -1517
  64. package/dist/llm-judge-DliimmRb.js.map +1 -0
  65. package/dist/{mint-vWOdD8Ae.js → mint-Cc1_zwRQ.js} +2 -2
  66. package/dist/{mint-vWOdD8Ae.js.map → mint-Cc1_zwRQ.js.map} +1 -1
  67. package/dist/openapi.json +1 -1
  68. package/dist/opencode-sqlite-CNw3vubS.js +145 -0
  69. package/dist/opencode-sqlite-CNw3vubS.js.map +1 -0
  70. package/dist/{produced-state-B8mw6zj9.js → produced-state-DrMqa2HD.js} +3 -2
  71. package/dist/{produced-state-B8mw6zj9.js.map → produced-state-DrMqa2HD.js.map} +1 -1
  72. package/dist/profile-cell.js +1 -268
  73. package/dist/{promotion-policy-LY9mVQ7W.js → promotion-policy-DWOm70gx.js} +2 -2
  74. package/dist/{promotion-policy-LY9mVQ7W.js.map → promotion-policy-DWOm70gx.js.map} +1 -1
  75. package/dist/{release-confidence-BsGEg_xg.js → release-confidence-BcGCclTB.js} +2 -2
  76. package/dist/{release-confidence-BsGEg_xg.js.map → release-confidence-BcGCclTB.js.map} +1 -1
  77. package/dist/report-command-DKlXfU5r.js +1528 -0
  78. package/dist/report-command-DKlXfU5r.js.map +1 -0
  79. package/dist/reporting.js +2 -2
  80. package/dist/{reward-hacking-CKW4teig.js → reward-hacking-D0XwhVWE.js} +2 -215
  81. package/dist/reward-hacking-D0XwhVWE.js.map +1 -0
  82. package/dist/rl.js +5 -4
  83. package/dist/rl.js.map +1 -1
  84. package/dist/rollout/index.js +4 -3
  85. package/dist/{rollout-C-znbbYg.js → rollout-DmoJVqrF.js} +4 -3
  86. package/dist/{rollout-C-znbbYg.js.map → rollout-DmoJVqrF.js.map} +1 -1
  87. package/dist/run-record-CR63CpHK.js +216 -0
  88. package/dist/run-record-CR63CpHK.js.map +1 -0
  89. package/dist/{run-record-ZIsR9Fif.js → run-record-DQpSf7t-.js} +2 -2
  90. package/dist/{run-record-ZIsR9Fif.js.map → run-record-DQpSf7t-.js.map} +1 -1
  91. package/dist/{semantic-concept-judge-E3s_fEjB.js → semantic-concept-judge-Dw-f7TEs.js} +3 -3
  92. package/dist/{semantic-concept-judge-E3s_fEjB.js.map → semantic-concept-judge-Dw-f7TEs.js.map} +1 -1
  93. package/dist/{sequential-B51qAYE4.js → sequential-B5gXgcyp.js} +3 -3
  94. package/dist/{sequential-B51qAYE4.js.map → sequential-B5gXgcyp.js.map} +1 -1
  95. package/dist/{skillopt-optimization-method-f7399oGb.js → skillopt-optimization-method-CV7go7ex.js} +6 -749
  96. package/dist/skillopt-optimization-method-CV7go7ex.js.map +1 -0
  97. package/dist/{statistical-heldout-Cqb73yE9.d.ts → statistical-heldout-Z9NROFFS.d.ts} +156 -3
  98. package/dist/statistical-heldout-Z9NROFFS.d.ts.map +1 -0
  99. package/dist/{summary-report-Bgh8CpNK.js → summary-report-B16xy9Kd.js} +2 -2
  100. package/dist/{summary-report-Bgh8CpNK.js.map → summary-report-B16xy9Kd.js.map} +1 -1
  101. package/dist/supervisor-run/index.d.ts +71 -6
  102. package/dist/supervisor-run/index.d.ts.map +1 -1
  103. package/dist/supervisor-run/index.js +6 -1357
  104. package/dist/supervisor-run/index.js.map +1 -1
  105. package/dist/terminal-record-Ce9_UjRz.js +539 -0
  106. package/dist/terminal-record-Ce9_UjRz.js.map +1 -0
  107. package/dist/traces.js +1 -1
  108. package/dist/{types-CoPUTiXb.d.ts → types-vUdAx2Cj.d.ts} +65 -3
  109. package/dist/types-vUdAx2Cj.d.ts.map +1 -0
  110. package/docs/public-api.md +62 -39
  111. package/docs/search-history-receipts.md +48 -1
  112. package/package.json +1 -1
  113. package/dist/attestation-XSUpbc4o.js.map +0 -1
  114. package/dist/attestation-c1QvaBdX.d.ts.map +0 -1
  115. package/dist/define-agent-eval-V1jQyCDR.d.ts.map +0 -1
  116. package/dist/define-agent-eval-ox5McL6e.js.map +0 -1
  117. package/dist/integrity-BWywb34E.js.map +0 -1
  118. package/dist/llm-judge-DmNaBrXB.js.map +0 -1
  119. package/dist/opencode-sqlite-eK6HW6dr.js.map +0 -1
  120. package/dist/power-preflight-CFXm0Vjo.js +0 -502
  121. package/dist/power-preflight-CFXm0Vjo.js.map +0 -1
  122. package/dist/pre-registration-D94b7Of5.js +0 -110
  123. package/dist/pre-registration-D94b7Of5.js.map +0 -1
  124. package/dist/profile-cell.js.map +0 -1
  125. package/dist/reward-hacking-CKW4teig.js.map +0 -1
  126. package/dist/skillopt-optimization-method-f7399oGb.js.map +0 -1
  127. package/dist/statistical-heldout-Cqb73yE9.d.ts.map +0 -1
  128. package/dist/types-CoPUTiXb.d.ts.map +0 -1
@@ -1 +0,0 @@
1
- {"version":3,"file":"opencode-sqlite-eK6HW6dr.js","names":["isRecord"],"sources":["../src/rollout/readers/claude-jsonl.ts","../src/rollout/readers/opencode-sqlite.ts"],"sourcesContent":["/**\n * Backfill reader over Claude Code project transcripts\n * (~/.claude/projects/<cwd-slug>/<sessionId>.jsonl) → canonical\n * chat-with-tools messages plus per-session token usage.\n *\n * Transcript lines consumed: type:\"user\" (string content or content blocks —\n * text + tool_result) and type:\"assistant\" (content blocks — thinking, text,\n * tool_use; message.usage carries tokens). Sidechain lines (isSidechain=true,\n * subagent threads) are separate invocations and are excluded from the main\n * transcript. Everything else (queue-operation, attachment, last-prompt…) is\n * transport metadata, not conversation.\n */\n\nimport { readdir, readFile } from 'node:fs/promises'\nimport { homedir } from 'node:os'\nimport { join } from 'node:path'\nimport type { ChatMessage, ChatToolCall } from '../schema'\n\nexport const DEFAULT_CLAUDE_PROJECTS_DIR = join(homedir(), '.claude', 'projects')\n\n/** Claude Code's project-directory slug for a working directory. */\nexport function claudeProjectSlug(cwd: string): string {\n return cwd.replace(/[^a-zA-Z0-9-]/g, '-')\n}\n\nexport interface ClaudeTranscriptRef {\n sessionId: string\n path: string\n}\n\n/** Transcript files recorded for sessions launched from `cwd`. */\nexport async function findClaudeTranscripts(\n cwd: string,\n projectsDir: string = DEFAULT_CLAUDE_PROJECTS_DIR,\n): Promise<ClaudeTranscriptRef[]> {\n const dir = join(projectsDir, claudeProjectSlug(cwd))\n const names = await readdir(dir).catch(() => [])\n return names\n .filter((n) => n.endsWith('.jsonl'))\n .sort()\n .map((n) => ({ sessionId: n.replace(/\\.jsonl$/, ''), path: join(dir, n) }))\n}\n\nexport interface ClaudeUsageTotals {\n tokensIn: number\n tokensOut: number\n cacheRead: number\n cacheWrite: number\n}\n\nexport interface ClaudeTranscript {\n messages: ChatMessage[]\n usage: ClaudeUsageTotals\n /** Timestamp of the first conversation line; null = empty transcript. */\n startedAt: string | null\n endedAt: string | null\n model: string | null\n}\n\nconst isRecord = (v: unknown): v is Record<string, unknown> =>\n typeof v === 'object' && v !== null && !Array.isArray(v)\n\n/**\n * One conversation line of a transcript, still in Claude Code's own shape.\n *\n * This is the single line-level parse of the format. `readClaudeTranscript`\n * projects it to canonical messages + usage; the supervision-tree reader\n * (`src/supervisor-run/claude-code-reader.ts`) projects the SAME entries to\n * spawn/settle/steer instants. Two projections, one parser — a second\n * transcript parser is how the two views silently disagree.\n */\nexport interface ClaudeEntry {\n readonly type: 'user' | 'assistant'\n /** ISO instant of the line; null when the line carried none. */\n readonly timestamp: string | null\n /** The Anthropic message body (`role`, `content`, `model`, `usage`). */\n readonly message: Record<string, unknown>\n /** Claude Code's structured tool result, when the line carries one. */\n readonly toolUseResult: unknown\n /** True on subagent threads — a separate invocation, not this transcript's turn. */\n readonly isSidechain: boolean\n /** Subagent id Claude Code stamps on sidechain lines; null on main-thread lines. */\n readonly agentId: string | null\n}\n\n/** Parse transcript jsonl text into conversation lines. Non-conversation lines are dropped. */\nexport function parseClaudeEntries(raw: string): ClaudeEntry[] {\n const out: ClaudeEntry[] = []\n for (const line of raw.split('\\n')) {\n if (!line.trim()) continue\n let entry: Record<string, unknown>\n try {\n const parsed: unknown = JSON.parse(line)\n if (!isRecord(parsed)) continue\n entry = parsed\n } catch {\n continue\n }\n if (entry.type !== 'user' && entry.type !== 'assistant') continue\n const message = entry.message\n if (!isRecord(message)) continue\n out.push({\n type: entry.type,\n timestamp: typeof entry.timestamp === 'string' ? entry.timestamp : null,\n message,\n toolUseResult: entry.toolUseResult,\n isSidechain: entry.isSidechain === true,\n agentId: typeof entry.agentId === 'string' ? entry.agentId : null,\n })\n }\n return out\n}\n\nexport interface ReadClaudeTranscriptOptions {\n /**\n * Read the sidechain (subagent) thread instead of skipping it. Subagent\n * transcripts under `<session>/subagents/agent-<id>.jsonl` are sidechain\n * lines end to end, so their usage is invisible without this.\n */\n readonly includeSidechain?: boolean\n}\n\nfunction blockText(content: unknown): string {\n if (typeof content === 'string') return content\n if (!Array.isArray(content)) return ''\n return content\n .filter(\n (b): b is Record<string, unknown> =>\n isRecord(b) && b.type === 'text' && typeof b.text === 'string',\n )\n .map((b) => b.text as string)\n .join('\\n')\n}\n\n/** Parse one transcript jsonl into canonical messages + usage totals. */\nexport async function readClaudeTranscript(\n path: string,\n options: ReadClaudeTranscriptOptions = {},\n): Promise<ClaudeTranscript> {\n return transcriptFromEntries(parseClaudeEntries(await readFile(path, 'utf8')), options)\n}\n\n/** The messages+usage projection of already-parsed entries. */\nexport function transcriptFromEntries(\n entries: readonly ClaudeEntry[],\n options: ReadClaudeTranscriptOptions = {},\n): ClaudeTranscript {\n const wantSidechain = options.includeSidechain === true\n const messages: ChatMessage[] = []\n const usage: ClaudeUsageTotals = { tokensIn: 0, tokensOut: 0, cacheRead: 0, cacheWrite: 0 }\n let startedAt: string | null = null\n let endedAt: string | null = null\n let model: string | null = null\n // Claude Code writes one jsonl line PER CONTENT BLOCK of an API message,\n // repeating message.id and usage on each — merge blocks into one canonical\n // assistant turn and count usage once per API message id.\n let lastAssistantApiId: string | null = null\n let lastAssistantIndex = -1\n\n for (const entry of entries) {\n if (entry.isSidechain !== wantSidechain) continue\n const message = entry.message\n if (entry.timestamp !== null) {\n if (startedAt === null) startedAt = entry.timestamp\n endedAt = entry.timestamp\n }\n\n if (entry.type === 'user') {\n lastAssistantApiId = null\n lastAssistantIndex = -1\n const content = message.content\n if (typeof content === 'string') {\n messages.push({ role: 'user', content })\n continue\n }\n if (!Array.isArray(content)) continue\n // A user line may interleave tool_result blocks (answers to the prior\n // assistant tool_use) with plain text; preserve order.\n let userText = ''\n for (const block of content) {\n if (!isRecord(block)) continue\n if (block.type === 'tool_result' && typeof block.tool_use_id === 'string') {\n messages.push({\n role: 'tool',\n tool_call_id: block.tool_use_id,\n content:\n blockText(block.content) || (typeof block.content === 'string' ? block.content : ''),\n })\n } else if (block.type === 'text' && typeof block.text === 'string') {\n userText += (userText.length > 0 ? '\\n' : '') + block.text\n }\n }\n if (userText.length > 0) messages.push({ role: 'user', content: userText })\n continue\n }\n\n // assistant\n if (typeof message.model === 'string') model = message.model\n const apiId = typeof message.id === 'string' ? message.id : null\n const continuesTurn = apiId !== null && apiId === lastAssistantApiId && lastAssistantIndex >= 0\n const msgUsage = message.usage\n if (isRecord(msgUsage) && !continuesTurn) {\n usage.tokensIn += typeof msgUsage.input_tokens === 'number' ? msgUsage.input_tokens : 0\n usage.tokensOut += typeof msgUsage.output_tokens === 'number' ? msgUsage.output_tokens : 0\n usage.cacheRead +=\n typeof msgUsage.cache_read_input_tokens === 'number' ? msgUsage.cache_read_input_tokens : 0\n usage.cacheWrite +=\n typeof msgUsage.cache_creation_input_tokens === 'number'\n ? msgUsage.cache_creation_input_tokens\n : 0\n }\n const content = message.content\n if (!Array.isArray(content)) continue\n let reasoning = ''\n let text = ''\n const toolCalls: ChatToolCall[] = []\n for (const block of content) {\n if (!isRecord(block)) continue\n if (\n block.type === 'thinking' &&\n typeof block.thinking === 'string' &&\n block.thinking.length > 0\n ) {\n reasoning += (reasoning.length > 0 ? '\\n' : '') + block.thinking\n } else if (block.type === 'text' && typeof block.text === 'string') {\n text += (text.length > 0 ? '\\n' : '') + block.text\n } else if (block.type === 'tool_use' && typeof block.id === 'string') {\n toolCalls.push({\n id: block.id,\n type: 'function',\n function: {\n name: typeof block.name === 'string' ? block.name : 'unknown',\n arguments: JSON.stringify(block.input ?? {}),\n },\n })\n }\n }\n if (reasoning.length === 0 && text.length === 0 && toolCalls.length === 0) continue\n if (continuesTurn) {\n const prev = messages[lastAssistantIndex]!\n if (text.length > 0) prev.content = prev.content === null ? text : `${prev.content}\\n${text}`\n if (reasoning.length > 0) {\n prev.reasoning_content =\n prev.reasoning_content === undefined\n ? reasoning\n : `${prev.reasoning_content}\\n${reasoning}`\n }\n if (toolCalls.length > 0) prev.tool_calls = [...(prev.tool_calls ?? []), ...toolCalls]\n continue\n }\n messages.push({\n role: 'assistant',\n content: text.length > 0 ? text : null,\n ...(reasoning.length > 0 ? { reasoning_content: reasoning } : {}),\n ...(toolCalls.length > 0 ? { tool_calls: toolCalls } : {}),\n })\n lastAssistantApiId = apiId\n lastAssistantIndex = messages.length - 1\n }\n\n return { messages, usage, startedAt, endedAt, model }\n}\n","/**\n * Read-only backfill reader over the opencode sqlite store\n * (~/.local/share/opencode/opencode.db) → canonical chat-with-tools messages.\n *\n * Schema consumed (observed, 2026-07): `session` rows carry directory /\n * parent_id / agent / model / cost / tokens_*; `message` rows carry a JSON\n * `data` blob ({role, modelID, providerID, tokens, cost, finish}); `part`\n * rows carry the actual content ({type: text|reasoning|tool|step-start|\n * step-finish|snapshot…}). Tool parts hold {callID, state:{input, output,\n * status}} — both the call and its result, which we split into an assistant\n * tool_call plus a role:\"tool\" result message.\n *\n * The store is mutable and can be corrupt (a `.corrupt-bak` sibling ships\n * next to it in the wild), so `openOpencodeDb` returns null instead of\n * throwing — callers record a gap line, never crash the backfill.\n */\n\nimport { createRequire } from 'node:module'\nimport { homedir } from 'node:os'\nimport { join } from 'node:path'\nimport type { DatabaseSync } from 'node:sqlite'\nimport type { ChatMessage, ChatToolCall } from '../schema'\n\nexport const DEFAULT_OPENCODE_DB = join(homedir(), '.local', 'share', 'opencode', 'opencode.db')\n\nexport interface OpencodeSessionRow {\n id: string\n parentId: string | null\n directory: string\n agent: string | null\n /** Raw session.model JSON: {id, providerID, variant} where present. */\n model: { id?: string; providerID?: string } | null\n costUsd: number\n tokensInput: number\n tokensOutput: number\n tokensReasoning: number\n tokensCacheRead: number\n tokensCacheWrite: number\n timeCreated: number\n timeUpdated: number\n}\n\n// `node:sqlite` is loaded through CommonJS `require`, not `import()`. esbuild\n// (bundling) and Vite (tests) both rewrite a dynamic import and strip the\n// `node:` prefix under an es20xx target, turning this builtin into a bogus\n// \"sqlite\" package lookup; composing the specifier at runtime does not reliably\n// defeat that (it still resolved through Vite's transform in some workers, so\n// the failure moved around as test files were added). A require obtained from\n// `createRequire` is not an analyzable module reference in either tool, so\n// neither can rewrite it.\n/** Open the store read-only; null = unavailable/corrupt (caller records a gap). */\nexport async function openOpencodeDb(\n path: string = DEFAULT_OPENCODE_DB,\n): Promise<DatabaseSync | null> {\n try {\n // Keep Node-only module initialization inside the Node-only operation.\n // Root imports are shared with edge consumers that do not define import.meta.url.\n const nodeRequire = createRequire(import.meta.url)\n const { DatabaseSync } = nodeRequire('node:sqlite') as typeof import('node:sqlite')\n const db = new DatabaseSync(path, { readOnly: true })\n // Probe: a corrupt store can open() fine and fail on first page read.\n db.prepare('SELECT id FROM session LIMIT 1').get()\n return db\n } catch {\n return null\n }\n}\n\nconst isRecord = (v: unknown): v is Record<string, unknown> =>\n typeof v === 'object' && v !== null && !Array.isArray(v)\n\nfunction parseSessionRow(row: Record<string, unknown>): OpencodeSessionRow {\n let model: OpencodeSessionRow['model'] = null\n if (typeof row.model === 'string' && row.model.length > 0) {\n try {\n const parsed: unknown = JSON.parse(row.model)\n if (isRecord(parsed)) model = parsed as { id?: string; providerID?: string }\n } catch {\n model = null\n }\n }\n return {\n id: String(row.id),\n parentId: row.parent_id === null || row.parent_id === undefined ? null : String(row.parent_id),\n directory: String(row.directory),\n agent: row.agent === null || row.agent === undefined ? null : String(row.agent),\n model,\n costUsd: Number(row.cost ?? 0),\n tokensInput: Number(row.tokens_input ?? 0),\n tokensOutput: Number(row.tokens_output ?? 0),\n tokensReasoning: Number(row.tokens_reasoning ?? 0),\n tokensCacheRead: Number(row.tokens_cache_read ?? 0),\n tokensCacheWrite: Number(row.tokens_cache_write ?? 0),\n timeCreated: Number(row.time_created ?? 0),\n timeUpdated: Number(row.time_updated ?? 0),\n }\n}\n\nconst SESSION_COLUMNS =\n 'id, parent_id, directory, agent, model, cost, tokens_input, tokens_output, tokens_reasoning, tokens_cache_read, tokens_cache_write, time_created, time_updated'\n\n/** Sessions whose cwd is `directory` (the worker-clone join key). */\nexport function findOpencodeSessionsByDirectory(\n db: DatabaseSync,\n directory: string,\n): OpencodeSessionRow[] {\n const rows = db\n .prepare(`SELECT ${SESSION_COLUMNS} FROM session WHERE directory = ? ORDER BY time_created`)\n .all(directory) as Array<Record<string, unknown>>\n return rows.map(parseSessionRow)\n}\n\ninterface OpencodePart {\n type?: string\n text?: string\n tool?: string\n callID?: string\n state?: { status?: string; input?: unknown; output?: unknown }\n}\n\nfunction toolResultContent(output: unknown): string {\n if (typeof output === 'string') return output\n if (output === null || output === undefined) return ''\n return JSON.stringify(output)\n}\n\n/**\n * Convert one session's message+part rows into canonical messages.\n * An opencode assistant message row spans several model steps; each step's\n * parts (reasoning → text → tool …) become one assistant message followed by\n * the role:\"tool\" results of its calls, preserving order.\n */\nexport function readOpencodeSessionMessages(db: DatabaseSync, sessionId: string): ChatMessage[] {\n const messageRows = db\n .prepare('SELECT id, data FROM message WHERE session_id = ? ORDER BY time_created, id')\n .all(sessionId) as Array<{ id: string; data: string }>\n const partsStmt = db.prepare('SELECT data FROM part WHERE message_id = ? ORDER BY id')\n\n const messages: ChatMessage[] = []\n for (const messageRow of messageRows) {\n let data: Record<string, unknown>\n try {\n const parsed: unknown = JSON.parse(messageRow.data)\n if (!isRecord(parsed)) continue\n data = parsed\n } catch {\n continue\n }\n const parts: OpencodePart[] = []\n for (const row of partsStmt.all(messageRow.id) as Array<{ data: string }>) {\n try {\n const parsed: unknown = JSON.parse(row.data)\n if (isRecord(parsed)) parts.push(parsed as OpencodePart)\n } catch {\n // Malformed part payload: skip the part, keep the message.\n }\n }\n\n if (data.role === 'user') {\n const text = parts\n .filter((p) => p.type === 'text' && typeof p.text === 'string')\n .map((p) => p.text as string)\n .join('\\n')\n messages.push({ role: 'user', content: text })\n continue\n }\n if (data.role !== 'assistant') continue\n\n // Split the row into steps at step-start boundaries; parts before the\n // first step-start (none observed, but tolerated) form an implicit step.\n const steps: OpencodePart[][] = []\n let current: OpencodePart[] = []\n for (const part of parts) {\n if (part.type === 'step-start') {\n if (current.length > 0) steps.push(current)\n current = []\n continue\n }\n if (part.type === 'step-finish' || part.type === 'snapshot' || part.type === 'patch') continue\n current.push(part)\n }\n if (current.length > 0) steps.push(current)\n\n for (const step of steps) {\n const reasoning = step\n .filter((p) => p.type === 'reasoning' && typeof p.text === 'string' && p.text.length > 0)\n .map((p) => p.text as string)\n .join('\\n')\n const text = step\n .filter((p) => p.type === 'text' && typeof p.text === 'string')\n .map((p) => p.text as string)\n .join('\\n')\n const toolParts = step.filter((p) => p.type === 'tool' && typeof p.callID === 'string')\n const toolCalls: ChatToolCall[] = toolParts.map((p) => ({\n id: p.callID as string,\n type: 'function',\n function: {\n name: p.tool ?? 'unknown',\n arguments: JSON.stringify(p.state?.input ?? {}),\n },\n }))\n if (reasoning.length === 0 && text.length === 0 && toolCalls.length === 0) continue\n messages.push({\n role: 'assistant',\n content: text.length > 0 ? text : null,\n ...(reasoning.length > 0 ? { reasoning_content: reasoning } : {}),\n ...(toolCalls.length > 0 ? { tool_calls: toolCalls } : {}),\n })\n for (const p of toolParts) {\n messages.push({\n role: 'tool',\n tool_call_id: p.callID as string,\n name: p.tool ?? 'unknown',\n content: toolResultContent(p.state?.output),\n })\n }\n }\n }\n return messages\n}\n"],"mappings":";;;;;;;;;;;;;;;;;AAkBA,MAAa,8BAA8B,KAAK,QAAQ,GAAG,WAAW,UAAU;;AAGhF,SAAgB,kBAAkB,KAAqB;CACrD,OAAO,IAAI,QAAQ,kBAAkB,GAAG;AAC1C;;AAQA,eAAsB,sBACpB,KACA,cAAsB,6BACU;CAChC,MAAM,MAAM,KAAK,aAAa,kBAAkB,GAAG,CAAC;CAEpD,QAAO,MADa,QAAQ,GAAG,CAAC,CAAC,YAAY,CAAC,CAAC,EAAA,CAE5C,QAAQ,MAAM,EAAE,SAAS,QAAQ,CAAC,CAAC,CACnC,KAAK,CAAC,CACN,KAAK,OAAO;EAAE,WAAW,EAAE,QAAQ,YAAY,EAAE;EAAG,MAAM,KAAK,KAAK,CAAC;CAAE,EAAE;AAC9E;AAkBA,MAAMA,cAAY,MAChB,OAAO,MAAM,YAAY,MAAM,QAAQ,CAAC,MAAM,QAAQ,CAAC;;AA0BzD,SAAgB,mBAAmB,KAA4B;CAC7D,MAAM,MAAqB,CAAC;CAC5B,KAAK,MAAM,QAAQ,IAAI,MAAM,IAAI,GAAG;EAClC,IAAI,CAAC,KAAK,KAAK,GAAG;EAClB,IAAI;EACJ,IAAI;GACF,MAAM,SAAkB,KAAK,MAAM,IAAI;GACvC,IAAI,CAACA,WAAS,MAAM,GAAG;GACvB,QAAQ;EACV,QAAQ;GACN;EACF;EACA,IAAI,MAAM,SAAS,UAAU,MAAM,SAAS,aAAa;EACzD,MAAM,UAAU,MAAM;EACtB,IAAI,CAACA,WAAS,OAAO,GAAG;EACxB,IAAI,KAAK;GACP,MAAM,MAAM;GACZ,WAAW,OAAO,MAAM,cAAc,WAAW,MAAM,YAAY;GACnE;GACA,eAAe,MAAM;GACrB,aAAa,MAAM,gBAAgB;GACnC,SAAS,OAAO,MAAM,YAAY,WAAW,MAAM,UAAU;EAC/D,CAAC;CACH;CACA,OAAO;AACT;AAWA,SAAS,UAAU,SAA0B;CAC3C,IAAI,OAAO,YAAY,UAAU,OAAO;CACxC,IAAI,CAAC,MAAM,QAAQ,OAAO,GAAG,OAAO;CACpC,OAAO,QACJ,QACE,MACCA,WAAS,CAAC,KAAK,EAAE,SAAS,UAAU,OAAO,EAAE,SAAS,QAC1D,CAAC,CACA,KAAK,MAAM,EAAE,IAAc,CAAC,CAC5B,KAAK,IAAI;AACd;;AAGA,eAAsB,qBACpB,MACA,UAAuC,CAAC,GACb;CAC3B,OAAO,sBAAsB,mBAAmB,MAAM,SAAS,MAAM,MAAM,CAAC,GAAG,OAAO;AACxF;;AAGA,SAAgB,sBACd,SACA,UAAuC,CAAC,GACtB;CAClB,MAAM,gBAAgB,QAAQ,qBAAqB;CACnD,MAAM,WAA0B,CAAC;CACjC,MAAM,QAA2B;EAAE,UAAU;EAAG,WAAW;EAAG,WAAW;EAAG,YAAY;CAAE;CAC1F,IAAI,YAA2B;CAC/B,IAAI,UAAyB;CAC7B,IAAI,QAAuB;CAI3B,IAAI,qBAAoC;CACxC,IAAI,qBAAqB;CAEzB,KAAK,MAAM,SAAS,SAAS;EAC3B,IAAI,MAAM,gBAAgB,eAAe;EACzC,MAAM,UAAU,MAAM;EACtB,IAAI,MAAM,cAAc,MAAM;GAC5B,IAAI,cAAc,MAAM,YAAY,MAAM;GAC1C,UAAU,MAAM;EAClB;EAEA,IAAI,MAAM,SAAS,QAAQ;GACzB,qBAAqB;GACrB,qBAAqB;GACrB,MAAM,UAAU,QAAQ;GACxB,IAAI,OAAO,YAAY,UAAU;IAC/B,SAAS,KAAK;KAAE,MAAM;KAAQ;IAAQ,CAAC;IACvC;GACF;GACA,IAAI,CAAC,MAAM,QAAQ,OAAO,GAAG;GAG7B,IAAI,WAAW;GACf,KAAK,MAAM,SAAS,SAAS;IAC3B,IAAI,CAACA,WAAS,KAAK,GAAG;IACtB,IAAI,MAAM,SAAS,iBAAiB,OAAO,MAAM,gBAAgB,UAC/D,SAAS,KAAK;KACZ,MAAM;KACN,cAAc,MAAM;KACpB,SACE,UAAU,MAAM,OAAO,MAAM,OAAO,MAAM,YAAY,WAAW,MAAM,UAAU;IACrF,CAAC;SACI,IAAI,MAAM,SAAS,UAAU,OAAO,MAAM,SAAS,UACxD,aAAa,SAAS,SAAS,IAAI,OAAO,MAAM,MAAM;GAE1D;GACA,IAAI,SAAS,SAAS,GAAG,SAAS,KAAK;IAAE,MAAM;IAAQ,SAAS;GAAS,CAAC;GAC1E;EACF;EAGA,IAAI,OAAO,QAAQ,UAAU,UAAU,QAAQ,QAAQ;EACvD,MAAM,QAAQ,OAAO,QAAQ,OAAO,WAAW,QAAQ,KAAK;EAC5D,MAAM,gBAAgB,UAAU,QAAQ,UAAU,sBAAsB,sBAAsB;EAC9F,MAAM,WAAW,QAAQ;EACzB,IAAIA,WAAS,QAAQ,KAAK,CAAC,eAAe;GACxC,MAAM,YAAY,OAAO,SAAS,iBAAiB,WAAW,SAAS,eAAe;GACtF,MAAM,aAAa,OAAO,SAAS,kBAAkB,WAAW,SAAS,gBAAgB;GACzF,MAAM,aACJ,OAAO,SAAS,4BAA4B,WAAW,SAAS,0BAA0B;GAC5F,MAAM,cACJ,OAAO,SAAS,gCAAgC,WAC5C,SAAS,8BACT;EACR;EACA,MAAM,UAAU,QAAQ;EACxB,IAAI,CAAC,MAAM,QAAQ,OAAO,GAAG;EAC7B,IAAI,YAAY;EAChB,IAAI,OAAO;EACX,MAAM,YAA4B,CAAC;EACnC,KAAK,MAAM,SAAS,SAAS;GAC3B,IAAI,CAACA,WAAS,KAAK,GAAG;GACtB,IACE,MAAM,SAAS,cACf,OAAO,MAAM,aAAa,YAC1B,MAAM,SAAS,SAAS,GAExB,cAAc,UAAU,SAAS,IAAI,OAAO,MAAM,MAAM;QACnD,IAAI,MAAM,SAAS,UAAU,OAAO,MAAM,SAAS,UACxD,SAAS,KAAK,SAAS,IAAI,OAAO,MAAM,MAAM;QACzC,IAAI,MAAM,SAAS,cAAc,OAAO,MAAM,OAAO,UAC1D,UAAU,KAAK;IACb,IAAI,MAAM;IACV,MAAM;IACN,UAAU;KACR,MAAM,OAAO,MAAM,SAAS,WAAW,MAAM,OAAO;KACpD,WAAW,KAAK,UAAU,MAAM,SAAS,CAAC,CAAC;IAC7C;GACF,CAAC;EAEL;EACA,IAAI,UAAU,WAAW,KAAK,KAAK,WAAW,KAAK,UAAU,WAAW,GAAG;EAC3E,IAAI,eAAe;GACjB,MAAM,OAAO,SAAS;GACtB,IAAI,KAAK,SAAS,GAAG,KAAK,UAAU,KAAK,YAAY,OAAO,OAAO,GAAG,KAAK,QAAQ,IAAI;GACvF,IAAI,UAAU,SAAS,GACrB,KAAK,oBACH,KAAK,sBAAsB,KAAA,IACvB,YACA,GAAG,KAAK,kBAAkB,IAAI;GAEtC,IAAI,UAAU,SAAS,GAAG,KAAK,aAAa,CAAC,GAAI,KAAK,cAAc,CAAC,GAAI,GAAG,SAAS;GACrF;EACF;EACA,SAAS,KAAK;GACZ,MAAM;GACN,SAAS,KAAK,SAAS,IAAI,OAAO;GAClC,GAAI,UAAU,SAAS,IAAI,EAAE,mBAAmB,UAAU,IAAI,CAAC;GAC/D,GAAI,UAAU,SAAS,IAAI,EAAE,YAAY,UAAU,IAAI,CAAC;EAC1D,CAAC;EACD,qBAAqB;EACrB,qBAAqB,SAAS,SAAS;CACzC;CAEA,OAAO;EAAE;EAAU;EAAO;EAAW;EAAS;CAAM;AACtD;;;;;;;;;;;;;;;;;;;AC9OA,MAAa,sBAAsB,KAAK,QAAQ,GAAG,UAAU,SAAS,YAAY,aAAa;;AA4B/F,eAAsB,eACpB,OAAe,qBACe;CAC9B,IAAI;EAIF,MAAM,EAAE,iBADY,cAAc,OAAO,KAAK,GACX,CAAC,CAAC,aAAa;EAClD,MAAM,KAAK,IAAI,aAAa,MAAM,EAAE,UAAU,KAAK,CAAC;EAEpD,GAAG,QAAQ,gCAAgC,CAAC,CAAC,IAAI;EACjD,OAAO;CACT,QAAQ;EACN,OAAO;CACT;AACF;AAEA,MAAM,YAAY,MAChB,OAAO,MAAM,YAAY,MAAM,QAAQ,CAAC,MAAM,QAAQ,CAAC;AAEzD,SAAS,gBAAgB,KAAkD;CACzE,IAAI,QAAqC;CACzC,IAAI,OAAO,IAAI,UAAU,YAAY,IAAI,MAAM,SAAS,GACtD,IAAI;EACF,MAAM,SAAkB,KAAK,MAAM,IAAI,KAAK;EAC5C,IAAI,SAAS,MAAM,GAAG,QAAQ;CAChC,QAAQ;EACN,QAAQ;CACV;CAEF,OAAO;EACL,IAAI,OAAO,IAAI,EAAE;EACjB,UAAU,IAAI,cAAc,QAAQ,IAAI,cAAc,KAAA,IAAY,OAAO,OAAO,IAAI,SAAS;EAC7F,WAAW,OAAO,IAAI,SAAS;EAC/B,OAAO,IAAI,UAAU,QAAQ,IAAI,UAAU,KAAA,IAAY,OAAO,OAAO,IAAI,KAAK;EAC9E;EACA,SAAS,OAAO,IAAI,QAAQ,CAAC;EAC7B,aAAa,OAAO,IAAI,gBAAgB,CAAC;EACzC,cAAc,OAAO,IAAI,iBAAiB,CAAC;EAC3C,iBAAiB,OAAO,IAAI,oBAAoB,CAAC;EACjD,iBAAiB,OAAO,IAAI,qBAAqB,CAAC;EAClD,kBAAkB,OAAO,IAAI,sBAAsB,CAAC;EACpD,aAAa,OAAO,IAAI,gBAAgB,CAAC;EACzC,aAAa,OAAO,IAAI,gBAAgB,CAAC;CAC3C;AACF;AAEA,MAAM,kBACJ;;AAGF,SAAgB,gCACd,IACA,WACsB;CAItB,OAHa,GACV,QAAQ,UAAU,gBAAgB,wDAAwD,CAAC,CAC3F,IAAI,SACG,CAAC,CAAC,IAAI,eAAe;AACjC;AAUA,SAAS,kBAAkB,QAAyB;CAClD,IAAI,OAAO,WAAW,UAAU,OAAO;CACvC,IAAI,WAAW,QAAQ,WAAW,KAAA,GAAW,OAAO;CACpD,OAAO,KAAK,UAAU,MAAM;AAC9B;;;;;;;AAQA,SAAgB,4BAA4B,IAAkB,WAAkC;CAC9F,MAAM,cAAc,GACjB,QAAQ,6EAA6E,CAAC,CACtF,IAAI,SAAS;CAChB,MAAM,YAAY,GAAG,QAAQ,wDAAwD;CAErF,MAAM,WAA0B,CAAC;CACjC,KAAK,MAAM,cAAc,aAAa;EACpC,IAAI;EACJ,IAAI;GACF,MAAM,SAAkB,KAAK,MAAM,WAAW,IAAI;GAClD,IAAI,CAAC,SAAS,MAAM,GAAG;GACvB,OAAO;EACT,QAAQ;GACN;EACF;EACA,MAAM,QAAwB,CAAC;EAC/B,KAAK,MAAM,OAAO,UAAU,IAAI,WAAW,EAAE,GAC3C,IAAI;GACF,MAAM,SAAkB,KAAK,MAAM,IAAI,IAAI;GAC3C,IAAI,SAAS,MAAM,GAAG,MAAM,KAAK,MAAsB;EACzD,QAAQ,CAER;EAGF,IAAI,KAAK,SAAS,QAAQ;GACxB,MAAM,OAAO,MACV,QAAQ,MAAM,EAAE,SAAS,UAAU,OAAO,EAAE,SAAS,QAAQ,CAAC,CAC9D,KAAK,MAAM,EAAE,IAAc,CAAC,CAC5B,KAAK,IAAI;GACZ,SAAS,KAAK;IAAE,MAAM;IAAQ,SAAS;GAAK,CAAC;GAC7C;EACF;EACA,IAAI,KAAK,SAAS,aAAa;EAI/B,MAAM,QAA0B,CAAC;EACjC,IAAI,UAA0B,CAAC;EAC/B,KAAK,MAAM,QAAQ,OAAO;GACxB,IAAI,KAAK,SAAS,cAAc;IAC9B,IAAI,QAAQ,SAAS,GAAG,MAAM,KAAK,OAAO;IAC1C,UAAU,CAAC;IACX;GACF;GACA,IAAI,KAAK,SAAS,iBAAiB,KAAK,SAAS,cAAc,KAAK,SAAS,SAAS;GACtF,QAAQ,KAAK,IAAI;EACnB;EACA,IAAI,QAAQ,SAAS,GAAG,MAAM,KAAK,OAAO;EAE1C,KAAK,MAAM,QAAQ,OAAO;GACxB,MAAM,YAAY,KACf,QAAQ,MAAM,EAAE,SAAS,eAAe,OAAO,EAAE,SAAS,YAAY,EAAE,KAAK,SAAS,CAAC,CAAC,CACxF,KAAK,MAAM,EAAE,IAAc,CAAC,CAC5B,KAAK,IAAI;GACZ,MAAM,OAAO,KACV,QAAQ,MAAM,EAAE,SAAS,UAAU,OAAO,EAAE,SAAS,QAAQ,CAAC,CAC9D,KAAK,MAAM,EAAE,IAAc,CAAC,CAC5B,KAAK,IAAI;GACZ,MAAM,YAAY,KAAK,QAAQ,MAAM,EAAE,SAAS,UAAU,OAAO,EAAE,WAAW,QAAQ;GACtF,MAAM,YAA4B,UAAU,KAAK,OAAO;IACtD,IAAI,EAAE;IACN,MAAM;IACN,UAAU;KACR,MAAM,EAAE,QAAQ;KAChB,WAAW,KAAK,UAAU,EAAE,OAAO,SAAS,CAAC,CAAC;IAChD;GACF,EAAE;GACF,IAAI,UAAU,WAAW,KAAK,KAAK,WAAW,KAAK,UAAU,WAAW,GAAG;GAC3E,SAAS,KAAK;IACZ,MAAM;IACN,SAAS,KAAK,SAAS,IAAI,OAAO;IAClC,GAAI,UAAU,SAAS,IAAI,EAAE,mBAAmB,UAAU,IAAI,CAAC;IAC/D,GAAI,UAAU,SAAS,IAAI,EAAE,YAAY,UAAU,IAAI,CAAC;GAC1D,CAAC;GACD,KAAK,MAAM,KAAK,WACd,SAAS,KAAK;IACZ,MAAM;IACN,cAAc,EAAE;IAChB,MAAM,EAAE,QAAQ;IAChB,SAAS,kBAAkB,EAAE,OAAO,MAAM;GAC5C,CAAC;EAEL;CACF;CACA,OAAO;AACT"}
@@ -1,502 +0,0 @@
1
- import { g as pairedRiskDifferenceScore, h as pairedRiskDifferenceExact, p as pairedBinaryScale } from "./paired-arms-D4aeIHUy.js";
2
- import { a as pairedSignTest, i as pairedDeltaTieFraction, r as pairedBootstrap } from "./paired-tests-C8iCsioC.js";
3
- //#region src/paired-delta-test.ts
4
- /** Smallest all-positive sample that can clear a one-sided exact sign test. */
5
- function minimumPairsForPairedDeltaTest(confidence = .95) {
6
- if (!Number.isFinite(confidence) || confidence <= 0 || confidence >= 1) throw new Error(`minimumPairsForPairedDeltaTest: confidence must be in (0,1), got ${confidence}`);
7
- const oneSidedAlpha = (1 - confidence) / 2;
8
- return Math.ceil(Math.log2(1 / oneSidedAlpha));
9
- }
10
- /**
11
- * Tests whether a paired candidate-minus-baseline delta clears a threshold.
12
- *
13
- * At 20 or more pairs, the percentile bootstrap lower bound carries the
14
- * decision. Below that point the interval is descriptive only, so the function
15
- * switches to a pre-registered one-sided exact sign test. The exact path is
16
- * deliberately conservative: it requires both a point estimate above the
17
- * threshold and enough consistently positive paired differences.
18
- *
19
- * ## A zero-width interval is never significant
20
- *
21
- * When every paired delta is identical the resample distribution is a point
22
- * mass and the interval collapses: `[0, 0]` when all pairs tie, `[g, g]` on n
23
- * identical deltas of g. Neither says the effect is certain — both say the
24
- * sample carries no information about how far the estimate could be wrong, and
25
- * `low > threshold` then answers on the point estimate alone. It fails in both
26
- * directions: `[0, 0]` clears every NEGATIVE threshold, which is how a
27
- * tie-dominated pass/fail comparison laundered a regression into a
28
- * noninferiority pass, and `[g, g]` clears every threshold below g with no
29
- * spread behind it. Under a bounded asymmetric null whose true mean paired
30
- * delta is exactly 0 — 2 % of pairs dropping by 1.0, the rest gaining 0.0204 —
31
- * every sample that misses the drop is exactly that shape, and deciding on
32
- * `low > 0` promoted 65.65 % of samples at n = 20 against a nominal 5 %.
33
- *
34
- * So `indeterminate` is reported and `significant` is false whenever the
35
- * interval has zero width, on BOTH paths: at small n the exact sign test is a
36
- * test of the MEDIAN and a zero-spread sample is precisely where it stops
37
- * saying anything about the mean the caller is thresholding.
38
- *
39
- * `threshold` may be negative — that is a noninferiority margin, and it is the
40
- * regime the zero-width hole is worst in. For a two-point (pass/fail) outcome
41
- * the percentile bootstrap is not a valid interval at a nonzero margin at all;
42
- * use {@link decidePairedPromotion}, which routes those to Tango's score
43
- * interval, rather than thresholding this function's bootstrap directly.
44
- */
45
- function pairedDeltaTest(before, after, options = {}) {
46
- const threshold = options.threshold ?? 0;
47
- if (!Number.isFinite(threshold)) throw new Error(`pairedDeltaTest: threshold must be finite, got ${threshold}`);
48
- const exactMinimum = minimumPairsForPairedDeltaTest(options.confidence ?? .95);
49
- const requestedMinimum = options.minPairs ?? exactMinimum;
50
- if (!Number.isInteger(requestedMinimum) || requestedMinimum < 1) throw new Error(`pairedDeltaTest: minPairs must be a positive integer, got ${requestedMinimum}`);
51
- const minimumPairs = Math.max(requestedMinimum, exactMinimum);
52
- const bootstrap = pairedBootstrap(before, after, options);
53
- const sufficient = bootstrap.n >= minimumPairs;
54
- const indeterminate = bootstrap.n > 0 && (!Number.isFinite(bootstrap.low) || !Number.isFinite(bootstrap.high) || bootstrap.low === bootstrap.high);
55
- if (bootstrap.gateEligible) return {
56
- bootstrap,
57
- method: "bootstrap-ci",
58
- pValue: null,
59
- minimumPairs,
60
- sufficient,
61
- indeterminate,
62
- significant: sufficient && !indeterminate && bootstrap.low > threshold
63
- };
64
- const exact = pairedSignTest(before.map((value, index) => after[index] - value - threshold), "greater");
65
- const estimate = options.statistic === "mean" ? bootstrap.mean : bootstrap.median;
66
- return {
67
- bootstrap,
68
- method: "exact-sign",
69
- pValue: exact.pValue,
70
- minimumPairs,
71
- sufficient,
72
- indeterminate,
73
- significant: sufficient && !indeterminate && estimate > threshold && exact.pValue <= (1 - bootstrap.confidence) / 2
74
- };
75
- }
76
- //#endregion
77
- //#region src/paired-promotion-decision.ts
78
- /**
79
- * @module
80
- * ONE rule for "does this paired interval clear a promotion threshold".
81
- *
82
- * The rule below was derived on `HeldOutGate` (#479) after the same estimator
83
- * bug shipped twice. It then turned out that a SECOND gate — the composable
84
- * `heldOutGate`, plus everything else routed through `heldoutSignificance` —
85
- * still carried the original defect, because the rule had been written into one
86
- * gate's method body rather than into a shared function. Two copies of a
87
- * statistical rule is how a defect survives in one of them, so there is now
88
- * exactly one copy and both gates call it.
89
- *
90
- * Three things the rule does that a bare `pairedBootstrap(...).low > threshold`
91
- * does not:
92
- *
93
- * 1. **Two-point (pass/fail) outcomes decide on Tango's SCORE interval.** On a
94
- * pass/fail eval the paired delta vector is dominated by ties, so the
95
- * bootstrap of the mean is a resample of a lattice with three atoms and its
96
- * percentile interval is not valid at a nonzero margin. The score interval
97
- * (`pairedRiskDifferenceScore`) re-estimates the nuisance loss rate under
98
- * each hypothesised margin instead of fixing it at the observed value, which
99
- * is the only construction that stays a confidence interval as the margin
100
- * moves off zero — the regime every noninferiority threshold lives in.
101
- * Measured on the composable gate before this change, at a true risk
102
- * difference sitting exactly on the production caller's -0.05 margin and a
103
- * nominal 5 %: 14.60 % false promotion at n = 40 and 10.10 % at n = 76.
104
- * 2. **McNemar's exact test holds a VETO at every non-negative threshold.**
105
- * Redundant with the interval by construction and kept anyway, so that
106
- * swapping the estimator for one without that duality cannot silently
107
- * reintroduce "promotes what the exact test refuses". Witness: n = 6, b = 5,
108
- * c = 0 — no exact argument reaches alpha = 0.05 with 5 discordant pairs
109
- * (two-sided floor 2/2^5 = 0.0625), whatever an interval says. A NEGATIVE
110
- * threshold is a noninferiority question, which McNemar's test of "no
111
- * difference" is not the right test for, so the veto does not apply there.
112
- * 3. **A ZERO-WIDTH interval is refused, wherever it sits.** At [0, 0] it
113
- * cannot tell a gain from a regression and clears every negative threshold.
114
- * Away from zero it fails the opposite way: n identical positive deltas give
115
- * [g, g], which clears threshold 0 on no spread at all. Both are an absence
116
- * of evidence. Measured on the composable gate before this change, under a
117
- * bounded asymmetric null whose true mean paired delta is exactly 0: 88.50 %
118
- * false promotion at n = 6 and 65.65 % at n = 20 against a nominal 5 %.
119
- *
120
- * Orthogonal to the small-sample switch inside {@link pairedDeltaTest}: that
121
- * picks the TEST from the sample size (bootstrap CI at n >= 20, pre-registered
122
- * exact sign test below it), this picks the ESTIMATOR from the outcome's shape.
123
- * Both are needed — an exact sign test applied to a tie-pinned median is still
124
- * blind, and a mean bootstrap CI at n = 6 is still not a valid test.
125
- */
126
- /**
127
- * Which estimator {@link decidePairedPromotion} would use on this data, and the
128
- * shape facts behind it — for callers that must report the shape on a path
129
- * where no interval is computed at all (an early rejection, or zero pairs).
130
- * Cheap: no bootstrap, no interval.
131
- */
132
- function pairedDecisionShape(before, after, statistic = "mean") {
133
- const tieFraction = before.length === 0 ? null : pairedDeltaTieFraction(before, after);
134
- if (statistic === "median") return {
135
- statistic: "median_bootstrap",
136
- binaryScale: null,
137
- tieFraction
138
- };
139
- const binaryScale = pairedBinaryScale(before, after);
140
- if (binaryScale !== null) return {
141
- statistic: "paired_risk_difference",
142
- binaryScale,
143
- tieFraction
144
- };
145
- return {
146
- statistic: "mean_bootstrap",
147
- binaryScale: null,
148
- tieFraction
149
- };
150
- }
151
- /**
152
- * Decide whether a paired candidate-minus-baseline delta clears a promotion
153
- * threshold. `before` is the baseline arm, `after` the candidate arm, paired by
154
- * position. Throws on unequal lengths.
155
- */
156
- function decidePairedPromotion(before, after, options = {}) {
157
- if (before.length !== after.length) throw new Error(`decidePairedPromotion: unequal sample sizes (${before.length} vs ${after.length})`);
158
- const threshold = options.threshold ?? 0;
159
- if (!Number.isFinite(threshold)) throw new Error(`decidePairedPromotion: threshold must be finite, got ${threshold}`);
160
- const confidence = options.confidence ?? .95;
161
- const exactMinimum = minimumPairsForPairedDeltaTest(confidence);
162
- const requestedMinimum = options.minPairs ?? exactMinimum;
163
- if (!Number.isInteger(requestedMinimum) || requestedMinimum < 1) throw new Error(`decidePairedPromotion: minPairs must be a positive integer, got ${requestedMinimum}`);
164
- const minimumPairs = Math.max(requestedMinimum, exactMinimum);
165
- const n = before.length;
166
- const sufficient = n >= minimumPairs;
167
- const { binaryScale, tieFraction } = pairedDecisionShape(before, after, options.statistic);
168
- let core;
169
- if (binaryScale !== null) {
170
- const unitControl = before.map((v) => v / binaryScale);
171
- const unitTreatment = after.map((v) => v / binaryScale);
172
- const exact = pairedRiskDifferenceExact(unitControl, unitTreatment, confidence);
173
- const score = pairedRiskDifferenceScore(unitControl, unitTreatment, confidence);
174
- const low = score.lower * binaryScale;
175
- core = {
176
- statistic: "paired_risk_difference",
177
- method: "score-interval",
178
- delta: score.riskDifference * binaryScale,
179
- low,
180
- high: score.upper * binaryScale,
181
- bootstrap: null,
182
- mcnemar: {
183
- b: exact.b,
184
- c: exact.c,
185
- nDiscordant: exact.nDiscordant,
186
- pValue: exact.pValue
187
- },
188
- pValue: null,
189
- clearsThreshold: low > threshold,
190
- label: "success-rate",
191
- methodDetail: ""
192
- };
193
- } else {
194
- const bootstrapStatistic = options.statistic === "median" ? "median" : "mean";
195
- const test = pairedDeltaTest(before, after, {
196
- confidence,
197
- resamples: options.resamples,
198
- statistic: bootstrapStatistic,
199
- seed: options.seed,
200
- threshold,
201
- minPairs: options.minPairs
202
- });
203
- const ci = test.bootstrap;
204
- core = {
205
- statistic: bootstrapStatistic === "mean" ? "mean_bootstrap" : "median_bootstrap",
206
- method: test.method,
207
- delta: bootstrapStatistic === "mean" ? ci.mean : ci.median,
208
- low: ci.low,
209
- high: ci.high,
210
- bootstrap: ci,
211
- mcnemar: null,
212
- pValue: test.pValue,
213
- clearsThreshold: test.significant,
214
- label: bootstrapStatistic,
215
- methodDetail: test.method === "exact-sign" ? ` Below ${test.minimumPairs} pairs the interval is descriptive only; the decision is the exact one-sided sign test, p=${fmt(test.pValue ?? 1)}.` : ""
216
- };
217
- }
218
- const indeterminate = !Number.isFinite(core.low) || !Number.isFinite(core.high) || core.low === core.high;
219
- const indeterminateCause = !indeterminate ? "" : tieFraction === 1 ? "every paired delta is an exact tie" : core.mcnemar !== null && core.mcnemar.nDiscordant === 0 ? "every pair is concordant (0 discordant pairs)" : `the ${core.label} CI collapsed to a point at ${fmt(core.low)}`;
220
- const exactTestVetoes = core.mcnemar !== null && threshold >= 0 && !(core.mcnemar.pValue < 1 - confidence);
221
- return {
222
- n,
223
- threshold,
224
- confidence,
225
- binaryScale,
226
- tieFraction,
227
- minimumPairs,
228
- sufficient,
229
- indeterminate,
230
- indeterminateCause,
231
- exactTestVetoes,
232
- promote: sufficient && !indeterminate && core.clearsThreshold && !exactTestVetoes,
233
- ...core
234
- };
235
- }
236
- function fmt(x) {
237
- return x.toFixed(4);
238
- }
239
- //#endregion
240
- //#region src/campaign/gates/statistical-heldout.ts
241
- /**
242
- * Statistical held-out promotion machinery — the trustworthy core the
243
- * point-estimate `heldout-delta` gate lacked.
244
- *
245
- * The shipped false positive it prevents: a winner re-scored against the
246
- * baseline on the holdout read run-to-run model NOISE (e.g. 91 vs 95) as a
247
- * "+4 lift" and shipped, because the gate compared point estimates with no
248
- * confidence interval. Here we pair candidate vs baseline holdout observations
249
- * and bootstrap a CI on the paired delta — a candidate ships only when the CI
250
- * lower bound clears the effect-size threshold (the gain is real at the
251
- * confidence level, not noise), and is blocked when a critical dimension
252
- * (e.g. `hallucination_free` for a legal agent) significantly regresses even if
253
- * the net composite rose (anti-Goodhart).
254
- *
255
- * Two traps this module is built around (both produce a NEW false positive if
256
- * gotten wrong):
257
- * 1. PAIRING GRANULARITY — pairs by FULL `cellId` (`scenario:rep`), never by
258
- * `scenarioId` (which averages reps away and destroys the within-pair
259
- * variance reduction that makes a paired bootstrap tighter than unpaired).
260
- * One paired observation per cell ⇒ reps multiply n.
261
- * 2. SCALE — a judge may emit composites/dimensions on [0,1] or 0-100. The
262
- * threshold + tolerance are interpreted in the judge's NATIVE scale; the
263
- * per-dimension tolerance auto-scales off the observed baseline magnitudes
264
- * so `-0.10` on [0,1] doesn't silently become a no-op on a 0-100 dimension.
265
- */
266
- /** Tie fraction at/above which a gate annotates its verdict with the tie share.
267
- * Tie-domination of the median bites structurally at >= 0.5 (the median is then
268
- * 0 by construction); 0.4 is a softer warn threshold that flags a run APPROACHING
269
- * that regime, so an operator sees it before the median goes fully blind. */
270
- const TIE_WARN_FRACTION = .4;
271
- /**
272
- * Pair candidate vs baseline holdout observations by FULL cellId. `select`
273
- * pulls the scalar from a cell's judge reports (composite, or a named
274
- * dimension); a cell contributes the mean of `select` across its judges. Cells
275
- * whose scenario is not in `scenarioIds`, or where `select` is undefined for
276
- * every judge on either side, are skipped on BOTH sides so the arrays stay
277
- * paired. Throws when the two maps disagree on which holdout cells exist — a
278
- * load-bearing invariant: the baseline + winner holdout campaigns run the same
279
- * scenarios with the same seed base, so their cellIds MUST align; a mismatch
280
- * means a silent pairing bug, not a soft fallback.
281
- */
282
- function pairHoldout(candidate, baseline, scenarioIds, select) {
283
- const cellValue = (byCell, cellId) => {
284
- const scores = byCell.get(cellId);
285
- if (!scores) return void 0;
286
- const vals = [];
287
- for (const s of Object.values(scores)) {
288
- if (s.failed === true) throw new Error(`pairHoldout: cell '${cellId}' contains a failed judge score`);
289
- const v = select(s);
290
- if (typeof v === "number" && !Number.isFinite(v)) throw new Error(`pairHoldout: cell '${cellId}' contains a non-finite selected score`);
291
- if (typeof v === "number") vals.push(v);
292
- }
293
- if (vals.length === 0) return void 0;
294
- return vals.reduce((a, b) => a + b, 0) / vals.length;
295
- };
296
- const inScope = (cellId) => scenarioIds.has(cellId.split(":")[0] ?? "");
297
- const candCells = [...candidate.keys()].filter(inScope).sort();
298
- const baseCells = [...baseline.keys()].filter(inScope).sort();
299
- if (candCells.length !== baseCells.length || candCells.some((c, i) => c !== baseCells[i])) throw new Error(`pairHoldout: candidate/baseline holdout cells do not align — candidate=[${candCells.join(",")}] baseline=[${baseCells.join(",")}]. Both holdout campaigns must run the same scenarios with the same seed base.`);
300
- const before = [];
301
- const after = [];
302
- const cellIds = [];
303
- for (const cellId of candCells) {
304
- const b = cellValue(baseline, cellId);
305
- const a = cellValue(candidate, cellId);
306
- if (b === void 0 && a === void 0) continue;
307
- if (b === void 0 || a === void 0) throw new Error(`pairHoldout: cell '${cellId}' has a selected score on only one arm`);
308
- before.push(b);
309
- after.push(a);
310
- cellIds.push(cellId);
311
- }
312
- return {
313
- before,
314
- after,
315
- cellIds
316
- };
317
- }
318
- /**
319
- * Significance of the held-out composite lift: ship only when the lower bound
320
- * of the interval the outcome's shape ADMITS exceeds `deltaThreshold` (default
321
- * 0 ⇒ "confidently positive"). Interpret `deltaThreshold` in the judge's native
322
- * scale.
323
- *
324
- * The decision is delegated whole to {@link decidePairedPromotion}, the one
325
- * copy of the rule (`src/paired-promotion-decision.ts`), which `HeldOutGate`
326
- * also calls. That module's header carries the measurements; the short version
327
- * is three guards a bare `bootstrap.low > threshold` does not have:
328
- *
329
- * - a two-point (pass/fail) outcome decides on Tango's SCORE interval, the
330
- * only paired-binary construction that stays valid at a nonzero margin;
331
- * - McNemar's exact test VETOES at any non-negative threshold;
332
- * - a ZERO-WIDTH interval is refused rather than promoted, in either
333
- * direction — [0,0] clears every negative threshold and [g,g] clears every
334
- * threshold below g, and both are an absence of evidence, not a result.
335
- *
336
- * Measured on this function before those guards landed, at a nominal 5 %:
337
- * 14.60 % false promotion at n = 40 on a paired-binary noninferiority boundary,
338
- * and 88.50 % at n = 6 under a bounded asymmetric null whose true mean paired
339
- * delta is exactly 0.
340
- *
341
- * At small n, where the percentile bootstrap is descriptive only, a
342
- * pre-registered exact sign test still carries the bootstrap path.
343
- */
344
- function heldoutSignificance(paired, opts = {}) {
345
- const deltaThreshold = opts.deltaThreshold ?? 0;
346
- const confidence = opts.confidence ?? .95;
347
- const resamples = opts.resamples ?? 2e3;
348
- const seed = opts.seed ?? 1337;
349
- const statistic = opts.statistic ?? "mean";
350
- const decision = decidePairedPromotion(paired.before, paired.after, {
351
- confidence,
352
- resamples,
353
- statistic,
354
- seed,
355
- threshold: deltaThreshold,
356
- minPairs: opts.minProductiveRuns
357
- });
358
- const bootstrap = decision.bootstrap ?? pairedBootstrap(paired.before, paired.after, {
359
- confidence,
360
- resamples,
361
- statistic,
362
- seed
363
- });
364
- const medianBootstrap = statistic === "median" ? bootstrap : pairedBootstrap(paired.before, paired.after, {
365
- confidence,
366
- resamples,
367
- statistic: "median",
368
- seed
369
- });
370
- const n = paired.before.length;
371
- let ties = 0;
372
- for (let i = 0; i < n; i += 1) {
373
- const after = paired.after[i] ?? 0;
374
- const before = paired.before[i] ?? 0;
375
- if (Math.abs(after - before) < 1e-9) ties += 1;
376
- }
377
- const tieFraction = n === 0 ? 0 : ties / n;
378
- return {
379
- paired,
380
- bootstrap,
381
- medianBootstrap,
382
- decision,
383
- decisionStatistic: decision.statistic,
384
- mcnemar: decision.mcnemar,
385
- tieFraction,
386
- n,
387
- minimumRequired: decision.minimumPairs,
388
- decisionMethod: decision.method,
389
- pValue: decision.pValue,
390
- significant: decision.promote,
391
- fewRuns: !decision.sufficient
392
- };
393
- }
394
- /** Detect the native scale of a set of scores: 0-100 when any magnitude clears
395
- * 1.5, else [0,1]. Used to auto-scale the regression tolerance so a default
396
- * expressed for [0,1] is not silently a no-op on a 0-100 dimension. */
397
- function detectScale(values) {
398
- return values.some((v) => Math.abs(v) > 1.5) ? 100 : 1;
399
- }
400
- /** Per-critical-dimension regression guard. For each dimension, pair the
401
- * candidate vs baseline values by full cellId and bootstrap the paired delta;
402
- * a dimension is "regressed" when the CI lower bound < −tolerance (conservative
403
- * — blocks if the credible worst case exceeds tolerance, which is the right
404
- * posture for safety dimensions like `hallucination_free`). When `tolerance`
405
- * is omitted it auto-scales: 0.05 on [0,1], 5 on 0-100.
406
- *
407
- * The interval comes from {@link decidePairedPromotion}, so a pass/fail
408
- * dimension is judged on Tango's score interval rather than a percentile
409
- * bootstrap of the mean — `tolerance` is a NONZERO margin, and the bootstrap
410
- * is not a valid interval at one. That matters most here because this guard
411
- * fails OPEN by construction: `tolerance` is positive, so an interval pinned at
412
- * [0,0] never satisfies `low < −tolerance` and a real regression on a safety
413
- * dimension would be reported as `regressed: false`. On the median it fails the
414
- * same way for the same reason — when most pairs tie, which is automatic for a
415
- * pass/fail dimension on {0,1} and on the 0-100 encoding `detectScale` exists
416
- * to support, the median CI collapses to [0,0]. Pass `statistic: 'median'` to
417
- * restore the pre-0.134 behaviour. */
418
- function dimensionRegressions(candidate, baseline, scenarioIds, criticalDimensions, opts = {}) {
419
- const out = [];
420
- for (const dim of criticalDimensions) {
421
- const paired = pairHoldout(candidate, baseline, scenarioIds, (s) => s.dimensions[dim]);
422
- if (paired.before.length === 0) continue;
423
- const tolerance = opts.tolerance ?? .05 * detectScale([...paired.before, ...paired.after]);
424
- const bootstrapStatistic = opts.statistic ?? "mean";
425
- const shared = {
426
- confidence: opts.confidence ?? .95,
427
- resamples: opts.resamples ?? 2e3,
428
- statistic: bootstrapStatistic,
429
- seed: opts.seed ?? 1337
430
- };
431
- const guard = decidePairedPromotion(paired.before, paired.after, shared);
432
- const regression = decidePairedPromotion(paired.after, paired.before, {
433
- ...shared,
434
- threshold: tolerance
435
- });
436
- const bootstrap = guard.bootstrap ?? pairedBootstrap(paired.before, paired.after, shared);
437
- out.push({
438
- dimension: dim,
439
- bootstrap,
440
- bootstrapStatistic,
441
- ci: {
442
- low: guard.low,
443
- high: guard.high
444
- },
445
- decisionStatistic: guard.statistic,
446
- mcnemar: guard.mcnemar,
447
- indeterminate: guard.indeterminate,
448
- regressed: bootstrap.low < -tolerance || regression.promote,
449
- tolerance,
450
- n: paired.before.length
451
- });
452
- }
453
- return out;
454
- }
455
- //#endregion
456
- //#region src/campaign/gates/power-preflight.ts
457
- /** Two-sided z for the common confidence levels; interpolation is overkill here. */
458
- function zFor(confidence) {
459
- if (confidence >= .99) return 2.576;
460
- if (confidence >= .95) return 1.96;
461
- if (confidence >= .9) return 1.645;
462
- return 1.282;
463
- }
464
- /** Estimate the minimum detectable lift a paired-holdout improvement run can
465
- * ship at a given budget, from the baseline holdout composites — call it BEFORE
466
- * spending a search to learn whether the effect you are hunting is even
467
- * observable at this holdout size and worker variance. */
468
- function powerPreflight(opts) {
469
- const composites = opts.baselineComposites.filter((v) => Number.isFinite(v));
470
- if (composites.length < 3) throw new Error(`powerPreflight: need >= 3 finite baseline composites to estimate variance, got ${composites.length}`);
471
- const deltaThreshold = opts.deltaThreshold ?? .05;
472
- const confidence = opts.confidence ?? .95;
473
- const n = opts.pairedN ?? composites.length;
474
- if (n < 2) throw new Error(`powerPreflight: pairedN must be >= 2, got ${n}`);
475
- const mean = composites.reduce((a, b) => a + b, 0) / composites.length;
476
- const variance = composites.reduce((a, b) => a + (b - mean) * (b - mean), 0) / (composites.length - 1);
477
- const sd = Math.sqrt(variance);
478
- const z = zFor(confidence);
479
- const mde = deltaThreshold + z * Math.SQRT2 * sd / Math.sqrt(n);
480
- const scaleAssumed = composites.every((v) => v >= -.001 && v <= 1.5);
481
- const headroom = Math.max(0, 1 - mean);
482
- const underpowered = scaleAssumed && mde > headroom;
483
- const sharedChannelCaveat = opts.sharedScorerChannel ? "Holdout and gate share one scoring channel: raising n/reps reduces only idiosyncratic noise — systematic judge bias remains and this MDE is a lower bound. Full debiasing needs an independent second scoring channel (different judge/benchmark family)." : void 0;
484
- const recommendation = underpowered ? `UNDERPOWERED: minimum detectable lift ${mde.toFixed(3)} exceeds the ${headroom.toFixed(3)} headroom above the baseline (${mean.toFixed(3)}) — no achievable effect can ship at this budget. Raise paired n (scenarios x reps) to ~${Math.ceil((z * Math.SQRT2 * sd / Math.max(headroom - deltaThreshold, .01)) ** 2)} or reduce worker variance before searching.` : `Minimum detectable lift at n=${n}: ${mde.toFixed(3)} (baseline sd ${sd.toFixed(3)}). Effects smaller than this cannot clear the gate; budget the search for effects you believe exceed it.`;
485
- return {
486
- n,
487
- sd,
488
- mde,
489
- baselineMean: mean,
490
- headroom,
491
- underpowered,
492
- scaleAssumed,
493
- deltaThreshold,
494
- confidence,
495
- ...sharedChannelCaveat ? { sharedChannelCaveat } : {},
496
- recommendation: sharedChannelCaveat ? `${recommendation} ${sharedChannelCaveat}` : recommendation
497
- };
498
- }
499
- //#endregion
500
- export { heldoutSignificance as a, pairedDecisionShape as c, dimensionRegressions as i, minimumPairsForPairedDeltaTest as l, TIE_WARN_FRACTION as n, pairHoldout as o, detectScale as r, decidePairedPromotion as s, powerPreflight as t, pairedDeltaTest as u };
501
-
502
- //# sourceMappingURL=power-preflight-CFXm0Vjo.js.map