@tangle-network/agent-eval 0.173.3 → 0.174.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (125) hide show
  1. package/CHANGELOG.md +31 -0
  2. package/dist/{proposal-findings-bko3GGy-.js → abort-signal-CtzAM_sJ.js} +11 -11
  3. package/dist/abort-signal-CtzAM_sJ.js.map +1 -0
  4. package/dist/adapters/http.d.ts +2 -2
  5. package/dist/agent-profile-_xPxqVJt.d.ts +488 -0
  6. package/dist/agent-profile-_xPxqVJt.d.ts.map +1 -0
  7. package/dist/analyst/index.d.ts +7 -9
  8. package/dist/analyst/index.d.ts.map +1 -1
  9. package/dist/analyst/index.js +8 -8
  10. package/dist/{benchmark-C4wk_Sjr.js → benchmark-DQKzykkO.js} +2 -2
  11. package/dist/{benchmark-C4wk_Sjr.js.map → benchmark-DQKzykkO.js.map} +1 -1
  12. package/dist/{benchmark-command-BY9oscke.js → benchmark-command-mZIlR-ra.js} +13 -13
  13. package/dist/{benchmark-command-BY9oscke.js.map → benchmark-command-mZIlR-ra.js.map} +1 -1
  14. package/dist/benchmarks/index.d.ts +3 -4
  15. package/dist/benchmarks/index.d.ts.map +1 -1
  16. package/dist/benchmarks/index.js +3 -3
  17. package/dist/campaign/index.d.ts +5 -9
  18. package/dist/campaign/index.js +7 -7
  19. package/dist/{campaign-B3kPMU8S.js → campaign-BzMSCejE.js} +8 -8
  20. package/dist/{campaign-B3kPMU8S.js.map → campaign-BzMSCejE.js.map} +1 -1
  21. package/dist/cli.js +1 -1
  22. package/dist/{client-DlqdbM7n.d.ts → client-vyYQg3bm.d.ts} +2 -2
  23. package/dist/{client-DlqdbM7n.d.ts.map → client-vyYQg3bm.d.ts.map} +1 -1
  24. package/dist/contract/index.d.ts +9 -10
  25. package/dist/contract/index.js +8 -8
  26. package/dist/{default-registry-B0bKikCb.js → default-registry-CrAp0pYq.js} +4 -4
  27. package/dist/{default-registry-B0bKikCb.js.map → default-registry-CrAp0pYq.js.map} +1 -1
  28. package/dist/{default-registry-BKwc8bN5.d.ts → default-registry-FfNzaUHV.d.ts} +3 -3
  29. package/dist/{default-registry-BKwc8bN5.d.ts.map → default-registry-FfNzaUHV.d.ts.map} +1 -1
  30. package/dist/{define-agent-eval-CY6qdlGV.d.ts → define-agent-eval-V1jQyCDR.d.ts} +102 -11
  31. package/dist/define-agent-eval-V1jQyCDR.d.ts.map +1 -0
  32. package/dist/{define-agent-eval-8h3lXXee.js → define-agent-eval-ox5McL6e.js} +331 -144
  33. package/dist/define-agent-eval-ox5McL6e.js.map +1 -0
  34. package/dist/{dspy-rlm-engine-CF0t2ITD.js → dspy-rlm-engine-Caz2pl4L.js} +3 -3
  35. package/dist/{dspy-rlm-engine-CF0t2ITD.js.map → dspy-rlm-engine-Caz2pl4L.js.map} +1 -1
  36. package/dist/{engine-DhFir3Ys.d.ts → engine-CvW_I72-.d.ts} +2 -2
  37. package/dist/{engine-DhFir3Ys.d.ts.map → engine-CvW_I72-.d.ts.map} +1 -1
  38. package/dist/experiment/index.d.ts +1 -4
  39. package/dist/experiment/index.d.ts.map +1 -1
  40. package/dist/{external-optimizer-process-BwITA9Jp.js → external-optimizer-process-CxnFL1hd.js} +2 -2
  41. package/dist/{external-optimizer-process-BwITA9Jp.js.map → external-optimizer-process-CxnFL1hd.js.map} +1 -1
  42. package/dist/{external-optimizer-subprocess-wBWeoG6A.js → external-optimizer-subprocess-CQi27uEI.js} +2 -2
  43. package/dist/{external-optimizer-subprocess-wBWeoG6A.js.map → external-optimizer-subprocess-CQi27uEI.js.map} +1 -1
  44. package/dist/fuzz.js +1 -1
  45. package/dist/fuzz.js.map +1 -1
  46. package/dist/hosted/index.d.ts +1 -1
  47. package/dist/{index-BQqOjerE.d.ts → index-BTrx5s8m.d.ts} +8 -9
  48. package/dist/index-BTrx5s8m.d.ts.map +1 -0
  49. package/dist/{index-D0Db5X-4.d.ts → index-Bn-nlnSV.d.ts} +4 -4
  50. package/dist/{index-D0Db5X-4.d.ts.map → index-Bn-nlnSV.d.ts.map} +1 -1
  51. package/dist/index-DKXuBPXf.d.ts +3840 -0
  52. package/dist/index-DKXuBPXf.d.ts.map +1 -0
  53. package/dist/index.d.ts +11 -13
  54. package/dist/index.d.ts.map +1 -1
  55. package/dist/index.js +10 -10
  56. package/dist/{kind-factory-gP6lDySe.js → kind-factory-BLvL-E44.js} +2 -2
  57. package/dist/{kind-factory-gP6lDySe.js.map → kind-factory-BLvL-E44.js.map} +1 -1
  58. package/dist/{llm-judge-BfqMFo4h.js → llm-judge-DmNaBrXB.js} +2541 -2435
  59. package/dist/llm-judge-DmNaBrXB.js.map +1 -0
  60. package/dist/{matrix-DGu8KhSs.d.ts → matrix-CJtXz1ky.d.ts} +2 -2
  61. package/dist/{matrix-DGu8KhSs.d.ts.map → matrix-CJtXz1ky.d.ts.map} +1 -1
  62. package/dist/multishot/golden/index.d.ts +1 -1
  63. package/dist/multishot/index.d.ts +2 -2
  64. package/dist/openapi.json +1 -1
  65. package/dist/{produced-state-D91uDvQw.js → produced-state-B8mw6zj9.js} +2 -2
  66. package/dist/{produced-state-D91uDvQw.js.map → produced-state-B8mw6zj9.js.map} +1 -1
  67. package/dist/rl.d.ts +1 -1
  68. package/dist/rl.d.ts.map +1 -1
  69. package/dist/rl.js.map +1 -1
  70. package/dist/{semantic-concept-judge-Dok7_35a.js → semantic-concept-judge-E3s_fEjB.js} +3 -3
  71. package/dist/{semantic-concept-judge-Dok7_35a.js.map → semantic-concept-judge-E3s_fEjB.js.map} +1 -1
  72. package/dist/{skillopt-optimization-method-DDw3v3gA.js → skillopt-optimization-method-f7399oGb.js} +5 -5
  73. package/dist/{skillopt-optimization-method-DDw3v3gA.js.map → skillopt-optimization-method-f7399oGb.js.map} +1 -1
  74. package/dist/statistical-heldout-Cqb73yE9.d.ts +1127 -0
  75. package/dist/statistical-heldout-Cqb73yE9.d.ts.map +1 -0
  76. package/dist/{store-otlp-Dow0pk_5.js → store-otlp-DV_H2HDu.js} +2 -2
  77. package/dist/{store-otlp-Dow0pk_5.js.map → store-otlp-DV_H2HDu.js.map} +1 -1
  78. package/dist/{store-tool-spans-CCZNsihA.d.ts → store-tool-spans-4o55ABER.d.ts} +3 -3
  79. package/dist/{store-tool-spans-CCZNsihA.d.ts.map → store-tool-spans-4o55ABER.d.ts.map} +1 -1
  80. package/dist/{store-tool-spans-CeNj_m2L.js → store-tool-spans-B9tjys_h.js} +3 -3
  81. package/dist/{store-tool-spans-CeNj_m2L.js.map → store-tool-spans-B9tjys_h.js.map} +1 -1
  82. package/dist/{task-failure-attributes-CZjZeBsY.js → task-failure-attributes-CUy9mkIY.js} +2 -2
  83. package/dist/{task-failure-attributes-CZjZeBsY.js.map → task-failure-attributes-CUy9mkIY.js.map} +1 -1
  84. package/dist/{tool-groups-Cp4Xdzrp.d.ts → tool-groups-DAe1t6zb.d.ts} +2 -2
  85. package/dist/tool-groups-DAe1t6zb.d.ts.map +1 -0
  86. package/dist/trace-repair/index.d.ts +1 -1
  87. package/dist/traces.d.ts +2 -2
  88. package/dist/traces.js +4 -4
  89. package/dist/{types-Ba5UQyVD.d.ts → types-BJz2CPTM.d.ts} +2 -2
  90. package/dist/{types-Ba5UQyVD.d.ts.map → types-BJz2CPTM.d.ts.map} +1 -1
  91. package/dist/{types-CiWITkGo.js → types-DQ0e2E7y.js} +2 -2
  92. package/dist/types-DQ0e2E7y.js.map +1 -0
  93. package/dist/{types-BDV4PiMR.d.ts → types-Dd1ejaeI.d.ts} +2 -2
  94. package/dist/{types-BDV4PiMR.d.ts.map → types-Dd1ejaeI.d.ts.map} +1 -1
  95. package/docs/campaign-proposers.md +42 -0
  96. package/package.json +1 -1
  97. package/dist/agent-profile-B9_GGsG8.d.ts +0 -84
  98. package/dist/agent-profile-B9_GGsG8.d.ts.map +0 -1
  99. package/dist/backend-integrity-CeuTgqsd.d.ts +0 -280
  100. package/dist/backend-integrity-CeuTgqsd.d.ts.map +0 -1
  101. package/dist/benchmark-BjLGkfnN.d.ts +0 -236
  102. package/dist/benchmark-BjLGkfnN.d.ts.map +0 -1
  103. package/dist/define-agent-eval-8h3lXXee.js.map +0 -1
  104. package/dist/define-agent-eval-CY6qdlGV.d.ts.map +0 -1
  105. package/dist/external-optimizer-contracts-CQCpyrIL.d.ts +0 -172
  106. package/dist/external-optimizer-contracts-CQCpyrIL.d.ts.map +0 -1
  107. package/dist/heldout-gate-Df5hsqmm.d.ts +0 -453
  108. package/dist/heldout-gate-Df5hsqmm.d.ts.map +0 -1
  109. package/dist/index-BQqOjerE.d.ts.map +0 -1
  110. package/dist/index-CFDffsKz.d.ts +0 -1135
  111. package/dist/index-CFDffsKz.d.ts.map +0 -1
  112. package/dist/llm-judge-BfqMFo4h.js.map +0 -1
  113. package/dist/power-preflight-Ptse_Kq7.d.ts +0 -117
  114. package/dist/power-preflight-Ptse_Kq7.d.ts.map +0 -1
  115. package/dist/pre-registration-BoI4ucR3.d.ts +0 -592
  116. package/dist/pre-registration-BoI4ucR3.d.ts.map +0 -1
  117. package/dist/promotion-policy-CvMda3kU.d.ts +0 -134
  118. package/dist/promotion-policy-CvMda3kU.d.ts.map +0 -1
  119. package/dist/proposal-findings-bko3GGy-.js.map +0 -1
  120. package/dist/provenance-CRY67X50.d.ts +0 -1995
  121. package/dist/provenance-CRY67X50.d.ts.map +0 -1
  122. package/dist/statistical-heldout-DTyB_6-1.d.ts +0 -295
  123. package/dist/statistical-heldout-DTyB_6-1.d.ts.map +0 -1
  124. package/dist/tool-groups-Cp4Xdzrp.d.ts.map +0 -1
  125. package/dist/types-CiWITkGo.js.map +0 -1
@@ -0,0 +1 @@
1
+ {"version":3,"file":"types-DQ0e2E7y.js","names":[],"sources":["../src/ledger-core/deep-freeze.ts","../src/analyst/usage-receipt.ts","../src/analyst/types.ts"],"sourcesContent":["/** Freeze a detached canonical-JSON graph. Canonicalization has already ruled out cycles.\n *\n * Lives outside canonical.ts so the analyst-benchmark implementation digest,\n * which covers canonical.ts, stays bound to the published benchmark evidence. */\nexport function deepFreezeCanonicalJson<T>(value: T): T {\n if (value && typeof value === 'object' && !Object.isFrozen(value)) {\n Object.freeze(value)\n for (const nested of Object.values(value)) deepFreezeCanonicalJson(nested)\n }\n return value\n}\n","import type { CostChannel, CostLedgerFilter, CostLedgerHandle } from '../cost-ledger'\nimport type { AnalystUsageReceipt } from './types'\n\nexport const DEFAULT_USAGE_SETTLEMENT_TIMEOUT_MS = 5_000\n\n/** Convert one ledger channel's complete call set into one analyst receipt. */\nexport function usageReceiptFromCostLedger(\n ledger: CostLedgerHandle,\n filter: CostChannel | CostLedgerFilter = 'analyst',\n): AnalystUsageReceipt {\n const resolvedFilter = typeof filter === 'string' ? { channel: filter } : filter\n const summary = ledger.summary(resolvedFilter)\n const receipts = ledger.list(resolvedFilter)\n const hasReasoningUsage = receipts.some((receipt) => receipt.reasoningTokens !== undefined)\n const hasCacheWriteUsage = receipts.some((receipt) => receipt.cacheWriteTokens !== undefined)\n const cost = summary.costProvenance\n return {\n calls: summary.totalCalls + summary.pendingCalls,\n tokens: summary.usageComplete\n ? {\n input: summary.inputTokens,\n output: summary.outputTokens,\n ...(hasReasoningUsage ? { reasoning: summary.reasoningTokens ?? 0 } : {}),\n ...(summary.cachedTokens > 0 ? { cached: summary.cachedTokens } : {}),\n ...(hasCacheWriteUsage ? { cacheWrite: summary.cacheWriteTokens ?? 0 } : {}),\n }\n : null,\n cost,\n ...(cost.kind === 'uncaptured' ? { knownCostUsd: summary.totalCostUsd } : {}),\n }\n}\n\nexport interface SettledUsageReceipt {\n settled: boolean\n pendingCalls: number\n receipt: AnalystUsageReceipt\n}\n\n/** Wait a bounded time for late provider receipts, then take one immutable snapshot. */\nexport async function settleUsageReceiptFromCostLedger(\n ledger: CostLedgerHandle,\n options: CostLedgerFilter & { timeoutMs?: number } = {},\n): Promise<SettledUsageReceipt> {\n const { timeoutMs: requestedTimeoutMs, ...requestedFilter } = options\n const filter: CostLedgerFilter = {\n channel: requestedFilter.channel ?? 'analyst',\n ...(requestedFilter.phase === undefined ? {} : { phase: requestedFilter.phase }),\n ...(requestedFilter.tags === undefined ? {} : { tags: requestedFilter.tags }),\n }\n const timeoutMs = validateUsageSettlementTimeout(requestedTimeoutMs)\n const initial = ledger.summary(filter)\n const waitResult =\n initial.pendingCalls === 0\n ? true\n : ledger.waitForIdle\n ? await ledger.waitForIdle({ timeoutMs })\n : false\n const pendingCalls = ledger.summary(filter).pendingCalls\n return {\n settled: waitResult && pendingCalls === 0,\n pendingCalls,\n receipt: usageReceiptFromCostLedger(ledger, filter),\n }\n}\n\nexport function validateUsageSettlementTimeout(timeoutMs?: number): number {\n const resolved = timeoutMs ?? DEFAULT_USAGE_SETTLEMENT_TIMEOUT_MS\n if (!Number.isSafeInteger(resolved) || resolved < 0 || resolved > 2_147_483_647) {\n throw new TypeError(\n 'settlementTimeoutMs must be a non-negative safe integer no greater than 2147483647',\n )\n }\n return resolved\n}\n\nexport function assertValidAnalystUsageReceipt(\n receipt: AnalystUsageReceipt,\n context = 'AnalystContext.recordUsage',\n): void {\n if (receipt.calls !== null && (!Number.isSafeInteger(receipt.calls) || receipt.calls < 0)) {\n throw new Error(`${context}: calls must be a non-negative safe integer or null`)\n }\n if (receipt.tokens) {\n assertNonNegativeSafeInteger(receipt.tokens.input, 'tokens.input', context)\n assertNonNegativeSafeInteger(receipt.tokens.output, 'tokens.output', context)\n if (receipt.tokens.reasoning !== undefined) {\n assertNonNegativeSafeInteger(receipt.tokens.reasoning, 'tokens.reasoning', context)\n if (receipt.tokens.reasoning > receipt.tokens.output) {\n throw new Error(`${context}: tokens.reasoning must not exceed tokens.output`)\n }\n }\n if (receipt.tokens.cached !== undefined) {\n assertNonNegativeSafeInteger(receipt.tokens.cached, 'tokens.cached', context)\n }\n if (receipt.tokens.cacheWrite !== undefined) {\n assertNonNegativeSafeInteger(receipt.tokens.cacheWrite, 'tokens.cacheWrite', context)\n }\n }\n if (receipt.cost.kind !== 'uncaptured') {\n assertNonNegativeFinite(receipt.cost.usd, 'cost.usd', context)\n } else if (receipt.cost.usd !== null) {\n throw new Error(`${context}: uncaptured cost.usd must be null`)\n }\n if (receipt.knownCostUsd !== undefined) {\n assertNonNegativeFinite(receipt.knownCostUsd, 'knownCostUsd', context)\n }\n if (receipt.partialTokens) {\n const { input, output } = receipt.partialTokens\n if (receipt.tokens) {\n throw new Error(`${context}: partialTokens must be absent when tokens is complete`)\n }\n if (input === null && output === null) {\n throw new Error(`${context}: partialTokens must carry at least one reported side`)\n }\n if (input !== null) assertNonNegativeSafeInteger(input, 'partialTokens.input', context)\n if (output !== null) assertNonNegativeSafeInteger(output, 'partialTokens.output', context)\n }\n}\n\nfunction assertNonNegativeSafeInteger(value: number, field: string, context: string): void {\n if (!Number.isSafeInteger(value) || value < 0) {\n throw new Error(`${context}: ${field} must be a non-negative safe integer`)\n }\n}\n\nfunction assertNonNegativeFinite(value: number, field: string, context: string): void {\n if (!Number.isFinite(value) || value < 0) {\n throw new Error(`${context}: ${field} must be a non-negative finite number`)\n }\n}\n","/**\n * Analyst contract — the missing orchestration layer over agent-eval's\n * existing analyzers (analyzeTraces, MultiLayerVerifier,\n * SemanticConceptJudge, JudgeFn, ...).\n *\n * Each existing primitive returns its own output shape. The Analyst\n * contract is the single envelope every primitive lifts into, so a\n * registry can run N analysts against a run and a single renderer can\n * compose findings without knowing which analyzer produced them.\n *\n * The contract is intentionally domain-agnostic: nothing here knows\n * about code, voice, RAG, or any particular agent stack. Analysts\n * declare what INPUT KIND they need (a trace store, an artifact dir,\n * a RunRecord, a JudgeInput, or `custom`), and the registry routes\n * the matching input from `AnalystRunInputs`.\n */\n\nimport { createHash } from 'node:crypto'\nimport type { CostLedgerHandle } from '../cost-ledger'\nimport type { RunCostProvenance, RunRecord, RunTokenUsage } from '../run-record'\nimport type { TraceAnalysisStore } from '../trace-analyst/store'\nimport type { JudgeInput } from '../types'\nimport type { ChatClient } from './chat-client'\n\n/**\n * Unified envelope every analyst emits. Schema-versioned so renderers\n * and time-series diffs survive future field additions.\n */\nexport interface AnalystFinding {\n schema_version: '1.0.0'\n /**\n * Stable hash over identity-defining fields (analyst_id + canonical\n * claim + area + optional subject). Two findings from two runs that\n * \"are the same finding\" share this id — that's what `diffFindings`\n * uses to compute appeared/disappeared sets across runs.\n */\n finding_id: string\n analyst_id: string\n produced_at: string\n severity: AnalystSeverity\n /**\n * Coarse classification. Renderers group by this. Free-form so\n * domain-specific analysts can introduce categories without a\n * schema change ('agent-reasoning', 'verification', 'cost',\n * 'tool-use', 'safety', 'latency', 'data-quality', ...).\n */\n area: string\n claim: string\n rationale?: string\n evidence_refs: EvidenceRef[]\n recommended_action?: string\n validation_plan?: string\n /** 0..1 — the analyst's own confidence. Not calibrated across analysts. */\n confidence: number\n /**\n * Optional subject the finding is about — leaf id, agent id, request\n * id. Included in finding_id when present so per-subject findings\n * diff cleanly across runs.\n */\n subject?: string\n /** True when this finding was lifted from a judge result rather than observed\n * directly in a trace or artifact. Descriptive only: proposal access is\n * controlled by `ProposalFinding.proposal_origin`. */\n derived_from_judge?: boolean\n /** Analyst-private extras; renderers ignore unless they know the analyst. */\n metadata?: Record<string, unknown>\n}\n\n/** Finding severity. `AnalystSeverity` derives from this array, so the type\n * and the schema that validates a finding cannot name different levels. */\nexport const ANALYST_SEVERITIES = ['critical', 'high', 'medium', 'low', 'info'] as const\n\nexport type AnalystSeverity = (typeof ANALYST_SEVERITIES)[number]\n\n/** Data sources that candidate generation may intentionally learn from. */\nexport type ProposalFindingOrigin = 'search' | 'production'\n\n/** A finding explicitly admitted as candidate-generation input. */\nexport type ProposalFinding = AnalystFinding & {\n readonly proposal_origin: ProposalFindingOrigin\n}\n\nexport interface EvidenceRef {\n /**\n * Where the evidence lives. `span` and `event` refer to OTLP trace\n * elements; `artifact` to a file inside the run's artifact tree;\n * `finding` to another AnalystFinding (cross-analyst chaining);\n * `metric` to a named scalar reading the renderer knows how to read.\n */\n kind: 'span' | 'event' | 'artifact' | 'finding' | 'metric'\n uri: string\n excerpt?: string\n}\n\n// ── Analyst contract ─────────────────────────────────────────────────\n\n/**\n * The discriminator the registry uses to pass the right input.\n * `custom` is the escape hatch — analysts that need something else\n * (e.g. an embedding cache, a partner SDK handle) read it from\n * `AnalystRunInputs.custom[<analyst id>]`.\n */\nexport type AnalystInputKind =\n | 'trace-store'\n | 'artifact-dir'\n | 'run-record'\n | 'judge-input'\n | 'custom'\n\nexport interface AnalystCost {\n /** `deterministic` analysts MUST NOT call the LLM. */\n kind: 'deterministic' | 'llm'\n /** Optional declared upper bound; the registry can enforce a budget. */\n est_usd_per_run?: number\n /** Models the analyst expects to use (informational). */\n models?: string[]\n /** Maximum post-cancellation wait for provider usage. Model analysts default to 5 seconds. */\n settlement_timeout_ms?: number\n}\n\nexport interface AnalystRequirements {\n /** Min number of shots / samples the analyst needs to produce signal. */\n min_shots?: number\n /** Capabilities the runtime must supply (e.g. ['network', 'gpu']). */\n capabilities?: string[]\n}\n\n/**\n * What's passed to every analyst call. The registry resolves which\n * field the analyst's `inputKind` selects and asserts it's present.\n */\nexport interface AnalystRunInputs {\n traceStore?: TraceAnalysisStore\n artifactDir?: string\n runRecord?: RunRecord\n judgeInput?: JudgeInput\n /** Keyed by analyst id; populated by callers that registered custom analysts. */\n custom?: Record<string, unknown>\n}\n\nexport interface AnalystContext {\n runId: string\n /** Stable correlation id so logs from a single registry.run() share a tag. */\n correlationId: string\n /** Enforced wall-clock deadline (epoch ms). */\n deadlineMs?: number\n /** Per-analyst USD budget. Analysts MAY check before issuing LLM calls. */\n budgetUsd?: number\n /** Shared paid-call account when the analyst runs inside a larger campaign. */\n costLedger?: CostLedgerHandle\n /** Attribution phase used when writing to the shared paid-call account. */\n costPhase?: string\n /**\n * Shared chat client. Analysts that call an LLM go through this so\n * the operator picks transport (sandbox-sdk | router | cli-bridge |\n * direct-provider | mock) at the registry boundary without touching\n * analyst code.\n */\n chat?: ChatClient\n /**\n * Findings from a prior run the operator wants the analyst to see as\n * retrieval context. Kinds that take advantage of cross-run memory\n * (failure-mode \"I saw this cluster last run\", knowledge-gap \"the wiki\n * page I asked for is still missing\") render these into the actor's\n * working set. Filtering is the operator's job: pass the slice that\n * matches the analyst's id, or pass everything and let the kind\n * filter. Empty / absent means no cross-run context.\n */\n priorFindings?: ReadonlyArray<AnalystFinding>\n /**\n * Findings emitted by analysts that completed earlier in this registry run.\n * This is separate from `priorFindings`: upstream findings are dependency\n * context for the current pass, while prior findings are cross-run memory.\n * The registry populates this only when `RegistryRunOpts.chainFindings` is on.\n */\n upstreamFindings?: ReadonlyArray<AnalystFinding>\n /**\n * Report metered work independently of findings. This keeps an empty finding\n * set from erasing token/cost telemetry. Multiple receipts are accumulated.\n */\n recordUsage?: (receipt: AnalystUsageReceipt) => void\n /** Free-form runtime tags (env, host, op). Findings can echo these into metadata. */\n tags?: Record<string, string>\n /** Logger callback — analysts SHOULD prefer this over console.* for testability. */\n log?: (msg: string, fields?: Record<string, unknown>) => void\n /** Optional abort signal. Analysts SHOULD pass it through to LLM calls. */\n signal?: AbortSignal\n /**\n * Optional live-execution port. A runtime that owns a sandbox or checkout\n * fills it so an analyst can execute a bounded probe against the run's\n * produced state instead of reasoning about it from the trace alone. This\n * package defines only the port: no field here reaches for an agent loop,\n * and an absent probe means the analyst works from recorded evidence.\n */\n probe?: ExecutionProbe\n}\n\n// ── Live-execution port ─────────────────────────────────────────────\n\n/** One bounded command an analyst asks the probe to run. */\nexport interface ExecutionProbeRequest {\n command: string\n /** Working directory inside the probed environment. */\n cwd?: string\n /** Hard wall-clock deadline for this one execution. */\n timeoutMs: number\n /** Bytes of combined output retained; the prober truncates beyond it. */\n maxOutputBytes?: number\n signal?: AbortSignal\n}\n\n/**\n * Typed outcome of one probe execution. `succeeded: false` is a PROBE failure\n * (the environment could not run the command); a command that ran and exited\n * non-zero is a successful observation with a non-zero `exitCode`.\n */\nexport type ExecutionProbeOutcome =\n | {\n succeeded: true\n exitCode: number\n stdout: string\n stderr: string\n durationMs: number\n /** True when output was cut at `maxOutputBytes`. */\n truncated: boolean\n }\n | { succeeded: false; error: { class: string; message: string } }\n\n/**\n * The seam a runtime fills to let analysts observe produced state live.\n * Implementations own sandboxing, credentials, and cleanup; analysts only\n * submit bounded requests and read typed outcomes.\n */\nexport interface ExecutionProbe {\n /** One plain sentence naming what is being probed (e.g. a sandbox id). */\n readonly description: string\n execute(request: ExecutionProbeRequest): Promise<ExecutionProbeOutcome>\n}\n\n/**\n * The minimal contract. Concrete analysts can refine `TInput` so\n * implementations stay type-safe (e.g. a trace analyst's `TInput` is\n * `TraceAnalysisStore`); the registry passes the right field from\n * `AnalystRunInputs` based on `inputKind`.\n */\nexport interface Analyst<TInput = unknown> {\n /** Stable identifier — appears in finding_id, telemetry, and registry exclusion lists. */\n readonly id: string\n /** Human-readable. One sentence. */\n readonly description: string\n readonly inputKind: AnalystInputKind\n readonly cost: AnalystCost\n readonly requires?: AnalystRequirements\n /** Bump on breaking changes to claim wording or area so old finding_ids don't collide. */\n readonly version: string\n analyze(input: TInput, ctx: AnalystContext): Promise<AnalystFinding[]>\n}\n\n/** Metered work performed by one analyst call. */\nexport interface AnalystUsageReceipt {\n /** Number of model-usage records observed at the provider boundary. */\n calls: number | null\n /** Null when the provider did not return token accounting. */\n tokens: RunTokenUsage | null\n /** Observed, estimated, or explicitly uncaptured dollar cost. */\n cost: RunCostProvenance\n /** Known lower bound when one or more calls have uncaptured cost. */\n knownCostUsd?: number\n /**\n * Token counts the provider reported on only one side. Present exactly when\n * `tokens` is null and at least one side WAS reported: `RunTokenUsage` has no\n * nullable side, so a one-sided count cannot live in `tokens` without writing\n * a zero nobody measured. Read it as a lower bound, never as a total — the\n * field exists so a null `tokens` cannot hide a real count.\n */\n partialTokens?: { input: number | null; output: number | null }\n /**\n * True when the token counts were DERIVED by the transport (from character\n * lengths, say) rather than measured by the model provider. `cost.kind` is\n * `estimated` both for a rate estimate over exact tokens and for one over\n * derived tokens; this is the field that separates them.\n */\n tokensEstimated?: boolean\n}\n\n// ── finding_id stability ─────────────────────────────────────────────\n\n/**\n * Compute the stable finding_id from the identity-defining fields.\n * Default implementation hashes {analyst_id, area, subject, normalized claim}.\n * Analysts that emit findings whose claim text varies per run (timestamps,\n * counts) SHOULD either: (a) pass an explicit `id_basis` to fix the hash,\n * or (b) move the variable part into `rationale`/`metadata` and keep the\n * `claim` static.\n */\nexport function computeFindingId(input: {\n analyst_id: string\n area: string\n subject?: string\n claim: string\n /** Override the claim for hashing — use when the displayed claim has run-specific bits. */\n id_basis?: string\n}): string {\n const basis = JSON.stringify({\n a: input.analyst_id,\n r: input.area,\n s: input.subject ?? '',\n c: normalizeClaim(input.id_basis ?? input.claim),\n })\n return `f_${createHash('sha256').update(basis).digest('hex').slice(0, 20)}`\n}\n\nfunction normalizeClaim(c: string): string {\n // Lowercase, collapse whitespace, strip trailing punctuation. Goal:\n // \"Leaf X failed install\" and \"Leaf X failed install.\" hash the same.\n return c\n .toLowerCase()\n .replace(/\\s+/g, ' ')\n .replace(/[.!?;:,]+$/g, '')\n .trim()\n}\n\n/**\n * Convenience factory: produce a fully-formed AnalystFinding with the\n * id computed automatically. Analyst code stays terse.\n */\nexport function makeFinding(\n init: Omit<AnalystFinding, 'schema_version' | 'finding_id' | 'produced_at'> & {\n id_basis?: string\n produced_at?: string\n },\n): AnalystFinding {\n const { id_basis, produced_at, ...rest } = init\n return {\n schema_version: '1.0.0',\n finding_id: computeFindingId({\n analyst_id: rest.analyst_id,\n area: rest.area,\n subject: rest.subject,\n claim: rest.claim,\n id_basis,\n }),\n produced_at: produced_at ?? new Date().toISOString(),\n ...rest,\n }\n}\n\n/** Build a finding whose source is explicitly allowed during candidate generation. */\nexport function makeProposalFinding(\n init: Omit<ProposalFinding, 'schema_version' | 'finding_id' | 'produced_at'> & {\n id_basis?: string\n produced_at?: string\n },\n): ProposalFinding {\n const { proposal_origin, ...finding } = init\n return { ...makeFinding(finding), proposal_origin }\n}\n\n// ── Registry result envelope ────────────────────────────────────────\n\nexport interface AnalystRunSummary {\n analyst_id: string\n status: 'ok' | 'skipped' | 'failed'\n /** Why skipped — missing input, budget exceeded, capability unmet. */\n reason?: string\n findings_count: number\n latency_ms: number\n /** Additive model usage and cost provenance for this analyst. */\n usage: AnalystUsageReceipt\n /** When `status='failed'`: the error class + message, never the full stack. */\n error?: { class: string; message: string }\n}\n\nexport interface AnalystRunResult {\n run_id: string\n correlation_id: string\n started_at: string\n ended_at: string\n findings: AnalystFinding[]\n per_analyst: AnalystRunSummary[]\n /** Total LLM cost in USD across all analysts in this registry.run(). */\n total_cost_usd: number\n /**\n * Provenance for `total_cost_usd`. When uncaptured, the numeric field is only\n * the known subtotal and must not be treated as the run's total spend.\n */\n total_cost_provenance?: RunCostProvenance\n}\n\n// ── Streaming event envelope ────────────────────────────────────────\n\n/**\n * Events emitted by `AnalystRegistry.runStream(...)` in real time as\n * the registry executes. UIs subscribe via `for await (const ev of\n * registry.runStream(...))`; `registry.run(...)` is a thin collector\n * over the same stream, so the two surfaces share their invariants.\n *\n * Per-finding events are intentionally omitted — analyzers are batch\n * operations (a recursive engine returns the full `findings:json[]` at the\n * end of the responder), so streaming inside one analyst would only\n * emit partial JSON consumers can't render. The kind-completion event\n * is the right granularity; subscribers wanting per-finding rendering\n * iterate `event.findings` themselves.\n */\nexport type AnalystRunEvent =\n | {\n type: 'run-started'\n run_id: string\n correlation_id: string\n started_at: string\n /** The ordered list of analyst ids the registry will run. */\n analyst_ids: ReadonlyArray<string>\n }\n | {\n type: 'analyst-skipped'\n summary: AnalystRunSummary\n }\n | {\n type: 'analyst-started'\n analyst_id: string\n started_at: string\n }\n | {\n type: 'analyst-completed'\n /** `summary.status` is `'ok'` for clean completion or `'failed'` for thrown analysts. */\n summary: AnalystRunSummary\n findings: ReadonlyArray<AnalystFinding>\n }\n | {\n type: 'run-completed'\n result: AnalystRunResult\n }\n"],"mappings":";;;;;;AAIA,SAAgB,wBAA2B,OAAa;CACtD,IAAI,SAAS,OAAO,UAAU,YAAY,CAAC,OAAO,SAAS,KAAK,GAAG;EACjE,OAAO,OAAO,KAAK;EACnB,KAAK,MAAM,UAAU,OAAO,OAAO,KAAK,GAAG,wBAAwB,MAAM;CAC3E;CACA,OAAO;AACT;;ACJA,SAAgB,2BACd,QACA,SAAyC,WACpB;CACrB,MAAM,iBAAiB,OAAO,WAAW,WAAW,EAAE,SAAS,OAAO,IAAI;CAC1E,MAAM,UAAU,OAAO,QAAQ,cAAc;CAC7C,MAAM,WAAW,OAAO,KAAK,cAAc;CAC3C,MAAM,oBAAoB,SAAS,MAAM,YAAY,QAAQ,oBAAoB,KAAA,CAAS;CAC1F,MAAM,qBAAqB,SAAS,MAAM,YAAY,QAAQ,qBAAqB,KAAA,CAAS;CAC5F,MAAM,OAAO,QAAQ;CACrB,OAAO;EACL,OAAO,QAAQ,aAAa,QAAQ;EACpC,QAAQ,QAAQ,gBACZ;GACE,OAAO,QAAQ;GACf,QAAQ,QAAQ;GAChB,GAAI,oBAAoB,EAAE,WAAW,QAAQ,mBAAmB,EAAE,IAAI,CAAC;GACvE,GAAI,QAAQ,eAAe,IAAI,EAAE,QAAQ,QAAQ,aAAa,IAAI,CAAC;GACnE,GAAI,qBAAqB,EAAE,YAAY,QAAQ,oBAAoB,EAAE,IAAI,CAAC;EAC5E,IACA;EACJ;EACA,GAAI,KAAK,SAAS,eAAe,EAAE,cAAc,QAAQ,aAAa,IAAI,CAAC;CAC7E;AACF;;AASA,eAAsB,iCACpB,QACA,UAAqD,CAAC,GACxB;CAC9B,MAAM,EAAE,WAAW,oBAAoB,GAAG,oBAAoB;CAC9D,MAAM,SAA2B;EAC/B,SAAS,gBAAgB,WAAW;EACpC,GAAI,gBAAgB,UAAU,KAAA,IAAY,CAAC,IAAI,EAAE,OAAO,gBAAgB,MAAM;EAC9E,GAAI,gBAAgB,SAAS,KAAA,IAAY,CAAC,IAAI,EAAE,MAAM,gBAAgB,KAAK;CAC7E;CACA,MAAM,YAAY,+BAA+B,kBAAkB;CAEnE,MAAM,aADU,OAAO,QAAQ,MAEvB,CAAC,CAAC,iBAAiB,IACrB,OACA,OAAO,cACL,MAAM,OAAO,YAAY,EAAE,UAAU,CAAC,IACtC;CACR,MAAM,eAAe,OAAO,QAAQ,MAAM,CAAC,CAAC;CAC5C,OAAO;EACL,SAAS,cAAc,iBAAiB;EACxC;EACA,SAAS,2BAA2B,QAAQ,MAAM;CACpD;AACF;AAEA,SAAgB,+BAA+B,WAA4B;CACzE,MAAM,WAAW,aAAA;CACjB,IAAI,CAAC,OAAO,cAAc,QAAQ,KAAK,WAAW,KAAK,WAAW,YAChE,MAAM,IAAI,UACR,oFACF;CAEF,OAAO;AACT;AAEA,SAAgB,+BACd,SACA,UAAU,8BACJ;CACN,IAAI,QAAQ,UAAU,SAAS,CAAC,OAAO,cAAc,QAAQ,KAAK,KAAK,QAAQ,QAAQ,IACrF,MAAM,IAAI,MAAM,GAAG,QAAQ,oDAAoD;CAEjF,IAAI,QAAQ,QAAQ;EAClB,6BAA6B,QAAQ,OAAO,OAAO,gBAAgB,OAAO;EAC1E,6BAA6B,QAAQ,OAAO,QAAQ,iBAAiB,OAAO;EAC5E,IAAI,QAAQ,OAAO,cAAc,KAAA,GAAW;GAC1C,6BAA6B,QAAQ,OAAO,WAAW,oBAAoB,OAAO;GAClF,IAAI,QAAQ,OAAO,YAAY,QAAQ,OAAO,QAC5C,MAAM,IAAI,MAAM,GAAG,QAAQ,iDAAiD;EAEhF;EACA,IAAI,QAAQ,OAAO,WAAW,KAAA,GAC5B,6BAA6B,QAAQ,OAAO,QAAQ,iBAAiB,OAAO;EAE9E,IAAI,QAAQ,OAAO,eAAe,KAAA,GAChC,6BAA6B,QAAQ,OAAO,YAAY,qBAAqB,OAAO;CAExF;CACA,IAAI,QAAQ,KAAK,SAAS,cACxB,wBAAwB,QAAQ,KAAK,KAAK,YAAY,OAAO;MACxD,IAAI,QAAQ,KAAK,QAAQ,MAC9B,MAAM,IAAI,MAAM,GAAG,QAAQ,mCAAmC;CAEhE,IAAI,QAAQ,iBAAiB,KAAA,GAC3B,wBAAwB,QAAQ,cAAc,gBAAgB,OAAO;CAEvE,IAAI,QAAQ,eAAe;EACzB,MAAM,EAAE,OAAO,WAAW,QAAQ;EAClC,IAAI,QAAQ,QACV,MAAM,IAAI,MAAM,GAAG,QAAQ,uDAAuD;EAEpF,IAAI,UAAU,QAAQ,WAAW,MAC/B,MAAM,IAAI,MAAM,GAAG,QAAQ,sDAAsD;EAEnF,IAAI,UAAU,MAAM,6BAA6B,OAAO,uBAAuB,OAAO;EACtF,IAAI,WAAW,MAAM,6BAA6B,QAAQ,wBAAwB,OAAO;CAC3F;AACF;AAEA,SAAS,6BAA6B,OAAe,OAAe,SAAuB;CACzF,IAAI,CAAC,OAAO,cAAc,KAAK,KAAK,QAAQ,GAC1C,MAAM,IAAI,MAAM,GAAG,QAAQ,IAAI,MAAM,qCAAqC;AAE9E;AAEA,SAAS,wBAAwB,OAAe,OAAe,SAAuB;CACpF,IAAI,CAAC,OAAO,SAAS,KAAK,KAAK,QAAQ,GACrC,MAAM,IAAI,MAAM,GAAG,QAAQ,IAAI,MAAM,sCAAsC;AAE/E;;;;;;;;;;;;;;;;;;;;;AC3DA,MAAa,qBAAqB;CAAC;CAAY;CAAQ;CAAU;CAAO;AAAM;;;;;;;;;AAiO9E,SAAgB,iBAAiB,OAOtB;CACT,MAAM,QAAQ,KAAK,UAAU;EAC3B,GAAG,MAAM;EACT,GAAG,MAAM;EACT,GAAG,MAAM,WAAW;EACpB,GAAG,eAAe,MAAM,YAAY,MAAM,KAAK;CACjD,CAAC;CACD,OAAO,KAAK,WAAW,QAAQ,CAAC,CAAC,OAAO,KAAK,CAAC,CAAC,OAAO,KAAK,CAAC,CAAC,MAAM,GAAG,EAAE;AAC1E;AAEA,SAAS,eAAe,GAAmB;CAGzC,OAAO,EACJ,YAAY,CAAC,CACb,QAAQ,QAAQ,GAAG,CAAC,CACpB,QAAQ,eAAe,EAAE,CAAC,CAC1B,KAAK;AACV;;;;;AAMA,SAAgB,YACd,MAIgB;CAChB,MAAM,EAAE,UAAU,aAAa,GAAG,SAAS;CAC3C,OAAO;EACL,gBAAgB;EAChB,YAAY,iBAAiB;GAC3B,YAAY,KAAK;GACjB,MAAM,KAAK;GACX,SAAS,KAAK;GACd,OAAO,KAAK;GACZ;EACF,CAAC;EACD,aAAa,gCAAe,IAAI,KAAK,EAAA,CAAE,YAAY;EACnD,GAAG;CACL;AACF;;AAGA,SAAgB,oBACd,MAIiB;CACjB,MAAM,EAAE,iBAAiB,GAAG,YAAY;CACxC,OAAO;EAAE,GAAG,YAAY,OAAO;EAAG;CAAgB;AACpD"}
@@ -1,5 +1,5 @@
1
1
  import { s as RunSplitTag } from "./run-record-DTv1MdjK.js";
2
- import { R as Scenario, d as DispatchContext } from "./types-Ba5UQyVD.js";
2
+ import { R as Scenario, d as DispatchContext } from "./types-BJz2CPTM.js";
3
3
  //#region src/benchmarks/types.d.ts
4
4
  type BenchmarkTaskKind = 'retrieval' | 'rag-answer' | 'hallucination' | 'kb-improvement' | 'routing' | 'custom';
5
5
  type BenchmarkFamily = 'beir' | 'mteb-retrieval' | 'msmarco' | 'trec-dl' | 'miracl' | 'lotte' | 'bright' | 'crag' | 'hotpotqa' | 'kilt' | 'ragtruth' | 'faithbench' | 'first-party' | 'custom';
@@ -90,4 +90,4 @@ declare const BENCHMARK_SPLIT_SEED = "agent-eval-v1";
90
90
  declare function deterministicSplit(itemId: string, seed?: string): RunSplitTag;
91
91
  //#endregion
92
92
  export { BenchmarkFamily as a, BenchmarkSource as c, BenchmarkEvaluation as i, BenchmarkTaskKind as l, BenchmarkAdapter as n, BenchmarkResponder as o, BenchmarkDatasetItem as r, BenchmarkScenario as s, BENCHMARK_SPLIT_SEED as t, deterministicSplit as u };
93
- //# sourceMappingURL=types-BDV4PiMR.d.ts.map
93
+ //# sourceMappingURL=types-Dd1ejaeI.d.ts.map
@@ -1 +1 @@
1
- {"version":3,"file":"types-BDV4PiMR.d.ts","names":[],"sources":["../src/benchmarks/types.ts"],"mappings":";;;KASY;KAQA;UAgBK,qBAAqB;;;EAGpC;;EAEA,SAAS;;EAET,QAAQ;;EAER,SAAS;;EAET,WAAW;;EAEX;;EAEA,SAAS;EACT,WAAW;;UAGI;;;;EAIf;;EAEA;;EAEA,aAAa;;;EAGb,KAAK;EACL;;UAGe;EACf;EACA;EACA;EACA;EACA;;;UAQe,iBAAiB,kBAAkB,oBAAoB;;EAEtE;EACA,SAAS;EACT,WAAW;EACX;EACA,SAAS;EACT;;;;;EAKA,YAAY,OAAO,cAAc,QAAQ,qBAAqB;;EAE9D,SAAS,MAAM,qBAAqB,WAAW,UAAU,YAAY,QAAQ;;;;EAI7E,YAAY,iBAAiB;;UAGd,kBAAkB,4BAA4B;EAC7D;EACA;EACA,QAAQ;EACR,UAAU;EACV,UAAU;EACV,MAAM,qBAAqB;;KAGjB,mBAAmB,oBAAoB,uBAAuB;EACxE,UAAU,kBAAkB;EAC5B,MAAM,qBAAqB;EAC3B,SAAS;MACL,QAAQ,aAAa;;;cAoBd;;;;;;;;;iBAUG,mBACd,gBACA,gBACC"}
1
+ {"version":3,"file":"types-Dd1ejaeI.d.ts","names":[],"sources":["../src/benchmarks/types.ts"],"mappings":";;;KASY;KAQA;UAgBK,qBAAqB;;;EAGpC;;EAEA,SAAS;;EAET,QAAQ;;EAER,SAAS;;EAET,WAAW;;EAEX;;EAEA,SAAS;EACT,WAAW;;UAGI;;;;EAIf;;EAEA;;EAEA,aAAa;;;EAGb,KAAK;EACL;;UAGe;EACf;EACA;EACA;EACA;EACA;;;UAQe,iBAAiB,kBAAkB,oBAAoB;;EAEtE;EACA,SAAS;EACT,WAAW;EACX;EACA,SAAS;EACT;;;;;EAKA,YAAY,OAAO,cAAc,QAAQ,qBAAqB;;EAE9D,SAAS,MAAM,qBAAqB,WAAW,UAAU,YAAY,QAAQ;;;;EAI7E,YAAY,iBAAiB;;UAGd,kBAAkB,4BAA4B;EAC7D;EACA;EACA,QAAQ;EACR,UAAU;EACV,UAAU;EACV,MAAM,qBAAqB;;KAGjB,mBAAmB,oBAAoB,uBAAuB;EACxE,UAAU,kBAAkB;EAC5B,MAAM,qBAAqB;EAC3B,SAAS;MACL,QAAQ,aAAa;;;cAoBd;;;;;;;;;iBAUG,mBACd,gBACA,gBACC"}
@@ -20,6 +20,48 @@ Use it when one surface must get better.
20
20
  Use it when two or more methods must be compared at equal budget.
21
21
  Runnable versions: [`examples/self-improve-optimizer`](../examples/self-improve-optimizer/) and [`examples/compare-optimization-methods`](../examples/compare-optimization-methods/).
22
22
 
23
+ ## Read An Improvement Result
24
+
25
+ `selfImprove({ method })` executes the complete method once and measures its selected surface on final cases.
26
+ The method may select the unchanged baseline; that result returns `gateDecision: 'hold'` and an empty diff.
27
+ Agent Eval does not score train and selection cases again or choose a different surface after the method finishes.
28
+
29
+ The result type has two modes:
30
+
31
+ | Mode | Result | Search evidence | Cost |
32
+ |---|---|---|---|
33
+ | `proposer` | `SelfImproveProposerResult` | Native `raw.generations`, `generationsExplored`, and optional `searchHistory` | Shared `cost` ledger summary |
34
+ | `method` | `SelfImproveMethodResult` | Actual `raw.method` and its optional `searchHistory` | Combined method and final `cost`; receipt breakdown in `ledgerCost` |
35
+
36
+ Both types are exported from the package root and `/contract`.
37
+ `SelfImproveResult` is their union; branch on `result.mode` before reading mode-specific fields.
38
+ Calls with a concrete `method` or `proposer` infer the corresponding result type.
39
+ Method mode has no native generation count or fabricated native search measurements.
40
+ Its durable `method-provenance.json` uses schema `tangle.method-improvement` and records partition, measurement, and cost-receipt digests.
41
+ Proposer mode retains `LoopProvenanceRecord`.
42
+
43
+ When method holdout is deferred, `baseline` and `winner.compositeMean` are `null`, `lift` is absent, and the decision is `hold`.
44
+ The selected surface remains available in `winner.surface`.
45
+ Method cost preserves the larger of reported search spend and newly recorded search receipts, then adds final measurements without counting receipts twice.
46
+ Underreported spending and incomplete receipts remain explicit; `raw.method.cost` retains the original report.
47
+ Inspect `cost.accountingComplete` and `cost.incompleteReasons` before treating the known subtotal as complete spending.
48
+ The shared dollar limit controls calls admitted through the cost ledger; arbitrary off-ledger callbacks must enforce their own spending limits.
49
+
50
+ Native generation records report `ci95: null` because search does not estimate candidate uncertainty.
51
+ Final comparisons retain their independently computed statistics.
52
+ Every final case and replica must have complete execution and judge results before comparison.
53
+
54
+ ## Bind Cached Measurements To Their Evaluator
55
+
56
+ Candidate surface content is part of native search and final measurement identity.
57
+ Pass a stable `dispatchRef` for execution behavior outside that surface, such as the worker revision and tool configuration.
58
+ Change it when that behavior changes; function names cannot identify captured state.
59
+ Set `judgeVersion` when a judge's scoring behavior changes.
60
+
61
+ To reuse `premeasuredBaseline` in proposer mode, measure the same train cases, seed, replicas, execution revision, and judges.
62
+ The standalone campaign must use `dispatchRef: surfaceDispatchRef(baselineSurface, dispatchRef)` from `/campaign`.
63
+ Agent Eval refuses a prior baseline whose evaluator manifest differs.
64
+
23
65
  ## Adapt A Third-Party Text Optimizer
24
66
 
25
67
  `externalTextOptimizationMethod()` is the general adapter for a package that already owns text or component search.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@tangle-network/agent-eval",
3
- "version": "0.173.3",
3
+ "version": "0.174.0",
4
4
  "description": "Evaluate and improve AI agents from runs, traces, judges, and feedback. Compare candidates, cluster failures, measure lift, and gate releases.",
5
5
  "homepage": "https://github.com/tangle-network/agent-eval#readme",
6
6
  "repository": {
@@ -1,84 +0,0 @@
1
- import { AgentProfile, AgentProfile as AgentProfile$1, HarnessType, HarnessType as HarnessType$1 } from "@tangle-network/agent-interface";
2
- //#region src/agent-profile.d.ts
3
- /**
4
- * The agentic coding harnesses an eval sweeps by default — the ones we care about
5
- * ranking. This is the SINGLE source of that list; consumers import it instead of
6
- * re-declaring their own (a re-declared list is how the fleet drifts). Pass an
7
- * explicit `harnesses` (e.g. `harnessTypeSchema.options` for literally every known
8
- * harness) to widen beyond these.
9
- */
10
- declare const CODING_HARNESSES: readonly HarnessType[];
11
- interface ProfileAxisSpec {
12
- /** The domain profile to sweep. Its prompt/tools/skills are held fixed; only the
13
- * harness and model vary. `model.default` is the fallback model. */
14
- base: AgentProfile;
15
- /** Harnesses to cross. Default: {@link CODING_HARNESSES}. */
16
- harnesses?: readonly HarnessType[];
17
- /** Models to cross. Default: `[base.model.default]` — one model, i.e. today's
18
- * single-model behaviour, so omitting this never changes an existing run. */
19
- models?: readonly string[];
20
- /** Force every (harness, model) pair verbatim, even ones the harness can't run —
21
- * for deliberately testing failure modes. Default (false): SNAP instead — a
22
- * vendor-locked harness runs only the swept models in its family, or its native
23
- * default when it supports none, so no harness is dropped and none gets a
24
- * guaranteed-failing foreign-model cell. */
25
- keepIncompatible?: boolean;
26
- }
27
- /** Model sentinel for a vendor-locked harness that supports none of the swept models:
28
- * it carries no provider prefix, so `harnessSupportsModel` accepts it and the harness
29
- * resolves it to its own native default model at runtime (e.g. kimi-code → its Kimi
30
- * model). Lets `expandProfileAxes` snap-instead-of-drop without a per-harness flagship
31
- * table that would rot as router catalogs change. */
32
- declare const HARNESS_NATIVE_MODEL = "default";
33
- /**
34
- * Expand a base profile across the harness × model matrix into the `AgentProfile[]`
35
- * that `runProfileMatrix` / `selfImprove` score — the ONE place "which harnesses ×
36
- * which models do we evaluate" lives, so no product hand-rolls its own harness list
37
- * or column→profile mapping (the pattern that let those copies drift and silently
38
- * break the harness pivot).
39
- *
40
- * Each cell clones `base`, sets the canonical top-level `harness` and `model.default`,
41
- * and stamps `metadata.harness` + `metadata.harnessModel` for matrix grouping. Both
42
- * metadata fields are hash-bearing, so every cell gets a distinct `agentProfileId` row
43
- * and results join back by harness/model via {@link harnessAxisOf} with no
44
- * hand-recomputed key. A vendor-locked harness snaps to its family's swept models — or
45
- * its native default ({@link HARNESS_NATIVE_MODEL}) when it supports none — so every
46
- * requested harness runs; `keepIncompatible` forces every pair verbatim.
47
- *
48
- * Omit `harnesses`/`models` to sweep the full default set — the "turn it on for
49
- * everything we care about" switch, identical in shape whether one harness or all.
50
- */
51
- declare function expandProfileAxes(spec: ProfileAxisSpec): AgentProfile[];
52
- /**
53
- * Read the (harness, model) a matrix cell ran under, off a profile or a result row's
54
- * profile — the join-back for a `byHarness` pivot. Returns undefined when the profile
55
- * wasn't produced by {@link expandProfileAxes}. Callers group `result.byProfile` by
56
- * this instead of recomputing an id (recomputing the wrong key is what broke the pivot
57
- * in the hand-rolled copies).
58
- */
59
- declare function harnessAxisOf(profile: Pick<AgentProfile, 'metadata'>): {
60
- harness: HarnessType;
61
- model: string;
62
- } | undefined;
63
- /**
64
- * Collision-resistant, path-safe, human-readable profile id for eval artifacts.
65
- * Scorecard joins still use `agentProfileHash`; this id is for run ids, matrix
66
- * keys, and directory names where two profiles must not collapse onto one row.
67
- * The suffix is the first 64 bits of the behaviour hash, enough for ordinary
68
- * eval matrices while keeping filenames readable.
69
- */
70
- declare function agentProfileId(profile: AgentProfile): string;
71
- /**
72
- * Deterministic behaviour identity for the canonical
73
- * `@tangle-network/agent-interface` AgentProfile.
74
- *
75
- * `name` and `description` are labels and do not affect the hash. Profile
76
- * `version`, prompt, model hints, tools, resources, hooks, modes, permissions,
77
- * and extensions do affect the hash. Resource array order is hash-bearing
78
- * because mount order can change agent behaviour. Undefined fields are treated
79
- * as absent; explicit `null` fields remain hash-bearing.
80
- */
81
- declare function agentProfileHash(profile: AgentProfile): string;
82
- //#endregion
83
- export { ProfileAxisSpec as a, expandProfileAxes as c, HarnessType$1 as i, harnessAxisOf as l, CODING_HARNESSES as n, agentProfileHash as o, HARNESS_NATIVE_MODEL as r, agentProfileId as s, AgentProfile$1 as t };
84
- //# sourceMappingURL=agent-profile-B9_GGsG8.d.ts.map
@@ -1 +0,0 @@
1
- {"version":3,"file":"agent-profile-B9_GGsG8.d.ts","names":[],"sources":["../src/agent-profile.ts"],"mappings":";;;;;;;;;cAea,2BAA2B;UAOvB;;;EAGf,MAAM;;EAEN,qBAAqB;;;EAGrB;;;;;;EAMA;;;;;;;cAQW;;;;;;;;;;;;;;;;;;;iBAoBG,kBAAkB,MAAM,kBAAkB;;;;;;;;iBAwD1C,cACd,SAAS,KAAK;EACX,SAAS;EAAa;;;;;;;;;iBAiBX,eAAe,SAAS;;;;;;;;;;;iBAmDxB,iBAAiB,SAAS"}
@@ -1,280 +0,0 @@
1
- import { t as AgentEvalError } from "./errors-DEE6u6ot.js";
2
- import { c as CostLedgerHandle } from "./cost-ledger-DbQdN3nO.js";
3
- import { a as RunRecord } from "./run-record-DTv1MdjK.js";
4
- import { Y as RawProviderSink, p as ChatClient } from "./types-gvRsyJLh.js";
5
- import { a as CheckerIdentity, n as VerdictCertification, t as DefaultVerdict, x as VerificationStrategySource } from "./verdict-E4eRNf7-.js";
6
- //#region src/artifact-validator.d.ts
7
- /**
8
- * Artifact validators.
9
- *
10
- * Generic "score a produced artifact" primitive. Tax uses it for PDF form
11
- * correctness, research for sourced briefs, browser for task assertions, coding
12
- * for social posts. One interface, many validators.
13
- *
14
- * A validator receives an `Artifact` (file on disk, JSON blob, text, binary)
15
- * plus a `ValidationContext` (scenario id, the turns that produced it) and
16
- * returns a `ValidationResult` with pass/fail + 0..1 score + structured
17
- * issues.
18
- */
19
- interface Artifact {
20
- /** Logical kind — validators type-guard on this */
21
- kind: 'file' | 'json' | 'text' | 'binary' | string;
22
- /** Filesystem-style path, optional */
23
- path?: string;
24
- /** String content for text/json/file kinds */
25
- content?: string;
26
- /** Binary content (if kind === 'binary') */
27
- bytes?: Uint8Array;
28
- /** Caller-supplied metadata (mimeType, sha256, size, etc.) */
29
- metadata?: Record<string, unknown>;
30
- }
31
- //#endregion
32
- //#region src/completion-verifier.d.ts
33
- /** What kind of produced state can satisfy a requirement structurally. */
34
- type SatisfiedBy = 'artifact' | 'proposal' | 'tool-call' | 'any';
35
- interface CompletionRequirement {
36
- /** Stable id from the task gold (e.g. a persona's `expected_requirements[].req_id`). */
37
- reqId: string;
38
- /** Human-readable description of the required deliverable. */
39
- title: string;
40
- /** Optional kind/category hint, matched against a produced item's kind. */
41
- category?: string;
42
- /** What produced state satisfies this requirement. Defaults to 'any'. */
43
- satisfiedBy?: SatisfiedBy;
44
- }
45
- interface TaskGold {
46
- taskId: string;
47
- requirements: CompletionRequirement[];
48
- }
49
- interface ProducedProposal {
50
- id: string;
51
- title: string;
52
- status: 'pending' | 'approved' | 'rejected';
53
- /** Optional persisted body — when present, enables a correctness check. */
54
- content?: string;
55
- }
56
- /** Everything observable about what a run actually produced. */
57
- interface ProducedState {
58
- /** Persisted vault artifacts. Reuses the shared `Artifact` shape. */
59
- artifacts: Artifact[];
60
- /** Proposals / filings the agent created. */
61
- proposals: ProducedProposal[];
62
- /** Names of tools the agent invoked. */
63
- toolCalls: string[];
64
- }
65
- interface RequirementCheck {
66
- reqId: string;
67
- title: string;
68
- /** A produced item of the right kind matched the requirement, non-empty. */
69
- structurallyPresent: boolean;
70
- /**
71
- * Whether the matched item actually fulfils the requirement. `null` when
72
- * not structurally present, when the matched item carries no content
73
- * to assess, or when the correctness check itself failed (`unmeasured`).
74
- */
75
- correct: boolean | null;
76
- /** structurallyPresent && !unmeasured && correct !== false. */
77
- satisfied: boolean;
78
- /**
79
- * Set when the correctness check itself errored (LLM call failure or an
80
- * unparseable response after retry). The requirement's fulfilment is
81
- * UNKNOWN — `correct` stays null, `satisfied` is false, and
82
- * `completionVerdict` excludes the row from `completionRate`'s
83
- * denominator. Never folded into a zero: a synthetic zero is
84
- * indistinguishable from a real failure (see `JudgeParseError`).
85
- */
86
- unmeasured?: true;
87
- /** Why the correctness check could not be measured (present iff `unmeasured`). */
88
- unmeasuredReason?: string;
89
- /** Human-readable evidence for the verdict. */
90
- evidence: string[];
91
- }
92
- /** Extends the substrate verdict spine: `valid` = `fullyComplete` and
93
- * `score` = `completionRate` — derived in `completionVerdict()`, the one
94
- * place those equalities hold by construction. */
95
- interface CompletionVerdict extends DefaultVerdict {
96
- taskId: string;
97
- requirements: RequirementCheck[];
98
- /** satisfied / MEASURABLE requirements (unmeasured rows leave the denominator). */
99
- completionRate: number;
100
- /** Every measurable requirement satisfied (false when anything is unmeasured). */
101
- fullyComplete: boolean;
102
- /** Requirements whose correctness check errored — reported, never scored as zero. */
103
- unmeasuredCount: number;
104
- }
105
- /**
106
- * Construct a `CompletionVerdict` from the per-requirement checks, deriving
107
- * `completionRate` / `fullyComplete` and the spine fields (`valid` =
108
- * `fullyComplete`, `score` = `completionRate`) in one place. Throws on zero
109
- * requirements — a verdict over nothing is a misconfiguration, mirroring
110
- * `verifyCompletion`'s gold-spec guard.
111
- */
112
- declare function completionVerdict(input: {
113
- taskId: string;
114
- requirements: RequirementCheck[];
115
- /** What certified the correctness stage, when anything did. Omitted =
116
- * an uncertified verdict — the honest default for a bare checker. */
117
- certification?: VerdictCertification;
118
- }): CompletionVerdict;
119
- /**
120
- * What a correctness checker declares about itself so the completion
121
- * verdict can carry a certification: which strategy member it discharges,
122
- * its exact identity, and the steps its answers rest on unverified.
123
- */
124
- interface CorrectnessCheckerAttestation {
125
- strategy: VerificationStrategySource;
126
- checker: CheckerIdentity;
127
- assumptions: string[];
128
- }
129
- /**
130
- * Decides whether a produced item's content actually fulfils a requirement.
131
- * Injected so the structural verifier stays pure and unit-testable; the
132
- * production implementation is `createLlmCorrectnessChecker`.
133
- *
134
- * `attestation` is optional metadata on the function value: a checker that
135
- * carries one yields certified completion verdicts; a bare function yields
136
- * the same verdict uncertified. A plain arrow function remains a valid
137
- * checker.
138
- */
139
- interface CorrectnessChecker {
140
- (requirement: CompletionRequirement, content: string): Promise<{
141
- correct: boolean;
142
- reason: string;
143
- }>;
144
- attestation?: CorrectnessCheckerAttestation;
145
- }
146
- /**
147
- * Verify whether a run completed the task. `checkCorrectness` is injected —
148
- * `createLlmCorrectnessChecker` for production, a deterministic stub in tests.
149
- *
150
- * Throws on a gold spec with no requirements: an eval task that requires
151
- * nothing is a misconfiguration, not a vacuously-complete task.
152
- */
153
- declare function verifyCompletion(gold: TaskGold, state: ProducedState, checkCorrectness: CorrectnessChecker): Promise<CompletionVerdict>;
154
- interface LlmCorrectnessCheckerOpts {
155
- model?: string;
156
- /** Optional ledger for direct use. */
157
- costLedger?: CostLedgerHandle;
158
- costPhase?: string;
159
- costTags?: Record<string, string>;
160
- signal?: AbortSignal;
161
- /** Max chars of artifact content sent to the checker. */
162
- maxContentChars?: number;
163
- /**
164
- * Checker LLM calls per requirement before giving up (parse failures and
165
- * call errors both consume attempts). The failure then surfaces as an
166
- * `unmeasured` requirement, never a zero.
167
- */
168
- maxAttempts?: number;
169
- /**
170
- * Forensic capture of every checker request/response/error — without it a
171
- * checker failure is unauditable (the agent-turn raws never contain the
172
- * checker's own calls). Same sink contract as `LlmClient`.
173
- */
174
- rawSink?: RawProviderSink;
175
- }
176
- /**
177
- * Production `CorrectnessChecker` — one LLM call per matched artifact,
178
- * deterministic (temperature 0), structured JSON out. Judges fulfilment
179
- * only: a plan, a gesture, or a description of what should be done does not
180
- * fulfil a requirement — the artifact must BE the deliverable.
181
- */
182
- declare function createLlmCorrectnessChecker(chat: ChatClient, opts?: LlmCorrectnessCheckerOpts): CorrectnessChecker;
183
- //#endregion
184
- //#region src/produced-state.d.ts
185
- /** A tool the agent invoked. */
186
- interface ToolCallEventLike {
187
- type: 'tool_call';
188
- toolName: string;
189
- }
190
- /**
191
- * An artifact the agent produced. `content` is the enriched field — the
192
- * runtime's base `artifact` event carries only metadata; the completion
193
- * oracle needs the body to verify the deliverable, so the runtime emits it.
194
- */
195
- interface ArtifactEventLike {
196
- type: 'artifact';
197
- artifactId: string;
198
- name?: string;
199
- mimeType?: string;
200
- uri?: string;
201
- content?: string;
202
- }
203
- /** A proposal / filing the agent created. */
204
- interface ProposalEventLike {
205
- type: 'proposal_created';
206
- proposalId: string;
207
- title: string;
208
- status?: 'pending' | 'approved' | 'rejected';
209
- content?: string;
210
- }
211
- /**
212
- * The subset of runtime stream events `extractProducedState` consumes.
213
- * agent-runtime's full `RuntimeStreamEvent` union satisfies this structurally;
214
- * the `{ type: string }` catch-all keeps the input permissive so callers can
215
- * pass the whole unfiltered telemetry stream — unrecognized events are skipped.
216
- */
217
- type RuntimeEventLike = ToolCallEventLike | ArtifactEventLike | ProposalEventLike | {
218
- type: string;
219
- };
220
- /**
221
- * Normalize a run's runtime event stream into `ProducedState`.
222
- *
223
- * Pure and total — unrecognized event types are skipped. `toolCalls` is
224
- * deduplicated by name in first-seen order (completion cares about a tool's
225
- * presence, not its call count). An artifact with neither a name nor a uri
226
- * still yields an entry keyed by its `artifactId` so it is never silently
227
- * dropped; an artifact with no `content` yields empty content, which the
228
- * completion oracle's structural check then rejects on its own.
229
- */
230
- declare function extractProducedState(events: readonly RuntimeEventLike[]): ProducedState;
231
- //#endregion
232
- //#region src/integrity/backend-integrity.d.ts
233
- interface BackendIntegrityReport {
234
- /** Total records inspected. */
235
- totalRecords: number;
236
- /** Records with input=0 AND output=0 (a stub fingerprint). */
237
- stubRecords: number;
238
- /** Records with nonzero token usage (real LLM activity). */
239
- realRecords: number;
240
- /** Records where output>0 but costUsd=0 (real LLM, broken cost ledger). */
241
- uncostedRecords: number;
242
- /** Sum of input tokens across all records. */
243
- totalInputTokens: number;
244
- /** Sum of output tokens across all records. */
245
- totalOutputTokens: number;
246
- /** Sum of costUsd across all records. */
247
- totalCostUsd: number;
248
- /** Worst-case integrity verdict. */
249
- verdict: 'real' | 'mixed' | 'stub';
250
- /** Human-readable diagnosis suitable for terminal output. */
251
- diagnosis: string;
252
- }
253
- /**
254
- * Error thrown when an integrity assertion fails. Caller can pattern-match
255
- * by `code === 'AGENT_EVAL_BACKEND_STUB'` to differentiate from other
256
- * errors.
257
- */
258
- declare class BackendIntegrityError extends AgentEvalError {
259
- readonly report: BackendIntegrityReport;
260
- constructor(message: string, report: BackendIntegrityReport);
261
- }
262
- /**
263
- * Inspect a batch of RunRecords and return an integrity report. Pure
264
- * function — no I/O, no logging. The caller decides what to do with the
265
- * verdict (print warning, throw, gate CI, etc.).
266
- */
267
- declare function summarizeBackendIntegrity(records: ReadonlyArray<RunRecord>): BackendIntegrityReport;
268
- /**
269
- * Throw BackendIntegrityError if the verdict is 'stub' — i.e. every record
270
- * shows zero LLM activity. Non-strict callers can pass `{ allowMixed: false }`
271
- * to also reject mixed verdicts (recommended for CI gates).
272
- *
273
- * Real backends pass through silently.
274
- */
275
- declare function assertRealBackend(records: ReadonlyArray<RunRecord>, opts?: {
276
- allowMixed?: boolean;
277
- }): BackendIntegrityReport;
278
- //#endregion
279
- export { SatisfiedBy as _, ArtifactEventLike as a, createLlmCorrectnessChecker as b, ToolCallEventLike as c, CompletionVerdict as d, CorrectnessChecker as f, RequirementCheck as g, ProducedState as h, summarizeBackendIntegrity as i, extractProducedState as l, ProducedProposal as m, BackendIntegrityReport as n, ProposalEventLike as o, LlmCorrectnessCheckerOpts as p, assertRealBackend as r, RuntimeEventLike as s, BackendIntegrityError as t, CompletionRequirement as u, TaskGold as v, verifyCompletion as x, completionVerdict as y };
280
- //# sourceMappingURL=backend-integrity-CeuTgqsd.d.ts.map
@@ -1 +0,0 @@
1
- {"version":3,"file":"backend-integrity-CeuTgqsd.d.ts","names":[],"sources":["../src/artifact-validator.ts","../src/completion-verifier.ts","../src/produced-state.ts","../src/integrity/backend-integrity.ts"],"mappings":";;;;;;;;;;;;;;;;;;UAaiB;;EAEf;;EAEA;;EAEA;;EAEA,QAAQ;;EAER,WAAW;;;;;KCsBD;UAEK;;EAEf;;EAEA;;EAEA;;EAEA,cAAc;;UAGC;EACf;EACA,cAAc;;UAGC;EACf;EACA;EACA;;EAEA;;;UAIe;;EAEf,WAAW;;EAEX,WAAW;;EAEX;;UAGe;EACf;EACA;;EAEA;;;;;;EAMA;;EAEA;;;;;;;;;EASA;;EAEA;;EAEA;;;;;UAMe,0BAA0B;EACzC;EACA,cAAc;;EAEd;;EAEA;;EAEA;;;;;;;;;iBAUc,kBAAkB;EAChC;EACA,cAAc;;;EAGd,gBAAgB;IACd;;;;;;UAqCa;EACf,UAAU;EACV,SAAS;EACT;;;;;;;;;;;;UAae;GAEb,aAAa,uBACb,kBACC;IAAU;IAAkB;;EAC/B,cAAc;;;;;;;;;iBAsLM,iBACpB,MAAM,UACN,OAAO,eACP,kBAAkB,qBACjB,QAAQ;UAyGM;EACf;;EAEA,aAAa;EACb;EACA,WAAW;EACX,SAAS;;EAET;;;;;;EAMA;;;;;;EAMA,UAAU;;;;;;;;iBA2CI,4BACd,MAAM,YACN,OAAM,4BACL;;;;UCnhBc;EACf;EACA;;;;;;;UAQe;EACf;EACA;EACA;EACA;EACA;EACA;;;UAIe;EACf;EACA;EACA;EACA;EAIA;;;;;;;;KASU,mBACR,oBACA,oBACA;EACE;;;;;;;;;;;;iBAmBU,qBAAqB,iBAAiB,qBAAqB;;;UCpD1D;;EAEf;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;;;;;;cAQW,8BAA8B;WAGvB,QAAQ;EAF1B,YACE,iBACgB,QAAQ;;;;;;;iBAWZ,0BACd,SAAS,cAAc,aACtB;;;;;;;;iBAuHa,kBACd,SAAS,cAAc,YACvB;EAAQ;IACP"}