@tangle-network/agent-eval 0.171.0 → 0.172.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (114) hide show
  1. package/CHANGELOG.md +55 -0
  2. package/README.md +3 -0
  3. package/dist/adapters/http.d.ts +2 -2
  4. package/dist/analyst/index.d.ts +5 -5
  5. package/dist/analyst/index.js +2 -2
  6. package/dist/{experiment-tracker-Dm8yQMqb.d.ts → attestation-CJBGmMVh.d.ts} +78 -2
  7. package/dist/attestation-CJBGmMVh.d.ts.map +1 -0
  8. package/dist/{experiment-tracker-BKEumQug.js → attestation-XSUpbc4o.js} +96 -2
  9. package/dist/attestation-XSUpbc4o.js.map +1 -0
  10. package/dist/{benchmark-command-D8k3Gf0J.js → benchmark-command-DoFcisuM.js} +2 -2
  11. package/dist/{benchmark-command-D8k3Gf0J.js.map → benchmark-command-DoFcisuM.js.map} +1 -1
  12. package/dist/benchmarks/index.d.ts +4 -4
  13. package/dist/benchmarks/index.js +2 -2
  14. package/dist/bounded-process-CVOC_D3H.js +181 -0
  15. package/dist/bounded-process-CVOC_D3H.js.map +1 -0
  16. package/dist/builder-eval/index.js +39 -98
  17. package/dist/builder-eval/index.js.map +1 -1
  18. package/dist/campaign/index.d.ts +8 -7
  19. package/dist/campaign/index.js +4 -4
  20. package/dist/{campaign-B72njjHj.js → campaign-Dp35pBbS.js} +4 -4
  21. package/dist/{campaign-B72njjHj.js.map → campaign-Dp35pBbS.js.map} +1 -1
  22. package/dist/canonical-CFpojCN5.d.ts +31 -0
  23. package/dist/canonical-CFpojCN5.d.ts.map +1 -0
  24. package/dist/{chat-client-DEtybj5i.js → chat-client-DI79OPye.js} +2 -2
  25. package/dist/{chat-client-DEtybj5i.js.map → chat-client-DI79OPye.js.map} +1 -1
  26. package/dist/cli.js +1 -1
  27. package/dist/{client-CDtcZ3p9.d.ts → client-Df7wdslk.d.ts} +2 -2
  28. package/dist/{client-CDtcZ3p9.d.ts.map → client-Df7wdslk.d.ts.map} +1 -1
  29. package/dist/contract/index.d.ts +9 -9
  30. package/dist/contract/index.js +4 -4
  31. package/dist/{default-registry-B0s2zU-s.d.ts → default-registry-XxedTLwu.d.ts} +3 -3
  32. package/dist/{default-registry-B0s2zU-s.d.ts.map → default-registry-XxedTLwu.d.ts.map} +1 -1
  33. package/dist/{define-agent-eval-Cjy2yhqP.d.ts → define-agent-eval-0wW7gFhr.d.ts} +4 -4
  34. package/dist/{define-agent-eval-Cjy2yhqP.d.ts.map → define-agent-eval-0wW7gFhr.d.ts.map} +1 -1
  35. package/dist/{define-agent-eval-Dy8QgxAI.js → define-agent-eval-jS8xj_Q_.js} +2 -2
  36. package/dist/{define-agent-eval-Dy8QgxAI.js.map → define-agent-eval-jS8xj_Q_.js.map} +1 -1
  37. package/dist/descriptive-B2iPaT9J.d.ts +89 -0
  38. package/dist/descriptive-B2iPaT9J.d.ts.map +1 -0
  39. package/dist/{engine-CAmTUk52.d.ts → engine-BfRay1qD.d.ts} +2 -2
  40. package/dist/{engine-CAmTUk52.d.ts.map → engine-BfRay1qD.d.ts.map} +1 -1
  41. package/dist/experiment/index.d.ts +3 -54
  42. package/dist/experiment/index.d.ts.map +1 -1
  43. package/dist/experiment/index.js +1 -95
  44. package/dist/experiment/index.js.map +1 -1
  45. package/dist/{heldout-gate-Dh2b62w8.d.ts → heldout-gate-JgNRDZwZ.d.ts} +3 -3
  46. package/dist/{heldout-gate-Dh2b62w8.d.ts.map → heldout-gate-JgNRDZwZ.d.ts.map} +1 -1
  47. package/dist/hosted/index.d.ts +1 -1
  48. package/dist/{index-lfaSeKSD.d.ts → index-D-UdhAmg.d.ts} +3 -31
  49. package/dist/index-D-UdhAmg.d.ts.map +1 -0
  50. package/dist/{index-DT73JraI.d.ts → index-DDAPhUJJ.d.ts} +4 -4
  51. package/dist/{index-DT73JraI.d.ts.map → index-DDAPhUJJ.d.ts.map} +1 -1
  52. package/dist/{index-fNXZMCzX.d.ts → index-DnglhM0A.d.ts} +9 -9
  53. package/dist/{index-fNXZMCzX.d.ts.map → index-DnglhM0A.d.ts.map} +1 -1
  54. package/dist/{index-8VIogTyS.d.ts → index-_vPrVMRX.d.ts} +6 -6
  55. package/dist/{index-8VIogTyS.d.ts.map → index-_vPrVMRX.d.ts.map} +1 -1
  56. package/dist/index.d.ts +116 -15
  57. package/dist/index.d.ts.map +1 -1
  58. package/dist/index.js +7 -6
  59. package/dist/index.js.map +1 -1
  60. package/dist/{integrity-CyWSSoQS.js → integrity-BWywb34E.js} +34 -11
  61. package/dist/{integrity-CyWSSoQS.js.map → integrity-BWywb34E.js.map} +1 -1
  62. package/dist/ledger-core/index.d.ts +2 -1
  63. package/dist/{llm-judge-BtJ2Sfk_.js → llm-judge-aQHIk5_-.js} +16 -10
  64. package/dist/llm-judge-aQHIk5_-.js.map +1 -0
  65. package/dist/{matrix-DiHmUobV.d.ts → matrix-Ch8JO1pG.d.ts} +2 -2
  66. package/dist/{matrix-DiHmUobV.d.ts.map → matrix-Ch8JO1pG.d.ts.map} +1 -1
  67. package/dist/meta-eval/index.d.ts +162 -2
  68. package/dist/meta-eval/index.d.ts.map +1 -1
  69. package/dist/meta-eval/index.js +287 -2
  70. package/dist/meta-eval/index.js.map +1 -1
  71. package/dist/multishot/golden/index.d.ts +1 -1
  72. package/dist/multishot/index.d.ts +2 -2
  73. package/dist/openapi.json +1 -1
  74. package/dist/{produced-state-Cm6DU_Ao.js → produced-state-CxmbFxFd.js} +2 -2
  75. package/dist/{produced-state-Cm6DU_Ao.js.map → produced-state-CxmbFxFd.js.map} +1 -1
  76. package/dist/{promotion-policy-WSXtBgBb.d.ts → promotion-policy-BBBcz5_3.d.ts} +2 -2
  77. package/dist/{promotion-policy-WSXtBgBb.d.ts.map → promotion-policy-BBBcz5_3.d.ts.map} +1 -1
  78. package/dist/{provenance-CafMdZKM.d.ts → provenance-Dp-vvyrU.d.ts} +13 -5
  79. package/dist/provenance-Dp-vvyrU.d.ts.map +1 -0
  80. package/dist/rl.d.ts +1 -1
  81. package/dist/sandbox-harness-BlSOu4LX.d.ts.map +1 -1
  82. package/dist/{skillopt-optimization-method-DzlF2RM7.js → skillopt-optimization-method-LHi02MzH.js} +2 -2
  83. package/dist/{skillopt-optimization-method-DzlF2RM7.js.map → skillopt-optimization-method-LHi02MzH.js.map} +1 -1
  84. package/dist/{statistical-heldout-UhiexnjU.d.ts → statistical-heldout-Yldkntvy.d.ts} +2 -2
  85. package/dist/{statistical-heldout-UhiexnjU.d.ts.map → statistical-heldout-Yldkntvy.d.ts.map} +1 -1
  86. package/dist/{store-tool-spans-D_qMl2__.d.ts → store-tool-spans-BvdUbeOB.d.ts} +3 -3
  87. package/dist/{store-tool-spans-D_qMl2__.d.ts.map → store-tool-spans-BvdUbeOB.d.ts.map} +1 -1
  88. package/dist/supervisor-run/index.d.ts +35 -2
  89. package/dist/supervisor-run/index.d.ts.map +1 -1
  90. package/dist/supervisor-run/index.js +83 -38
  91. package/dist/supervisor-run/index.js.map +1 -1
  92. package/dist/{tool-groups-RGYfVWpc.d.ts → tool-groups-DjwlMBvW.d.ts} +2 -2
  93. package/dist/tool-groups-DjwlMBvW.d.ts.map +1 -0
  94. package/dist/trace-repair/index.d.ts +1 -1
  95. package/dist/traces.d.ts +2 -2
  96. package/dist/{types-CMyW4GnH.d.ts → types-CCZ34qmV.d.ts} +2 -2
  97. package/dist/{types-CMyW4GnH.d.ts.map → types-CCZ34qmV.d.ts.map} +1 -1
  98. package/dist/{types-DeIUdzNd.d.ts → types-CoPUTiXb.d.ts} +23 -92
  99. package/dist/types-CoPUTiXb.d.ts.map +1 -0
  100. package/dist/{types-JHMOqZI4.d.ts → types-nokrtr7M.d.ts} +11 -1
  101. package/dist/types-nokrtr7M.d.ts.map +1 -0
  102. package/docs/eval-surface-map.md +36 -0
  103. package/docs/insight-report.md +19 -0
  104. package/docs/plants.md +123 -0
  105. package/docs/public-api.md +45 -20
  106. package/package.json +1 -1
  107. package/dist/experiment-tracker-BKEumQug.js.map +0 -1
  108. package/dist/experiment-tracker-Dm8yQMqb.d.ts.map +0 -1
  109. package/dist/index-lfaSeKSD.d.ts.map +0 -1
  110. package/dist/llm-judge-BtJ2Sfk_.js.map +0 -1
  111. package/dist/provenance-CafMdZKM.d.ts.map +0 -1
  112. package/dist/tool-groups-RGYfVWpc.d.ts.map +0 -1
  113. package/dist/types-DeIUdzNd.d.ts.map +0 -1
  114. package/dist/types-JHMOqZI4.d.ts.map +0 -1
@@ -1,5 +1,5 @@
1
1
  import { w as TraceAnalysisStore } from "./types-DMoNFDWi.js";
2
- import { m as TraceAnalysisToolDescriptor } from "./engine-CAmTUk52.js";
2
+ import { m as TraceAnalysisToolDescriptor } from "./engine-BfRay1qD.js";
3
3
  //#region src/analyst/tool-groups.d.ts
4
4
  /** Named tool sets. Kinds pass `tools: TRACE_TOOL_GROUPS.failureForensics` etc. */
5
5
  type TraceToolGroupName =
@@ -25,4 +25,4 @@ type TraceToolGroupName =
25
25
  declare function buildTraceToolsForGroup(group: TraceToolGroupName, store: TraceAnalysisStore): TraceAnalysisToolDescriptor[];
26
26
  //#endregion
27
27
  export { buildTraceToolsForGroup as n, TraceToolGroupName as t };
28
- //# sourceMappingURL=tool-groups-RGYfVWpc.d.ts.map
28
+ //# sourceMappingURL=tool-groups-DjwlMBvW.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"tool-groups-DjwlMBvW.d.ts","names":[],"sources":["../src/analyst/tool-groups.ts"],"mappings":";;;;KAmBY;;;;;;;;;;;;;;;;;;;;iBAgDI,wBACd,OAAO,oBACP,OAAO,qBACN"}
@@ -2,7 +2,7 @@ import { r as CaptureIntegrityError } from "../errors-DEE6u6ot.js";
2
2
  import { b as CustomTokenPricing, c as CostLedgerHandle, p as CostProvenance } from "../cost-ledger-DbQdN3nO.js";
3
3
  import { u as RunTokenUsage } from "../run-record-DQjRcYwA.js";
4
4
  import { t as DefaultVerdict } from "../verdict-E4eRNf7-.js";
5
- import { i as TraceAnalystLimits, t as TraceAnalysisEngine } from "../engine-CAmTUk52.js";
5
+ import { i as TraceAnalystLimits, t as TraceAnalysisEngine } from "../engine-BfRay1qD.js";
6
6
  import { t as PrimeBridgeTransport } from "../prime-bridge-transport-6feEglLf.js";
7
7
  import { a as RecordedTrajectoryStep, h as isRecordedTimeout } from "../steps-CiNVJry_.js";
8
8
  //#region src/trace-repair/mini-swe-scaffold.d.ts
package/dist/traces.d.ts CHANGED
@@ -2,10 +2,10 @@ import { c as ValidationError, o as LimitExceededError, r as CaptureIntegrityErr
2
2
  import { C as TraceEvent, E as isToolSpan, S as ToolSpan, T as isLlmSpan, _ as Span, a as FAILURE_CLASSES, b as SpanStatus, c as JudgeSpan, d as RetrievalSpan, f as Run, g as SandboxSpan, h as RunStatus, i as EventKind, l as LlmSpan, m as RunOutcome, n as BudgetLedgerEntry, o as FailureClass, p as RunLayer, r as BudgetSpec, s as GenericSpan, t as Artifact, u as Message, v as SpanBase, w as isJudgeSpan, x as TRACE_SCHEMA_VERSION, y as SpanKind } from "./schema-DID1Cqct.js";
3
3
  import { a as RunRecord, l as RunTerminalOutcome, s as RunSplitTag, u as RunTokenUsage } from "./run-record-DQjRcYwA.js";
4
4
  import { A as SearchSpanResult, B as ViewSpansResult, C as TRACE_ANALYSIS_LIMITS, D as DatasetOverview, E as DEFAULT_TRACE_ANALYST_BUDGETS, F as TraceAnalystFilters, H as ViewTraceResult, I as TraceAnalystSpan, L as TraceAnalystSpanKind, M as SpanMatchRecord, N as TRACE_ANALYST_TRUNCATION_MARKER_PREFIX, O as ErrorCluster, P as TraceAnalystByteBudgets, R as TraceAnalystSpanStatus, S as BoundedTraceAnalysisStoreOptions, T as TraceAnalysisStoreContext, V as ViewTraceOversized, j as SearchTraceResult, k as QueryTracesPage, w as TraceAnalysisStore, z as TraceAnalystTraceSummary } from "./types-DMoNFDWi.js";
5
- import { A as REDACTION_VERSION, B as exportRunAsOtlp, C as scoreTraceInsightReadiness, D as AnalyzeTracesResult, E as AnalyzeTracesOptions, F as OtlpFlatLine, G as ExtractedUsage, H as CaptureFetchOptions, I as OTEL_AGENT_EVAL_SCOPE, J as extractUsageFromResponse, K as SseUsageMode, L as OtlpExport, M as RedactionRule, N as redactString, O as analyzeTraces, P as redactValue, R as OtlpResourceSpans, S as planTraceInsightQuestions, T as AnalyzeTracesInput, U as captureFetchToRawSink, V as CaptureFetchContext, W as ExtractUsageFromSseOptions, X as createBoundedTraceAnalysisStore, Y as extractUsageFromSse, _ as buildTraceInsightPrompt, a as ToolSpansToTraceAnalysisStoreOptions, b as domainEvidencePattern, c as TraceInsightFinding, d as TraceInsightQualityGate, f as TraceInsightQuestion, g as buildTraceInsightContext, h as TraceInsightTask, i as OtlpFileTraceStoreOptions, j as RedactionReport, k as DEFAULT_REDACTION_RULES, l as TraceInsightPanelRole, m as TraceInsightSuite, n as toolSpansToTraceAnalysisStore, o as otlpTextToTraceAnalysisStore, p as TraceInsightReadiness, q as extractUsage, r as OtlpFileTraceStore, s as TraceInsightContext, t as ToolTraceMissingError, u as TraceInsightPromptInput, v as defaultTraceInsightPanel, w as tokenizeDomainWords, x as inferDomainKeywords, y as describeTraceInsightScope, z as OtlpSpan } from "./store-tool-spans-D_qMl2__.js";
5
+ import { A as REDACTION_VERSION, B as exportRunAsOtlp, C as scoreTraceInsightReadiness, D as AnalyzeTracesResult, E as AnalyzeTracesOptions, F as OtlpFlatLine, G as ExtractedUsage, H as CaptureFetchOptions, I as OTEL_AGENT_EVAL_SCOPE, J as extractUsageFromResponse, K as SseUsageMode, L as OtlpExport, M as RedactionRule, N as redactString, O as analyzeTraces, P as redactValue, R as OtlpResourceSpans, S as planTraceInsightQuestions, T as AnalyzeTracesInput, U as captureFetchToRawSink, V as CaptureFetchContext, W as ExtractUsageFromSseOptions, X as createBoundedTraceAnalysisStore, Y as extractUsageFromSse, _ as buildTraceInsightPrompt, a as ToolSpansToTraceAnalysisStoreOptions, b as domainEvidencePattern, c as TraceInsightFinding, d as TraceInsightQualityGate, f as TraceInsightQuestion, g as buildTraceInsightContext, h as TraceInsightTask, i as OtlpFileTraceStoreOptions, j as RedactionReport, k as DEFAULT_REDACTION_RULES, l as TraceInsightPanelRole, m as TraceInsightSuite, n as toolSpansToTraceAnalysisStore, o as otlpTextToTraceAnalysisStore, p as TraceInsightReadiness, q as extractUsage, r as OtlpFileTraceStore, s as TraceInsightContext, t as ToolTraceMissingError, u as TraceInsightPromptInput, v as defaultTraceInsightPanel, w as tokenizeDomainWords, x as inferDomainKeywords, y as describeTraceInsightScope, z as OtlpSpan } from "./store-tool-spans-BvdUbeOB.js";
6
6
  import { a as NoopRawProviderSink, c as RawProviderEvent, d as defaultProviderRedactor, f as providerFromBaseUrl, i as InMemoryRawProviderSinkOptions, l as RawProviderSink, n as FileSystemRawProviderSinkOptions, o as ProviderRedactor, r as InMemoryRawProviderSink, s as RawProviderDirection, t as FileSystemRawProviderSink, u as RawProviderSinkFilter } from "./raw-provider-sink-BU29Sh8h.js";
7
7
  import { a as RunFilter, i as InMemoryTraceStore, n as FileSystemTraceStore, o as SpanFilter, r as FileSystemTraceStoreOptions, s as TraceStore, t as EventFilter } from "./store-Cq9oOrI1.js";
8
- import { f as BuildTraceAnalysisToolsOptions, g as traceAnalystFunctionGroup, h as buildTraceAnalysisToolDescriptors, m as TraceAnalysisToolDescriptor, p as TRACE_ANALYST_TOOL_NAMESPACE } from "./engine-CAmTUk52.js";
8
+ import { f as BuildTraceAnalysisToolsOptions, g as traceAnalystFunctionGroup, h as buildTraceAnalysisToolDescriptors, m as TraceAnalysisToolDescriptor, p as TRACE_ANALYST_TOOL_NAMESPACE } from "./engine-BfRay1qD.js";
9
9
  import { a as TraceEmitterOptions, i as TraceEmitter, n as RunCompleteHookContext, r as SpanHandle, t as RunCompleteHook } from "./emitter-Bvnu0VzL.js";
10
10
  import { C as TOOL_LATENCY_MS, D as asNumber, E as applyLlmSpanOtlpAttributes, O as contextInputTokens, S as TOOL_ARGS_CAPTURED, T as TOOL_NAME_ATTR_KEYS, _ as LlmSpanOtlpInput, a as LLM_CACHE_WRITE_TOKEN_ATTR_KEYS, b as RUN_COST_ATTR_KEYS, c as LLM_COST_USD, d as LLM_MODEL_ATTR_KEYS, f as LLM_MODEL_NAME, g as LLM_REASONING_TOKEN_ATTR_KEYS, h as LLM_REASONING_TOKENS, i as LLM_CACHE_WRITE_TOKENS, k as firstNumberAttr, l as LLM_INPUT_TOKENS, m as LLM_OUTPUT_TOKEN_ATTR_KEYS, n as LLM_CACHED_TOKENS, o as LLM_CONTEXT_TOKENS, p as LLM_OUTPUT_TOKENS, r as LLM_CACHED_TOKEN_ATTR_KEYS, s as LLM_COST_ATTR_KEYS, t as INPUT_VALUE, u as LLM_INPUT_TOKEN_ATTR_KEYS, v as OPENINFERENCE_SPAN_KIND, w as TOOL_NAME, x as SPAN_KIND_ATTR_KEYS, y as OUTPUT_VALUE } from "./attribute-vocabulary-DLJ6303h.js";
11
11
  import { a as RunIntegrityReport, i as RunIntegrityIssueCode, n as RunIntegrityExpectations, o as assertRunCaptured, r as RunIntegrityIssue, s as throwIfRunIncomplete, t as RunIntegrityError } from "./integrity-B_EDELom.js";
@@ -1,5 +1,5 @@
1
1
  import { s as RunSplitTag } from "./run-record-DQjRcYwA.js";
2
- import { R as Scenario, d as DispatchContext } from "./types-JHMOqZI4.js";
2
+ import { R as Scenario, d as DispatchContext } from "./types-nokrtr7M.js";
3
3
  //#region src/benchmarks/types.d.ts
4
4
  type BenchmarkTaskKind = 'retrieval' | 'rag-answer' | 'hallucination' | 'kb-improvement' | 'routing' | 'custom';
5
5
  type BenchmarkFamily = 'beir' | 'mteb-retrieval' | 'msmarco' | 'trec-dl' | 'miracl' | 'lotte' | 'bright' | 'crag' | 'hotpotqa' | 'kilt' | 'ragtruth' | 'faithbench' | 'first-party' | 'custom';
@@ -90,4 +90,4 @@ declare const BENCHMARK_SPLIT_SEED = "agent-eval-v1";
90
90
  declare function deterministicSplit(itemId: string, seed?: string): RunSplitTag;
91
91
  //#endregion
92
92
  export { BenchmarkFamily as a, BenchmarkSource as c, BenchmarkEvaluation as i, BenchmarkTaskKind as l, BenchmarkAdapter as n, BenchmarkResponder as o, BenchmarkDatasetItem as r, BenchmarkScenario as s, BENCHMARK_SPLIT_SEED as t, deterministicSplit as u };
93
- //# sourceMappingURL=types-CMyW4GnH.d.ts.map
93
+ //# sourceMappingURL=types-CCZ34qmV.d.ts.map
@@ -1 +1 @@
1
- {"version":3,"file":"types-CMyW4GnH.d.ts","names":[],"sources":["../src/benchmarks/types.ts"],"mappings":";;;KASY;KAQA;UAgBK,qBAAqB;;;EAGpC;;EAEA,SAAS;;EAET,QAAQ;;EAER,SAAS;;EAET,WAAW;;EAEX;;EAEA,SAAS;EACT,WAAW;;UAGI;;;;EAIf;;EAEA;;EAEA,aAAa;;;EAGb,KAAK;EACL;;UAGe;EACf;EACA;EACA;EACA;EACA;;;UAQe,iBAAiB,kBAAkB,oBAAoB;;EAEtE;EACA,SAAS;EACT,WAAW;EACX;EACA,SAAS;EACT;;;;;EAKA,YAAY,OAAO,cAAc,QAAQ,qBAAqB;;EAE9D,SAAS,MAAM,qBAAqB,WAAW,UAAU,YAAY,QAAQ;;;;EAI7E,YAAY,iBAAiB;;UAGd,kBAAkB,4BAA4B;EAC7D;EACA;EACA,QAAQ;EACR,UAAU;EACV,UAAU;EACV,MAAM,qBAAqB;;KAGjB,mBAAmB,oBAAoB,uBAAuB;EACxE,UAAU,kBAAkB;EAC5B,MAAM,qBAAqB;EAC3B,SAAS;MACL,QAAQ,aAAa;;;cAoBd;;;;;;;;;iBAUG,mBACd,gBACA,gBACC"}
1
+ {"version":3,"file":"types-CCZ34qmV.d.ts","names":[],"sources":["../src/benchmarks/types.ts"],"mappings":";;;KASY;KAQA;UAgBK,qBAAqB;;;EAGpC;;EAEA,SAAS;;EAET,QAAQ;;EAER,SAAS;;EAET,WAAW;;EAEX;;EAEA,SAAS;EACT,WAAW;;UAGI;;;;EAIf;;EAEA;;EAEA,aAAa;;;EAGb,KAAK;EACL;;UAGe;EACf;EACA;EACA;EACA;EACA;;;UAQe,iBAAiB,kBAAkB,oBAAoB;;EAEtE;EACA,SAAS;EACT,WAAW;EACX;EACA,SAAS;EACT;;;;;EAKA,YAAY,OAAO,cAAc,QAAQ,qBAAqB;;EAE9D,SAAS,MAAM,qBAAqB,WAAW,UAAU,YAAY,QAAQ;;;;EAI7E,YAAY,iBAAiB;;UAGd,kBAAkB,4BAA4B;EAC7D;EACA;EACA,QAAQ;EACR,UAAU;EACV,UAAU;EACV,MAAM,qBAAqB;;KAGjB,mBAAmB,oBAAoB,uBAAuB;EACxE,UAAU,kBAAkB;EAC5B,MAAM,qBAAqB;EAC3B,SAAS;MACL,QAAQ,aAAa;;;cAoBd;;;;;;;;;iBAUG,mBACd,gBACA,gBACC"}
@@ -1,91 +1,5 @@
1
+ import { t as SeriesDistribution } from "./descriptive-B2iPaT9J.js";
1
2
  import { m as RolloutLine } from "./schema-BzWDXhOR.js";
2
- //#region src/statistics/descriptive.d.ts
3
- /**
4
- * Descriptive statistics: means, bootstrap spread, correlation, and the
5
- * weighted judge-dimension composite. Nothing here is a significance test.
6
- */
7
- /** Weighted mean — falls back to uniform weights when omitted */
8
- declare function weightedMean(scores: {
9
- score: number;
10
- weight?: number;
11
- }[]): number;
12
- /**
13
- * Percentile bootstrap confidence interval on the mean of `scores`.
14
- *
15
- * Descriptive spread. It is not a significance test, and at small n its bounds
16
- * are anti-conservative in the same way {@link pairedBootstrap}'s are — see
17
- * {@link BOOTSTRAP_GATE_MIN_N}. With no `seed` the resampling is seeded from
18
- * the scores themselves, so the interval is reproducible either way.
19
- */
20
- declare function confidenceInterval(scores: number[], confidence?: number, opts?: {
21
- seed?: number;
22
- resamples?: number;
23
- }): {
24
- mean: number;
25
- lower: number;
26
- upper: number;
27
- };
28
- /** Partial credit: returns 0-1 ratio of current toward target */
29
- declare function partialCredit(current: number, target: number): number;
30
- /** Distribution summary of a number series: count, extremes, quantiles, sum. */
31
- interface SeriesDistribution {
32
- readonly n: number;
33
- readonly min: number;
34
- readonly p50: number;
35
- readonly p90: number;
36
- readonly max: number;
37
- readonly sum: number;
38
- }
39
- /**
40
- * Fold a number series into its distribution summary. Quantiles use the
41
- * nearest-rank definition — the `ceil(q·n)`-th order statistic — so every
42
- * reported quantile is a value from the series. Returns `null` for an empty
43
- * series: an empty series has no distribution, and a zero-filled summary
44
- * would read as a measured all-zero series.
45
- */
46
- declare function summarizeNumberSeries(values: readonly number[]): SeriesDistribution | null;
47
- /**
48
- * Average-rank-with-ties transform (1-indexed). Tied values receive the mean
49
- * of the ranks they span, the standard correction for Spearman's ρ.
50
- */
51
- declare function ranks(xs: number[]): number[];
52
- /**
53
- * Pearson product-moment correlation coefficient r ∈ [-1, 1] between two
54
- * equal-length series. See the edge-case contract above: NaN for n < 2 or
55
- * unequal lengths, 1 when both series are constant, 0 when exactly one is.
56
- */
57
- declare function pearsonR(a: number[], b: number[]): number;
58
- /**
59
- * Spearman's rank correlation ρ — Pearson over the average-rank-with-ties
60
- * transform of each series. Same edge-case contract as {@link pearsonR}.
61
- */
62
- declare function spearmanR(a: number[], b: number[]): number;
63
- interface WeightedCompositeInput {
64
- /** Per-dimension scores (typically 0..1). */
65
- dims: Record<string, number>;
66
- /** Weight per dimension. Every weighted dimension MUST be present in
67
- * `dims` — a weight for an absent dimension is a config error and throws,
68
- * because silently dropping it would renormalise the composite onto a
69
- * different denominator than intended. */
70
- weights: Record<string, number>;
71
- /** Optional pass threshold; when set, the result reports `pass`. */
72
- threshold?: number;
73
- }
74
- interface WeightedCompositeResult {
75
- composite: number;
76
- pass?: boolean;
77
- }
78
- /**
79
- * Weighted composite over judge dimensions: `Σ(score_d · w_d) / Σ(w_d)` across
80
- * the weighted dimensions. The canonical replacement for the per-consumer
81
- * hand-rolled composite math (tax/legal/creative/gtm each ship a copy).
82
- *
83
- * Fail-loud: throws if a weighted dimension is missing from `dims`, if any
84
- * weight is negative, or if the weights sum to 0 — none of which can produce
85
- * a meaningful composite.
86
- */
87
- declare function weightedComposite(input: WeightedCompositeInput): WeightedCompositeResult;
88
- //#endregion
89
3
  //#region src/supervisor-run/types.d.ts
90
4
  /** A metric that could not be computed, with the reason its artifact was missing. */
91
5
  interface Unavailable {
@@ -300,7 +214,12 @@ interface OrchestrationMetrics {
300
214
  interface DecisionMetrics {
301
215
  readonly settledByStatus: Measured<Record<string, number>>;
302
216
  readonly settledVerdicts: Measured<Record<string, number>>;
303
- /** Worker verified its own work green AND produced a patch. */
217
+ /**
218
+ * Workers whose recorded verdict was green. A store that retains a delivered patch also
219
+ * requires patch bytes; a store that retains none accepts the verdict alone, because the
220
+ * verdict IS the acceptance decision that store recorded. `emptyPass` — the split that
221
+ * needs patch bytes — is what reads unavailable there.
222
+ */
304
223
  readonly accepted: Measured<number>;
305
224
  /** Worker settled with a failing verify. */
306
225
  readonly rejected: Measured<number>;
@@ -364,6 +283,17 @@ interface SpendMeasurement {
364
283
  readonly usd: Measured<number>;
365
284
  /** Source records folded into `usd`; 0 when the measurement is unavailable. */
366
285
  readonly records: number;
286
+ /**
287
+ * Records the store wrote with `usdKnown: false` — work that HAPPENED at a price the
288
+ * provider never reported. Those records are NOT folded into `usd`, so a run with any
289
+ * of them has a `usd` that is a floor on real spend, never the measured total. Dropping
290
+ * the whole channel instead would discard the records that DID carry a price.
291
+ */
292
+ readonly unknownRecords: number;
293
+ /** True exactly when `unknownRecords > 0`: `usd` covers some of the run, not all of it. */
294
+ readonly partial: boolean;
295
+ /** Node ids behind `unknownRecords`, in journal order. Empty when none. */
296
+ readonly unknownNodes: readonly string[];
367
297
  }
368
298
  /**
369
299
  * The run's total inference spend, measured two ways.
@@ -399,8 +329,9 @@ interface EconomicsMetrics {
399
329
  /**
400
330
  * One collapsed number kept for existing consumers: the close record when
401
331
  * the store wrote one, else the journal-derived sum. `totalUsdSource` names
402
- * the pick. Prefer `spend` the collapse hides which accounting question
403
- * the number answers.
332
+ * the pick, and says so when the number is a partial floor because some
333
+ * records carried `usdKnown: false`. Prefer `spend` — the collapse hides
334
+ * which accounting question the number answers and how much of it is priced.
404
335
  */
405
336
  readonly totalUsd: Measured<number>;
406
337
  /**
@@ -518,5 +449,5 @@ interface SupervisorRunTreeGap {
518
449
  readonly count?: number;
519
450
  }
520
451
  //#endregion
521
- export { unavailable as A, weightedComposite as B, SupervisorRunTreeGap as C, WorkerLogSource as D, WallDistribution as E, partialCredit as F, pearsonR as I, ranks as L, WeightedCompositeInput as M, WeightedCompositeResult as N, isUnavailable as O, confidenceInterval as P, spearmanR as R, SupervisorRunTree as S, Unavailable as T, weightedMean as V, SupervisorRunNodeRole as _, OrchestrationMetrics as a, SupervisorRunRollup as b, PerWorkerRow as c, SUPERVISOR_RUN_ROLLUP_SCHEMA as d, SUPERVISOR_RUN_SCHEMA as f, SteerBreakdown as g, SpendMeasurements as h, NO_SOURCE_LIMITS as i, SeriesDistribution as j, showMeasured as k, RoleSpend as l, SpendMeasurement as m, EconomicsMetrics as n, OutcomeMetrics as o, SourceLimits as p, Measured as r, PatchStats as s, DecisionMetrics as t, RollupCellRow as u, SupervisorRunReader as v, SupervisorRunTreeGapCode as w, SupervisorRunSources as x, SupervisorRunReport as y, summarizeNumberSeries as z };
522
- //# sourceMappingURL=types-DeIUdzNd.d.ts.map
452
+ export { unavailable as A, SupervisorRunTreeGap as C, WorkerLogSource as D, WallDistribution as E, isUnavailable as O, SupervisorRunTree as S, Unavailable as T, SupervisorRunNodeRole as _, OrchestrationMetrics as a, SupervisorRunRollup as b, PerWorkerRow as c, SUPERVISOR_RUN_ROLLUP_SCHEMA as d, SUPERVISOR_RUN_SCHEMA as f, SteerBreakdown as g, SpendMeasurements as h, NO_SOURCE_LIMITS as i, showMeasured as k, RoleSpend as l, SpendMeasurement as m, EconomicsMetrics as n, OutcomeMetrics as o, SourceLimits as p, Measured as r, PatchStats as s, DecisionMetrics as t, RollupCellRow as u, SupervisorRunReader as v, SupervisorRunTreeGapCode as w, SupervisorRunSources as x, SupervisorRunReport as y };
453
+ //# sourceMappingURL=types-CoPUTiXb.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"types-CoPUTiXb.d.ts","names":[],"sources":["../src/supervisor-run/types.ts"],"mappings":";;;;UAkCiB;WACN;;;KAIC,SAAS,KAAK,IAAI;iBAEd,YAAY,iBAAiB;iBAI7B,cAAc,aAAa,KAAK;;iBAKhC,aAAa,GAAG;;KAWpB;;;;;UAMK;;;;;WAKN;;WAEA;;WAEA;;WAEA;;WAEA;;WAEA;;WAEA;;WAEA;WACA;WACA;WACA;;;;;;;;;;;;;UAcM;;WAEN;;WAEA;;WAEA;;WAEA;;WAEA;;;cAIE,kBAAkB;;;;;;;;;;UAiBd;;WAEN;WACA;;WAEA;;WAEA;;;;;;WAMA;;;;;;WAMA;;WAEA;;WAEA;;WAEA;;WAEA;;WAEA,kBAAkB;;WAElB;;WAEA;;;;;;;WAOA;;WAEA;;WAEA;;WAEA;;;;;WAKA;IACP;IACA;IACA;IACA;;IAEA;IACA;;WAEO;;WAEA,QAAQ;;;;;;WAMR;;;;;WAKA;;;;;;UAOM;;WAEN;EACT,QAAQ,QAAQ;;cAOL;cACA;UAEI;;WAEN;WACA;;WAEA;;WAEA;;UAGM;WACN,gBAAgB;WAChB,gBAAgB;WAChB,kBAAkB;;WAElB,QAAQ;WACR,iBAAiB;WACjB,gBAAgB,kBAAkB;;WAElC,kBAAkB;;;;;;WAMlB,OAAO;WACP,WAAW;WACX,gBAAgB;;WAEhB,UAAU;;WAEV,gBAAgB;;WAEhB,iBAAiB;WACjB,oBAAoB;WACpB,kBAAkB;;;;;;;;;WASlB,sBAAsB;;WAEtB,QAAQ;WACR,SAAS;;WAET,mBAAmB;;UAGb;WACN,iBAAiB,SAAS;WAC1B,iBAAiB,SAAS;;;;;;;WAO1B,UAAU;;WAEV,UAAU;;WAEV,WAAW;;WAEX,oBAAoB;;WAEpB,wBAAwB;;WAExB,eAAe;WACf,qBAAqB;;UAGf;WACN,UAAU;WACV,WAAW;;;;;;WAMX,WAAW;WACX,YAAY;WACZ,KAAK;WACL;;UAGM;;WAEN;WACA;;WAEA,MAAM;;WAEN;;WAEA;;WAEA;;WAEA;;WAEA;WACA;;WAEA;WACA;WACA;WACA;WACA;;WAEA;;;;;;;KAQC,mBAAmB;;UAGd;WACN,KAAK;;WAEL;;;;;;;WAOA;;WAEA;;WAEA;;;;;;;;;;;;;;UAeM;WACN,gBAAgB;WAChB,aAAa;;UAGP;;WAEN,OAAO;;;;;;;;WAQP,kBAAkB;;WAElB,SAAS;;WAET,OAAO;;;;;;;;WAQP,UAAU;;;;;;WAMV;WACA,yBAAyB;WACzB,0BAA0B,SAAS;WACnC,WAAW,kBAAkB;;UAGvB;WACN;WACA;WACA;WACA;;UAGM;WACN,WAAW;WACX,YAAY;WACZ,WAAW;WACX,eAAe;WACf,YAAY;WACZ,aAAa;WACb,YAAY;WACZ,YAAY;WACZ,UAAU;WACV,OAAO,SAAS;;WAEhB;;UAGM;WACN,eAAe;;WAEf;WACA;WACA;WACA,cAAc;WACd,yBAAyB;WACzB;WACA,eAAe;WACf,UAAU;WACV,WAAW;WACX,SAAS;;WAET;;WAEA;;UAGM;WACN;WACA;WACA,QAAQ;WACR,OAAO;WACP,aAAa;WACb,SAAS;WACT,UAAU;WACV,KAAK;;UAGC;WACN,eAAe;WACf;WACA,aAAa;WACb,iBAAiB;WACjB;WACA,WAAW;WACX,mBAAmB;WACnB,iBAAiB;WACjB,aAAa;WACb,qBAAqB;WACrB,eAAe;;;;;WAKf,UAAU;;;;;;;WAOV;aACE;eAA2B,OAAO;eAA2B;;aAC7D;eAAwB,OAAO;eAA2B;;;WAE5D,eAAe;WACf,kBAAkB;;;;;;;;UASZ;WACN;WACA,gBAAgB;;WAEhB,eAAe;;;KAId;UASK;WACN,MAAM;WACN;WACA;WACA"}
@@ -2,6 +2,7 @@ import { S as PaidCallResult, T as RunPaidCallInput, a as CostChannel, c as Cost
2
2
  import { u as RunTokenUsage } from "./run-record-DQjRcYwA.js";
3
3
  import { _ as ProposalFinding } from "./types-DMoNFDWi.js";
4
4
  import { C as LlmCallMetadata } from "./types-Bfk0uxRj.js";
5
+ import { t as SeriesDistribution } from "./descriptive-B2iPaT9J.js";
5
6
  //#region src/campaign/types.d.ts
6
7
  /** Stable identifier + kind tag for any scenario. Consumers
7
8
  * extend with their per-domain payload (persona, task, requirement, ...). */
@@ -542,11 +543,20 @@ interface JudgeAggregate {
542
543
  stdev: number;
543
544
  ci95: [number, number];
544
545
  n: number;
546
+ /** Order statistics over the exact scores `mean` was taken over, from the
547
+ * package's one distribution summary (`summarizeNumberSeries`). A mean and
548
+ * a CI alone cannot separate a bimodal judge from a tight one, and cannot
549
+ * show the outlier that carried the mean. Never null here: an aggregate is
550
+ * only recorded for a judge that produced at least one score. */
551
+ distribution: SeriesDistribution;
545
552
  }
546
553
  interface ScenarioAggregate {
547
554
  meanComposite: number;
548
555
  ci95: [number, number];
549
556
  n: number;
557
+ /** Order statistics over the per-cell composites this scenario produced.
558
+ * Same type and same contract as {@link JudgeAggregate.distribution}. */
559
+ distribution: SeriesDistribution;
550
560
  }
551
561
  interface GenerationRecord {
552
562
  generationIndex: number;
@@ -660,4 +670,4 @@ interface CampaignResult<TArtifact = unknown, TScenario extends Scenario = Scena
660
670
  }
661
671
  //#endregion
662
672
  export { LabeledScenarioWrite as A, ScoredSurfaceOutcome as B, JudgeDimension as C, LabeledScenarioSampleArgs as D, LabeledScenarioRecord as E, ProposeContext as F, labelTrustRank as G, SurfaceProposer as H, ProposedCandidate as I, RedactionStatus as L, OptimizerConfig as M, ParetoParent as N, LabeledScenarioSource as O, ProposalTrackContext as P, Scenario as R, JudgeConfig as S, LabelTrust as T, TraceSpan as U, SessionScript as V, isProposedCandidate as W, GateDecision as _, CampaignResult as a, GenerationRecord as b, CampaignTraceWriter as c, DispatchContext as d, DispatchFn as f, GateContribution as g, GateContext as h, CampaignCostMeter as i, MutableSurface as j, LabeledScenarioStore as k, CodeSurface as l, GateCheckStatus as m, CampaignArtifactWriter as n, CampaignScenarioIdentity as o, Gate as p, CampaignCellResult as r, CampaignTokenUsage as s, CampaignAggregates as t, ComponentSurface as u, GateResult as v, JudgeScore as w, JudgeAggregate as x, GenerationCandidate as y, ScenarioAggregate as z };
663
- //# sourceMappingURL=types-JHMOqZI4.d.ts.map
673
+ //# sourceMappingURL=types-nokrtr7M.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"types-nokrtr7M.d.ts","names":[],"sources":["../src/campaign/types.ts"],"mappings":";;;;;;;;UAiCiB;EACf;EACA;EACA;;;;;EAKA;;;UAIe,iCAAiC,KAAK;EACrD;;;;;UAMe;EACf;;EAEA;EACA;EACA;EACA;EACA,QAAQ;EACR,OAAO;EACP,WAAW;EACX,MAAM;;EAEN;;EAEA;;;;;;;;EAQA;;;;KAKU,WAAW,kBAAkB,UAAU,cACjD,UAAU,WACV,KAAK,oBACF,QAAQ;;;;UAOI,cAAc,WAAW;EACxC;EACA;EACA;;EAEA;;;EAGA,sBAAsB,UAAU,WAAW,sBAAsB,UAAU,cAAc;;UAK1E;;EAEf;;EAEA;;;;;;;;;UAUe,YAAY,WAAW,kBAAkB,WAAW;EACnE;EACA,YAAY;;;;EAIZ;;;EAGA,MAAM;IACJ,UAAU;IACV,UAAU;IACV,QAAQ;;IAER,aAAa;IACb;IACA,WAAW;MACT,aAAa,QAAQ;EACzB,aAAa,UAAU;;;;;;;;;;;UAYR;EACf,YAAY;EACZ;EACA;;EAEA,UAAU;;;;;;;EAOV;;;;EAIA,eAAe,eAAe;IAAQ;IAAe;;;;;EAIrD;;;EAGA;;EAEA;;EAEA,WAAW,eAAe;;;;;;;UAUX;WACN;;;WAGA;;WAEA;;WAEA;;WAEA;;WAEA;;WAEA;;;WAGA;aACE;aACA;aACA;;;WAGF;;;UAIM;WACN;WACA,YAAY,SAAS;;;;;;;;;;;KAYpB,0BAA0B,mBAAmB;;;;;;;UAQxC;EACf,SAAS;;EAET;;;;EAIA;;;;;;EAMA,cAAc,SAAS;;;;iBAKT,oBACd,OAAO,iBAAiB,oBACvB,SAAS;;;;;;;;;UAkBK;EACf,SAAS;EACT;;;EAGA,YAAY;;;EAGZ;;EAEA;EACA;EACA;;;UAIe;EACf;EACA;EACA;EACA;EACA;;;;;UAMe;;;EAGf;;EAEA;EACA;EACA;EACA,YAAY;;;;;EAKZ,WAAW;IAAQ;IAAoB;IAAmB;IAAgB;;EAC1E;IACE;IACA;;;;;UAMa,eAAe,YAAY;;;WAGjC,gBAAgB;WAChB,SAAS,cAAc;WACvB,UAAU,cAAc;;WAExB;WACA;WACA,QAAQ;;WAER,QAAQ;;;WAGR,kBAAkB;;;WAGlB,mBAAmB;;;;WAInB,gBAAgB;;;;WAIhB;;;;;;;WAOA,gBAAgB,cAAc;;WAE9B,aAAa;WACb;;;;;;;;;;;;;;UAeM,gBAAgB,YAAY;EAC3C;;;;;EAKA,QAAQ,KAAK,eAAe,aAAa,QAAQ,MAAM,iBAAiB;;;EAGxE,QAAQ;IAAQ,SAAS,cAAc;;IAAwB;IAAe;;;UAG/D;EACf;EACA;EACA,mBAAmB,qBAAqB;;UAGzB,wBAAwB;EACvC,UAAU;;;KAMA;;KAGA;UAEK;EACf;EACA,QAAQ;EACR;;UAGe,YAAY,WAAW,kBAAkB;EACxD,oBAAoB,YAAY;EAChC,oBAAoB,YAAY;;EAEhC,aAAa,YAAY,eAAe;;;;;EAKxC,sBAAsB,YAAY,eAAe;;;;;;;EAOjD,yBAAyB,YAAY,eAAe;;;EAGpD,uBAAuB,YAAY;EACnC,WAAW;EACX;IAAQ;IAAmB;;;EAE3B,aAAa;EACb;EACA,QAAQ;;UAGO;EACf,UAAU;EACV;EACA,mBAAmB;EACnB;;;UAIe,KAAK,qBAAqB,kBAAkB,WAAW;EACtE;EACA,OAAO,KAAK,YAAY,WAAW,aAAa,QAAQ;;;;UAOzC;EACf,KAAK,cAAc,aAAa,0BAA0B;EAC1D,SAAS;;UAGM;EACf,IAAI,aAAa;EACjB,aAAa,aAAa;;;;UAKX;EACf,MAAM,cAAc,kBAAkB,aAAa;EACnD,UAAU,cAAc,iBAAiB;;;;;;KAO/B,qBAAqB;;;;;UAMhB;;EAEf,YAAY,GACV,OAAO,KAAK,iBAAiB;IAC3B,UAAU;MAEX,QAAQ,eAAe;;;;;KAQhB;KAOA;;;;;;;;;;;;;;;KAgBA;;iBASI,eAAe,OAAO;;;;UAOrB,qBAAqB,kBAAkB,WAAW,UAAU;EAC3E,UAAU;EACV,UAAU;EACV,aAAa,eAAe;EAC5B,QAAQ;EACR;EACA;EACA,iBAAiB;;;;;EAKjB,aAAa;;EAEb;;UAGe,sBAAsB,kBAAkB,WAAW,UAAU,6BACpE,qBAAqB,WAAW;;EAExC;;;EAGA;;UAGe;EACf;;EAEA;;;;EAIA;EACA;IACE;IACA,SAAS,wBAAwB;IACjC;IACA;;;;;IAKA,WAAW;;;UAIE;EACf,QAAQ,OAAO,uBAAuB;EACtC,OAAO,MAAM,4BAA4B,QAAQ;EACjD,QAAQ;IACN;IACA;IACA,UAAU;;;IAGV,SAAS,OAAO;;;UAMH,mBAAmB;;;EAGlC;EACA;EACA;EACA;EACA;EACA,UAAU;EACV,aAAa,eAAe;;;EAG5B;;EAEA,gBAAgB;;EAEhB;;;EAGA,YAAY;;;EAGZ;;;EAGA;EACA;EACA;EACA;;;;EAIA;;EAEA;;EAEA;EACA;;UAGe;EACf;EACA;EACA;EACA;;;;;;EAMA,cAAc;;UAGC;EACf;EACA;EACA;;;EAGA,cAAc;;UAGC;EACf;EACA,YAAY;EACZ;;;;;;UAOe;EACf;;EAEA;;EAEA;;EAEA;;EAEA;;;EAGA;;;;EAIA;;;;EAIA;IACE;IACA;IACA,iBAAiB;MAAQ;MAAgB;;;;;EAI3C,YAAY;;;;;;;;;;EAUZ,WAAW;IAAQ;IAAoB;IAAmB;IAAgB;;;;EAG1E;;;;EAIA;;;;EAIA,cAAc,SAAS;;UAGR;EACf,SAAS,eAAe;EACxB,YAAY,eAAe;;EAE3B,MAAM;;EAEN;EACA;EACA;;EAEA;;EAEA;;EAEA;;EAEA;;UAGe,eAAe,qBAAqB,kBAAkB,WAAW;;EAEhF;;EAEA;EACA;;EAEA;EACA;EACA;EACA;EACA,OAAO,MAAM,mBAAmB;EAChC,YAAY;EACZ;IACE,aAAa;IACb;;EAEF,OAAO;EACP;EACA;EACA,iBAAiB;;;EAGjB,WAAW,MAAM,2BAA2B,KAAK"}
@@ -43,6 +43,42 @@ is `runCampaign` with capture inverted; `runAgentMatrix` is the scheduler undern
43
43
  Merging any two of these conflates distinct mental models (measure ≠ search ≠
44
44
  release-gate). Keep them separate; pick by the table.
45
45
 
46
+ ## What a campaign result reports: the mean and the spread
47
+
48
+ `CampaignResult.aggregates` carries two maps.
49
+ `byJudge` holds one `JudgeAggregate` per judge that produced at least one score.
50
+ `byScenario` holds one `ScenarioAggregate` per scenario that produced at least one composite.
51
+
52
+ Each aggregate reports a mean, a seeded bootstrap `ci95` band, `n`, and a `distribution`.
53
+ `distribution` is the `SeriesDistribution` value `summarizeNumberSeries` returns: `n`, `min`, `p50`, `p90`, `max`, and `sum` over the exact scores the mean was taken over.
54
+ Quantiles use the nearest-rank definition, so every reported quantile is a score the campaign measured.
55
+
56
+ Read the distribution before you read the mean.
57
+ A mean and an interval alone cannot separate a bimodal judge from a tight one, and cannot show the outlier that carried the mean.
58
+ Six cells scoring `0, 0, 0, 1, 1, 1` and six cells scoring `0.5` report the same mean; only `min` and `max` tell them apart.
59
+
60
+ A judge that produced no score has no entry at all.
61
+ An absent aggregate is the honest record of an unmeasured judge, and a zero-filled distribution would read as a measured all-zero series.
62
+
63
+ `SeriesDistribution` is the one distribution summary in this package.
64
+ It is not the `ScalarDistribution` the insight report uses; see `insight-report.md` for why those two shapes stay separate.
65
+
66
+ ## Planning the cell grid without a run directory
67
+
68
+ `buildCellSchedule(scenarios, seed, reps)` returns the `(scenario × rep)` fan-out: one `CellScheduleSlot` per cell, with its `cellId` and its per-cell seed.
69
+ It touches no filesystem, so a caller can size a design, or assert a design's cell count and seeds in a test, before a run directory exists.
70
+ Scenarios that share a `seedGroup` receive the same per-replicate seeds, which is what makes a paired comparison see common randomness.
71
+
72
+ Use `planCampaignRun` instead when you also need the cached, to-run, and blocked classification; that call needs a real run directory because it reads the durable cache.
73
+ `cellDirectory` and `cellCachePath` name a cell's location once a run directory is chosen.
74
+
75
+ ## Evidence receipts: `attest`
76
+
77
+ `attest(report, provenance)` content-addresses any serializable report and binds that address to the provenance needed to reproduce it: model versions, seeds, price-table hash, code SHA, and inputs hash.
78
+ `verifyAttestation(report, attested)` returns a typed outcome rather than throwing, so a pipeline records why a report failed to verify instead of dying.
79
+ `ATTESTATION_ALGORITHM` is the hash-scheme tag every attestation carries, and a verifier rejects an unknown algorithm instead of guessing.
80
+ Signing stays with the consumer: an `AttestedReport` is a stable byte-identical payload to sign, and this package never holds keys.
81
+
46
82
  ## Failed cells: receipts and bounded retry
47
83
 
48
84
  A failed cell writes `<cell>/failure-receipt.json` before the campaign can abort.
@@ -148,6 +148,25 @@ When a measured mean is below 0.5, inspect the lowest-scoring runs before tuning
148
148
 
149
149
  **Use the histogram for:** finding bimodal failure modes. A bin with `count > 0` near zero and another > 0 near 1 means your agent has two distinct behaviors, not one noisy one.
150
150
 
151
+ ### Why this is not the same shape as a campaign aggregate
152
+
153
+ This report uses `ScalarDistribution`.
154
+ A campaign aggregate (`CampaignResult.aggregates`) uses `SeriesDistribution`, the value `summarizeNumberSeries` returns.
155
+ The two shapes stay separate for three reasons, and none of them is an accident of history.
156
+
157
+ 1. `ScalarDistribution` is a wire contract.
158
+ `ScalarDistributionSchema` in `src/hosted/schemas.ts:45` is a strict Zod object.
159
+ It is embedded in `InsightReportSchema`, which is embedded in `EvalRunEventSchema`, which the hosted client validates every event against before it ships (`src/hosted/client.ts:182`).
160
+ A strict object rejects an unknown key, so adding or renaming a field breaks every event a consumer already sends.
161
+ 2. `ScalarDistribution` reports what a report needs and a series summary does not have: a histogram, the worst-N `tailRuns` by score, `p95` for a latency question, and `mean` plus `stddev` beside the order statistics.
162
+ `SeriesDistribution` is the in-memory summary of a plain number series with no run identity attached.
163
+ 3. The two answer at different `n = 0` boundaries.
164
+ `ScalarDistribution` represents an empty series as `n: 0` with every field `null`, because a report always has a slot for a metric it did not measure.
165
+ `summarizeNumberSeries` returns `null` for an empty series, because there is no distribution to report and a zero-filled summary would read as a measured all-zero series.
166
+
167
+ Both refuse to encode a missing measurement as a zero.
168
+ That is the shared rule; the shapes differ because the surfaces differ.
169
+
151
170
  ---
152
171
 
153
172
  ## `costQuality`: cost-vs-quality Pareto
package/docs/plants.md ADDED
@@ -0,0 +1,123 @@
1
+ # Plants: does the grader catch a wrong answer?
2
+
3
+ A grading run reports how the work scored.
4
+ It cannot report whether the grader would have noticed a wrong answer, because every item it saw was authored in good faith.
5
+ A plant closes that hole.
6
+ A plant is an item authored wrong by construction, mixed into the live grading set, graded by the same path as everything else.
7
+ The share of plants the grader refused is the **catch rate**.
8
+
9
+ The measured motive: a sibling lab ran a deliverable gate that accepted any non-empty submission.
10
+ It produced six false certifications in seventeen deliveries.
11
+ No agent lied — the gate never asked a question the format could fail.
12
+ A catch rate is the number that would have shown it on the first day.
13
+
14
+ Everything below lives in `src/meta-eval/plants.ts` and is published from `@tangle-network/agent-eval/meta-eval`.
15
+
16
+ ## Three calls
17
+
18
+ ```ts
19
+ import { catchRate, definePlant, seedPlants } from '@tangle-network/agent-eval/meta-eval'
20
+
21
+ const plants = [
22
+ definePlant({
23
+ id: 'plant-throughput',
24
+ kind: 'wrong-value',
25
+ item: { itemId: 'claim-118', humanScore: 0 },
26
+ expectedVerdict: 'reject',
27
+ }),
28
+ ]
29
+
30
+ const { items, manifest } = seedPlants(dataset, plants, { seed: 7, acceptThreshold: 0.5 })
31
+
32
+ // Publish manifest.seal now. Keep `manifest` itself out of the graded workspace.
33
+ const results = await grade(items.map((item) => item.itemId))
34
+
35
+ const report = catchRate(results, manifest)
36
+ ```
37
+
38
+ `report` is a `CatchRateReport`:
39
+
40
+ | Field | Meaning |
41
+ | --- | --- |
42
+ | `status` | `evaluated`, `incomplete`, or `not_evaluated`. Only `evaluated` carries a rate. |
43
+ | `reason` | Why the status is not `evaluated`. Absent when it is. |
44
+ | `seeded` / `caught` / `missed` | Plants armed, graded as the seed demands, and graded against it. |
45
+ | `indecisive` | The grader answered and declined to decide. Counted apart, in neither side of `rate`. |
46
+ | `rate` | `caught / (caught + missed)`, or `null`. |
47
+ | `byKind` | The same counts per plant kind. A kind nobody seeded is absent, never zero. |
48
+ | `missedIds` | Every plant id the grader got wrong, by name. |
49
+ | `missingIds` | Every plant id with no result at all. |
50
+ | `unseeded` | How the same grader treated the rest of the set. |
51
+
52
+ ## What each call composes
53
+
54
+ Plants add no parallel calibration machinery.
55
+ Each piece is an existing primitive doing its own job.
56
+
57
+ | Piece | Primitive it reuses |
58
+ | --- | --- |
59
+ | The seeded item | `GoldenItem` (`src/judge-calibration.ts`). Its `humanScore` is the grade a working grader owes the item, so the mixed set feeds `calibrateJudge` unchanged. |
60
+ | The grader's output | `CandidateScore[]`, the array `calibrateJudge` and `snapshotFromSentinelSet` already read. A `PlantOutcome` is that shape with `score: null` added for "ran, declined to decide". |
61
+ | "Caught" | The join in `snapshotFromSentinelSet` (`src/meta-eval/sentinel.ts`) with the labels inverted: the grade lands on the side of `acceptThreshold` the seed demands. |
62
+ | The seal | `hashCanonical` (`src/ledger-core/canonical.ts`), the RFC 8785 digest the sealed-experiment path uses. |
63
+ | The mix order | `mulberry32` (`src/statistics/random.ts`), the package's one seedable generator. |
64
+
65
+ `src/canary.ts` is deliberately not this.
66
+ Its three detectors — silent judge fallback, calibration drift, distribution shift — all watch the judge's own statistics.
67
+ None injects an item the judge can fail.
68
+
69
+ ## The four kinds
70
+
71
+ `kind` records how the item was authored wrong.
72
+ It changes nothing about how the item is graded; it exists so `byKind` can show which defect a grader is blind to.
73
+
74
+ | Kind | Authored by |
75
+ | --- | --- |
76
+ | `wrong-value` | Altering one load-bearing value: a number off by one, a comparison flipped. |
77
+ | `self-certifying` | Giving the item a check that passes without testing the claim. |
78
+ | `unreachable-input` | Pointing the check at an input that does not exist, so it cannot run at all. |
79
+ | `duplicate` | Copying an item already in the set, which is owed a duplicate flag rather than a second grade. |
80
+
81
+ ## Blindness has two halves
82
+
83
+ This module owns one half.
84
+ It never puts a plant flag on a graded item: `seedPlants` returns the mixed set and a manifest, and only the manifest knows which ids are seeded.
85
+ Every item in `items` carries the same two fields, so nothing a grader receives says which are plants.
86
+
87
+ The caller owns the other half, and it is the half that decides whether the measurement means anything.
88
+
89
+ - Keep the manifest out of the workspace the graded agents can read.
90
+ A truth label inside that tree is readable by the thing being measured.
91
+ - Publish `manifest.seal` **before** the grading run.
92
+ A manifest edited afterwards to match the results no longer hashes to its seal, and `catchRate` refuses it with a `CaptureIntegrityError`.
93
+ - Write the plant tag into a record only after the verdict is decided, so the tag cannot reach the verdict path.
94
+
95
+ ## What refuses, and why
96
+
97
+ A catch rate that cannot refuse is not a measurement.
98
+
99
+ | Condition | Result |
100
+ | --- | --- |
101
+ | A seeded plant has no result | `status: 'incomplete'`, `rate: null`, `missingIds` names them. A rate over the results that happened to come back is the number a broken run reports as success. |
102
+ | Zero plants were seeded | `status: 'not_evaluated'`, `rate: null`. Never 1.0: a grader asked nothing it could fail did not score perfectly. |
103
+ | Every seeded plant is indecisive | `status: 'not_evaluated'`, `rate: null`. No check tested the seeded defect, so nothing was measured. |
104
+ | The manifest no longer matches its seal | `CaptureIntegrityError`. |
105
+ | A result names an id the manifest never handed out, or repeats one | `ValidationError`. The results and the manifest describe different runs, and a rate across two runs is a fabrication. |
106
+ | A plant record's `expectedVerdict` disagrees with its `humanScore` | `ValidationError` at `definePlant`. The same item would read as wrong here and as correct to every instrument that joins on the id. |
107
+
108
+ ## Reading the number
109
+
110
+ `rate` alone can be gamed by a grader that refuses everything: it catches every reject-plant and scores 1.0.
111
+ `unseeded` is where that shows.
112
+ It reports how the same grader treated the items that were not seeded — `n`, `decided`, `rejected`, and `rejectionRate` — using no labels at all.
113
+
114
+ - Catch rate 1.0 with `unseeded.rejectionRate` near 1.0 is a refusal reflex, not discrimination.
115
+ - Catch rate 1.0 with `unseeded.rejectionRate` near 0 is a grader that separates the two classes.
116
+
117
+ The stronger form is to seed both directions.
118
+ A plant with `expectedVerdict: 'accept'` is a known-good item authored the same way, and a grader that refuses it is scored as a miss exactly like one that certifies a known-wrong item.
119
+
120
+ ## Related
121
+
122
+ - [`docs/verdicts.md`](./verdicts.md) — where every verifier in this package lands its result.
123
+ - [`docs/verification-strategies.md`](./verification-strategies.md) — certifying work that has no answer key at all.