@tangle-network/agent-eval 0.173.2 → 0.174.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (128) hide show
  1. package/CHANGELOG.md +40 -0
  2. package/dist/{proposal-findings-bko3GGy-.js → abort-signal-CtzAM_sJ.js} +11 -11
  3. package/dist/abort-signal-CtzAM_sJ.js.map +1 -0
  4. package/dist/adapters/http.d.ts +2 -2
  5. package/dist/agent-profile-_xPxqVJt.d.ts +488 -0
  6. package/dist/agent-profile-_xPxqVJt.d.ts.map +1 -0
  7. package/dist/analyst/index.d.ts +7 -9
  8. package/dist/analyst/index.d.ts.map +1 -1
  9. package/dist/analyst/index.js +8 -8
  10. package/dist/{benchmark-C4wk_Sjr.js → benchmark-DQKzykkO.js} +2 -2
  11. package/dist/{benchmark-C4wk_Sjr.js.map → benchmark-DQKzykkO.js.map} +1 -1
  12. package/dist/{benchmark-command-CS6gVHVq.js → benchmark-command-mZIlR-ra.js} +13 -13
  13. package/dist/{benchmark-command-CS6gVHVq.js.map → benchmark-command-mZIlR-ra.js.map} +1 -1
  14. package/dist/benchmarks/index.d.ts +3 -4
  15. package/dist/benchmarks/index.d.ts.map +1 -1
  16. package/dist/benchmarks/index.js +3 -3
  17. package/dist/campaign/index.d.ts +5 -9
  18. package/dist/campaign/index.js +7 -7
  19. package/dist/{campaign-B3kPMU8S.js → campaign-BzMSCejE.js} +8 -8
  20. package/dist/{campaign-B3kPMU8S.js.map → campaign-BzMSCejE.js.map} +1 -1
  21. package/dist/cli.js +1 -1
  22. package/dist/{client-DlqdbM7n.d.ts → client-vyYQg3bm.d.ts} +2 -2
  23. package/dist/{client-DlqdbM7n.d.ts.map → client-vyYQg3bm.d.ts.map} +1 -1
  24. package/dist/contract/index.d.ts +9 -10
  25. package/dist/contract/index.js +8 -8
  26. package/dist/{default-registry-B0bKikCb.js → default-registry-CrAp0pYq.js} +4 -4
  27. package/dist/{default-registry-B0bKikCb.js.map → default-registry-CrAp0pYq.js.map} +1 -1
  28. package/dist/{default-registry-BKwc8bN5.d.ts → default-registry-FfNzaUHV.d.ts} +3 -3
  29. package/dist/{default-registry-BKwc8bN5.d.ts.map → default-registry-FfNzaUHV.d.ts.map} +1 -1
  30. package/dist/{define-agent-eval-CY6qdlGV.d.ts → define-agent-eval-V1jQyCDR.d.ts} +102 -11
  31. package/dist/define-agent-eval-V1jQyCDR.d.ts.map +1 -0
  32. package/dist/{define-agent-eval-8h3lXXee.js → define-agent-eval-ox5McL6e.js} +331 -144
  33. package/dist/define-agent-eval-ox5McL6e.js.map +1 -0
  34. package/dist/{dspy-rlm-engine-CF0t2ITD.js → dspy-rlm-engine-Caz2pl4L.js} +3 -3
  35. package/dist/{dspy-rlm-engine-CF0t2ITD.js.map → dspy-rlm-engine-Caz2pl4L.js.map} +1 -1
  36. package/dist/{engine-DhFir3Ys.d.ts → engine-CvW_I72-.d.ts} +2 -2
  37. package/dist/{engine-DhFir3Ys.d.ts.map → engine-CvW_I72-.d.ts.map} +1 -1
  38. package/dist/experiment/index.d.ts +1 -4
  39. package/dist/experiment/index.d.ts.map +1 -1
  40. package/dist/{external-optimizer-process-BwITA9Jp.js → external-optimizer-process-CxnFL1hd.js} +2 -2
  41. package/dist/{external-optimizer-process-BwITA9Jp.js.map → external-optimizer-process-CxnFL1hd.js.map} +1 -1
  42. package/dist/{external-optimizer-subprocess-wBWeoG6A.js → external-optimizer-subprocess-CQi27uEI.js} +2 -2
  43. package/dist/{external-optimizer-subprocess-wBWeoG6A.js.map → external-optimizer-subprocess-CQi27uEI.js.map} +1 -1
  44. package/dist/fuzz.js +1 -1
  45. package/dist/fuzz.js.map +1 -1
  46. package/dist/hosted/index.d.ts +1 -1
  47. package/dist/{index-BQqOjerE.d.ts → index-BTrx5s8m.d.ts} +8 -9
  48. package/dist/index-BTrx5s8m.d.ts.map +1 -0
  49. package/dist/{index-D0Db5X-4.d.ts → index-Bn-nlnSV.d.ts} +4 -4
  50. package/dist/{index-D0Db5X-4.d.ts.map → index-Bn-nlnSV.d.ts.map} +1 -1
  51. package/dist/index-DKXuBPXf.d.ts +3840 -0
  52. package/dist/index-DKXuBPXf.d.ts.map +1 -0
  53. package/dist/index.d.ts +11 -13
  54. package/dist/index.d.ts.map +1 -1
  55. package/dist/index.js +10 -10
  56. package/dist/{kind-factory-gP6lDySe.js → kind-factory-BLvL-E44.js} +2 -2
  57. package/dist/{kind-factory-gP6lDySe.js.map → kind-factory-BLvL-E44.js.map} +1 -1
  58. package/dist/{llm-judge-BfqMFo4h.js → llm-judge-DmNaBrXB.js} +2541 -2435
  59. package/dist/llm-judge-DmNaBrXB.js.map +1 -0
  60. package/dist/{matrix-DGu8KhSs.d.ts → matrix-CJtXz1ky.d.ts} +2 -2
  61. package/dist/{matrix-DGu8KhSs.d.ts.map → matrix-CJtXz1ky.d.ts.map} +1 -1
  62. package/dist/multishot/golden/index.d.ts +1 -1
  63. package/dist/multishot/index.d.ts +2 -2
  64. package/dist/openapi.json +1 -1
  65. package/dist/{produced-state-D91uDvQw.js → produced-state-B8mw6zj9.js} +2 -2
  66. package/dist/{produced-state-D91uDvQw.js.map → produced-state-B8mw6zj9.js.map} +1 -1
  67. package/dist/rl.d.ts +1 -1
  68. package/dist/rl.d.ts.map +1 -1
  69. package/dist/rl.js.map +1 -1
  70. package/dist/{semantic-concept-judge-Dok7_35a.js → semantic-concept-judge-E3s_fEjB.js} +3 -3
  71. package/dist/{semantic-concept-judge-Dok7_35a.js.map → semantic-concept-judge-E3s_fEjB.js.map} +1 -1
  72. package/dist/{skillopt-optimization-method-DDw3v3gA.js → skillopt-optimization-method-f7399oGb.js} +5 -5
  73. package/dist/{skillopt-optimization-method-DDw3v3gA.js.map → skillopt-optimization-method-f7399oGb.js.map} +1 -1
  74. package/dist/statistical-heldout-Cqb73yE9.d.ts +1127 -0
  75. package/dist/statistical-heldout-Cqb73yE9.d.ts.map +1 -0
  76. package/dist/{store-otlp-Dow0pk_5.js → store-otlp-DV_H2HDu.js} +2 -2
  77. package/dist/{store-otlp-Dow0pk_5.js.map → store-otlp-DV_H2HDu.js.map} +1 -1
  78. package/dist/{store-tool-spans-CCZNsihA.d.ts → store-tool-spans-4o55ABER.d.ts} +3 -3
  79. package/dist/{store-tool-spans-CCZNsihA.d.ts.map → store-tool-spans-4o55ABER.d.ts.map} +1 -1
  80. package/dist/{store-tool-spans-CeNj_m2L.js → store-tool-spans-B9tjys_h.js} +3 -3
  81. package/dist/{store-tool-spans-CeNj_m2L.js.map → store-tool-spans-B9tjys_h.js.map} +1 -1
  82. package/dist/supervisor-run/index.d.ts.map +1 -1
  83. package/dist/supervisor-run/index.js +25 -7
  84. package/dist/supervisor-run/index.js.map +1 -1
  85. package/dist/{task-failure-attributes-CZjZeBsY.js → task-failure-attributes-CUy9mkIY.js} +2 -2
  86. package/dist/{task-failure-attributes-CZjZeBsY.js.map → task-failure-attributes-CUy9mkIY.js.map} +1 -1
  87. package/dist/{tool-groups-Cp4Xdzrp.d.ts → tool-groups-DAe1t6zb.d.ts} +2 -2
  88. package/dist/tool-groups-DAe1t6zb.d.ts.map +1 -0
  89. package/dist/trace-repair/index.d.ts +1 -1
  90. package/dist/traces.d.ts +2 -2
  91. package/dist/traces.js +4 -4
  92. package/dist/{types-Ba5UQyVD.d.ts → types-BJz2CPTM.d.ts} +2 -2
  93. package/dist/{types-Ba5UQyVD.d.ts.map → types-BJz2CPTM.d.ts.map} +1 -1
  94. package/dist/{types-CiWITkGo.js → types-DQ0e2E7y.js} +2 -2
  95. package/dist/types-DQ0e2E7y.js.map +1 -0
  96. package/dist/{types-BDV4PiMR.d.ts → types-Dd1ejaeI.d.ts} +2 -2
  97. package/dist/{types-BDV4PiMR.d.ts.map → types-Dd1ejaeI.d.ts.map} +1 -1
  98. package/docs/campaign-proposers.md +42 -0
  99. package/package.json +1 -1
  100. package/dist/agent-profile-B9_GGsG8.d.ts +0 -84
  101. package/dist/agent-profile-B9_GGsG8.d.ts.map +0 -1
  102. package/dist/backend-integrity-CeuTgqsd.d.ts +0 -280
  103. package/dist/backend-integrity-CeuTgqsd.d.ts.map +0 -1
  104. package/dist/benchmark-BjLGkfnN.d.ts +0 -236
  105. package/dist/benchmark-BjLGkfnN.d.ts.map +0 -1
  106. package/dist/define-agent-eval-8h3lXXee.js.map +0 -1
  107. package/dist/define-agent-eval-CY6qdlGV.d.ts.map +0 -1
  108. package/dist/external-optimizer-contracts-CQCpyrIL.d.ts +0 -172
  109. package/dist/external-optimizer-contracts-CQCpyrIL.d.ts.map +0 -1
  110. package/dist/heldout-gate-Df5hsqmm.d.ts +0 -453
  111. package/dist/heldout-gate-Df5hsqmm.d.ts.map +0 -1
  112. package/dist/index-BQqOjerE.d.ts.map +0 -1
  113. package/dist/index-CFDffsKz.d.ts +0 -1135
  114. package/dist/index-CFDffsKz.d.ts.map +0 -1
  115. package/dist/llm-judge-BfqMFo4h.js.map +0 -1
  116. package/dist/power-preflight-Ptse_Kq7.d.ts +0 -117
  117. package/dist/power-preflight-Ptse_Kq7.d.ts.map +0 -1
  118. package/dist/pre-registration-BoI4ucR3.d.ts +0 -592
  119. package/dist/pre-registration-BoI4ucR3.d.ts.map +0 -1
  120. package/dist/promotion-policy-CvMda3kU.d.ts +0 -134
  121. package/dist/promotion-policy-CvMda3kU.d.ts.map +0 -1
  122. package/dist/proposal-findings-bko3GGy-.js.map +0 -1
  123. package/dist/provenance-CRY67X50.d.ts +0 -1995
  124. package/dist/provenance-CRY67X50.d.ts.map +0 -1
  125. package/dist/statistical-heldout-DTyB_6-1.d.ts +0 -295
  126. package/dist/statistical-heldout-DTyB_6-1.d.ts.map +0 -1
  127. package/dist/tool-groups-Cp4Xdzrp.d.ts.map +0 -1
  128. package/dist/types-CiWITkGo.js.map +0 -1
@@ -1,280 +0,0 @@
1
- import { t as AgentEvalError } from "./errors-DEE6u6ot.js";
2
- import { c as CostLedgerHandle } from "./cost-ledger-DbQdN3nO.js";
3
- import { a as RunRecord } from "./run-record-DTv1MdjK.js";
4
- import { Y as RawProviderSink, p as ChatClient } from "./types-gvRsyJLh.js";
5
- import { a as CheckerIdentity, n as VerdictCertification, t as DefaultVerdict, x as VerificationStrategySource } from "./verdict-E4eRNf7-.js";
6
- //#region src/artifact-validator.d.ts
7
- /**
8
- * Artifact validators.
9
- *
10
- * Generic "score a produced artifact" primitive. Tax uses it for PDF form
11
- * correctness, research for sourced briefs, browser for task assertions, coding
12
- * for social posts. One interface, many validators.
13
- *
14
- * A validator receives an `Artifact` (file on disk, JSON blob, text, binary)
15
- * plus a `ValidationContext` (scenario id, the turns that produced it) and
16
- * returns a `ValidationResult` with pass/fail + 0..1 score + structured
17
- * issues.
18
- */
19
- interface Artifact {
20
- /** Logical kind — validators type-guard on this */
21
- kind: 'file' | 'json' | 'text' | 'binary' | string;
22
- /** Filesystem-style path, optional */
23
- path?: string;
24
- /** String content for text/json/file kinds */
25
- content?: string;
26
- /** Binary content (if kind === 'binary') */
27
- bytes?: Uint8Array;
28
- /** Caller-supplied metadata (mimeType, sha256, size, etc.) */
29
- metadata?: Record<string, unknown>;
30
- }
31
- //#endregion
32
- //#region src/completion-verifier.d.ts
33
- /** What kind of produced state can satisfy a requirement structurally. */
34
- type SatisfiedBy = 'artifact' | 'proposal' | 'tool-call' | 'any';
35
- interface CompletionRequirement {
36
- /** Stable id from the task gold (e.g. a persona's `expected_requirements[].req_id`). */
37
- reqId: string;
38
- /** Human-readable description of the required deliverable. */
39
- title: string;
40
- /** Optional kind/category hint, matched against a produced item's kind. */
41
- category?: string;
42
- /** What produced state satisfies this requirement. Defaults to 'any'. */
43
- satisfiedBy?: SatisfiedBy;
44
- }
45
- interface TaskGold {
46
- taskId: string;
47
- requirements: CompletionRequirement[];
48
- }
49
- interface ProducedProposal {
50
- id: string;
51
- title: string;
52
- status: 'pending' | 'approved' | 'rejected';
53
- /** Optional persisted body — when present, enables a correctness check. */
54
- content?: string;
55
- }
56
- /** Everything observable about what a run actually produced. */
57
- interface ProducedState {
58
- /** Persisted vault artifacts. Reuses the shared `Artifact` shape. */
59
- artifacts: Artifact[];
60
- /** Proposals / filings the agent created. */
61
- proposals: ProducedProposal[];
62
- /** Names of tools the agent invoked. */
63
- toolCalls: string[];
64
- }
65
- interface RequirementCheck {
66
- reqId: string;
67
- title: string;
68
- /** A produced item of the right kind matched the requirement, non-empty. */
69
- structurallyPresent: boolean;
70
- /**
71
- * Whether the matched item actually fulfils the requirement. `null` when
72
- * not structurally present, when the matched item carries no content
73
- * to assess, or when the correctness check itself failed (`unmeasured`).
74
- */
75
- correct: boolean | null;
76
- /** structurallyPresent && !unmeasured && correct !== false. */
77
- satisfied: boolean;
78
- /**
79
- * Set when the correctness check itself errored (LLM call failure or an
80
- * unparseable response after retry). The requirement's fulfilment is
81
- * UNKNOWN — `correct` stays null, `satisfied` is false, and
82
- * `completionVerdict` excludes the row from `completionRate`'s
83
- * denominator. Never folded into a zero: a synthetic zero is
84
- * indistinguishable from a real failure (see `JudgeParseError`).
85
- */
86
- unmeasured?: true;
87
- /** Why the correctness check could not be measured (present iff `unmeasured`). */
88
- unmeasuredReason?: string;
89
- /** Human-readable evidence for the verdict. */
90
- evidence: string[];
91
- }
92
- /** Extends the substrate verdict spine: `valid` = `fullyComplete` and
93
- * `score` = `completionRate` — derived in `completionVerdict()`, the one
94
- * place those equalities hold by construction. */
95
- interface CompletionVerdict extends DefaultVerdict {
96
- taskId: string;
97
- requirements: RequirementCheck[];
98
- /** satisfied / MEASURABLE requirements (unmeasured rows leave the denominator). */
99
- completionRate: number;
100
- /** Every measurable requirement satisfied (false when anything is unmeasured). */
101
- fullyComplete: boolean;
102
- /** Requirements whose correctness check errored — reported, never scored as zero. */
103
- unmeasuredCount: number;
104
- }
105
- /**
106
- * Construct a `CompletionVerdict` from the per-requirement checks, deriving
107
- * `completionRate` / `fullyComplete` and the spine fields (`valid` =
108
- * `fullyComplete`, `score` = `completionRate`) in one place. Throws on zero
109
- * requirements — a verdict over nothing is a misconfiguration, mirroring
110
- * `verifyCompletion`'s gold-spec guard.
111
- */
112
- declare function completionVerdict(input: {
113
- taskId: string;
114
- requirements: RequirementCheck[];
115
- /** What certified the correctness stage, when anything did. Omitted =
116
- * an uncertified verdict — the honest default for a bare checker. */
117
- certification?: VerdictCertification;
118
- }): CompletionVerdict;
119
- /**
120
- * What a correctness checker declares about itself so the completion
121
- * verdict can carry a certification: which strategy member it discharges,
122
- * its exact identity, and the steps its answers rest on unverified.
123
- */
124
- interface CorrectnessCheckerAttestation {
125
- strategy: VerificationStrategySource;
126
- checker: CheckerIdentity;
127
- assumptions: string[];
128
- }
129
- /**
130
- * Decides whether a produced item's content actually fulfils a requirement.
131
- * Injected so the structural verifier stays pure and unit-testable; the
132
- * production implementation is `createLlmCorrectnessChecker`.
133
- *
134
- * `attestation` is optional metadata on the function value: a checker that
135
- * carries one yields certified completion verdicts; a bare function yields
136
- * the same verdict uncertified. A plain arrow function remains a valid
137
- * checker.
138
- */
139
- interface CorrectnessChecker {
140
- (requirement: CompletionRequirement, content: string): Promise<{
141
- correct: boolean;
142
- reason: string;
143
- }>;
144
- attestation?: CorrectnessCheckerAttestation;
145
- }
146
- /**
147
- * Verify whether a run completed the task. `checkCorrectness` is injected —
148
- * `createLlmCorrectnessChecker` for production, a deterministic stub in tests.
149
- *
150
- * Throws on a gold spec with no requirements: an eval task that requires
151
- * nothing is a misconfiguration, not a vacuously-complete task.
152
- */
153
- declare function verifyCompletion(gold: TaskGold, state: ProducedState, checkCorrectness: CorrectnessChecker): Promise<CompletionVerdict>;
154
- interface LlmCorrectnessCheckerOpts {
155
- model?: string;
156
- /** Optional ledger for direct use. */
157
- costLedger?: CostLedgerHandle;
158
- costPhase?: string;
159
- costTags?: Record<string, string>;
160
- signal?: AbortSignal;
161
- /** Max chars of artifact content sent to the checker. */
162
- maxContentChars?: number;
163
- /**
164
- * Checker LLM calls per requirement before giving up (parse failures and
165
- * call errors both consume attempts). The failure then surfaces as an
166
- * `unmeasured` requirement, never a zero.
167
- */
168
- maxAttempts?: number;
169
- /**
170
- * Forensic capture of every checker request/response/error — without it a
171
- * checker failure is unauditable (the agent-turn raws never contain the
172
- * checker's own calls). Same sink contract as `LlmClient`.
173
- */
174
- rawSink?: RawProviderSink;
175
- }
176
- /**
177
- * Production `CorrectnessChecker` — one LLM call per matched artifact,
178
- * deterministic (temperature 0), structured JSON out. Judges fulfilment
179
- * only: a plan, a gesture, or a description of what should be done does not
180
- * fulfil a requirement — the artifact must BE the deliverable.
181
- */
182
- declare function createLlmCorrectnessChecker(chat: ChatClient, opts?: LlmCorrectnessCheckerOpts): CorrectnessChecker;
183
- //#endregion
184
- //#region src/produced-state.d.ts
185
- /** A tool the agent invoked. */
186
- interface ToolCallEventLike {
187
- type: 'tool_call';
188
- toolName: string;
189
- }
190
- /**
191
- * An artifact the agent produced. `content` is the enriched field — the
192
- * runtime's base `artifact` event carries only metadata; the completion
193
- * oracle needs the body to verify the deliverable, so the runtime emits it.
194
- */
195
- interface ArtifactEventLike {
196
- type: 'artifact';
197
- artifactId: string;
198
- name?: string;
199
- mimeType?: string;
200
- uri?: string;
201
- content?: string;
202
- }
203
- /** A proposal / filing the agent created. */
204
- interface ProposalEventLike {
205
- type: 'proposal_created';
206
- proposalId: string;
207
- title: string;
208
- status?: 'pending' | 'approved' | 'rejected';
209
- content?: string;
210
- }
211
- /**
212
- * The subset of runtime stream events `extractProducedState` consumes.
213
- * agent-runtime's full `RuntimeStreamEvent` union satisfies this structurally;
214
- * the `{ type: string }` catch-all keeps the input permissive so callers can
215
- * pass the whole unfiltered telemetry stream — unrecognized events are skipped.
216
- */
217
- type RuntimeEventLike = ToolCallEventLike | ArtifactEventLike | ProposalEventLike | {
218
- type: string;
219
- };
220
- /**
221
- * Normalize a run's runtime event stream into `ProducedState`.
222
- *
223
- * Pure and total — unrecognized event types are skipped. `toolCalls` is
224
- * deduplicated by name in first-seen order (completion cares about a tool's
225
- * presence, not its call count). An artifact with neither a name nor a uri
226
- * still yields an entry keyed by its `artifactId` so it is never silently
227
- * dropped; an artifact with no `content` yields empty content, which the
228
- * completion oracle's structural check then rejects on its own.
229
- */
230
- declare function extractProducedState(events: readonly RuntimeEventLike[]): ProducedState;
231
- //#endregion
232
- //#region src/integrity/backend-integrity.d.ts
233
- interface BackendIntegrityReport {
234
- /** Total records inspected. */
235
- totalRecords: number;
236
- /** Records with input=0 AND output=0 (a stub fingerprint). */
237
- stubRecords: number;
238
- /** Records with nonzero token usage (real LLM activity). */
239
- realRecords: number;
240
- /** Records where output>0 but costUsd=0 (real LLM, broken cost ledger). */
241
- uncostedRecords: number;
242
- /** Sum of input tokens across all records. */
243
- totalInputTokens: number;
244
- /** Sum of output tokens across all records. */
245
- totalOutputTokens: number;
246
- /** Sum of costUsd across all records. */
247
- totalCostUsd: number;
248
- /** Worst-case integrity verdict. */
249
- verdict: 'real' | 'mixed' | 'stub';
250
- /** Human-readable diagnosis suitable for terminal output. */
251
- diagnosis: string;
252
- }
253
- /**
254
- * Error thrown when an integrity assertion fails. Caller can pattern-match
255
- * by `code === 'AGENT_EVAL_BACKEND_STUB'` to differentiate from other
256
- * errors.
257
- */
258
- declare class BackendIntegrityError extends AgentEvalError {
259
- readonly report: BackendIntegrityReport;
260
- constructor(message: string, report: BackendIntegrityReport);
261
- }
262
- /**
263
- * Inspect a batch of RunRecords and return an integrity report. Pure
264
- * function — no I/O, no logging. The caller decides what to do with the
265
- * verdict (print warning, throw, gate CI, etc.).
266
- */
267
- declare function summarizeBackendIntegrity(records: ReadonlyArray<RunRecord>): BackendIntegrityReport;
268
- /**
269
- * Throw BackendIntegrityError if the verdict is 'stub' — i.e. every record
270
- * shows zero LLM activity. Non-strict callers can pass `{ allowMixed: false }`
271
- * to also reject mixed verdicts (recommended for CI gates).
272
- *
273
- * Real backends pass through silently.
274
- */
275
- declare function assertRealBackend(records: ReadonlyArray<RunRecord>, opts?: {
276
- allowMixed?: boolean;
277
- }): BackendIntegrityReport;
278
- //#endregion
279
- export { SatisfiedBy as _, ArtifactEventLike as a, createLlmCorrectnessChecker as b, ToolCallEventLike as c, CompletionVerdict as d, CorrectnessChecker as f, RequirementCheck as g, ProducedState as h, summarizeBackendIntegrity as i, extractProducedState as l, ProducedProposal as m, BackendIntegrityReport as n, ProposalEventLike as o, LlmCorrectnessCheckerOpts as p, assertRealBackend as r, RuntimeEventLike as s, BackendIntegrityError as t, CompletionRequirement as u, TaskGold as v, verifyCompletion as x, completionVerdict as y };
280
- //# sourceMappingURL=backend-integrity-CeuTgqsd.d.ts.map
@@ -1 +0,0 @@
1
- {"version":3,"file":"backend-integrity-CeuTgqsd.d.ts","names":[],"sources":["../src/artifact-validator.ts","../src/completion-verifier.ts","../src/produced-state.ts","../src/integrity/backend-integrity.ts"],"mappings":";;;;;;;;;;;;;;;;;;UAaiB;;EAEf;;EAEA;;EAEA;;EAEA,QAAQ;;EAER,WAAW;;;;;KCsBD;UAEK;;EAEf;;EAEA;;EAEA;;EAEA,cAAc;;UAGC;EACf;EACA,cAAc;;UAGC;EACf;EACA;EACA;;EAEA;;;UAIe;;EAEf,WAAW;;EAEX,WAAW;;EAEX;;UAGe;EACf;EACA;;EAEA;;;;;;EAMA;;EAEA;;;;;;;;;EASA;;EAEA;;EAEA;;;;;UAMe,0BAA0B;EACzC;EACA,cAAc;;EAEd;;EAEA;;EAEA;;;;;;;;;iBAUc,kBAAkB;EAChC;EACA,cAAc;;;EAGd,gBAAgB;IACd;;;;;;UAqCa;EACf,UAAU;EACV,SAAS;EACT;;;;;;;;;;;;UAae;GAEb,aAAa,uBACb,kBACC;IAAU;IAAkB;;EAC/B,cAAc;;;;;;;;;iBAsLM,iBACpB,MAAM,UACN,OAAO,eACP,kBAAkB,qBACjB,QAAQ;UAyGM;EACf;;EAEA,aAAa;EACb;EACA,WAAW;EACX,SAAS;;EAET;;;;;;EAMA;;;;;;EAMA,UAAU;;;;;;;;iBA2CI,4BACd,MAAM,YACN,OAAM,4BACL;;;;UCnhBc;EACf;EACA;;;;;;;UAQe;EACf;EACA;EACA;EACA;EACA;EACA;;;UAIe;EACf;EACA;EACA;EACA;EAIA;;;;;;;;KASU,mBACR,oBACA,oBACA;EACE;;;;;;;;;;;;iBAmBU,qBAAqB,iBAAiB,qBAAqB;;;UCpD1D;;EAEf;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;;;;;;cAQW,8BAA8B;WAGvB,QAAQ;EAF1B,YACE,iBACgB,QAAQ;;;;;;;iBAWZ,0BACd,SAAS,cAAc,aACtB;;;;;;;;iBAuHa,kBACd,SAAS,cAAc,YACvB;EAAQ;IACP"}
@@ -1,236 +0,0 @@
1
- import { c as AnalystRunInputs, f as AnalystUsageReceipt, i as AnalystFinding, p as EvidenceRef, w as TraceAnalysisStore } from "./types-DN2WdT5S.js";
2
- import { c as RegistryRunOpts, n as AnalystRegistry } from "./registry-7pOUBrtX.js";
3
- //#region src/analyst/benchmark-scoring.d.ts
4
- declare function scoreAnalystFindings(testCase: Pick<AnalystBenchmarkCase, 'id' | 'expectedIssues' | 'labeledEvidence'>, findings: readonly AnalystFinding[]): AnalystFindingScore;
5
- //#endregion
6
- //#region src/analyst/benchmark.d.ts
7
- interface AnalystEvidenceExpectation {
8
- uri: string;
9
- kind?: EvidenceRef['kind'];
10
- }
11
- interface AnalystIssueExpectation {
12
- id: string;
13
- findingIds?: readonly string[];
14
- areas?: readonly string[];
15
- subjects?: readonly string[];
16
- evidence?: readonly AnalystEvidenceExpectation[];
17
- evidenceMode?: 'any' | 'all';
18
- /** Exact evidence location for the first unrecoverable or causal step. */
19
- criticalEvidence?: readonly AnalystEvidenceExpectation[];
20
- }
21
- type AnalystBenchmarkLabelState = 'positive' | 'trusted-negative' | 'unlabeled';
22
- interface AnalystBenchmarkCase<TInput = unknown> {
23
- id: string;
24
- /** Independent source unit used for resampling, such as a task or incident. */
25
- clusterId: string;
26
- /** Whether labels prove an issue, prove no issue, or leave the outcome unknown. */
27
- labelState: AnalystBenchmarkLabelState;
28
- input: TInput;
29
- expectedIssues: readonly AnalystIssueExpectation[];
30
- /** Complete set of labeled locations used to measure label-location agreement. */
31
- labeledEvidence?: readonly AnalystEvidenceExpectation[];
32
- tags?: readonly string[];
33
- metadata?: Record<string, unknown>;
34
- }
35
- interface AnalystFindingScore {
36
- expectedIssueCount: number;
37
- matchedIssueIds: string[];
38
- missedIssueIds: string[];
39
- supportedFindingIndexes: number[];
40
- unsupportedFindingIndexes: number[];
41
- unlabeledEvidence: EvidenceRef[];
42
- issueRecall: number;
43
- findingPrecision: number;
44
- f1: number;
45
- criticalStepAccuracy: number | null;
46
- /** Share of findings that cite at least one evidence location. */
47
- citationCoverage: number | null;
48
- /** Share of citations that include a non-empty source excerpt. */
49
- citationExcerptCoverage: number | null;
50
- /** Share of citations that agree with a labeled case location. */
51
- citationLabelAgreement: number | null;
52
- predictionOnLabelEmptyCase: boolean;
53
- }
54
- interface AnalystEvidenceResolutionError {
55
- evidence: EvidenceRef;
56
- class: string;
57
- message: string;
58
- }
59
- interface AnalystEvidenceResolution {
60
- checked: number;
61
- resolved: number;
62
- unresolvedEvidence: EvidenceRef[];
63
- errors: AnalystEvidenceResolutionError[];
64
- /** Null when no citations were checked or any resolution attempt failed. */
65
- validity: number | null;
66
- }
67
- type AnalystEvidenceResolver<TInput = unknown> = (input: {
68
- caseId: string;
69
- caseInput: TInput;
70
- evidence: EvidenceRef;
71
- signal?: AbortSignal;
72
- }) => boolean | Promise<boolean>;
73
- /**
74
- * Resolve canonical `trace://<trace>/span/<span>` evidence against a trace store.
75
- * Other evidence kinds and URI schemes require a caller-supplied resolver.
76
- */
77
- declare function traceStoreEvidenceResolver<TInput>(getStore: (input: TInput) => TraceAnalysisStore): AnalystEvidenceResolver<TInput>;
78
- interface AnalystBenchmarkOutput {
79
- findings: readonly AnalystFinding[];
80
- usage?: AnalystUsageReceipt;
81
- metadata?: Record<string, unknown>;
82
- /**
83
- * End-to-end duration measured by an external runner before import.
84
- * Use null when the source explicitly did not capture duration.
85
- */
86
- observedLatencyMs?: number | null;
87
- /** Marks a completed transport as a failed analyst run while retaining usage and metadata. */
88
- error?: AnalystBenchmarkError;
89
- }
90
- interface AnalystBenchmarkError {
91
- class: string;
92
- message: string;
93
- code?: string;
94
- status?: number;
95
- }
96
- interface AnalystBenchmarkRunner<TInput = unknown> {
97
- id: string;
98
- analyze(input: TInput, context: {
99
- caseId: string;
100
- repetition: number;
101
- signal?: AbortSignal;
102
- }): AnalystBenchmarkOutput | Promise<AnalystBenchmarkOutput>;
103
- }
104
- interface AnalystBenchmarkObservation {
105
- runnerId: string;
106
- caseId: string;
107
- clusterId: string;
108
- labelState: AnalystBenchmarkLabelState;
109
- repetition: number;
110
- executionIndex: number;
111
- latencyMs: number | null;
112
- latencySource: 'benchmark-clock' | 'runner-reported' | 'uncaptured';
113
- findings: readonly AnalystFinding[];
114
- score: AnalystFindingScore;
115
- evidenceResolution?: AnalystEvidenceResolution;
116
- caseTags: readonly string[];
117
- caseMetadata?: Record<string, unknown>;
118
- usage?: AnalystUsageReceipt;
119
- runnerMetadata?: Record<string, unknown>;
120
- error?: AnalystBenchmarkError;
121
- }
122
- interface AnalystLatencyDistribution {
123
- min: number;
124
- mean: number;
125
- p50: number;
126
- p95: number;
127
- max: number;
128
- }
129
- interface AnalystBenchmarkSummary {
130
- runnerId: string;
131
- plannedRuns: number;
132
- completedRuns: number;
133
- failedRuns: number;
134
- issueBearingRuns: number;
135
- trustedNegativeRuns: number;
136
- unlabeledRuns: number;
137
- /** Pooled across all labeled issues and findings. */
138
- issueRecall: number | null;
139
- /** Pooled across all labeled issues and findings. */
140
- findingPrecision: number | null;
141
- /** Harmonic mean of the pooled precision and recall. */
142
- f1: number | null;
143
- /** Mean of per-case recall over issue-bearing runs. */
144
- macroIssueRecall: number | null;
145
- /** Mean of per-case precision over issue-bearing runs. */
146
- macroFindingPrecision: number | null;
147
- /** Mean of per-case F1 over issue-bearing runs. */
148
- macroF1: number | null;
149
- criticalStepAccuracy: number | null;
150
- citationCoverage: number | null;
151
- citationExcerptCoverage: number | null;
152
- citationLabelAgreement: number | null;
153
- citationResolution: number | null;
154
- citationResolutionUnknownRuns: number;
155
- unresolvedCitations: number;
156
- citationResolutionErrors: number;
157
- trustedNegativeFalsePositiveRate: number | null;
158
- trustedNegativeFailureRate: number | null;
159
- unlabeledPredictionRate: number | null;
160
- unlabeledFailureRate: number | null;
161
- /** Primary repeatability measure over complete finding identity and evidence. */
162
- predictionAgreement: number | null;
163
- /** Repeated cases contributing equally to predictionAgreement. */
164
- predictionAgreementCases: number;
165
- /** Secondary repeatability detail over matched expected labels. */
166
- matchedLabelAgreement: number | null;
167
- /** Positive repeated cases contributing equally to matchedLabelAgreement. */
168
- matchedLabelAgreementCases: number;
169
- latencyMs: AnalystLatencyDistribution | null;
170
- benchmarkClockLatencyRuns: number;
171
- runnerReportedLatencyRuns: number;
172
- latencyUnknownRuns: number;
173
- calls: number;
174
- callsUnknownRuns: number;
175
- inputTokens: number;
176
- outputTokens: number;
177
- reasoningTokens: number;
178
- cachedTokens: number;
179
- cacheWriteTokens: number;
180
- tokenUsageUnknownRuns: number;
181
- reasoningTokenUsageUnknownRuns: number;
182
- cachedTokenUsageUnknownRuns: number;
183
- cacheWriteTokenUsageUnknownRuns: number;
184
- knownCostUsd: number;
185
- costUnknownRuns: number;
186
- }
187
- interface AnalystBenchmarkDatasetRef {
188
- id: string;
189
- revision: string;
190
- split?: string;
191
- }
192
- interface AnalystBenchmarkDescriptor {
193
- id?: string;
194
- dataset?: AnalystBenchmarkDatasetRef;
195
- command?: string;
196
- environment?: Record<string, string>;
197
- metadata?: Record<string, unknown>;
198
- }
199
- interface AnalystBenchmarkProvenance extends AnalystBenchmarkDescriptor {
200
- startedAt: string;
201
- endedAt: string;
202
- caseCount: number;
203
- runnerIds: string[];
204
- repetitions: number;
205
- maxConcurrency: number;
206
- runnerOrderSeed: number;
207
- }
208
- interface AnalystBenchmarkResult {
209
- provenance: AnalystBenchmarkProvenance;
210
- observations: AnalystBenchmarkObservation[];
211
- summaries: AnalystBenchmarkSummary[];
212
- }
213
- interface RunAnalystBenchmarkOptions<TInput> {
214
- cases: readonly AnalystBenchmarkCase<TInput>[];
215
- runners: readonly AnalystBenchmarkRunner<TInput>[];
216
- repetitions?: number;
217
- maxConcurrency?: number;
218
- runnerOrderSeed?: number;
219
- resolveEvidence?: AnalystEvidenceResolver<TInput>;
220
- benchmark?: AnalystBenchmarkDescriptor;
221
- /** Previously persisted rows. Exact case, runner, repetition, and execution identities are required. */
222
- initialObservations?: readonly AnalystBenchmarkObservation[];
223
- onObservation?: (observation: AnalystBenchmarkObservation) => void | Promise<void>;
224
- signal?: AbortSignal;
225
- }
226
- declare function runAnalystBenchmark<TInput>(options: RunAnalystBenchmarkOptions<TInput>): Promise<AnalystBenchmarkResult>;
227
- declare function registryBenchmarkRunner(options: {
228
- id: string;
229
- registry: AnalystRegistry;
230
- runOptions?: Omit<RegistryRunOpts, 'signal'>;
231
- /** Count any selected analyst failure as a failed benchmark run. */
232
- failOnAnalystFailure?: boolean;
233
- }): AnalystBenchmarkRunner<AnalystRunInputs>;
234
- //#endregion
235
- export { scoreAnalystFindings as C, traceStoreEvidenceResolver as S, AnalystIssueExpectation as _, AnalystBenchmarkLabelState as a, registryBenchmarkRunner as b, AnalystBenchmarkProvenance as c, AnalystBenchmarkSummary as d, AnalystEvidenceExpectation as f, AnalystFindingScore as g, AnalystEvidenceResolver as h, AnalystBenchmarkError as i, AnalystBenchmarkResult as l, AnalystEvidenceResolutionError as m, AnalystBenchmarkDatasetRef as n, AnalystBenchmarkObservation as o, AnalystEvidenceResolution as p, AnalystBenchmarkDescriptor as r, AnalystBenchmarkOutput as s, AnalystBenchmarkCase as t, AnalystBenchmarkRunner as u, AnalystLatencyDistribution as v, runAnalystBenchmark as x, RunAnalystBenchmarkOptions as y };
236
- //# sourceMappingURL=benchmark-BjLGkfnN.d.ts.map
@@ -1 +0,0 @@
1
- {"version":3,"file":"benchmark-BjLGkfnN.d.ts","names":[],"sources":["../src/analyst/benchmark-scoring.ts","../src/analyst/benchmark.ts"],"mappings":";;;iBASgB,qBACd,UAAU,KAAK,oEACf,mBAAmB,mBAClB;;;UCIc;EACf;EACA,OAAO;;UAGQ;EACf;EACA;EACA;EACA;EACA,oBAAoB;EACpB;;EAEA,4BAA4B;;KAGlB;UAEK,qBAAqB;EACpC;;EAEA;;EAEA,YAAY;EACZ,OAAO;EACP,yBAAyB;;EAEzB,2BAA2B;EAC3B;EACA,WAAW;;UAGI;EACf;EACA;EACA;EACA;EACA;EACA,mBAAmB;EACnB;EACA;EACA;EACA;;EAEA;;EAEA;;EAEA;EACA;;UAGe;EACf,UAAU;EACV;EACA;;UAGe;EACf;EACA;EACA,oBAAoB;EACpB,QAAQ;;EAER;;KAGU,wBAAwB,qBAAqB;EACvD;EACA,WAAW;EACX,UAAU;EACV,SAAS;gBACK;;;;;iBAMA,2BAA2B,QACzC,WAAW,OAAO,WAAW,qBAC5B,wBAAwB;UAoBV;EACf,mBAAmB;EACnB,QAAQ;EACR,WAAW;;;;;EAKX;;EAEA,QAAQ;;UAGO;EACf;EACA;EACA;EACA;;UAGe,uBAAuB;EACtC;EACA,QACE,OAAO,QACP;IAAW;IAAgB;IAAoB,SAAS;MACvD,yBAAyB,QAAQ;;UAGrB;EACf;EACA;EACA;EACA,YAAY;EACZ;EACA;EACA;EACA;EACA,mBAAmB;EACnB,OAAO;EACP,qBAAqB;EACrB;EACA,eAAe;EACf,QAAQ;EACR,iBAAiB;EACjB,QAAQ;;UAGO;EACf;EACA;EACA;EACA;EACA;;UAGe;EACf;EACA;EACA;EACA;EACA;EACA;EACA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;;EAEA;;EAEA;;EAEA;;EAEA;EACA,WAAW;EACX;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;;UAGe;EACf;EACA;EACA;;UAGe;EACf;EACA,UAAU;EACV;EACA,cAAc;EACd,WAAW;;UAGI,mCAAmC;EAClD;EACA;EACA;EACA;EACA;EACA;EACA;;UAGe;EACf,YAAY;EACZ,cAAc;EACd,WAAW;;UAGI,2BAA2B;EAC1C,gBAAgB,qBAAqB;EACrC,kBAAkB,uBAAuB;EACzC;EACA;EACA;EACA,kBAAkB,wBAAwB;EAC1C,YAAY;;EAEZ,+BAA+B;EAC/B,iBAAiB,aAAa,uCAAuC;EACrE,SAAS;;iBAGW,oBAAoB,QACxC,SAAS,2BAA2B,UACnC,QAAQ;iBAyDK,wBAAwB;EACtC;EACA,UAAU;EACV,aAAa,KAAK;;EAElB;IACE,uBAAuB"}