@tangle-network/agent-eval 0.170.0 → 0.172.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +68 -0
- package/README.md +3 -0
- package/dist/adapters/http.d.ts +2 -2
- package/dist/analyst/index.d.ts +5 -5
- package/dist/analyst/index.js +2 -2
- package/dist/{experiment-tracker-Dm8yQMqb.d.ts → attestation-CJBGmMVh.d.ts} +78 -2
- package/dist/attestation-CJBGmMVh.d.ts.map +1 -0
- package/dist/{experiment-tracker-BKEumQug.js → attestation-XSUpbc4o.js} +96 -2
- package/dist/attestation-XSUpbc4o.js.map +1 -0
- package/dist/{benchmark-command-BA7qOdWw.js → benchmark-command-DoFcisuM.js} +2 -2
- package/dist/{benchmark-command-BA7qOdWw.js.map → benchmark-command-DoFcisuM.js.map} +1 -1
- package/dist/benchmarks/index.d.ts +4 -4
- package/dist/benchmarks/index.js +2 -2
- package/dist/bounded-process-CVOC_D3H.js +181 -0
- package/dist/bounded-process-CVOC_D3H.js.map +1 -0
- package/dist/builder-eval/index.js +39 -98
- package/dist/builder-eval/index.js.map +1 -1
- package/dist/campaign/index.d.ts +8 -7
- package/dist/campaign/index.js +4 -4
- package/dist/{campaign-BeCbxFqs.js → campaign-Dp35pBbS.js} +4 -4
- package/dist/{campaign-BeCbxFqs.js.map → campaign-Dp35pBbS.js.map} +1 -1
- package/dist/canonical-CFpojCN5.d.ts +31 -0
- package/dist/canonical-CFpojCN5.d.ts.map +1 -0
- package/dist/{chat-client-DEtybj5i.js → chat-client-DI79OPye.js} +2 -2
- package/dist/{chat-client-DEtybj5i.js.map → chat-client-DI79OPye.js.map} +1 -1
- package/dist/cli.js +1 -1
- package/dist/{client-_Fsa5c2_.d.ts → client-Df7wdslk.d.ts} +2 -2
- package/dist/{client-_Fsa5c2_.d.ts.map → client-Df7wdslk.d.ts.map} +1 -1
- package/dist/contract/index.d.ts +9 -9
- package/dist/contract/index.js +4 -4
- package/dist/{default-registry-ovxrOP0_.d.ts → default-registry-XxedTLwu.d.ts} +3 -3
- package/dist/{default-registry-ovxrOP0_.d.ts.map → default-registry-XxedTLwu.d.ts.map} +1 -1
- package/dist/{define-agent-eval-DVJm8Xlh.d.ts → define-agent-eval-0wW7gFhr.d.ts} +4 -4
- package/dist/{define-agent-eval-DVJm8Xlh.d.ts.map → define-agent-eval-0wW7gFhr.d.ts.map} +1 -1
- package/dist/{define-agent-eval-Clj-8igZ.js → define-agent-eval-jS8xj_Q_.js} +2 -2
- package/dist/{define-agent-eval-Clj-8igZ.js.map → define-agent-eval-jS8xj_Q_.js.map} +1 -1
- package/dist/descriptive-B2iPaT9J.d.ts +89 -0
- package/dist/descriptive-B2iPaT9J.d.ts.map +1 -0
- package/dist/{engine-D12Rb6WB.d.ts → engine-BfRay1qD.d.ts} +2 -2
- package/dist/{engine-D12Rb6WB.d.ts.map → engine-BfRay1qD.d.ts.map} +1 -1
- package/dist/experiment/index.d.ts +3 -54
- package/dist/experiment/index.d.ts.map +1 -1
- package/dist/experiment/index.js +1 -95
- package/dist/experiment/index.js.map +1 -1
- package/dist/{heldout-gate-Bn7_xWCv.d.ts → heldout-gate-JgNRDZwZ.d.ts} +3 -3
- package/dist/{heldout-gate-Bn7_xWCv.d.ts.map → heldout-gate-JgNRDZwZ.d.ts.map} +1 -1
- package/dist/hosted/index.d.ts +1 -1
- package/dist/{index-lfaSeKSD.d.ts → index-D-UdhAmg.d.ts} +3 -31
- package/dist/index-D-UdhAmg.d.ts.map +1 -0
- package/dist/{index-CM-SM00y.d.ts → index-DDAPhUJJ.d.ts} +4 -4
- package/dist/{index-CM-SM00y.d.ts.map → index-DDAPhUJJ.d.ts.map} +1 -1
- package/dist/{index-Bfs5aufo.d.ts → index-DnglhM0A.d.ts} +9 -9
- package/dist/{index-Bfs5aufo.d.ts.map → index-DnglhM0A.d.ts.map} +1 -1
- package/dist/{index-DBbivBNs.d.ts → index-_vPrVMRX.d.ts} +6 -6
- package/dist/{index-DBbivBNs.d.ts.map → index-_vPrVMRX.d.ts.map} +1 -1
- package/dist/index.d.ts +245 -15
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +712 -9
- package/dist/index.js.map +1 -1
- package/dist/{integrity-CyWSSoQS.js → integrity-BWywb34E.js} +34 -11
- package/dist/{integrity-CyWSSoQS.js.map → integrity-BWywb34E.js.map} +1 -1
- package/dist/ledger-core/index.d.ts +2 -1
- package/dist/{llm-judge-DbJdo8Nj.js → llm-judge-aQHIk5_-.js} +20 -11
- package/dist/llm-judge-aQHIk5_-.js.map +1 -0
- package/dist/{matrix-BpI5Trmo.d.ts → matrix-Ch8JO1pG.d.ts} +2 -2
- package/dist/{matrix-BpI5Trmo.d.ts.map → matrix-Ch8JO1pG.d.ts.map} +1 -1
- package/dist/meta-eval/index.d.ts +162 -2
- package/dist/meta-eval/index.d.ts.map +1 -1
- package/dist/meta-eval/index.js +287 -2
- package/dist/meta-eval/index.js.map +1 -1
- package/dist/multishot/golden/index.d.ts +1 -1
- package/dist/multishot/index.d.ts +2 -2
- package/dist/openapi.json +1 -1
- package/dist/{produced-state-CtSIp5cQ.js → produced-state-CxmbFxFd.js} +2 -2
- package/dist/{produced-state-CtSIp5cQ.js.map → produced-state-CxmbFxFd.js.map} +1 -1
- package/dist/{promotion-policy-CkXSgKkF.d.ts → promotion-policy-BBBcz5_3.d.ts} +2 -2
- package/dist/{promotion-policy-CkXSgKkF.d.ts.map → promotion-policy-BBBcz5_3.d.ts.map} +1 -1
- package/dist/{provenance-CIRUardl.d.ts → provenance-Dp-vvyrU.d.ts} +17 -5
- package/dist/provenance-Dp-vvyrU.d.ts.map +1 -0
- package/dist/rl.d.ts +1 -1
- package/dist/sandbox-harness-BlSOu4LX.d.ts.map +1 -1
- package/dist/{skillopt-optimization-method-B2R9C5aG.js → skillopt-optimization-method-LHi02MzH.js} +2 -2
- package/dist/{skillopt-optimization-method-B2R9C5aG.js.map → skillopt-optimization-method-LHi02MzH.js.map} +1 -1
- package/dist/{statistical-heldout-DFS7QGpS.d.ts → statistical-heldout-Yldkntvy.d.ts} +2 -2
- package/dist/{statistical-heldout-DFS7QGpS.d.ts.map → statistical-heldout-Yldkntvy.d.ts.map} +1 -1
- package/dist/{store-tool-spans-B2DJ_82T.d.ts → store-tool-spans-BvdUbeOB.d.ts} +3 -3
- package/dist/{store-tool-spans-B2DJ_82T.d.ts.map → store-tool-spans-BvdUbeOB.d.ts.map} +1 -1
- package/dist/supervisor-run/index.d.ts +35 -2
- package/dist/supervisor-run/index.d.ts.map +1 -1
- package/dist/supervisor-run/index.js +83 -38
- package/dist/supervisor-run/index.js.map +1 -1
- package/dist/{tool-groups-BnXlCJZQ.d.ts → tool-groups-DjwlMBvW.d.ts} +2 -2
- package/dist/tool-groups-DjwlMBvW.d.ts.map +1 -0
- package/dist/trace-repair/index.d.ts +1 -1
- package/dist/traces.d.ts +2 -2
- package/dist/{types-i21ccEkr.d.ts → types-CCZ34qmV.d.ts} +2 -2
- package/dist/{types-i21ccEkr.d.ts.map → types-CCZ34qmV.d.ts.map} +1 -1
- package/dist/{types-DeIUdzNd.d.ts → types-CoPUTiXb.d.ts} +23 -92
- package/dist/types-CoPUTiXb.d.ts.map +1 -0
- package/dist/{types-Dy237wiH.d.ts → types-nokrtr7M.d.ts} +21 -1
- package/dist/types-nokrtr7M.d.ts.map +1 -0
- package/dist/wire/index.d.ts +3 -3
- package/dist/wire/index.d.ts.map +1 -1
- package/docs/campaign-proposers.md +4 -0
- package/docs/eval-surface-map.md +36 -0
- package/docs/insight-report.md +19 -0
- package/docs/plants.md +123 -0
- package/docs/public-api.md +45 -20
- package/package.json +19 -18
- package/dist/experiment-tracker-BKEumQug.js.map +0 -1
- package/dist/experiment-tracker-Dm8yQMqb.d.ts.map +0 -1
- package/dist/index-lfaSeKSD.d.ts.map +0 -1
- package/dist/llm-judge-DbJdo8Nj.js.map +0 -1
- package/dist/provenance-CIRUardl.d.ts.map +0 -1
- package/dist/tool-groups-BnXlCJZQ.d.ts.map +0 -1
- package/dist/types-DeIUdzNd.d.ts.map +0 -1
- package/dist/types-Dy237wiH.d.ts.map +0 -1
package/CHANGELOG.md
CHANGED
|
@@ -4,6 +4,74 @@ All notable changes to `@tangle-network/agent-eval` and its sibling `agent-eval-
|
|
|
4
4
|
|
|
5
5
|
---
|
|
6
6
|
|
|
7
|
+
## [0.172.0] — 2026-09-01
|
|
8
|
+
|
|
9
|
+
### Added
|
|
10
|
+
|
|
11
|
+
- Campaign aggregates carry the spread beside the mean.
|
|
12
|
+
`JudgeAggregate` and `ScenarioAggregate` each gain a `distribution` field holding the `SeriesDistribution` value `summarizeNumberSeries` returns: `n`, `min`, `p50`, `p90`, `max`, and `sum` over the exact scores the mean was taken over.
|
|
13
|
+
A mean, a standard deviation, and a bootstrap interval cannot separate a bimodal judge from a tight one, and cannot show the outlier that carried the mean: six cells scoring `0, 0, 0, 1, 1, 1` and six cells scoring `0.5` reported identical aggregates before this change.
|
|
14
|
+
The values come from the package's existing helper, so a campaign aggregate and every other series summary here report the same fields under the same nearest-rank quantile definition.
|
|
15
|
+
The auto-PR by-judge table prints `min`, `p50`, `p90`, and `max` beside the mean for the same reason.
|
|
16
|
+
A judge that produced no score still has no entry at all, because an absent aggregate is the honest record of an unmeasured judge.
|
|
17
|
+
Closes #705.
|
|
18
|
+
- `buildCellSchedule` and its `CellScheduleSlot` type are exported from `@tangle-network/agent-eval/campaign`.
|
|
19
|
+
The function computes the `(scenario × rep)` fan-out with seed-group aware per-cell seeds and touches no filesystem, but it was reachable only through `planCampaignRun` and `runCampaign`, both of which refuse an empty run directory.
|
|
20
|
+
A caller that wants the grid alone — a size preview before a run directory is chosen, or a test that asserts a design's cell count and seeds — no longer has to invent a path it will not use.
|
|
21
|
+
Closes #707.
|
|
22
|
+
- `attest`, `verifyAttestation`, `ATTESTATION_ALGORITHM`, and the `AttestedReport`, `AttestationProvenance`, and `AttestationVerification` types are exported from the package root.
|
|
23
|
+
Reproducibility attestation has been implemented and tested in `src/attestation.ts` since #237 (2026-06-10), and was absent from every published entry, so a consumer building evidence receipts could not reach it from an installed copy.
|
|
24
|
+
|
|
25
|
+
### Changed
|
|
26
|
+
|
|
27
|
+
- `docs/eval-surface-map.md` states what `CampaignResult.aggregates` reports, how to read the cell grid without a run directory, and what `attest` binds.
|
|
28
|
+
`docs/insight-report.md` states why the report's `ScalarDistribution` and the campaign's `SeriesDistribution` stay separate shapes: the first is a strict published wire contract that also carries a histogram, `tailRuns`, and `p95`, and the two answer differently at `n = 0`.
|
|
29
|
+
- This release is a minor, not a patch. 0.171.1 stayed on the patch line because its export was purely additive, and this one is not: `JudgeAggregate` and `ScenarioAggregate` each gain a REQUIRED field, so any consumer that builds one of those values by hand must add it. A consumer that only reads a `CampaignResult` needs no change.
|
|
30
|
+
- Pin the four dependency manifests with digest `3f65588dbbec59d2c783edf245a9751653346fc67eae691d38af22c8b6b5b8ce`. Three of the four carry the version, so the lock digest moves with every release.
|
|
31
|
+
|
|
32
|
+
### Tests
|
|
33
|
+
|
|
34
|
+
- `src/paired-promotion-decision.test.ts` covers `decidePairedPromotion` directly for the first time.
|
|
35
|
+
The function decides every promotion in this package and seven modules depend on it, and it had no test of its own.
|
|
36
|
+
Twenty-two cases cover the three input validations, estimator selection across two-point, `0`-to-`100`, continuous, and forced-median outcomes, the `sufficient` pair-count floor and its rise with the confidence level, the zero-width interval refusal in both the all-tie and the identical-positive-delta form, McNemar's veto on the `n = 6, b = 5, c = 0` witness and its absence at a negative threshold, the exact-threshold boundary on both paths, and two candidates that must be refused.
|
|
37
|
+
- `src/campaign/cell-aggregates.test.ts` asserts that a bimodal cell and a tight cell with equal means produce different recorded aggregates.
|
|
38
|
+
- `src/campaign/cell-schedule.test.ts` asserts the grid, the shared seeds inside a `seedGroup`, and the cell-directory naming.
|
|
39
|
+
- The packed-export verifier binds `attest`, `verifyAttestation`, `buildCellSchedule`, and the campaign `SeriesDistribution` type from the packed tarball, and checks the two value exports are present at runtime.
|
|
40
|
+
|
|
41
|
+
---
|
|
42
|
+
|
|
43
|
+
## [0.171.1] — 2026-09-01
|
|
44
|
+
|
|
45
|
+
### Added
|
|
46
|
+
|
|
47
|
+
- `runBoundedProcess` is exported from the package entry. It runs one command under a wall-clock bound, kills the command's whole process group when that bound is reached, caps the captured output, and always resolves. The mechanics are the ones `SubprocessSandboxDriver` already had; the driver is not exported, so no consumer could reach them.
|
|
48
|
+
- The motive is measured, and it is not this package's own. `@tangle-network/agent-knowledge` graded claims through `execFile` with no process group of its own. A deadline killed the shell and left every descendant running, and those descendants held the stdout and stderr pipes, so `close` never fired and the deadline produced nothing the caller could see. Three CaDiCaL solver processes outlived their grading parent by five days. A grading loop hung twice on the same cause, for 8 hours and for 1.9 hours. agent-knowledge deletes its own `runBash` and calls this instead, which is the process half of tangle-network/agent-knowledge#181.
|
|
49
|
+
- The extraction adds the three inputs a claim grader needs and the driver never had. `shell` names the shell binary, because a check written for `bash` is a syntax error in `/bin/sh`. `envMode: 'replace'` passes exactly the named variables with no `process.env` merge, which is what a grader needs when the claim under test is about what the command could read. `signal` accepts an `AbortSignal`, kills the group the way the deadline does, and refuses to spawn at all when the signal is already aborted.
|
|
50
|
+
- A killed run never reports success. A SIGKILLed child can close with exit code 0, so a run killed by the deadline reports 124 and a run killed by an abort reports 137 whenever the closed code is 0 or absent. `killedByTimeout` and `killedBySignal` say which kill happened.
|
|
51
|
+
- `src/bounded-process.test.ts` carries the control that makes the guarantee falsifiable. It spawns the same `sleep 60 & wait` command **without** `detached`, kills the shell alone, and asserts that the descendant is still alive and that `close` never fired. That is the failure the group kill prevents, reproduced in the same file that proves the fix.
|
|
52
|
+
|
|
53
|
+
### Changed
|
|
54
|
+
|
|
55
|
+
- `SubprocessSandboxDriver.exec` calls `runBoundedProcess`. Every `SandboxResult` field, both defaults (a 10-minute deadline and a 16 MiB output cap), the env merge onto `process.env`, and the rule that a timed-out phase never reaches the test parser are unchanged. One implementation of the process-group kill now serves both the driver and any consumer.
|
|
56
|
+
- The group kill no longer logs an `ESRCH`. `ESRCH` means the group exited between the deadline and the signal, which `close` reports on its own. A kill that fails for any other reason still warns, because the group may still be running.
|
|
57
|
+
- This release is a patch. The export is additive and the package is pre-1.0, so the compatibility boundary sits at the minor: a `0.172.0` would fall outside the `>=0.171.0 <0.172.0` peer range consumers declare and would force an unrelated cohort move.
|
|
58
|
+
- Pin the four dependency manifests with digest `1ef007066bb5c110a51272c0e44cfbd3857ce028b60b96ed6b70d6dbef1d7d8e`. Three of the four carry the version, so the lock digest moves with every release.
|
|
59
|
+
|
|
60
|
+
---
|
|
61
|
+
|
|
62
|
+
## [0.171.0] — 2026-08-31
|
|
63
|
+
|
|
64
|
+
### Changed
|
|
65
|
+
|
|
66
|
+
- Compile and publish the exported `AgentProfile` contract against `@tangle-network/agent-interface` 2.
|
|
67
|
+
This keeps Eval, Runtime, and consumers on one canonical profile type.
|
|
68
|
+
- Refresh direct dependencies to their latest approved versions.
|
|
69
|
+
- Pin Zod 4.5.4 through a version-scoped release-age exception for this compatibility cohort.
|
|
70
|
+
- Set Node.js 20.19.0 as the minimum version required by the dependency graph.
|
|
71
|
+
- Pin the four dependency manifests with digest `43f2e94849ab4d777ac8c0c358c6c4f0687129c322c0565fb5cc8583922a433d`.
|
|
72
|
+
|
|
73
|
+
---
|
|
74
|
+
|
|
7
75
|
## [0.170.0] — 2026-08-21
|
|
8
76
|
|
|
9
77
|
### Added
|
package/README.md
CHANGED
|
@@ -105,6 +105,7 @@ Every row is a function you call. Each links to a runnable example.
|
|
|
105
105
|
| [`deltaRepair()`](./docs/trace-repair-grader.md) — a finding must be graded by executing the repair it proposes | a trajectory, an analyst finding, a sandbox | the repair's measured effect against a no-fix control |
|
|
106
106
|
| [`replayVerify()`](./docs/trajectory-replay.md) — you must know whether a recorded failure still reproduces | a recorded shell trajectory and its pinned image | a re-execution verdict and the divergences found |
|
|
107
107
|
| [`analyzeSupervisorRun()`](./docs/adapters-observability.md) — a recursive or supervised run directory must be read | a run directory | counts that stay missing when a measurement is missing, never zero |
|
|
108
|
+
| [`seedPlants()` / `catchRate()`](./docs/plants.md) — you must know whether the grader catches a wrong answer, not only how the work scored | a grading set and items authored wrong by construction | a sealed manifest, then a catch rate that refuses rather than guessing |
|
|
108
109
|
| [`buildRlDataset()`](./examples/publish-rl-dataset/) — scored runs should become training data | run records and preferences | reward, preference, and supervised rows |
|
|
109
110
|
|
|
110
111
|
## Configure Model Calls
|
|
@@ -148,6 +149,7 @@ and [DSPy](./docs/campaign-proposers.md#use-official-dspy-optimizers).
|
|
|
148
149
|
| `@tangle-network/agent-eval/traces` | Store, replay, and inspect structured traces. |
|
|
149
150
|
| `@tangle-network/agent-eval/reporting` | Statistical comparisons and report rendering. |
|
|
150
151
|
| `@tangle-network/agent-eval/supervisor-run` | Read recursive run directories without collapsing missing measurements to zero. |
|
|
152
|
+
| `@tangle-network/agent-eval/meta-eval` | Measure the grader itself: judge calibration, sentinels, and seeded known-wrong plants. |
|
|
151
153
|
| `@tangle-network/agent-eval/profile-cell` | Create and validate portable agent-profile identities. |
|
|
152
154
|
| `@tangle-network/agent-eval/ledger-core` | Append-only hash-chained journal with idempotent append and chain verification. |
|
|
153
155
|
| `@tangle-network/agent-eval/benchmarks` | Benchmark adapters and retrieval metrics. |
|
|
@@ -170,6 +172,7 @@ Use a subpath when you want an explicit capability boundary.
|
|
|
170
172
|
| How do I register an experiment as a sealed object? | [`docs/experiment.md`](./docs/experiment.md) |
|
|
171
173
|
| How is something certified without an answer key? | [`docs/verification-strategies.md`](./docs/verification-strategies.md) |
|
|
172
174
|
| Where does every verifier land its result? | [`docs/verdicts.md`](./docs/verdicts.md) |
|
|
175
|
+
| Does the grader catch a claim that is known to be wrong? | [`docs/plants.md`](./docs/plants.md) |
|
|
173
176
|
| How do I turn a coding-agent session log into runs? | [`docs/code-agent-intake.md`](./docs/code-agent-intake.md) |
|
|
174
177
|
| How do I score a string from another language? | [`docs/wire-protocol.md`](./docs/wire-protocol.md) |
|
|
175
178
|
|
package/dist/adapters/http.d.ts
CHANGED
|
@@ -1,5 +1,5 @@
|
|
|
1
|
-
import { R as Scenario, d as DispatchContext, f as DispatchFn } from "../types-
|
|
2
|
-
import "../index-
|
|
1
|
+
import { R as Scenario, d as DispatchContext, f as DispatchFn } from "../types-nokrtr7M.js";
|
|
2
|
+
import "../index-DnglhM0A.js";
|
|
3
3
|
//#region src/adapters/http.d.ts
|
|
4
4
|
interface HttpDispatchOptions<TScenario extends Scenario, _TArtifact> {
|
|
5
5
|
/** Static endpoint URL. Mutually exclusive with `resolveUrl`. */
|
package/dist/analyst/index.d.ts
CHANGED
|
@@ -2,12 +2,12 @@ import { b as CustomTokenPricing, c as CostLedgerHandle } from "../cost-ledger-D
|
|
|
2
2
|
import { I as TraceAnalystSpan, _ as ProposalFinding, a as AnalystInputKind, b as makeFinding, c as AnalystRunInputs, d as AnalystSeverity, f as AnalystUsageReceipt, g as ExecutionProbeRequest, h as ExecutionProbeOutcome, i as AnalystFinding, l as AnalystRunResult, m as ExecutionProbe, n as AnalystContext, o as AnalystRequirements, p as EvidenceRef, r as AnalystCost, s as AnalystRunEvent, t as Analyst, u as AnalystRunSummary, v as ProposalFindingOrigin, w as TraceAnalysisStore, x as makeProposalFinding, y as computeFindingId } from "../types-DMoNFDWi.js";
|
|
3
3
|
import { _ as CreateChatClientOpts, b as SandboxSdkTransportOpts, f as ChatCallOpts, g as ChatTransport, h as ChatResponse, m as ChatRequest, p as ChatClient, v as CustomTransportOpts, x as createChatClient, y as MockTransportOpts } from "../types-Bfk0uxRj.js";
|
|
4
4
|
import { f as ExternalOptimizerModelExecutionObservation, h as ExternalOptimizerRunnerCommand, l as ExternalOptimizerModelCall } from "../external-optimizer-contracts-szBJ_1vh.js";
|
|
5
|
-
import { a as createTraceAnalyst, c as BehavioralAnalystOptions, i as TraceAnalystDefinition, l as behavioralAnalyst, n as buildDefaultAnalystRegistry, o as renderPriorFindings, r as CreateTraceAnalystOptions, s as runTraceAnalyst, t as DefaultAnalystRegistryOptions, u as deriveEfficiencyFindings } from "../default-registry-
|
|
5
|
+
import { a as createTraceAnalyst, c as BehavioralAnalystOptions, i as TraceAnalystDefinition, l as behavioralAnalyst, n as buildDefaultAnalystRegistry, o as renderPriorFindings, r as CreateTraceAnalystOptions, s as runTraceAnalyst, t as DefaultAnalystRegistryOptions, u as deriveEfficiencyFindings } from "../default-registry-XxedTLwu.js";
|
|
6
6
|
import { a as ExactAnalystRunPolicySnapshot, c as ExactAnalystSnapshot, d as ExactExecutionComponentSnapshot, i as ExactAnalystRunEvent, l as ExactCapableAnalyst, n as ExactAnalystExecutionPlanSnapshot, o as ExactAnalystRunResult, r as ExactAnalystRunCompletion, s as ExactAnalystRunSummary, t as ExactAnalystBudgetSnapshot, u as ExactExecutionComponentIdentity } from "../exact-types-BEecmnWm.js";
|
|
7
7
|
import { a as ExactAnalystBudgetPolicy, c as RegistryRunOpts, i as BudgetPolicy, n as AnalystRegistry, o as ExactAnalystRunExecutionError, r as AnalystRegistryOptions, s as ExactRegistryRunOpts, t as AnalystHooks } from "../registry-xEb_xfns.js";
|
|
8
|
-
import { a as resolveTraceAnalystLimits, c as RawAnalystFinding, d as parseRawFinding, i as TraceAnalystLimits, l as RawAnalystFindingSchema, n as TraceAnalysisEngineRequest, o as RAW_FINDING_SCHEMA_PROMPT, r as TraceAnalysisEngineResult, s as RawAnalystEvidence, t as TraceAnalysisEngine, u as evidenceRefsFromRawFinding } from "../engine-
|
|
9
|
-
import { n as buildTraceToolsForGroup, t as TraceToolGroupName } from "../tool-groups-
|
|
10
|
-
import { C as DspyRlmTraceEngineOptions, O as SemanticConceptJudgeInput, S as renderFindingSubject, _ as FindingSubject, a as FAILURE_MODE_KIND_SPEC, b as findingSubjectGrammarPromptFor, c as emitControlIntegrityFindings, d as FindingsStore, f as PersistedFinding, g as FINDING_SUBJECT_SYNTAX, h as FINDING_SUBJECT_KINDS, i as IMPROVEMENT_KIND_SPEC, k as SemanticConceptJudgeOptions, l as DiffPolicy, m as diffFindings, n as KNOWLEDGE_POISONING_KIND_SPEC, o as CONTROL_INTEGRITY_ANALYST, p as defaultIsMaterial, r as KNOWLEDGE_GAP_KIND_SPEC, s as ControlIntegrityAnalyst, t as DEFAULT_TRACE_ANALYST_KINDS, u as FindingsDiff, v as FindingSubjectKind, w as createDspyRlmTraceEngine, x as parseFindingSubject, y as KIND_EXPECTED_SUBJECTS } from "../index-
|
|
8
|
+
import { a as resolveTraceAnalystLimits, c as RawAnalystFinding, d as parseRawFinding, i as TraceAnalystLimits, l as RawAnalystFindingSchema, n as TraceAnalysisEngineRequest, o as RAW_FINDING_SCHEMA_PROMPT, r as TraceAnalysisEngineResult, s as RawAnalystEvidence, t as TraceAnalysisEngine, u as evidenceRefsFromRawFinding } from "../engine-BfRay1qD.js";
|
|
9
|
+
import { n as buildTraceToolsForGroup, t as TraceToolGroupName } from "../tool-groups-DjwlMBvW.js";
|
|
10
|
+
import { C as DspyRlmTraceEngineOptions, O as SemanticConceptJudgeInput, S as renderFindingSubject, _ as FindingSubject, a as FAILURE_MODE_KIND_SPEC, b as findingSubjectGrammarPromptFor, c as emitControlIntegrityFindings, d as FindingsStore, f as PersistedFinding, g as FINDING_SUBJECT_SYNTAX, h as FINDING_SUBJECT_KINDS, i as IMPROVEMENT_KIND_SPEC, k as SemanticConceptJudgeOptions, l as DiffPolicy, m as diffFindings, n as KNOWLEDGE_POISONING_KIND_SPEC, o as CONTROL_INTEGRITY_ANALYST, p as defaultIsMaterial, r as KNOWLEDGE_GAP_KIND_SPEC, s as ControlIntegrityAnalyst, t as DEFAULT_TRACE_ANALYST_KINDS, u as FindingsDiff, v as FindingSubjectKind, w as createDspyRlmTraceEngine, x as parseFindingSubject, y as KIND_EXPECTED_SUBJECTS } from "../index-DDAPhUJJ.js";
|
|
11
11
|
import { C as scoreAnalystFindings, S as traceStoreEvidenceResolver, _ as AnalystIssueExpectation, a as AnalystBenchmarkLabelState, b as registryBenchmarkRunner, c as AnalystBenchmarkProvenance, d as AnalystBenchmarkSummary, f as AnalystEvidenceExpectation, g as AnalystFindingScore, h as AnalystEvidenceResolver, i as AnalystBenchmarkError, l as AnalystBenchmarkResult, m as AnalystEvidenceResolutionError, n as AnalystBenchmarkDatasetRef, o as AnalystBenchmarkObservation, p as AnalystEvidenceResolution, r as AnalystBenchmarkDescriptor, s as AnalystBenchmarkOutput, t as AnalystBenchmarkCase, u as AnalystBenchmarkRunner, v as AnalystLatencyDistribution, x as runAnalystBenchmark, y as RunAnalystBenchmarkOptions } from "../benchmark-h-h4bfqj.js";
|
|
12
12
|
import { t as AgentProfile } from "../agent-profile-B9_GGsG8.js";
|
|
13
13
|
import { i as nodeHttpPrimeBridgeTransport, n as PrimeBridgeTransportRequest, r as PrimeBridgeTransportResult, t as PrimeBridgeTransport } from "../prime-bridge-transport-6feEglLf.js";
|
|
@@ -1267,7 +1267,7 @@ declare function runAnalystBenchmarkCommand(argv: readonly string[], env?: NodeJ
|
|
|
1267
1267
|
declare const ANALYST_BENCHMARK_IMPLEMENTATION_DIGEST_ALGORITHM = "sha256-canonical-source-manifest";
|
|
1268
1268
|
declare const ANALYST_BENCHMARK_DEPENDENCY_LOCK_DIGEST_ALGORITHM = "sha256-canonical-file-manifest";
|
|
1269
1269
|
declare const ANALYST_BENCHMARK_DEPENDENCY_LOCK_FILES: readonly string[];
|
|
1270
|
-
declare const ANALYST_BENCHMARK_DEPENDENCY_LOCK_SHA256 = "
|
|
1270
|
+
declare const ANALYST_BENCHMARK_DEPENDENCY_LOCK_SHA256 = "3f65588dbbec59d2c783edf245a9751653346fc67eae691d38af22c8b6b5b8ce";
|
|
1271
1271
|
declare const ANALYST_BENCHMARK_EVIDENCE_DEPENDENCY_LOCK_SHA256 = "1e03f2daed356d60316aabefb407ec1e437ac94d408d61eea4ae096e9c6fbb5b";
|
|
1272
1272
|
declare const ANALYST_BENCHMARK_EVIDENCE_IMPLEMENTATION_SHA256 = "4dba263b6256a30d56c7fdb2d992d3a953c0035d731f359b704db806f68f75ac";
|
|
1273
1273
|
declare const ANALYST_BENCHMARK_IMPLEMENTATION_FILES: readonly string[];
|
package/dist/analyst/index.js
CHANGED
|
@@ -2,11 +2,11 @@ import { i as CostLedger } from "../cost-ledger-B1qx30B4.js";
|
|
|
2
2
|
import { n as isProposalFinding, t as assertProposalFindings } from "../proposal-findings-bko3GGy-.js";
|
|
3
3
|
import { c as validateUsageSettlementTimeout, i as makeProposalFinding, n as computeFindingId, o as settleUsageReceiptFromCostLedger, r as makeFinding } from "../types-CiWITkGo.js";
|
|
4
4
|
import { A as FINDING_SUBJECT_SYNTAX, C as RAW_FINDING_SCHEMA_PROMPT, D as coerceJson, E as parseRawFinding, F as resolveTraceAnalystLimits, M as findingSubjectGrammarPromptFor, N as parseFindingSubject, O as stripCodeFences, P as renderFindingSubject, T as evidenceRefsFromRawFinding, i as buildTraceToolsForGroup, j as KIND_EXPECTED_SUBJECTS, k as FINDING_SUBJECT_KINDS, n as renderPriorFindings, r as runTraceAnalyst, t as createTraceAnalyst, w as RawAnalystFindingSchema } from "../kind-factory-DMeEoMQZ.js";
|
|
5
|
-
import { a as DEFAULT_TRACE_ANALYST_KINDS, c as IMPROVEMENT_KIND_SPEC, d as ControlIntegrityAnalyst, f as emitControlIntegrityFindings, i as ExactAnalystRunExecutionError, l as FAILURE_MODE_KIND_SPEC, m as deriveEfficiencyFindings, n as buildDefaultAnalystRegistry, o as KNOWLEDGE_POISONING_KIND_SPEC, p as behavioralAnalyst, r as AnalystRegistry, s as KNOWLEDGE_GAP_KIND_SPEC, t as createChatClient, u as CONTROL_INTEGRITY_ANALYST } from "../chat-client-
|
|
5
|
+
import { a as DEFAULT_TRACE_ANALYST_KINDS, c as IMPROVEMENT_KIND_SPEC, d as ControlIntegrityAnalyst, f as emitControlIntegrityFindings, i as ExactAnalystRunExecutionError, l as FAILURE_MODE_KIND_SPEC, m as deriveEfficiencyFindings, n as buildDefaultAnalystRegistry, o as KNOWLEDGE_POISONING_KIND_SPEC, p as behavioralAnalyst, r as AnalystRegistry, s as KNOWLEDGE_GAP_KIND_SPEC, t as createChatClient, u as CONTROL_INTEGRITY_ANALYST } from "../chat-client-DI79OPye.js";
|
|
6
6
|
import { t as createDspyRlmTraceEngine } from "../dspy-rlm-engine-CS3qcCEk.js";
|
|
7
7
|
import { a as diffFindings, i as defaultIsMaterial, n as runSemanticConceptJudge, r as FindingsStore, t as SEMANTIC_CONCEPT_JUDGE_VERSION } from "../semantic-concept-judge-I36eejJx.js";
|
|
8
8
|
import { a as scoreAnalystFindings, i as summarizeAnalystBenchmarkRunner, n as runAnalystBenchmark, r as traceStoreEvidenceResolver, t as registryBenchmarkRunner } from "../benchmark-C4wk_Sjr.js";
|
|
9
|
-
import { $ as ANALYST_BENCHMARK_LOCAL_RECEIPT_FILE, A as effectiveAnalystProtocolSha256, B as loadCodeTraceVerificationArtifacts, C as adaptPublicBenchmarkFindings, D as renderCodeTraceCalibrationMarkdown, E as readAnalystBenchmarkArtifact, F as publicBenchmarkProtocolSha256, G as ANALYST_BENCHMARK_EVIDENCE_DEPENDENCY_LOCK_SHA256, H as ANALYST_BENCHMARK_DEPENDENCY_LOCK_DIGEST_ALGORITHM, I as publicBenchmarkRlmInstructions, J as ANALYST_BENCHMARK_IMPLEMENTATION_FILES, K as ANALYST_BENCHMARK_EVIDENCE_IMPLEMENTATION_SHA256, L as publicBenchmarkSystemPrompt, M as CODE_TRACE_BENCH_ANALYST_PROMPT, N as MAX_INCORRECT_BLOCKS, O as summarizeCodeTraceCalibration, P as MAX_INCORRECT_BLOCK_STEPS, Q as ANALYST_BENCHMARK_COST_LEDGER_FILE, R as DEFAULT_MAX_VERIFICATION_ARTIFACT_BYTES, S as analystDefinitionProtocolSha256, T as expandCodeTraceFailureBlocks, U as ANALYST_BENCHMARK_DEPENDENCY_LOCK_FILES, V as parseVerificationOutcome, W as ANALYST_BENCHMARK_DEPENDENCY_LOCK_SHA256, X as analystBenchmarkDependencyLockDigest, Y as ANALYST_BENCHMARK_IMPLEMENTATION_SHA256, Z as analystBenchmarkImplementationDigest, _ as createPublicBenchmarkDirectRunner, a as primeCodeTraceAnalystDefinition, at as summarizeAgentRxCalibration, b as AnalystExpressivenessError, c as loadPublicBenchmarkRows, ct as agentRxBenchmarkCase, d as publicBenchmarkSelectionReport, dt as roundAgentRxStep, et as ANALYST_BENCHMARK_MANIFEST_FILE, f as selectPublicBenchmarkRows, ft as normalizeBenchmarkLabel, g as runReplVariableAnalystDefinition, h as rlmEngineLimits, i as primeAnalystProtocolSha256, it as renderAgentRxCalibrationMarkdown, j as readAnalystInstructionsOverride, k as analystInstructionsOverrideFromText, l as preparePublicAnalystBenchmark, lt as agentRxPredictionsToFindings, m as publicRlmAnalystDefinition, n as renderAnalystBenchmarkMarkdown, nt as compareAnalystRunners, o as runInlineAnalystDefinition, ot as codeTraceBenchCase, p as createPublicBenchmarkRlmRunner, q as ANALYST_BENCHMARK_IMPLEMENTATION_DIGEST_ALGORITHM, r as createPrimeBenchmarkRunner, rt as AGENT_RX_UPSTREAM_REVISION, s as nodeHttpPrimeBridgeTransport, st as codeTracerPredictionsToFindings, t as runAnalystBenchmarkCommand, tt as ANALYST_BENCHMARK_OBSERVATIONS_FILE, u as publicBenchmarkDistributions, ut as normalizeAgentRxCategory, v as publicDirectAnalystDefinition, w as emptyPublicBenchmarkRunner, x as analystDefinitionAsymmetries, y as runChunkedAnalystDefinition, z as appendVerificationArtifactsToOtlp } from "../benchmark-command-
|
|
9
|
+
import { $ as ANALYST_BENCHMARK_LOCAL_RECEIPT_FILE, A as effectiveAnalystProtocolSha256, B as loadCodeTraceVerificationArtifacts, C as adaptPublicBenchmarkFindings, D as renderCodeTraceCalibrationMarkdown, E as readAnalystBenchmarkArtifact, F as publicBenchmarkProtocolSha256, G as ANALYST_BENCHMARK_EVIDENCE_DEPENDENCY_LOCK_SHA256, H as ANALYST_BENCHMARK_DEPENDENCY_LOCK_DIGEST_ALGORITHM, I as publicBenchmarkRlmInstructions, J as ANALYST_BENCHMARK_IMPLEMENTATION_FILES, K as ANALYST_BENCHMARK_EVIDENCE_IMPLEMENTATION_SHA256, L as publicBenchmarkSystemPrompt, M as CODE_TRACE_BENCH_ANALYST_PROMPT, N as MAX_INCORRECT_BLOCKS, O as summarizeCodeTraceCalibration, P as MAX_INCORRECT_BLOCK_STEPS, Q as ANALYST_BENCHMARK_COST_LEDGER_FILE, R as DEFAULT_MAX_VERIFICATION_ARTIFACT_BYTES, S as analystDefinitionProtocolSha256, T as expandCodeTraceFailureBlocks, U as ANALYST_BENCHMARK_DEPENDENCY_LOCK_FILES, V as parseVerificationOutcome, W as ANALYST_BENCHMARK_DEPENDENCY_LOCK_SHA256, X as analystBenchmarkDependencyLockDigest, Y as ANALYST_BENCHMARK_IMPLEMENTATION_SHA256, Z as analystBenchmarkImplementationDigest, _ as createPublicBenchmarkDirectRunner, a as primeCodeTraceAnalystDefinition, at as summarizeAgentRxCalibration, b as AnalystExpressivenessError, c as loadPublicBenchmarkRows, ct as agentRxBenchmarkCase, d as publicBenchmarkSelectionReport, dt as roundAgentRxStep, et as ANALYST_BENCHMARK_MANIFEST_FILE, f as selectPublicBenchmarkRows, ft as normalizeBenchmarkLabel, g as runReplVariableAnalystDefinition, h as rlmEngineLimits, i as primeAnalystProtocolSha256, it as renderAgentRxCalibrationMarkdown, j as readAnalystInstructionsOverride, k as analystInstructionsOverrideFromText, l as preparePublicAnalystBenchmark, lt as agentRxPredictionsToFindings, m as publicRlmAnalystDefinition, n as renderAnalystBenchmarkMarkdown, nt as compareAnalystRunners, o as runInlineAnalystDefinition, ot as codeTraceBenchCase, p as createPublicBenchmarkRlmRunner, q as ANALYST_BENCHMARK_IMPLEMENTATION_DIGEST_ALGORITHM, r as createPrimeBenchmarkRunner, rt as AGENT_RX_UPSTREAM_REVISION, s as nodeHttpPrimeBridgeTransport, st as codeTracerPredictionsToFindings, t as runAnalystBenchmarkCommand, tt as ANALYST_BENCHMARK_OBSERVATIONS_FILE, u as publicBenchmarkDistributions, ut as normalizeAgentRxCategory, v as publicDirectAnalystDefinition, w as emptyPublicBenchmarkRunner, x as analystDefinitionAsymmetries, y as runChunkedAnalystDefinition, z as appendVerificationArtifactsToOtlp } from "../benchmark-command-DoFcisuM.js";
|
|
10
10
|
import { a as extractPrimeJsonObject, c as primeProtocolSha256, d as runPrimeExchange, f as decodeReplyRows, i as emptyPrimeRawUsage, l as primeReplyDefect, n as buildPrimePrompt, o as mergePrimeRawUsage, r as buildPrimeRepairPrompt, s as normalizePrimeUsage, t as analystUsageReceiptFromPrimeUsage, u as projectPrimeTrajectory } from "../prime-protocol-6tZTVsWm.js";
|
|
11
11
|
import { existsSync, readFileSync, readdirSync, statSync } from "node:fs";
|
|
12
12
|
import { join } from "node:path";
|
|
@@ -233,5 +233,81 @@ declare class ExperimentTracker {
|
|
|
233
233
|
verdictFor(experimentId: string): Promise<ImprovementVerdictResult>;
|
|
234
234
|
}
|
|
235
235
|
//#endregion
|
|
236
|
-
|
|
237
|
-
|
|
236
|
+
//#region src/attestation.d.ts
|
|
237
|
+
/**
|
|
238
|
+
* Reproducibility attestation for any serializable report object.
|
|
239
|
+
*
|
|
240
|
+
* `attest()` binds a report to its content address (sha-256 over canonical
|
|
241
|
+
* JSON) AND binds that address to the provenance needed to reproduce it:
|
|
242
|
+
* model versions, seeds, price-table hash, code SHA, inputs hash. The outer
|
|
243
|
+
* `envelopeHash` prevents provenance from being rewritten while leaving the
|
|
244
|
+
* report hash valid.
|
|
245
|
+
*
|
|
246
|
+
* Layering: content-addressing is the substrate's job; cryptographic SIGNING
|
|
247
|
+
* (who vouches for the attestation, key management, transparency logs) is the
|
|
248
|
+
* consumer's layer on top. An `AttestedReport` is a stable byte-identical
|
|
249
|
+
* payload a consumer can sign — the substrate never holds keys.
|
|
250
|
+
*
|
|
251
|
+
* Generic by design: the report parameter is ANY value `canonicalJson`
|
|
252
|
+
* accepts (campaign results, fuzz capsules, scorecards, cost ledgers). Do not
|
|
253
|
+
* couple this module to a specific report schema.
|
|
254
|
+
*/
|
|
255
|
+
/** Hash scheme identifier carried by every attestation. A verifier rejects
|
|
256
|
+
* unknown algorithms instead of guessing. */
|
|
257
|
+
declare const ATTESTATION_ALGORITHM: 'sha256/canonical-json';
|
|
258
|
+
interface AttestationProvenance {
|
|
259
|
+
/** Every model involved in producing the report, name → version/id. */
|
|
260
|
+
modelVersions: Record<string, string>;
|
|
261
|
+
/** RNG seeds the run was driven by, when seeded. */
|
|
262
|
+
seeds?: number[];
|
|
263
|
+
/** Content hash of the price table used for cost figures — cost numbers
|
|
264
|
+
* are only reproducible against the same prices. */
|
|
265
|
+
priceTableHash?: string;
|
|
266
|
+
/** Git SHA of the code that produced the report. */
|
|
267
|
+
codeSha: string;
|
|
268
|
+
/** Content hash of the input set (scenarios, dataset manifest, ...). */
|
|
269
|
+
inputsHash?: string;
|
|
270
|
+
/** ISO-8601 timestamp, caller-supplied — the substrate stays clock-free
|
|
271
|
+
* so attestation is deterministic and testable. */
|
|
272
|
+
createdAt: string;
|
|
273
|
+
}
|
|
274
|
+
interface AttestedReport {
|
|
275
|
+
/** Hex sha-256 over the canonical JSON of the report. */
|
|
276
|
+
reportHash: string;
|
|
277
|
+
provenance: AttestationProvenance;
|
|
278
|
+
algorithm: typeof ATTESTATION_ALGORITHM;
|
|
279
|
+
/**
|
|
280
|
+
* Hex sha-256 over `{ reportHash, provenance, algorithm }`. New attestations
|
|
281
|
+
* always carry it. Optional only so persisted pre-envelope attestations can
|
|
282
|
+
* still be read and explicitly recognized as legacy by callers.
|
|
283
|
+
*/
|
|
284
|
+
envelopeHash?: string;
|
|
285
|
+
}
|
|
286
|
+
interface AttestationVerification {
|
|
287
|
+
valid: boolean;
|
|
288
|
+
/** Populated iff `valid` is false — names the exact mismatch. */
|
|
289
|
+
reason?: string;
|
|
290
|
+
/** True only for a valid pre-envelope attestation whose provenance is not cryptographically bound. */
|
|
291
|
+
legacyUnboundProvenance?: true;
|
|
292
|
+
}
|
|
293
|
+
/**
|
|
294
|
+
* Content-address a report and bind it to its provenance. Throws (via
|
|
295
|
+
* `canonicalJson`) if the report or provenance contains undefined / function /
|
|
296
|
+
* symbol / non-finite numbers — an attestation that cannot be unambiguously
|
|
297
|
+
* serialized cannot be trusted.
|
|
298
|
+
*/
|
|
299
|
+
declare function attest(report: unknown, provenance: AttestationProvenance): AttestedReport;
|
|
300
|
+
/**
|
|
301
|
+
* Verify a report against its attestation. Returns a typed outcome rather
|
|
302
|
+
* than throwing: an unverifiable report (e.g. one that no longer
|
|
303
|
+
* canonicalizes) is a verification failure with the cause in `reason`, not a
|
|
304
|
+
* crash — verifiers run in pipelines that must record WHY, not die.
|
|
305
|
+
*
|
|
306
|
+
* Legacy attestations without `envelopeHash` remain readable, but verification
|
|
307
|
+
* explicitly marks their provenance as unbound so a promotion path can refuse
|
|
308
|
+
* them instead of accidentally treating old metadata as cryptographic proof.
|
|
309
|
+
*/
|
|
310
|
+
declare function verifyAttestation(report: unknown, attested: AttestedReport): AttestationVerification;
|
|
311
|
+
//#endregion
|
|
312
|
+
export { requiredSampleSize as C, requiredPairedSampleSize as S, inMemoryExperimentStore as _, attest as a, mcnemarRequiredN as b, ExperimentRep as c, ExperimentVerdict as d, ImprovementThresholds as f, improvementVerdict as g, fileExperimentStore as h, AttestedReport as i, ExperimentStats as l, computeExperimentStats as m, AttestationProvenance as n, verifyAttestation as o, ImprovementVerdictResult as p, AttestationVerification as r, Experiment as s, ATTESTATION_ALGORITHM as t, ExperimentTracker as u, mulberry32 as v, pairedMde as x, mcnemarPower as y };
|
|
313
|
+
//# sourceMappingURL=attestation-CJBGmMVh.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"attestation-CJBGmMVh.d.ts","names":[],"sources":["../src/statistics/power-and-mde.ts","../src/statistics/random.ts","../src/experiment-tracker.ts","../src/attestation.ts"],"mappings":";;;;;;;;iBAWgB,mBAAmB;EACjC;EACA;EACA;EACA;;;;;;;;;;;iBAsBc,yBAAyB;EACvC;EACA;EACA;EACA;;;;;;;;iBAkBc,UAAU;EACxB;EACA;EACA;EACA;;;;;;;;;;;;;;;;iBAyBc,iBAAiB;EAC/B;EACA;EACA;EACA;EACA;;;;;;;iBAyBc,aAAa;EAC3B;EACA;EACA;EACA;EACA;;;;;;;;iBCrHc,WAAW;;;;;KCyBf;;UAGK;;EAEf;;EAEA;;EAEA;;;;UAKe;;EAEf;;EAEA;;EAEA;;EAEA;;EAEA,WAAW;;EAEX;;EAEA,UAAU;;UAGK;EACf;EACA;EACA;EACA;;EAEA;;EAEA;;;EAGA;;EAEA;;EAEA;;UAGe;;EAEf;;EAEA;;EAEA,YAAY;;EAEZ;;EAEA;EACA,MAAM;EACN,OAAO;EACP,SAAS;;EAET;;UAGe;;EAEf;;EAEA;;EAEA;;EAEA;;;EAGA;;UAGe;EACf,SAAS;;EAET;;EAEA;;;;;;;iBAkDc,uBACd,MAAM,iBACN,aAAa,wBACZ;;;;;;;iBAgDa,mBACd,WAAW,iBACX,QAAQ,wBACR,aAAa,wBACZ;;;KA6CS,yBAAyB,uBAAuB,QAAQ;;;UAInD;EACf,QAAQ,QAAQ;EAChB,KAAK,aAAa,eAAe;;;;iBAoBnB,wBAAwB,UAAS,eAAoB;;iBAarD,oBAAoB,eAAe;UAyBlC;EACf,QAAQ;EACR,mBAAmB;EACnB,aAAa;;EAEb;;UAGe;EACf;EACA;EACA;EACA;;EAEA,aAAa;;;;;;;;;cAUF;mBACM;mBACA;mBACA;mBACA;EAEjB,YAAY,UAAS;EAOf,OAAO,OAAO,wBAAwB,QAAQ;;;EA6B9C,OACJ,sBACA,KAAK,KAAK;IAAwC;IAAc;MAC/D,QAAQ;EAqCL,IAAI,uBAAuB,QAAQ;EAMnC,QAAQ,QAAQ;;EAKhB,WAAW,uBAAuB,QAAQ;;;;;;;;;;;;;;;;;;;;;;;;cCzarC;UAEI;;EAEf,eAAe;;EAEf;;;EAGA;;EAEA;;EAEA;;;EAGA;;UAGe;;EAEf;EACA,YAAY;EACZ,kBAAkB;;;;;;EAMlB;;UAGe;EACf;;EAEA;;EAEA;;;;;;;;iBAiBc,OAAO,iBAAiB,YAAY,wBAAwB;;;;;;;;;;;iBAqB5D,kBACd,iBACA,UAAU,iBACT"}
|
|
@@ -1,5 +1,6 @@
|
|
|
1
1
|
import { s as ValidationError } from "./errors-Dngq5h35.js";
|
|
2
2
|
import { n as iqr } from "./baseline-BC-eBZ7U.js";
|
|
3
|
+
import { n as contentHash } from "./verdict-cache-B3eCVQtY.js";
|
|
3
4
|
import { execSync } from "node:child_process";
|
|
4
5
|
//#region src/experiment-tracker.ts
|
|
5
6
|
/**
|
|
@@ -275,6 +276,99 @@ var ExperimentTracker = class {
|
|
|
275
276
|
}
|
|
276
277
|
};
|
|
277
278
|
//#endregion
|
|
278
|
-
|
|
279
|
+
//#region src/attestation.ts
|
|
280
|
+
/**
|
|
281
|
+
* Reproducibility attestation for any serializable report object.
|
|
282
|
+
*
|
|
283
|
+
* `attest()` binds a report to its content address (sha-256 over canonical
|
|
284
|
+
* JSON) AND binds that address to the provenance needed to reproduce it:
|
|
285
|
+
* model versions, seeds, price-table hash, code SHA, inputs hash. The outer
|
|
286
|
+
* `envelopeHash` prevents provenance from being rewritten while leaving the
|
|
287
|
+
* report hash valid.
|
|
288
|
+
*
|
|
289
|
+
* Layering: content-addressing is the substrate's job; cryptographic SIGNING
|
|
290
|
+
* (who vouches for the attestation, key management, transparency logs) is the
|
|
291
|
+
* consumer's layer on top. An `AttestedReport` is a stable byte-identical
|
|
292
|
+
* payload a consumer can sign — the substrate never holds keys.
|
|
293
|
+
*
|
|
294
|
+
* Generic by design: the report parameter is ANY value `canonicalJson`
|
|
295
|
+
* accepts (campaign results, fuzz capsules, scorecards, cost ledgers). Do not
|
|
296
|
+
* couple this module to a specific report schema.
|
|
297
|
+
*/
|
|
298
|
+
/** Hash scheme identifier carried by every attestation. A verifier rejects
|
|
299
|
+
* unknown algorithms instead of guessing. */
|
|
300
|
+
const ATTESTATION_ALGORITHM = "sha256/canonical-json";
|
|
301
|
+
function envelopeMaterial(reportHash, provenance, algorithm) {
|
|
302
|
+
return {
|
|
303
|
+
reportHash,
|
|
304
|
+
provenance,
|
|
305
|
+
algorithm
|
|
306
|
+
};
|
|
307
|
+
}
|
|
308
|
+
/**
|
|
309
|
+
* Content-address a report and bind it to its provenance. Throws (via
|
|
310
|
+
* `canonicalJson`) if the report or provenance contains undefined / function /
|
|
311
|
+
* symbol / non-finite numbers — an attestation that cannot be unambiguously
|
|
312
|
+
* serialized cannot be trusted.
|
|
313
|
+
*/
|
|
314
|
+
function attest(report, provenance) {
|
|
315
|
+
const reportHash = contentHash(report);
|
|
316
|
+
const algorithm = ATTESTATION_ALGORITHM;
|
|
317
|
+
return {
|
|
318
|
+
reportHash,
|
|
319
|
+
provenance,
|
|
320
|
+
algorithm,
|
|
321
|
+
envelopeHash: contentHash(envelopeMaterial(reportHash, provenance, algorithm))
|
|
322
|
+
};
|
|
323
|
+
}
|
|
324
|
+
/**
|
|
325
|
+
* Verify a report against its attestation. Returns a typed outcome rather
|
|
326
|
+
* than throwing: an unverifiable report (e.g. one that no longer
|
|
327
|
+
* canonicalizes) is a verification failure with the cause in `reason`, not a
|
|
328
|
+
* crash — verifiers run in pipelines that must record WHY, not die.
|
|
329
|
+
*
|
|
330
|
+
* Legacy attestations without `envelopeHash` remain readable, but verification
|
|
331
|
+
* explicitly marks their provenance as unbound so a promotion path can refuse
|
|
332
|
+
* them instead of accidentally treating old metadata as cryptographic proof.
|
|
333
|
+
*/
|
|
334
|
+
function verifyAttestation(report, attested) {
|
|
335
|
+
if (attested.algorithm !== "sha256/canonical-json") return {
|
|
336
|
+
valid: false,
|
|
337
|
+
reason: `unknown algorithm '${attested.algorithm}' — this verifier only checks '${ATTESTATION_ALGORITHM}'`
|
|
338
|
+
};
|
|
339
|
+
let recomputed;
|
|
340
|
+
try {
|
|
341
|
+
recomputed = contentHash(report);
|
|
342
|
+
} catch (err) {
|
|
343
|
+
return {
|
|
344
|
+
valid: false,
|
|
345
|
+
reason: `report is not canonicalizable: ${err instanceof Error ? err.message : String(err)}`
|
|
346
|
+
};
|
|
347
|
+
}
|
|
348
|
+
if (recomputed !== attested.reportHash) return {
|
|
349
|
+
valid: false,
|
|
350
|
+
reason: `report hash mismatch: attested ${attested.reportHash}, recomputed ${recomputed}`
|
|
351
|
+
};
|
|
352
|
+
if (attested.envelopeHash === void 0) return {
|
|
353
|
+
valid: true,
|
|
354
|
+
legacyUnboundProvenance: true
|
|
355
|
+
};
|
|
356
|
+
let envelopeHash;
|
|
357
|
+
try {
|
|
358
|
+
envelopeHash = contentHash(envelopeMaterial(attested.reportHash, attested.provenance, attested.algorithm));
|
|
359
|
+
} catch (err) {
|
|
360
|
+
return {
|
|
361
|
+
valid: false,
|
|
362
|
+
reason: `attestation provenance is not canonicalizable: ${err instanceof Error ? err.message : String(err)}`
|
|
363
|
+
};
|
|
364
|
+
}
|
|
365
|
+
if (envelopeHash !== attested.envelopeHash) return {
|
|
366
|
+
valid: false,
|
|
367
|
+
reason: `attestation envelope hash mismatch: attested ${attested.envelopeHash}, recomputed ${envelopeHash}`
|
|
368
|
+
};
|
|
369
|
+
return { valid: true };
|
|
370
|
+
}
|
|
371
|
+
//#endregion
|
|
372
|
+
export { computeExperimentStats as a, inMemoryExperimentStore as c, ExperimentTracker as i, attest as n, fileExperimentStore as o, verifyAttestation as r, improvementVerdict as s, ATTESTATION_ALGORITHM as t };
|
|
279
373
|
|
|
280
|
-
//# sourceMappingURL=
|
|
374
|
+
//# sourceMappingURL=attestation-XSUpbc4o.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"attestation-XSUpbc4o.js","names":[],"sources":["../src/experiment-tracker.ts","../src/attestation.ts"],"sourcesContent":["/**\n * Experiment tracker — git-provenanced experiment log with N-rep stats and a\n * KEEP / REGRESSION / NOISE verdict against a parent.\n *\n * Every loop the fleet runs reduces to the same question: \"I ran the candidate\n * N times — is the median measurably better than the parent, or is the delta\n * inside the noise band?\" The hand-rolled copies bake a fixed score scale\n * (percentage points), a fixed store path (`.evolve/experiments-v2.json`), and\n * `execSync('git …')` straight into the module. This is the canonical version:\n * provenance and persistence are injected, thresholds are configurable, and the\n * stats + verdict are pure functions you can unit-test without a git repo or a\n * filesystem.\n *\n * Stats per experiment: median / mean / min / max / iqr / stddev / passRate /\n * n, plus a `stable` flag (`iqr < iqrUnstableAbove && stddev < stddevUnstableAbove`).\n *\n * Verdict against a parent (both must have `n >= minRepsForVerdict`):\n * - NOISE — the candidate is too unstable to judge (`!stable`)\n * - KEEP — `medianDelta > keepThreshold`\n * - REGRESSION — `medianDelta < -regressionThreshold`\n * - NOISE — otherwise (delta inside the band)\n * With no parent (or insufficient reps) the verdict is the neutral ITERATE.\n */\n\nimport { execSync } from 'node:child_process'\nimport type { EvidenceRef } from './analyst/types'\nimport { iqr } from './baseline'\nimport { ValidationError } from './errors'\n\n/** Verdict for one experiment relative to its parent. ITERATE is the neutral\n * \"keep collecting reps / no parent to compare against\" state. */\nexport type ExperimentVerdict = 'KEEP' | 'ITERATE' | 'NOISE' | 'REGRESSION'\n\n/** Git provenance for the working tree an experiment was run from. */\nexport interface ExperimentProvenance {\n /** Commit sha (short or full — the tracker does not interpret it). */\n commit: string\n /** First line of the commit message. */\n message: string\n /** Files changed vs the parent commit, or a marker like 'uncommitted'. */\n changedFiles: string[]\n}\n\n/** A single repetition of an experiment, carrying the score the verdict is\n * computed on plus any free-form per-rep metrics the consumer wants kept. */\nexport interface ExperimentRep {\n /** 0-indexed repetition number within the experiment. */\n rep: number\n /** The score this rep is judged on (same scale as the thresholds). */\n score: number\n /** ISO timestamp the rep completed. */\n timestamp: string\n /** Stable execution/run identity that produced this score. */\n runId?: string\n /** Mechanically resolvable trace, artifact, metric, or finding evidence. */\n evidence?: EvidenceRef[]\n /** Whether this rep passed the consumer's own gate — folded into `passRate`. */\n passed?: boolean\n /** Free-form numeric metrics retained for later analysis. */\n metrics?: Record<string, number>\n}\n\nexport interface ExperimentStats {\n median: number\n mean: number\n min: number\n max: number\n /** Inter-quartile range of the rep scores. */\n iqr: number\n /** Population standard deviation of the rep scores. */\n stddev: number\n /** Fraction of reps with `passed === true`, over reps that set `passed`.\n * null when no rep declared a pass/fail outcome. */\n passRate: number | null\n /** Number of reps. */\n n: number\n /** True when the sample is tight enough to trust for a verdict. */\n stable: boolean\n}\n\nexport interface Experiment {\n /** Stable id for the experiment. */\n id: string\n /** Free-form label / config descriptor. */\n label: string\n /** Git provenance captured when the experiment was created. */\n provenance: ExperimentProvenance\n /** Parent experiment id this candidate is compared against, if any. */\n parentId?: string\n /** One-line summary of what changed from the parent. */\n changeSummary: string\n reps: ExperimentRep[]\n stats: ExperimentStats\n verdict: ExperimentVerdict\n /** ISO timestamp the experiment was created. */\n createdAt: string\n}\n\nexport interface ImprovementThresholds {\n /** medianDelta strictly above this ⇒ KEEP. Default 5. */\n keepThreshold?: number\n /** medianDelta strictly below the negative of this ⇒ REGRESSION. Default 5. */\n regressionThreshold?: number\n /** iqr at or above this ⇒ unstable. Default 10. */\n iqrUnstableAbove?: number\n /** stddev at or above this ⇒ unstable. Default Infinity (iqr-only stability). */\n stddevUnstableAbove?: number\n /** Reps required on BOTH candidate and parent before a verdict is rendered.\n * Default 3. */\n minRepsForVerdict?: number\n}\n\nexport interface ImprovementVerdictResult {\n verdict: ExperimentVerdict\n /** candidate.median − parent.median; null when no parent or insufficient reps. */\n medianDelta: number | null\n /** Human-readable reason for the verdict — for dashboards and logs. */\n reason: string\n}\n\nconst DEFAULTS: Required<ImprovementThresholds> = {\n keepThreshold: 5,\n regressionThreshold: 5,\n iqrUnstableAbove: 10,\n stddevUnstableAbove: Number.POSITIVE_INFINITY,\n minRepsForVerdict: 3,\n}\n\nfunction resolveThresholds(t: ImprovementThresholds | undefined): Required<ImprovementThresholds> {\n const r = { ...DEFAULTS, ...(t ?? {}) }\n if (r.keepThreshold < 0) {\n throw new ValidationError(\n `experiment-tracker: keepThreshold must be >= 0, got ${r.keepThreshold}`,\n )\n }\n if (r.regressionThreshold < 0) {\n throw new ValidationError(\n `experiment-tracker: regressionThreshold must be >= 0, got ${r.regressionThreshold}`,\n )\n }\n if (r.minRepsForVerdict < 1) {\n throw new ValidationError(\n `experiment-tracker: minRepsForVerdict must be >= 1, got ${r.minRepsForVerdict}`,\n )\n }\n return r\n}\n\nfunction median(sorted: number[]): number {\n const n = sorted.length\n if (n === 0) return 0\n const mid = Math.floor(n / 2)\n return n % 2 === 0 ? (sorted[mid - 1]! + sorted[mid]!) / 2 : sorted[mid]!\n}\n\n/** Population standard deviation (÷n). 0 for fewer than 2 values. */\nfunction stddev(values: number[], mean: number): number {\n if (values.length < 2) return 0\n const variance = values.reduce((acc, v) => acc + (v - mean) ** 2, 0) / values.length\n return Math.sqrt(variance)\n}\n\n/**\n * Compute the N-rep statistics for a set of reps. Pure — no I/O. The `stable`\n * flag is the trust gate the verdict depends on: a sample whose spread exceeds\n * the configured bounds can't distinguish a real delta from run-to-run noise.\n */\nexport function computeExperimentStats(\n reps: ExperimentRep[],\n thresholds?: ImprovementThresholds,\n): ExperimentStats {\n const t = resolveThresholds(thresholds)\n const n = reps.length\n if (n === 0) {\n return {\n median: 0,\n mean: 0,\n min: 0,\n max: 0,\n iqr: 0,\n stddev: 0,\n passRate: null,\n n: 0,\n stable: false,\n }\n }\n const scores = reps.map((r) => {\n if (!Number.isFinite(r.score)) {\n throw new ValidationError(`experiment-tracker: rep ${r.rep} has non-finite score ${r.score}`)\n }\n return r.score\n })\n const sorted = [...scores].sort((a, b) => a - b)\n const mean = scores.reduce((s, v) => s + v, 0) / n\n const sd = stddev(scores, mean)\n const spread = iqr(scores)\n const rated = reps.filter((r) => typeof r.passed === 'boolean')\n const passRate = rated.length === 0 ? null : rated.filter((r) => r.passed).length / rated.length\n const stable = spread < t.iqrUnstableAbove && sd < t.stddevUnstableAbove\n return {\n median: median(sorted),\n mean,\n min: sorted[0]!,\n max: sorted[n - 1]!,\n iqr: spread,\n stddev: sd,\n passRate,\n n,\n stable,\n }\n}\n\n/**\n * Verdict for a candidate against its parent. Pure — operates on already-computed\n * stats. KEEP/REGRESSION require both sides to have `>= minRepsForVerdict` reps\n * AND the candidate to be `stable`; otherwise the result is NOISE (unstable) or\n * ITERATE (not enough reps / no parent).\n */\nexport function improvementVerdict(\n candidate: ExperimentStats,\n parent: ExperimentStats | null,\n thresholds?: ImprovementThresholds,\n): ImprovementVerdictResult {\n const t = resolveThresholds(thresholds)\n if (!parent) {\n return {\n verdict: 'ITERATE',\n medianDelta: null,\n reason: 'no parent experiment to compare against',\n }\n }\n if (candidate.n < t.minRepsForVerdict || parent.n < t.minRepsForVerdict) {\n return {\n verdict: 'ITERATE',\n medianDelta: null,\n reason: `need >= ${t.minRepsForVerdict} reps on both sides (candidate n=${candidate.n}, parent n=${parent.n})`,\n }\n }\n if (!candidate.stable) {\n return {\n verdict: 'NOISE',\n medianDelta: candidate.median - parent.median,\n reason: `candidate unstable (iqr=${candidate.iqr}, stddev=${candidate.stddev.toFixed(2)})`,\n }\n }\n const medianDelta = candidate.median - parent.median\n if (medianDelta > t.keepThreshold) {\n return { verdict: 'KEEP', medianDelta, reason: `median +${medianDelta} > +${t.keepThreshold}` }\n }\n if (medianDelta < -t.regressionThreshold) {\n return {\n verdict: 'REGRESSION',\n medianDelta,\n reason: `median ${medianDelta} < -${t.regressionThreshold}`,\n }\n }\n return {\n verdict: 'NOISE',\n medianDelta,\n reason: `median delta ${medianDelta} inside noise band [-${t.regressionThreshold}, +${t.keepThreshold}]`,\n }\n}\n\n// ── Provenance + persistence seams ───────────────────────────────────\n\n/** Reads git provenance for the working tree. Inject a fake in tests; the\n * default implementation shells out to `git`. */\nexport type ProvenanceReader = () => ExperimentProvenance | Promise<ExperimentProvenance>\n\n/** Persistence seam for the experiment log. Inject in-memory in tests; the\n * filesystem implementation is `fileExperimentStore`. */\nexport interface ExperimentStore {\n load(): Promise<Experiment[]>\n save(experiments: Experiment[]): Promise<void>\n}\n\n/**\n * Default provenance reader: `git rev-parse HEAD`, the subject line, and the\n * files changed vs `HEAD~1`. Fail-loud — a tracker that silently logs\n * `commit: 'unknown'` corrupts the provenance the whole point of the log is to\n * carry. When the working tree genuinely has no parent commit, pass an override.\n */\nexport const gitProvenanceReader: ProvenanceReader = () => {\n const run = (cmd: string): string => execSync(cmd, { encoding: 'utf8' }).trim()\n const commit = run('git rev-parse --short HEAD')\n const message = run('git log -1 --format=%s')\n const changedRaw = run('git diff --name-only HEAD~1')\n const changedFiles = changedRaw.length === 0 ? [] : changedRaw.split('\\n').filter(Boolean)\n return { commit, message, changedFiles }\n}\n\n/** In-memory store — the default when no persistence is wanted (tests, ephemeral\n * runs). State lives on the instance. */\nexport function inMemoryExperimentStore(initial: Experiment[] = []): ExperimentStore {\n let state = initial.map((e) => structuredClone(e))\n return {\n async load() {\n return state.map((e) => structuredClone(e))\n },\n async save(experiments) {\n state = experiments.map((e) => structuredClone(e))\n },\n }\n}\n\n/** Filesystem store — a single JSON array at `path`, created on first save. */\nexport function fileExperimentStore(path: string): ExperimentStore {\n return {\n async load() {\n const fs = await import('node:fs/promises')\n try {\n const raw = await fs.readFile(path, 'utf8')\n const parsed = JSON.parse(raw)\n if (!Array.isArray(parsed)) {\n throw new ValidationError(`experiment-tracker: store at ${path} is not a JSON array`)\n }\n return parsed as Experiment[]\n } catch (err) {\n if ((err as NodeJS.ErrnoException).code === 'ENOENT') return []\n throw err\n }\n },\n async save(experiments) {\n const fs = await import('node:fs/promises')\n const pathMod = await import('node:path')\n await fs.mkdir(pathMod.dirname(path), { recursive: true })\n await fs.writeFile(path, JSON.stringify(experiments, null, 2), 'utf8')\n },\n }\n}\n\nexport interface ExperimentTrackerOptions {\n store?: ExperimentStore\n provenanceReader?: ProvenanceReader\n thresholds?: ImprovementThresholds\n /** Clock seam for deterministic timestamps in tests. Default `Date.now`. */\n now?: () => number\n}\n\nexport interface CreateExperimentInput {\n id: string\n label: string\n changeSummary: string\n parentId?: string\n /** Override provenance instead of reading from git (e.g. CI metadata). */\n provenance?: ExperimentProvenance\n}\n\n/**\n * Stateful tracker over an `ExperimentStore`. Create an experiment (provenance\n * is captured once), append reps as they complete (stats + verdict recompute on\n * every append), and read the log back for a dashboard. All persistence and git\n * access flow through the injected seams, so the tracker is fully testable\n * without a repo or disk.\n */\nexport class ExperimentTracker {\n private readonly store: ExperimentStore\n private readonly provenanceReader: ProvenanceReader\n private readonly thresholds: Required<ImprovementThresholds>\n private readonly now: () => number\n\n constructor(options: ExperimentTrackerOptions = {}) {\n this.store = options.store ?? inMemoryExperimentStore()\n this.provenanceReader = options.provenanceReader ?? gitProvenanceReader\n this.thresholds = resolveThresholds(options.thresholds)\n this.now = options.now ?? Date.now\n }\n\n async create(input: CreateExperimentInput): Promise<Experiment> {\n const experiments = await this.store.load()\n if (experiments.some((e) => e.id === input.id)) {\n throw new ValidationError(`experiment-tracker: experiment id \"${input.id}\" already exists`)\n }\n if (input.parentId && !experiments.some((e) => e.id === input.parentId)) {\n throw new ValidationError(\n `experiment-tracker: parent experiment \"${input.parentId}\" not found`,\n )\n }\n const provenance = input.provenance ?? (await this.provenanceReader())\n const experiment: Experiment = {\n id: input.id,\n label: input.label,\n provenance,\n parentId: input.parentId,\n changeSummary: input.changeSummary,\n reps: [],\n stats: computeExperimentStats([], this.thresholds),\n verdict: 'ITERATE',\n createdAt: new Date(this.now()).toISOString(),\n }\n experiments.push(experiment)\n await this.store.save(experiments)\n return structuredClone(experiment)\n }\n\n /** Append a rep (its `rep` index defaults to the current rep count) and\n * recompute stats + verdict. Returns the updated experiment. */\n async addRep(\n experimentId: string,\n rep: Omit<ExperimentRep, 'rep' | 'timestamp'> & { rep?: number; timestamp?: string },\n ): Promise<Experiment> {\n const experiments = await this.store.load()\n const exp = experiments.find((e) => e.id === experimentId)\n if (!exp)\n throw new ValidationError(`experiment-tracker: experiment \"${experimentId}\" not found`)\n if (rep.runId !== undefined && rep.runId.trim().length === 0) {\n throw new ValidationError('experiment-tracker: rep runId must be non-empty when present')\n }\n const evidence = rep.evidence?.map((reference, index) => {\n if (reference.uri.trim().length === 0) {\n throw new ValidationError(\n `experiment-tracker: rep evidence[${index}].uri must be non-empty`,\n )\n }\n return {\n kind: reference.kind,\n uri: reference.uri,\n ...(reference.excerpt === undefined ? {} : { excerpt: reference.excerpt }),\n }\n })\n const fullRep: ExperimentRep = {\n rep: rep.rep ?? exp.reps.length,\n score: rep.score,\n timestamp: rep.timestamp ?? new Date(this.now()).toISOString(),\n ...(rep.runId === undefined ? {} : { runId: rep.runId }),\n ...(evidence === undefined ? {} : { evidence }),\n ...(rep.passed === undefined ? {} : { passed: rep.passed }),\n ...(rep.metrics === undefined ? {} : { metrics: { ...rep.metrics } }),\n }\n exp.reps.push(fullRep)\n exp.stats = computeExperimentStats(exp.reps, this.thresholds)\n const parent = exp.parentId ? experiments.find((e) => e.id === exp.parentId) : undefined\n exp.verdict = improvementVerdict(exp.stats, parent?.stats ?? null, this.thresholds).verdict\n await this.store.save(experiments)\n return structuredClone(exp)\n }\n\n async get(experimentId: string): Promise<Experiment | undefined> {\n const experiments = await this.store.load()\n const found = experiments.find((e) => e.id === experimentId)\n return found ? structuredClone(found) : undefined\n }\n\n async list(): Promise<Experiment[]> {\n return this.store.load()\n }\n\n /** Full verdict (not just the enum) for an experiment vs its parent. */\n async verdictFor(experimentId: string): Promise<ImprovementVerdictResult> {\n const experiments = await this.store.load()\n const exp = experiments.find((e) => e.id === experimentId)\n if (!exp)\n throw new ValidationError(`experiment-tracker: experiment \"${experimentId}\" not found`)\n const parent = exp.parentId ? experiments.find((e) => e.id === exp.parentId) : undefined\n return improvementVerdict(exp.stats, parent?.stats ?? null, this.thresholds)\n }\n}\n","/**\n * Reproducibility attestation for any serializable report object.\n *\n * `attest()` binds a report to its content address (sha-256 over canonical\n * JSON) AND binds that address to the provenance needed to reproduce it:\n * model versions, seeds, price-table hash, code SHA, inputs hash. The outer\n * `envelopeHash` prevents provenance from being rewritten while leaving the\n * report hash valid.\n *\n * Layering: content-addressing is the substrate's job; cryptographic SIGNING\n * (who vouches for the attestation, key management, transparency logs) is the\n * consumer's layer on top. An `AttestedReport` is a stable byte-identical\n * payload a consumer can sign — the substrate never holds keys.\n *\n * Generic by design: the report parameter is ANY value `canonicalJson`\n * accepts (campaign results, fuzz capsules, scorecards, cost ledgers). Do not\n * couple this module to a specific report schema.\n */\n\nimport { contentHash } from './verdict-cache'\n\n/** Hash scheme identifier carried by every attestation. A verifier rejects\n * unknown algorithms instead of guessing. */\nexport const ATTESTATION_ALGORITHM = 'sha256/canonical-json' as const\n\nexport interface AttestationProvenance {\n /** Every model involved in producing the report, name → version/id. */\n modelVersions: Record<string, string>\n /** RNG seeds the run was driven by, when seeded. */\n seeds?: number[]\n /** Content hash of the price table used for cost figures — cost numbers\n * are only reproducible against the same prices. */\n priceTableHash?: string\n /** Git SHA of the code that produced the report. */\n codeSha: string\n /** Content hash of the input set (scenarios, dataset manifest, ...). */\n inputsHash?: string\n /** ISO-8601 timestamp, caller-supplied — the substrate stays clock-free\n * so attestation is deterministic and testable. */\n createdAt: string\n}\n\nexport interface AttestedReport {\n /** Hex sha-256 over the canonical JSON of the report. */\n reportHash: string\n provenance: AttestationProvenance\n algorithm: typeof ATTESTATION_ALGORITHM\n /**\n * Hex sha-256 over `{ reportHash, provenance, algorithm }`. New attestations\n * always carry it. Optional only so persisted pre-envelope attestations can\n * still be read and explicitly recognized as legacy by callers.\n */\n envelopeHash?: string\n}\n\nexport interface AttestationVerification {\n valid: boolean\n /** Populated iff `valid` is false — names the exact mismatch. */\n reason?: string\n /** True only for a valid pre-envelope attestation whose provenance is not cryptographically bound. */\n legacyUnboundProvenance?: true\n}\n\nfunction envelopeMaterial(\n reportHash: string,\n provenance: AttestationProvenance,\n algorithm: typeof ATTESTATION_ALGORITHM,\n): object {\n return { reportHash, provenance, algorithm }\n}\n\n/**\n * Content-address a report and bind it to its provenance. Throws (via\n * `canonicalJson`) if the report or provenance contains undefined / function /\n * symbol / non-finite numbers — an attestation that cannot be unambiguously\n * serialized cannot be trusted.\n */\nexport function attest(report: unknown, provenance: AttestationProvenance): AttestedReport {\n const reportHash = contentHash(report)\n const algorithm = ATTESTATION_ALGORITHM\n return {\n reportHash,\n provenance,\n algorithm,\n envelopeHash: contentHash(envelopeMaterial(reportHash, provenance, algorithm)),\n }\n}\n\n/**\n * Verify a report against its attestation. Returns a typed outcome rather\n * than throwing: an unverifiable report (e.g. one that no longer\n * canonicalizes) is a verification failure with the cause in `reason`, not a\n * crash — verifiers run in pipelines that must record WHY, not die.\n *\n * Legacy attestations without `envelopeHash` remain readable, but verification\n * explicitly marks their provenance as unbound so a promotion path can refuse\n * them instead of accidentally treating old metadata as cryptographic proof.\n */\nexport function verifyAttestation(\n report: unknown,\n attested: AttestedReport,\n): AttestationVerification {\n if (attested.algorithm !== ATTESTATION_ALGORITHM) {\n return {\n valid: false,\n reason: `unknown algorithm '${attested.algorithm}' — this verifier only checks '${ATTESTATION_ALGORITHM}'`,\n }\n }\n let recomputed: string\n try {\n recomputed = contentHash(report)\n } catch (err) {\n return {\n valid: false,\n reason: `report is not canonicalizable: ${err instanceof Error ? err.message : String(err)}`,\n }\n }\n if (recomputed !== attested.reportHash) {\n return {\n valid: false,\n reason: `report hash mismatch: attested ${attested.reportHash}, recomputed ${recomputed}`,\n }\n }\n\n if (attested.envelopeHash === undefined) {\n return { valid: true, legacyUnboundProvenance: true }\n }\n\n let envelopeHash: string\n try {\n envelopeHash = contentHash(\n envelopeMaterial(attested.reportHash, attested.provenance, attested.algorithm),\n )\n } catch (err) {\n return {\n valid: false,\n reason: `attestation provenance is not canonicalizable: ${err instanceof Error ? err.message : String(err)}`,\n }\n }\n if (envelopeHash !== attested.envelopeHash) {\n return {\n valid: false,\n reason: `attestation envelope hash mismatch: attested ${attested.envelopeHash}, recomputed ${envelopeHash}`,\n }\n }\n\n return { valid: true }\n}\n"],"mappings":";;;;;;;;;;;;;;;;;;;;;;;;;;;;AAwHA,MAAM,WAA4C;CAChD,eAAe;CACf,qBAAqB;CACrB,kBAAkB;CAClB,qBAAqB,OAAO;CAC5B,mBAAmB;AACrB;AAEA,SAAS,kBAAkB,GAAuE;CAChG,MAAM,IAAI;EAAE,GAAG;EAAU,GAAI,KAAK,CAAC;CAAG;CACtC,IAAI,EAAE,gBAAgB,GACpB,MAAM,IAAI,gBACR,uDAAuD,EAAE,eAC3D;CAEF,IAAI,EAAE,sBAAsB,GAC1B,MAAM,IAAI,gBACR,6DAA6D,EAAE,qBACjE;CAEF,IAAI,EAAE,oBAAoB,GACxB,MAAM,IAAI,gBACR,2DAA2D,EAAE,mBAC/D;CAEF,OAAO;AACT;AAEA,SAAS,OAAO,QAA0B;CACxC,MAAM,IAAI,OAAO;CACjB,IAAI,MAAM,GAAG,OAAO;CACpB,MAAM,MAAM,KAAK,MAAM,IAAI,CAAC;CAC5B,OAAO,IAAI,MAAM,KAAK,OAAO,MAAM,KAAM,OAAO,QAAS,IAAI,OAAO;AACtE;;AAGA,SAAS,OAAO,QAAkB,MAAsB;CACtD,IAAI,OAAO,SAAS,GAAG,OAAO;CAC9B,MAAM,WAAW,OAAO,QAAQ,KAAK,MAAM,OAAO,IAAI,SAAS,GAAG,CAAC,IAAI,OAAO;CAC9E,OAAO,KAAK,KAAK,QAAQ;AAC3B;;;;;;AAOA,SAAgB,uBACd,MACA,YACiB;CACjB,MAAM,IAAI,kBAAkB,UAAU;CACtC,MAAM,IAAI,KAAK;CACf,IAAI,MAAM,GACR,OAAO;EACL,QAAQ;EACR,MAAM;EACN,KAAK;EACL,KAAK;EACL,KAAK;EACL,QAAQ;EACR,UAAU;EACV,GAAG;EACH,QAAQ;CACV;CAEF,MAAM,SAAS,KAAK,KAAK,MAAM;EAC7B,IAAI,CAAC,OAAO,SAAS,EAAE,KAAK,GAC1B,MAAM,IAAI,gBAAgB,2BAA2B,EAAE,IAAI,wBAAwB,EAAE,OAAO;EAE9F,OAAO,EAAE;CACX,CAAC;CACD,MAAM,SAAS,CAAC,GAAG,MAAM,CAAC,CAAC,MAAM,GAAG,MAAM,IAAI,CAAC;CAC/C,MAAM,OAAO,OAAO,QAAQ,GAAG,MAAM,IAAI,GAAG,CAAC,IAAI;CACjD,MAAM,KAAK,OAAO,QAAQ,IAAI;CAC9B,MAAM,SAAS,IAAI,MAAM;CACzB,MAAM,QAAQ,KAAK,QAAQ,MAAM,OAAO,EAAE,WAAW,SAAS;CAC9D,MAAM,WAAW,MAAM,WAAW,IAAI,OAAO,MAAM,QAAQ,MAAM,EAAE,MAAM,CAAC,CAAC,SAAS,MAAM;CAC1F,MAAM,SAAS,SAAS,EAAE,oBAAoB,KAAK,EAAE;CACrD,OAAO;EACL,QAAQ,OAAO,MAAM;EACrB;EACA,KAAK,OAAO;EACZ,KAAK,OAAO,IAAI;EAChB,KAAK;EACL,QAAQ;EACR;EACA;EACA;CACF;AACF;;;;;;;AAQA,SAAgB,mBACd,WACA,QACA,YAC0B;CAC1B,MAAM,IAAI,kBAAkB,UAAU;CACtC,IAAI,CAAC,QACH,OAAO;EACL,SAAS;EACT,aAAa;EACb,QAAQ;CACV;CAEF,IAAI,UAAU,IAAI,EAAE,qBAAqB,OAAO,IAAI,EAAE,mBACpD,OAAO;EACL,SAAS;EACT,aAAa;EACb,QAAQ,WAAW,EAAE,kBAAkB,mCAAmC,UAAU,EAAE,aAAa,OAAO,EAAE;CAC9G;CAEF,IAAI,CAAC,UAAU,QACb,OAAO;EACL,SAAS;EACT,aAAa,UAAU,SAAS,OAAO;EACvC,QAAQ,2BAA2B,UAAU,IAAI,WAAW,UAAU,OAAO,QAAQ,CAAC,EAAE;CAC1F;CAEF,MAAM,cAAc,UAAU,SAAS,OAAO;CAC9C,IAAI,cAAc,EAAE,eAClB,OAAO;EAAE,SAAS;EAAQ;EAAa,QAAQ,WAAW,YAAY,MAAM,EAAE;CAAgB;CAEhG,IAAI,cAAc,CAAC,EAAE,qBACnB,OAAO;EACL,SAAS;EACT;EACA,QAAQ,UAAU,YAAY,MAAM,EAAE;CACxC;CAEF,OAAO;EACL,SAAS;EACT;EACA,QAAQ,gBAAgB,YAAY,uBAAuB,EAAE,oBAAoB,KAAK,EAAE,cAAc;CACxG;AACF;;;;;;;AAqBA,MAAa,4BAA8C;CACzD,MAAM,OAAO,QAAwB,SAAS,KAAK,EAAE,UAAU,OAAO,CAAC,CAAC,CAAC,KAAK;CAC9E,MAAM,SAAS,IAAI,4BAA4B;CAC/C,MAAM,UAAU,IAAI,wBAAwB;CAC5C,MAAM,aAAa,IAAI,6BAA6B;CAEpD,OAAO;EAAE;EAAQ;EAAS,cADL,WAAW,WAAW,IAAI,CAAC,IAAI,WAAW,MAAM,IAAI,CAAC,CAAC,OAAO,OAAO;CAClD;AACzC;;;AAIA,SAAgB,wBAAwB,UAAwB,CAAC,GAAoB;CACnF,IAAI,QAAQ,QAAQ,KAAK,MAAM,gBAAgB,CAAC,CAAC;CACjD,OAAO;EACL,MAAM,OAAO;GACX,OAAO,MAAM,KAAK,MAAM,gBAAgB,CAAC,CAAC;EAC5C;EACA,MAAM,KAAK,aAAa;GACtB,QAAQ,YAAY,KAAK,MAAM,gBAAgB,CAAC,CAAC;EACnD;CACF;AACF;;AAGA,SAAgB,oBAAoB,MAA+B;CACjE,OAAO;EACL,MAAM,OAAO;GACX,MAAM,KAAK,MAAM,OAAO;GACxB,IAAI;IACF,MAAM,MAAM,MAAM,GAAG,SAAS,MAAM,MAAM;IAC1C,MAAM,SAAS,KAAK,MAAM,GAAG;IAC7B,IAAI,CAAC,MAAM,QAAQ,MAAM,GACvB,MAAM,IAAI,gBAAgB,gCAAgC,KAAK,qBAAqB;IAEtF,OAAO;GACT,SAAS,KAAK;IACZ,IAAK,IAA8B,SAAS,UAAU,OAAO,CAAC;IAC9D,MAAM;GACR;EACF;EACA,MAAM,KAAK,aAAa;GACtB,MAAM,KAAK,MAAM,OAAO;GACxB,MAAM,UAAU,MAAM,OAAO;GAC7B,MAAM,GAAG,MAAM,QAAQ,QAAQ,IAAI,GAAG,EAAE,WAAW,KAAK,CAAC;GACzD,MAAM,GAAG,UAAU,MAAM,KAAK,UAAU,aAAa,MAAM,CAAC,GAAG,MAAM;EACvE;CACF;AACF;;;;;;;;AA0BA,IAAa,oBAAb,MAA+B;CAC7B;CACA;CACA;CACA;CAEA,YAAY,UAAoC,CAAC,GAAG;EAClD,KAAK,QAAQ,QAAQ,SAAS,wBAAwB;EACtD,KAAK,mBAAmB,QAAQ,oBAAoB;EACpD,KAAK,aAAa,kBAAkB,QAAQ,UAAU;EACtD,KAAK,MAAM,QAAQ,OAAO,KAAK;CACjC;CAEA,MAAM,OAAO,OAAmD;EAC9D,MAAM,cAAc,MAAM,KAAK,MAAM,KAAK;EAC1C,IAAI,YAAY,MAAM,MAAM,EAAE,OAAO,MAAM,EAAE,GAC3C,MAAM,IAAI,gBAAgB,sCAAsC,MAAM,GAAG,iBAAiB;EAE5F,IAAI,MAAM,YAAY,CAAC,YAAY,MAAM,MAAM,EAAE,OAAO,MAAM,QAAQ,GACpE,MAAM,IAAI,gBACR,0CAA0C,MAAM,SAAS,YAC3D;EAEF,MAAM,aAAa,MAAM,cAAe,MAAM,KAAK,iBAAiB;EACpE,MAAM,aAAyB;GAC7B,IAAI,MAAM;GACV,OAAO,MAAM;GACb;GACA,UAAU,MAAM;GAChB,eAAe,MAAM;GACrB,MAAM,CAAC;GACP,OAAO,uBAAuB,CAAC,GAAG,KAAK,UAAU;GACjD,SAAS;GACT,WAAW,IAAI,KAAK,KAAK,IAAI,CAAC,CAAC,CAAC,YAAY;EAC9C;EACA,YAAY,KAAK,UAAU;EAC3B,MAAM,KAAK,MAAM,KAAK,WAAW;EACjC,OAAO,gBAAgB,UAAU;CACnC;;;CAIA,MAAM,OACJ,cACA,KACqB;EACrB,MAAM,cAAc,MAAM,KAAK,MAAM,KAAK;EAC1C,MAAM,MAAM,YAAY,MAAM,MAAM,EAAE,OAAO,YAAY;EACzD,IAAI,CAAC,KACH,MAAM,IAAI,gBAAgB,mCAAmC,aAAa,YAAY;EACxF,IAAI,IAAI,UAAU,KAAA,KAAa,IAAI,MAAM,KAAK,CAAC,CAAC,WAAW,GACzD,MAAM,IAAI,gBAAgB,8DAA8D;EAE1F,MAAM,WAAW,IAAI,UAAU,KAAK,WAAW,UAAU;GACvD,IAAI,UAAU,IAAI,KAAK,CAAC,CAAC,WAAW,GAClC,MAAM,IAAI,gBACR,oCAAoC,MAAM,wBAC5C;GAEF,OAAO;IACL,MAAM,UAAU;IAChB,KAAK,UAAU;IACf,GAAI,UAAU,YAAY,KAAA,IAAY,CAAC,IAAI,EAAE,SAAS,UAAU,QAAQ;GAC1E;EACF,CAAC;EACD,MAAM,UAAyB;GAC7B,KAAK,IAAI,OAAO,IAAI,KAAK;GACzB,OAAO,IAAI;GACX,WAAW,IAAI,aAAa,IAAI,KAAK,KAAK,IAAI,CAAC,CAAC,CAAC,YAAY;GAC7D,GAAI,IAAI,UAAU,KAAA,IAAY,CAAC,IAAI,EAAE,OAAO,IAAI,MAAM;GACtD,GAAI,aAAa,KAAA,IAAY,CAAC,IAAI,EAAE,SAAS;GAC7C,GAAI,IAAI,WAAW,KAAA,IAAY,CAAC,IAAI,EAAE,QAAQ,IAAI,OAAO;GACzD,GAAI,IAAI,YAAY,KAAA,IAAY,CAAC,IAAI,EAAE,SAAS,EAAE,GAAG,IAAI,QAAQ,EAAE;EACrE;EACA,IAAI,KAAK,KAAK,OAAO;EACrB,IAAI,QAAQ,uBAAuB,IAAI,MAAM,KAAK,UAAU;EAC5D,MAAM,SAAS,IAAI,WAAW,YAAY,MAAM,MAAM,EAAE,OAAO,IAAI,QAAQ,IAAI,KAAA;EAC/E,IAAI,UAAU,mBAAmB,IAAI,OAAO,QAAQ,SAAS,MAAM,KAAK,UAAU,CAAC,CAAC;EACpF,MAAM,KAAK,MAAM,KAAK,WAAW;EACjC,OAAO,gBAAgB,GAAG;CAC5B;CAEA,MAAM,IAAI,cAAuD;EAE/D,MAAM,SAAQ,MADY,KAAK,MAAM,KAAK,EAAA,CAChB,MAAM,MAAM,EAAE,OAAO,YAAY;EAC3D,OAAO,QAAQ,gBAAgB,KAAK,IAAI,KAAA;CAC1C;CAEA,MAAM,OAA8B;EAClC,OAAO,KAAK,MAAM,KAAK;CACzB;;CAGA,MAAM,WAAW,cAAyD;EACxE,MAAM,cAAc,MAAM,KAAK,MAAM,KAAK;EAC1C,MAAM,MAAM,YAAY,MAAM,MAAM,EAAE,OAAO,YAAY;EACzD,IAAI,CAAC,KACH,MAAM,IAAI,gBAAgB,mCAAmC,aAAa,YAAY;EACxF,MAAM,SAAS,IAAI,WAAW,YAAY,MAAM,MAAM,EAAE,OAAO,IAAI,QAAQ,IAAI,KAAA;EAC/E,OAAO,mBAAmB,IAAI,OAAO,QAAQ,SAAS,MAAM,KAAK,UAAU;CAC7E;AACF;;;;;;;;;;;;;;;;;;;;;;;ACjbA,MAAa,wBAAwB;AAwCrC,SAAS,iBACP,YACA,YACA,WACQ;CACR,OAAO;EAAE;EAAY;EAAY;CAAU;AAC7C;;;;;;;AAQA,SAAgB,OAAO,QAAiB,YAAmD;CACzF,MAAM,aAAa,YAAY,MAAM;CACrC,MAAM,YAAY;CAClB,OAAO;EACL;EACA;EACA;EACA,cAAc,YAAY,iBAAiB,YAAY,YAAY,SAAS,CAAC;CAC/E;AACF;;;;;;;;;;;AAYA,SAAgB,kBACd,QACA,UACyB;CACzB,IAAI,SAAS,cAAA,yBACX,OAAO;EACL,OAAO;EACP,QAAQ,sBAAsB,SAAS,UAAU,iCAAiC,sBAAsB;CAC1G;CAEF,IAAI;CACJ,IAAI;EACF,aAAa,YAAY,MAAM;CACjC,SAAS,KAAK;EACZ,OAAO;GACL,OAAO;GACP,QAAQ,kCAAkC,eAAe,QAAQ,IAAI,UAAU,OAAO,GAAG;EAC3F;CACF;CACA,IAAI,eAAe,SAAS,YAC1B,OAAO;EACL,OAAO;EACP,QAAQ,kCAAkC,SAAS,WAAW,eAAe;CAC/E;CAGF,IAAI,SAAS,iBAAiB,KAAA,GAC5B,OAAO;EAAE,OAAO;EAAM,yBAAyB;CAAK;CAGtD,IAAI;CACJ,IAAI;EACF,eAAe,YACb,iBAAiB,SAAS,YAAY,SAAS,YAAY,SAAS,SAAS,CAC/E;CACF,SAAS,KAAK;EACZ,OAAO;GACL,OAAO;GACP,QAAQ,kDAAkD,eAAe,QAAQ,IAAI,UAAU,OAAO,GAAG;EAC3G;CACF;CACA,IAAI,iBAAiB,SAAS,cAC5B,OAAO;EACL,OAAO;EACP,QAAQ,gDAAgD,SAAS,aAAa,eAAe;CAC/F;CAGF,OAAO,EAAE,OAAO,KAAK;AACvB"}
|
|
@@ -1412,7 +1412,7 @@ const ANALYST_BENCHMARK_DEPENDENCY_LOCK_FILES = Object.freeze([
|
|
|
1412
1412
|
"package.json",
|
|
1413
1413
|
"pnpm-lock.yaml"
|
|
1414
1414
|
]);
|
|
1415
|
-
const ANALYST_BENCHMARK_DEPENDENCY_LOCK_SHA256 = "
|
|
1415
|
+
const ANALYST_BENCHMARK_DEPENDENCY_LOCK_SHA256 = "3f65588dbbec59d2c783edf245a9751653346fc67eae691d38af22c8b6b5b8ce";
|
|
1416
1416
|
const ANALYST_BENCHMARK_EVIDENCE_DEPENDENCY_LOCK_SHA256 = "1e03f2daed356d60316aabefb407ec1e437ac94d408d61eea4ae096e9c6fbb5b";
|
|
1417
1417
|
const ANALYST_BENCHMARK_EVIDENCE_IMPLEMENTATION_SHA256 = "4dba263b6256a30d56c7fdb2d992d3a953c0035d731f359b704db806f68f75ac";
|
|
1418
1418
|
const ANALYST_BENCHMARK_IMPLEMENTATION_FILES = Object.freeze([
|
|
@@ -6351,4 +6351,4 @@ function shellQuote(value) {
|
|
|
6351
6351
|
//#endregion
|
|
6352
6352
|
export { ANALYST_BENCHMARK_LOCAL_RECEIPT_FILE as $, effectiveAnalystProtocolSha256 as A, loadCodeTraceVerificationArtifacts as B, adaptPublicBenchmarkFindings as C, renderCodeTraceCalibrationMarkdown as D, readAnalystBenchmarkArtifact as E, publicBenchmarkProtocolSha256 as F, ANALYST_BENCHMARK_EVIDENCE_DEPENDENCY_LOCK_SHA256 as G, ANALYST_BENCHMARK_DEPENDENCY_LOCK_DIGEST_ALGORITHM as H, publicBenchmarkRlmInstructions as I, ANALYST_BENCHMARK_IMPLEMENTATION_FILES as J, ANALYST_BENCHMARK_EVIDENCE_IMPLEMENTATION_SHA256 as K, publicBenchmarkSystemPrompt as L, CODE_TRACE_BENCH_ANALYST_PROMPT as M, MAX_INCORRECT_BLOCKS as N, summarizeCodeTraceCalibration as O, MAX_INCORRECT_BLOCK_STEPS as P, ANALYST_BENCHMARK_COST_LEDGER_FILE as Q, DEFAULT_MAX_VERIFICATION_ARTIFACT_BYTES as R, analystDefinitionProtocolSha256 as S, expandCodeTraceFailureBlocks as T, ANALYST_BENCHMARK_DEPENDENCY_LOCK_FILES as U, parseVerificationOutcome as V, ANALYST_BENCHMARK_DEPENDENCY_LOCK_SHA256 as W, analystBenchmarkDependencyLockDigest as X, ANALYST_BENCHMARK_IMPLEMENTATION_SHA256 as Y, analystBenchmarkImplementationDigest as Z, createPublicBenchmarkDirectRunner as _, primeCodeTraceAnalystDefinition as a, summarizeAgentRxCalibration as at, AnalystExpressivenessError as b, loadPublicBenchmarkRows as c, agentRxBenchmarkCase as ct, publicBenchmarkSelectionReport as d, roundAgentRxStep as dt, ANALYST_BENCHMARK_MANIFEST_FILE as et, selectPublicBenchmarkRows as f, normalizeBenchmarkLabel as ft, runReplVariableAnalystDefinition as g, rlmEngineLimits as h, primeAnalystProtocolSha256 as i, renderAgentRxCalibrationMarkdown as it, readAnalystInstructionsOverride as j, analystInstructionsOverrideFromText as k, preparePublicAnalystBenchmark as l, agentRxPredictionsToFindings as lt, publicRlmAnalystDefinition as m, renderAnalystBenchmarkMarkdown as n, compareAnalystRunners as nt, runInlineAnalystDefinition as o, codeTraceBenchCase as ot, createPublicBenchmarkRlmRunner as p, ANALYST_BENCHMARK_IMPLEMENTATION_DIGEST_ALGORITHM as q, createPrimeBenchmarkRunner as r, AGENT_RX_UPSTREAM_REVISION as rt, nodeHttpPrimeBridgeTransport as s, codeTracerPredictionsToFindings as st, runAnalystBenchmarkCommand as t, ANALYST_BENCHMARK_OBSERVATIONS_FILE as tt, publicBenchmarkDistributions as u, normalizeAgentRxCategory as ut, publicDirectAnalystDefinition as v, emptyPublicBenchmarkRunner as w, analystDefinitionAsymmetries as x, runChunkedAnalystDefinition as y, appendVerificationArtifactsToOtlp as z };
|
|
6353
6353
|
|
|
6354
|
-
//# sourceMappingURL=benchmark-command-
|
|
6354
|
+
//# sourceMappingURL=benchmark-command-DoFcisuM.js.map
|