@tangle-network/agent-eval 0.173.3 → 0.175.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +54 -0
- package/README.md +1 -1
- package/dist/{proposal-findings-bko3GGy-.js → abort-signal-CtzAM_sJ.js} +11 -11
- package/dist/abort-signal-CtzAM_sJ.js.map +1 -0
- package/dist/adapters/http.d.ts +2 -2
- package/dist/agent-profile-_xPxqVJt.d.ts +488 -0
- package/dist/agent-profile-_xPxqVJt.d.ts.map +1 -0
- package/dist/analyst/index.d.ts +7 -9
- package/dist/analyst/index.d.ts.map +1 -1
- package/dist/analyst/index.js +8 -8
- package/dist/{benchmark-C4wk_Sjr.js → benchmark-DQKzykkO.js} +2 -2
- package/dist/{benchmark-C4wk_Sjr.js.map → benchmark-DQKzykkO.js.map} +1 -1
- package/dist/{benchmark-command-BY9oscke.js → benchmark-command-D_5xG9LG.js} +13 -13
- package/dist/{benchmark-command-BY9oscke.js.map → benchmark-command-D_5xG9LG.js.map} +1 -1
- package/dist/benchmarks/index.d.ts +3 -4
- package/dist/benchmarks/index.d.ts.map +1 -1
- package/dist/benchmarks/index.js +3 -3
- package/dist/campaign/index.d.ts +5 -9
- package/dist/campaign/index.js +7 -7
- package/dist/{campaign-B3kPMU8S.js → campaign-BzMSCejE.js} +8 -8
- package/dist/{campaign-B3kPMU8S.js.map → campaign-BzMSCejE.js.map} +1 -1
- package/dist/{opencode-sqlite-eK6HW6dr.js → claude-jsonl-CxZZrDJ3.js} +9 -149
- package/dist/claude-jsonl-CxZZrDJ3.js.map +1 -0
- package/dist/cli.js +9 -2
- package/dist/cli.js.map +1 -1
- package/dist/{client-DlqdbM7n.d.ts → client-vyYQg3bm.d.ts} +2 -2
- package/dist/{client-DlqdbM7n.d.ts.map → client-vyYQg3bm.d.ts.map} +1 -1
- package/dist/contract/index.d.ts +9 -10
- package/dist/contract/index.js +8 -8
- package/dist/{default-registry-B0bKikCb.js → default-registry-DBqVI4pq.js} +5 -5
- package/dist/{default-registry-B0bKikCb.js.map → default-registry-DBqVI4pq.js.map} +1 -1
- package/dist/{default-registry-BKwc8bN5.d.ts → default-registry-FfNzaUHV.d.ts} +3 -3
- package/dist/{default-registry-BKwc8bN5.d.ts.map → default-registry-FfNzaUHV.d.ts.map} +1 -1
- package/dist/{define-agent-eval-CY6qdlGV.d.ts → define-agent-eval-V1jQyCDR.d.ts} +102 -11
- package/dist/define-agent-eval-V1jQyCDR.d.ts.map +1 -0
- package/dist/{define-agent-eval-8h3lXXee.js → define-agent-eval-ox5McL6e.js} +331 -144
- package/dist/define-agent-eval-ox5McL6e.js.map +1 -0
- package/dist/{dspy-rlm-engine-CF0t2ITD.js → dspy-rlm-engine-Caz2pl4L.js} +3 -3
- package/dist/{dspy-rlm-engine-CF0t2ITD.js.map → dspy-rlm-engine-Caz2pl4L.js.map} +1 -1
- package/dist/{engine-DhFir3Ys.d.ts → engine-CvW_I72-.d.ts} +2 -2
- package/dist/{engine-DhFir3Ys.d.ts.map → engine-CvW_I72-.d.ts.map} +1 -1
- package/dist/experiment/index.d.ts +1 -4
- package/dist/experiment/index.d.ts.map +1 -1
- package/dist/{external-optimizer-process-BwITA9Jp.js → external-optimizer-process-CxnFL1hd.js} +2 -2
- package/dist/{external-optimizer-process-BwITA9Jp.js.map → external-optimizer-process-CxnFL1hd.js.map} +1 -1
- package/dist/{external-optimizer-subprocess-wBWeoG6A.js → external-optimizer-subprocess-CQi27uEI.js} +2 -2
- package/dist/{external-optimizer-subprocess-wBWeoG6A.js.map → external-optimizer-subprocess-CQi27uEI.js.map} +1 -1
- package/dist/fuzz.js +1 -1
- package/dist/fuzz.js.map +1 -1
- package/dist/hosted/index.d.ts +1 -1
- package/dist/{index-D0Db5X-4.d.ts → index-BAAiSF3_.d.ts} +5 -5
- package/dist/{index-D0Db5X-4.d.ts.map → index-BAAiSF3_.d.ts.map} +1 -1
- package/dist/{index-BQqOjerE.d.ts → index-BTrx5s8m.d.ts} +8 -9
- package/dist/index-BTrx5s8m.d.ts.map +1 -0
- package/dist/index-DKXuBPXf.d.ts +3840 -0
- package/dist/index-DKXuBPXf.d.ts.map +1 -0
- package/dist/index.d.ts +11 -13
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +11 -11
- package/dist/{integrity-BWywb34E.js → integrity-DsHWCebQ.js} +11 -435
- package/dist/integrity-DsHWCebQ.js.map +1 -0
- package/dist/{kind-factory-gP6lDySe.js → kind-factory-BLvL-E44.js} +2 -2
- package/dist/{kind-factory-gP6lDySe.js.map → kind-factory-BLvL-E44.js.map} +1 -1
- package/dist/{llm-judge-BfqMFo4h.js → llm-judge-DmNaBrXB.js} +2541 -2435
- package/dist/llm-judge-DmNaBrXB.js.map +1 -0
- package/dist/{matrix-DGu8KhSs.d.ts → matrix-CJtXz1ky.d.ts} +2 -2
- package/dist/{matrix-DGu8KhSs.d.ts.map → matrix-CJtXz1ky.d.ts.map} +1 -1
- package/dist/multishot/golden/index.d.ts +1 -1
- package/dist/multishot/index.d.ts +2 -2
- package/dist/openapi.json +1 -1
- package/dist/opencode-sqlite-CNw3vubS.js +145 -0
- package/dist/opencode-sqlite-CNw3vubS.js.map +1 -0
- package/dist/{produced-state-D91uDvQw.js → produced-state-B8mw6zj9.js} +2 -2
- package/dist/{produced-state-D91uDvQw.js.map → produced-state-B8mw6zj9.js.map} +1 -1
- package/dist/report-command-DKlXfU5r.js +1528 -0
- package/dist/report-command-DKlXfU5r.js.map +1 -0
- package/dist/rl.d.ts +1 -1
- package/dist/rl.d.ts.map +1 -1
- package/dist/rl.js.map +1 -1
- package/dist/rollout/index.js +3 -2
- package/dist/{rollout-C-znbbYg.js → rollout-CGlDq1GI.js} +3 -2
- package/dist/{rollout-C-znbbYg.js.map → rollout-CGlDq1GI.js.map} +1 -1
- package/dist/{semantic-concept-judge-Dok7_35a.js → semantic-concept-judge-E3s_fEjB.js} +3 -3
- package/dist/{semantic-concept-judge-Dok7_35a.js.map → semantic-concept-judge-E3s_fEjB.js.map} +1 -1
- package/dist/{skillopt-optimization-method-DDw3v3gA.js → skillopt-optimization-method-f7399oGb.js} +5 -5
- package/dist/{skillopt-optimization-method-DDw3v3gA.js.map → skillopt-optimization-method-f7399oGb.js.map} +1 -1
- package/dist/statistical-heldout-Cqb73yE9.d.ts +1127 -0
- package/dist/statistical-heldout-Cqb73yE9.d.ts.map +1 -0
- package/dist/{store-otlp-Dow0pk_5.js → store-otlp-DV_H2HDu.js} +2 -2
- package/dist/{store-otlp-Dow0pk_5.js.map → store-otlp-DV_H2HDu.js.map} +1 -1
- package/dist/{store-tool-spans-CCZNsihA.d.ts → store-tool-spans-4o55ABER.d.ts} +3 -3
- package/dist/{store-tool-spans-CCZNsihA.d.ts.map → store-tool-spans-4o55ABER.d.ts.map} +1 -1
- package/dist/{store-tool-spans-CeNj_m2L.js → store-tool-spans-B9tjys_h.js} +3 -3
- package/dist/{store-tool-spans-CeNj_m2L.js.map → store-tool-spans-B9tjys_h.js.map} +1 -1
- package/dist/supervisor-run/index.d.ts +71 -6
- package/dist/supervisor-run/index.d.ts.map +1 -1
- package/dist/supervisor-run/index.js +6 -1357
- package/dist/supervisor-run/index.js.map +1 -1
- package/dist/{task-failure-attributes-CZjZeBsY.js → task-failure-attributes-CUy9mkIY.js} +2 -2
- package/dist/{task-failure-attributes-CZjZeBsY.js.map → task-failure-attributes-CUy9mkIY.js.map} +1 -1
- package/dist/terminal-record-Ce9_UjRz.js +539 -0
- package/dist/terminal-record-Ce9_UjRz.js.map +1 -0
- package/dist/{tool-groups-Cp4Xdzrp.d.ts → tool-groups-DAe1t6zb.d.ts} +2 -2
- package/dist/tool-groups-DAe1t6zb.d.ts.map +1 -0
- package/dist/trace-repair/index.d.ts +1 -1
- package/dist/traces.d.ts +2 -2
- package/dist/traces.js +4 -4
- package/dist/{types-Ba5UQyVD.d.ts → types-BJz2CPTM.d.ts} +2 -2
- package/dist/{types-Ba5UQyVD.d.ts.map → types-BJz2CPTM.d.ts.map} +1 -1
- package/dist/{types-CiWITkGo.js → types-DQ0e2E7y.js} +2 -2
- package/dist/types-DQ0e2E7y.js.map +1 -0
- package/dist/{types-BDV4PiMR.d.ts → types-Dd1ejaeI.d.ts} +2 -2
- package/dist/{types-BDV4PiMR.d.ts.map → types-Dd1ejaeI.d.ts.map} +1 -1
- package/dist/{types-CoPUTiXb.d.ts → types-vUdAx2Cj.d.ts} +65 -3
- package/dist/types-vUdAx2Cj.d.ts.map +1 -0
- package/docs/campaign-proposers.md +42 -0
- package/package.json +1 -1
- package/dist/agent-profile-B9_GGsG8.d.ts +0 -84
- package/dist/agent-profile-B9_GGsG8.d.ts.map +0 -1
- package/dist/backend-integrity-CeuTgqsd.d.ts +0 -280
- package/dist/backend-integrity-CeuTgqsd.d.ts.map +0 -1
- package/dist/benchmark-BjLGkfnN.d.ts +0 -236
- package/dist/benchmark-BjLGkfnN.d.ts.map +0 -1
- package/dist/define-agent-eval-8h3lXXee.js.map +0 -1
- package/dist/define-agent-eval-CY6qdlGV.d.ts.map +0 -1
- package/dist/external-optimizer-contracts-CQCpyrIL.d.ts +0 -172
- package/dist/external-optimizer-contracts-CQCpyrIL.d.ts.map +0 -1
- package/dist/heldout-gate-Df5hsqmm.d.ts +0 -453
- package/dist/heldout-gate-Df5hsqmm.d.ts.map +0 -1
- package/dist/index-BQqOjerE.d.ts.map +0 -1
- package/dist/index-CFDffsKz.d.ts +0 -1135
- package/dist/index-CFDffsKz.d.ts.map +0 -1
- package/dist/integrity-BWywb34E.js.map +0 -1
- package/dist/llm-judge-BfqMFo4h.js.map +0 -1
- package/dist/opencode-sqlite-eK6HW6dr.js.map +0 -1
- package/dist/power-preflight-Ptse_Kq7.d.ts +0 -117
- package/dist/power-preflight-Ptse_Kq7.d.ts.map +0 -1
- package/dist/pre-registration-BoI4ucR3.d.ts +0 -592
- package/dist/pre-registration-BoI4ucR3.d.ts.map +0 -1
- package/dist/promotion-policy-CvMda3kU.d.ts +0 -134
- package/dist/promotion-policy-CvMda3kU.d.ts.map +0 -1
- package/dist/proposal-findings-bko3GGy-.js.map +0 -1
- package/dist/provenance-CRY67X50.d.ts +0 -1995
- package/dist/provenance-CRY67X50.d.ts.map +0 -1
- package/dist/statistical-heldout-DTyB_6-1.d.ts +0 -295
- package/dist/statistical-heldout-DTyB_6-1.d.ts.map +0 -1
- package/dist/tool-groups-Cp4Xdzrp.d.ts.map +0 -1
- package/dist/types-CiWITkGo.js.map +0 -1
- package/dist/types-CoPUTiXb.d.ts.map +0 -1
|
@@ -1,84 +0,0 @@
|
|
|
1
|
-
import { AgentProfile, AgentProfile as AgentProfile$1, HarnessType, HarnessType as HarnessType$1 } from "@tangle-network/agent-interface";
|
|
2
|
-
//#region src/agent-profile.d.ts
|
|
3
|
-
/**
|
|
4
|
-
* The agentic coding harnesses an eval sweeps by default — the ones we care about
|
|
5
|
-
* ranking. This is the SINGLE source of that list; consumers import it instead of
|
|
6
|
-
* re-declaring their own (a re-declared list is how the fleet drifts). Pass an
|
|
7
|
-
* explicit `harnesses` (e.g. `harnessTypeSchema.options` for literally every known
|
|
8
|
-
* harness) to widen beyond these.
|
|
9
|
-
*/
|
|
10
|
-
declare const CODING_HARNESSES: readonly HarnessType[];
|
|
11
|
-
interface ProfileAxisSpec {
|
|
12
|
-
/** The domain profile to sweep. Its prompt/tools/skills are held fixed; only the
|
|
13
|
-
* harness and model vary. `model.default` is the fallback model. */
|
|
14
|
-
base: AgentProfile;
|
|
15
|
-
/** Harnesses to cross. Default: {@link CODING_HARNESSES}. */
|
|
16
|
-
harnesses?: readonly HarnessType[];
|
|
17
|
-
/** Models to cross. Default: `[base.model.default]` — one model, i.e. today's
|
|
18
|
-
* single-model behaviour, so omitting this never changes an existing run. */
|
|
19
|
-
models?: readonly string[];
|
|
20
|
-
/** Force every (harness, model) pair verbatim, even ones the harness can't run —
|
|
21
|
-
* for deliberately testing failure modes. Default (false): SNAP instead — a
|
|
22
|
-
* vendor-locked harness runs only the swept models in its family, or its native
|
|
23
|
-
* default when it supports none, so no harness is dropped and none gets a
|
|
24
|
-
* guaranteed-failing foreign-model cell. */
|
|
25
|
-
keepIncompatible?: boolean;
|
|
26
|
-
}
|
|
27
|
-
/** Model sentinel for a vendor-locked harness that supports none of the swept models:
|
|
28
|
-
* it carries no provider prefix, so `harnessSupportsModel` accepts it and the harness
|
|
29
|
-
* resolves it to its own native default model at runtime (e.g. kimi-code → its Kimi
|
|
30
|
-
* model). Lets `expandProfileAxes` snap-instead-of-drop without a per-harness flagship
|
|
31
|
-
* table that would rot as router catalogs change. */
|
|
32
|
-
declare const HARNESS_NATIVE_MODEL = "default";
|
|
33
|
-
/**
|
|
34
|
-
* Expand a base profile across the harness × model matrix into the `AgentProfile[]`
|
|
35
|
-
* that `runProfileMatrix` / `selfImprove` score — the ONE place "which harnesses ×
|
|
36
|
-
* which models do we evaluate" lives, so no product hand-rolls its own harness list
|
|
37
|
-
* or column→profile mapping (the pattern that let those copies drift and silently
|
|
38
|
-
* break the harness pivot).
|
|
39
|
-
*
|
|
40
|
-
* Each cell clones `base`, sets the canonical top-level `harness` and `model.default`,
|
|
41
|
-
* and stamps `metadata.harness` + `metadata.harnessModel` for matrix grouping. Both
|
|
42
|
-
* metadata fields are hash-bearing, so every cell gets a distinct `agentProfileId` row
|
|
43
|
-
* and results join back by harness/model via {@link harnessAxisOf} with no
|
|
44
|
-
* hand-recomputed key. A vendor-locked harness snaps to its family's swept models — or
|
|
45
|
-
* its native default ({@link HARNESS_NATIVE_MODEL}) when it supports none — so every
|
|
46
|
-
* requested harness runs; `keepIncompatible` forces every pair verbatim.
|
|
47
|
-
*
|
|
48
|
-
* Omit `harnesses`/`models` to sweep the full default set — the "turn it on for
|
|
49
|
-
* everything we care about" switch, identical in shape whether one harness or all.
|
|
50
|
-
*/
|
|
51
|
-
declare function expandProfileAxes(spec: ProfileAxisSpec): AgentProfile[];
|
|
52
|
-
/**
|
|
53
|
-
* Read the (harness, model) a matrix cell ran under, off a profile or a result row's
|
|
54
|
-
* profile — the join-back for a `byHarness` pivot. Returns undefined when the profile
|
|
55
|
-
* wasn't produced by {@link expandProfileAxes}. Callers group `result.byProfile` by
|
|
56
|
-
* this instead of recomputing an id (recomputing the wrong key is what broke the pivot
|
|
57
|
-
* in the hand-rolled copies).
|
|
58
|
-
*/
|
|
59
|
-
declare function harnessAxisOf(profile: Pick<AgentProfile, 'metadata'>): {
|
|
60
|
-
harness: HarnessType;
|
|
61
|
-
model: string;
|
|
62
|
-
} | undefined;
|
|
63
|
-
/**
|
|
64
|
-
* Collision-resistant, path-safe, human-readable profile id for eval artifacts.
|
|
65
|
-
* Scorecard joins still use `agentProfileHash`; this id is for run ids, matrix
|
|
66
|
-
* keys, and directory names where two profiles must not collapse onto one row.
|
|
67
|
-
* The suffix is the first 64 bits of the behaviour hash, enough for ordinary
|
|
68
|
-
* eval matrices while keeping filenames readable.
|
|
69
|
-
*/
|
|
70
|
-
declare function agentProfileId(profile: AgentProfile): string;
|
|
71
|
-
/**
|
|
72
|
-
* Deterministic behaviour identity for the canonical
|
|
73
|
-
* `@tangle-network/agent-interface` AgentProfile.
|
|
74
|
-
*
|
|
75
|
-
* `name` and `description` are labels and do not affect the hash. Profile
|
|
76
|
-
* `version`, prompt, model hints, tools, resources, hooks, modes, permissions,
|
|
77
|
-
* and extensions do affect the hash. Resource array order is hash-bearing
|
|
78
|
-
* because mount order can change agent behaviour. Undefined fields are treated
|
|
79
|
-
* as absent; explicit `null` fields remain hash-bearing.
|
|
80
|
-
*/
|
|
81
|
-
declare function agentProfileHash(profile: AgentProfile): string;
|
|
82
|
-
//#endregion
|
|
83
|
-
export { ProfileAxisSpec as a, expandProfileAxes as c, HarnessType$1 as i, harnessAxisOf as l, CODING_HARNESSES as n, agentProfileHash as o, HARNESS_NATIVE_MODEL as r, agentProfileId as s, AgentProfile$1 as t };
|
|
84
|
-
//# sourceMappingURL=agent-profile-B9_GGsG8.d.ts.map
|
|
@@ -1 +0,0 @@
|
|
|
1
|
-
{"version":3,"file":"agent-profile-B9_GGsG8.d.ts","names":[],"sources":["../src/agent-profile.ts"],"mappings":";;;;;;;;;cAea,2BAA2B;UAOvB;;;EAGf,MAAM;;EAEN,qBAAqB;;;EAGrB;;;;;;EAMA;;;;;;;cAQW;;;;;;;;;;;;;;;;;;;iBAoBG,kBAAkB,MAAM,kBAAkB;;;;;;;;iBAwD1C,cACd,SAAS,KAAK;EACX,SAAS;EAAa;;;;;;;;;iBAiBX,eAAe,SAAS;;;;;;;;;;;iBAmDxB,iBAAiB,SAAS"}
|
|
@@ -1,280 +0,0 @@
|
|
|
1
|
-
import { t as AgentEvalError } from "./errors-DEE6u6ot.js";
|
|
2
|
-
import { c as CostLedgerHandle } from "./cost-ledger-DbQdN3nO.js";
|
|
3
|
-
import { a as RunRecord } from "./run-record-DTv1MdjK.js";
|
|
4
|
-
import { Y as RawProviderSink, p as ChatClient } from "./types-gvRsyJLh.js";
|
|
5
|
-
import { a as CheckerIdentity, n as VerdictCertification, t as DefaultVerdict, x as VerificationStrategySource } from "./verdict-E4eRNf7-.js";
|
|
6
|
-
//#region src/artifact-validator.d.ts
|
|
7
|
-
/**
|
|
8
|
-
* Artifact validators.
|
|
9
|
-
*
|
|
10
|
-
* Generic "score a produced artifact" primitive. Tax uses it for PDF form
|
|
11
|
-
* correctness, research for sourced briefs, browser for task assertions, coding
|
|
12
|
-
* for social posts. One interface, many validators.
|
|
13
|
-
*
|
|
14
|
-
* A validator receives an `Artifact` (file on disk, JSON blob, text, binary)
|
|
15
|
-
* plus a `ValidationContext` (scenario id, the turns that produced it) and
|
|
16
|
-
* returns a `ValidationResult` with pass/fail + 0..1 score + structured
|
|
17
|
-
* issues.
|
|
18
|
-
*/
|
|
19
|
-
interface Artifact {
|
|
20
|
-
/** Logical kind — validators type-guard on this */
|
|
21
|
-
kind: 'file' | 'json' | 'text' | 'binary' | string;
|
|
22
|
-
/** Filesystem-style path, optional */
|
|
23
|
-
path?: string;
|
|
24
|
-
/** String content for text/json/file kinds */
|
|
25
|
-
content?: string;
|
|
26
|
-
/** Binary content (if kind === 'binary') */
|
|
27
|
-
bytes?: Uint8Array;
|
|
28
|
-
/** Caller-supplied metadata (mimeType, sha256, size, etc.) */
|
|
29
|
-
metadata?: Record<string, unknown>;
|
|
30
|
-
}
|
|
31
|
-
//#endregion
|
|
32
|
-
//#region src/completion-verifier.d.ts
|
|
33
|
-
/** What kind of produced state can satisfy a requirement structurally. */
|
|
34
|
-
type SatisfiedBy = 'artifact' | 'proposal' | 'tool-call' | 'any';
|
|
35
|
-
interface CompletionRequirement {
|
|
36
|
-
/** Stable id from the task gold (e.g. a persona's `expected_requirements[].req_id`). */
|
|
37
|
-
reqId: string;
|
|
38
|
-
/** Human-readable description of the required deliverable. */
|
|
39
|
-
title: string;
|
|
40
|
-
/** Optional kind/category hint, matched against a produced item's kind. */
|
|
41
|
-
category?: string;
|
|
42
|
-
/** What produced state satisfies this requirement. Defaults to 'any'. */
|
|
43
|
-
satisfiedBy?: SatisfiedBy;
|
|
44
|
-
}
|
|
45
|
-
interface TaskGold {
|
|
46
|
-
taskId: string;
|
|
47
|
-
requirements: CompletionRequirement[];
|
|
48
|
-
}
|
|
49
|
-
interface ProducedProposal {
|
|
50
|
-
id: string;
|
|
51
|
-
title: string;
|
|
52
|
-
status: 'pending' | 'approved' | 'rejected';
|
|
53
|
-
/** Optional persisted body — when present, enables a correctness check. */
|
|
54
|
-
content?: string;
|
|
55
|
-
}
|
|
56
|
-
/** Everything observable about what a run actually produced. */
|
|
57
|
-
interface ProducedState {
|
|
58
|
-
/** Persisted vault artifacts. Reuses the shared `Artifact` shape. */
|
|
59
|
-
artifacts: Artifact[];
|
|
60
|
-
/** Proposals / filings the agent created. */
|
|
61
|
-
proposals: ProducedProposal[];
|
|
62
|
-
/** Names of tools the agent invoked. */
|
|
63
|
-
toolCalls: string[];
|
|
64
|
-
}
|
|
65
|
-
interface RequirementCheck {
|
|
66
|
-
reqId: string;
|
|
67
|
-
title: string;
|
|
68
|
-
/** A produced item of the right kind matched the requirement, non-empty. */
|
|
69
|
-
structurallyPresent: boolean;
|
|
70
|
-
/**
|
|
71
|
-
* Whether the matched item actually fulfils the requirement. `null` when
|
|
72
|
-
* not structurally present, when the matched item carries no content
|
|
73
|
-
* to assess, or when the correctness check itself failed (`unmeasured`).
|
|
74
|
-
*/
|
|
75
|
-
correct: boolean | null;
|
|
76
|
-
/** structurallyPresent && !unmeasured && correct !== false. */
|
|
77
|
-
satisfied: boolean;
|
|
78
|
-
/**
|
|
79
|
-
* Set when the correctness check itself errored (LLM call failure or an
|
|
80
|
-
* unparseable response after retry). The requirement's fulfilment is
|
|
81
|
-
* UNKNOWN — `correct` stays null, `satisfied` is false, and
|
|
82
|
-
* `completionVerdict` excludes the row from `completionRate`'s
|
|
83
|
-
* denominator. Never folded into a zero: a synthetic zero is
|
|
84
|
-
* indistinguishable from a real failure (see `JudgeParseError`).
|
|
85
|
-
*/
|
|
86
|
-
unmeasured?: true;
|
|
87
|
-
/** Why the correctness check could not be measured (present iff `unmeasured`). */
|
|
88
|
-
unmeasuredReason?: string;
|
|
89
|
-
/** Human-readable evidence for the verdict. */
|
|
90
|
-
evidence: string[];
|
|
91
|
-
}
|
|
92
|
-
/** Extends the substrate verdict spine: `valid` = `fullyComplete` and
|
|
93
|
-
* `score` = `completionRate` — derived in `completionVerdict()`, the one
|
|
94
|
-
* place those equalities hold by construction. */
|
|
95
|
-
interface CompletionVerdict extends DefaultVerdict {
|
|
96
|
-
taskId: string;
|
|
97
|
-
requirements: RequirementCheck[];
|
|
98
|
-
/** satisfied / MEASURABLE requirements (unmeasured rows leave the denominator). */
|
|
99
|
-
completionRate: number;
|
|
100
|
-
/** Every measurable requirement satisfied (false when anything is unmeasured). */
|
|
101
|
-
fullyComplete: boolean;
|
|
102
|
-
/** Requirements whose correctness check errored — reported, never scored as zero. */
|
|
103
|
-
unmeasuredCount: number;
|
|
104
|
-
}
|
|
105
|
-
/**
|
|
106
|
-
* Construct a `CompletionVerdict` from the per-requirement checks, deriving
|
|
107
|
-
* `completionRate` / `fullyComplete` and the spine fields (`valid` =
|
|
108
|
-
* `fullyComplete`, `score` = `completionRate`) in one place. Throws on zero
|
|
109
|
-
* requirements — a verdict over nothing is a misconfiguration, mirroring
|
|
110
|
-
* `verifyCompletion`'s gold-spec guard.
|
|
111
|
-
*/
|
|
112
|
-
declare function completionVerdict(input: {
|
|
113
|
-
taskId: string;
|
|
114
|
-
requirements: RequirementCheck[];
|
|
115
|
-
/** What certified the correctness stage, when anything did. Omitted =
|
|
116
|
-
* an uncertified verdict — the honest default for a bare checker. */
|
|
117
|
-
certification?: VerdictCertification;
|
|
118
|
-
}): CompletionVerdict;
|
|
119
|
-
/**
|
|
120
|
-
* What a correctness checker declares about itself so the completion
|
|
121
|
-
* verdict can carry a certification: which strategy member it discharges,
|
|
122
|
-
* its exact identity, and the steps its answers rest on unverified.
|
|
123
|
-
*/
|
|
124
|
-
interface CorrectnessCheckerAttestation {
|
|
125
|
-
strategy: VerificationStrategySource;
|
|
126
|
-
checker: CheckerIdentity;
|
|
127
|
-
assumptions: string[];
|
|
128
|
-
}
|
|
129
|
-
/**
|
|
130
|
-
* Decides whether a produced item's content actually fulfils a requirement.
|
|
131
|
-
* Injected so the structural verifier stays pure and unit-testable; the
|
|
132
|
-
* production implementation is `createLlmCorrectnessChecker`.
|
|
133
|
-
*
|
|
134
|
-
* `attestation` is optional metadata on the function value: a checker that
|
|
135
|
-
* carries one yields certified completion verdicts; a bare function yields
|
|
136
|
-
* the same verdict uncertified. A plain arrow function remains a valid
|
|
137
|
-
* checker.
|
|
138
|
-
*/
|
|
139
|
-
interface CorrectnessChecker {
|
|
140
|
-
(requirement: CompletionRequirement, content: string): Promise<{
|
|
141
|
-
correct: boolean;
|
|
142
|
-
reason: string;
|
|
143
|
-
}>;
|
|
144
|
-
attestation?: CorrectnessCheckerAttestation;
|
|
145
|
-
}
|
|
146
|
-
/**
|
|
147
|
-
* Verify whether a run completed the task. `checkCorrectness` is injected —
|
|
148
|
-
* `createLlmCorrectnessChecker` for production, a deterministic stub in tests.
|
|
149
|
-
*
|
|
150
|
-
* Throws on a gold spec with no requirements: an eval task that requires
|
|
151
|
-
* nothing is a misconfiguration, not a vacuously-complete task.
|
|
152
|
-
*/
|
|
153
|
-
declare function verifyCompletion(gold: TaskGold, state: ProducedState, checkCorrectness: CorrectnessChecker): Promise<CompletionVerdict>;
|
|
154
|
-
interface LlmCorrectnessCheckerOpts {
|
|
155
|
-
model?: string;
|
|
156
|
-
/** Optional ledger for direct use. */
|
|
157
|
-
costLedger?: CostLedgerHandle;
|
|
158
|
-
costPhase?: string;
|
|
159
|
-
costTags?: Record<string, string>;
|
|
160
|
-
signal?: AbortSignal;
|
|
161
|
-
/** Max chars of artifact content sent to the checker. */
|
|
162
|
-
maxContentChars?: number;
|
|
163
|
-
/**
|
|
164
|
-
* Checker LLM calls per requirement before giving up (parse failures and
|
|
165
|
-
* call errors both consume attempts). The failure then surfaces as an
|
|
166
|
-
* `unmeasured` requirement, never a zero.
|
|
167
|
-
*/
|
|
168
|
-
maxAttempts?: number;
|
|
169
|
-
/**
|
|
170
|
-
* Forensic capture of every checker request/response/error — without it a
|
|
171
|
-
* checker failure is unauditable (the agent-turn raws never contain the
|
|
172
|
-
* checker's own calls). Same sink contract as `LlmClient`.
|
|
173
|
-
*/
|
|
174
|
-
rawSink?: RawProviderSink;
|
|
175
|
-
}
|
|
176
|
-
/**
|
|
177
|
-
* Production `CorrectnessChecker` — one LLM call per matched artifact,
|
|
178
|
-
* deterministic (temperature 0), structured JSON out. Judges fulfilment
|
|
179
|
-
* only: a plan, a gesture, or a description of what should be done does not
|
|
180
|
-
* fulfil a requirement — the artifact must BE the deliverable.
|
|
181
|
-
*/
|
|
182
|
-
declare function createLlmCorrectnessChecker(chat: ChatClient, opts?: LlmCorrectnessCheckerOpts): CorrectnessChecker;
|
|
183
|
-
//#endregion
|
|
184
|
-
//#region src/produced-state.d.ts
|
|
185
|
-
/** A tool the agent invoked. */
|
|
186
|
-
interface ToolCallEventLike {
|
|
187
|
-
type: 'tool_call';
|
|
188
|
-
toolName: string;
|
|
189
|
-
}
|
|
190
|
-
/**
|
|
191
|
-
* An artifact the agent produced. `content` is the enriched field — the
|
|
192
|
-
* runtime's base `artifact` event carries only metadata; the completion
|
|
193
|
-
* oracle needs the body to verify the deliverable, so the runtime emits it.
|
|
194
|
-
*/
|
|
195
|
-
interface ArtifactEventLike {
|
|
196
|
-
type: 'artifact';
|
|
197
|
-
artifactId: string;
|
|
198
|
-
name?: string;
|
|
199
|
-
mimeType?: string;
|
|
200
|
-
uri?: string;
|
|
201
|
-
content?: string;
|
|
202
|
-
}
|
|
203
|
-
/** A proposal / filing the agent created. */
|
|
204
|
-
interface ProposalEventLike {
|
|
205
|
-
type: 'proposal_created';
|
|
206
|
-
proposalId: string;
|
|
207
|
-
title: string;
|
|
208
|
-
status?: 'pending' | 'approved' | 'rejected';
|
|
209
|
-
content?: string;
|
|
210
|
-
}
|
|
211
|
-
/**
|
|
212
|
-
* The subset of runtime stream events `extractProducedState` consumes.
|
|
213
|
-
* agent-runtime's full `RuntimeStreamEvent` union satisfies this structurally;
|
|
214
|
-
* the `{ type: string }` catch-all keeps the input permissive so callers can
|
|
215
|
-
* pass the whole unfiltered telemetry stream — unrecognized events are skipped.
|
|
216
|
-
*/
|
|
217
|
-
type RuntimeEventLike = ToolCallEventLike | ArtifactEventLike | ProposalEventLike | {
|
|
218
|
-
type: string;
|
|
219
|
-
};
|
|
220
|
-
/**
|
|
221
|
-
* Normalize a run's runtime event stream into `ProducedState`.
|
|
222
|
-
*
|
|
223
|
-
* Pure and total — unrecognized event types are skipped. `toolCalls` is
|
|
224
|
-
* deduplicated by name in first-seen order (completion cares about a tool's
|
|
225
|
-
* presence, not its call count). An artifact with neither a name nor a uri
|
|
226
|
-
* still yields an entry keyed by its `artifactId` so it is never silently
|
|
227
|
-
* dropped; an artifact with no `content` yields empty content, which the
|
|
228
|
-
* completion oracle's structural check then rejects on its own.
|
|
229
|
-
*/
|
|
230
|
-
declare function extractProducedState(events: readonly RuntimeEventLike[]): ProducedState;
|
|
231
|
-
//#endregion
|
|
232
|
-
//#region src/integrity/backend-integrity.d.ts
|
|
233
|
-
interface BackendIntegrityReport {
|
|
234
|
-
/** Total records inspected. */
|
|
235
|
-
totalRecords: number;
|
|
236
|
-
/** Records with input=0 AND output=0 (a stub fingerprint). */
|
|
237
|
-
stubRecords: number;
|
|
238
|
-
/** Records with nonzero token usage (real LLM activity). */
|
|
239
|
-
realRecords: number;
|
|
240
|
-
/** Records where output>0 but costUsd=0 (real LLM, broken cost ledger). */
|
|
241
|
-
uncostedRecords: number;
|
|
242
|
-
/** Sum of input tokens across all records. */
|
|
243
|
-
totalInputTokens: number;
|
|
244
|
-
/** Sum of output tokens across all records. */
|
|
245
|
-
totalOutputTokens: number;
|
|
246
|
-
/** Sum of costUsd across all records. */
|
|
247
|
-
totalCostUsd: number;
|
|
248
|
-
/** Worst-case integrity verdict. */
|
|
249
|
-
verdict: 'real' | 'mixed' | 'stub';
|
|
250
|
-
/** Human-readable diagnosis suitable for terminal output. */
|
|
251
|
-
diagnosis: string;
|
|
252
|
-
}
|
|
253
|
-
/**
|
|
254
|
-
* Error thrown when an integrity assertion fails. Caller can pattern-match
|
|
255
|
-
* by `code === 'AGENT_EVAL_BACKEND_STUB'` to differentiate from other
|
|
256
|
-
* errors.
|
|
257
|
-
*/
|
|
258
|
-
declare class BackendIntegrityError extends AgentEvalError {
|
|
259
|
-
readonly report: BackendIntegrityReport;
|
|
260
|
-
constructor(message: string, report: BackendIntegrityReport);
|
|
261
|
-
}
|
|
262
|
-
/**
|
|
263
|
-
* Inspect a batch of RunRecords and return an integrity report. Pure
|
|
264
|
-
* function — no I/O, no logging. The caller decides what to do with the
|
|
265
|
-
* verdict (print warning, throw, gate CI, etc.).
|
|
266
|
-
*/
|
|
267
|
-
declare function summarizeBackendIntegrity(records: ReadonlyArray<RunRecord>): BackendIntegrityReport;
|
|
268
|
-
/**
|
|
269
|
-
* Throw BackendIntegrityError if the verdict is 'stub' — i.e. every record
|
|
270
|
-
* shows zero LLM activity. Non-strict callers can pass `{ allowMixed: false }`
|
|
271
|
-
* to also reject mixed verdicts (recommended for CI gates).
|
|
272
|
-
*
|
|
273
|
-
* Real backends pass through silently.
|
|
274
|
-
*/
|
|
275
|
-
declare function assertRealBackend(records: ReadonlyArray<RunRecord>, opts?: {
|
|
276
|
-
allowMixed?: boolean;
|
|
277
|
-
}): BackendIntegrityReport;
|
|
278
|
-
//#endregion
|
|
279
|
-
export { SatisfiedBy as _, ArtifactEventLike as a, createLlmCorrectnessChecker as b, ToolCallEventLike as c, CompletionVerdict as d, CorrectnessChecker as f, RequirementCheck as g, ProducedState as h, summarizeBackendIntegrity as i, extractProducedState as l, ProducedProposal as m, BackendIntegrityReport as n, ProposalEventLike as o, LlmCorrectnessCheckerOpts as p, assertRealBackend as r, RuntimeEventLike as s, BackendIntegrityError as t, CompletionRequirement as u, TaskGold as v, verifyCompletion as x, completionVerdict as y };
|
|
280
|
-
//# sourceMappingURL=backend-integrity-CeuTgqsd.d.ts.map
|
|
@@ -1 +0,0 @@
|
|
|
1
|
-
{"version":3,"file":"backend-integrity-CeuTgqsd.d.ts","names":[],"sources":["../src/artifact-validator.ts","../src/completion-verifier.ts","../src/produced-state.ts","../src/integrity/backend-integrity.ts"],"mappings":";;;;;;;;;;;;;;;;;;UAaiB;;EAEf;;EAEA;;EAEA;;EAEA,QAAQ;;EAER,WAAW;;;;;KCsBD;UAEK;;EAEf;;EAEA;;EAEA;;EAEA,cAAc;;UAGC;EACf;EACA,cAAc;;UAGC;EACf;EACA;EACA;;EAEA;;;UAIe;;EAEf,WAAW;;EAEX,WAAW;;EAEX;;UAGe;EACf;EACA;;EAEA;;;;;;EAMA;;EAEA;;;;;;;;;EASA;;EAEA;;EAEA;;;;;UAMe,0BAA0B;EACzC;EACA,cAAc;;EAEd;;EAEA;;EAEA;;;;;;;;;iBAUc,kBAAkB;EAChC;EACA,cAAc;;;EAGd,gBAAgB;IACd;;;;;;UAqCa;EACf,UAAU;EACV,SAAS;EACT;;;;;;;;;;;;UAae;GAEb,aAAa,uBACb,kBACC;IAAU;IAAkB;;EAC/B,cAAc;;;;;;;;;iBAsLM,iBACpB,MAAM,UACN,OAAO,eACP,kBAAkB,qBACjB,QAAQ;UAyGM;EACf;;EAEA,aAAa;EACb;EACA,WAAW;EACX,SAAS;;EAET;;;;;;EAMA;;;;;;EAMA,UAAU;;;;;;;;iBA2CI,4BACd,MAAM,YACN,OAAM,4BACL;;;;UCnhBc;EACf;EACA;;;;;;;UAQe;EACf;EACA;EACA;EACA;EACA;EACA;;;UAIe;EACf;EACA;EACA;EACA;EAIA;;;;;;;;KASU,mBACR,oBACA,oBACA;EACE;;;;;;;;;;;;iBAmBU,qBAAqB,iBAAiB,qBAAqB;;;UCpD1D;;EAEf;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;;;;;;cAQW,8BAA8B;WAGvB,QAAQ;EAF1B,YACE,iBACgB,QAAQ;;;;;;;iBAWZ,0BACd,SAAS,cAAc,aACtB;;;;;;;;iBAuHa,kBACd,SAAS,cAAc,YACvB;EAAQ;IACP"}
|
|
@@ -1,236 +0,0 @@
|
|
|
1
|
-
import { c as AnalystRunInputs, f as AnalystUsageReceipt, i as AnalystFinding, p as EvidenceRef, w as TraceAnalysisStore } from "./types-DN2WdT5S.js";
|
|
2
|
-
import { c as RegistryRunOpts, n as AnalystRegistry } from "./registry-7pOUBrtX.js";
|
|
3
|
-
//#region src/analyst/benchmark-scoring.d.ts
|
|
4
|
-
declare function scoreAnalystFindings(testCase: Pick<AnalystBenchmarkCase, 'id' | 'expectedIssues' | 'labeledEvidence'>, findings: readonly AnalystFinding[]): AnalystFindingScore;
|
|
5
|
-
//#endregion
|
|
6
|
-
//#region src/analyst/benchmark.d.ts
|
|
7
|
-
interface AnalystEvidenceExpectation {
|
|
8
|
-
uri: string;
|
|
9
|
-
kind?: EvidenceRef['kind'];
|
|
10
|
-
}
|
|
11
|
-
interface AnalystIssueExpectation {
|
|
12
|
-
id: string;
|
|
13
|
-
findingIds?: readonly string[];
|
|
14
|
-
areas?: readonly string[];
|
|
15
|
-
subjects?: readonly string[];
|
|
16
|
-
evidence?: readonly AnalystEvidenceExpectation[];
|
|
17
|
-
evidenceMode?: 'any' | 'all';
|
|
18
|
-
/** Exact evidence location for the first unrecoverable or causal step. */
|
|
19
|
-
criticalEvidence?: readonly AnalystEvidenceExpectation[];
|
|
20
|
-
}
|
|
21
|
-
type AnalystBenchmarkLabelState = 'positive' | 'trusted-negative' | 'unlabeled';
|
|
22
|
-
interface AnalystBenchmarkCase<TInput = unknown> {
|
|
23
|
-
id: string;
|
|
24
|
-
/** Independent source unit used for resampling, such as a task or incident. */
|
|
25
|
-
clusterId: string;
|
|
26
|
-
/** Whether labels prove an issue, prove no issue, or leave the outcome unknown. */
|
|
27
|
-
labelState: AnalystBenchmarkLabelState;
|
|
28
|
-
input: TInput;
|
|
29
|
-
expectedIssues: readonly AnalystIssueExpectation[];
|
|
30
|
-
/** Complete set of labeled locations used to measure label-location agreement. */
|
|
31
|
-
labeledEvidence?: readonly AnalystEvidenceExpectation[];
|
|
32
|
-
tags?: readonly string[];
|
|
33
|
-
metadata?: Record<string, unknown>;
|
|
34
|
-
}
|
|
35
|
-
interface AnalystFindingScore {
|
|
36
|
-
expectedIssueCount: number;
|
|
37
|
-
matchedIssueIds: string[];
|
|
38
|
-
missedIssueIds: string[];
|
|
39
|
-
supportedFindingIndexes: number[];
|
|
40
|
-
unsupportedFindingIndexes: number[];
|
|
41
|
-
unlabeledEvidence: EvidenceRef[];
|
|
42
|
-
issueRecall: number;
|
|
43
|
-
findingPrecision: number;
|
|
44
|
-
f1: number;
|
|
45
|
-
criticalStepAccuracy: number | null;
|
|
46
|
-
/** Share of findings that cite at least one evidence location. */
|
|
47
|
-
citationCoverage: number | null;
|
|
48
|
-
/** Share of citations that include a non-empty source excerpt. */
|
|
49
|
-
citationExcerptCoverage: number | null;
|
|
50
|
-
/** Share of citations that agree with a labeled case location. */
|
|
51
|
-
citationLabelAgreement: number | null;
|
|
52
|
-
predictionOnLabelEmptyCase: boolean;
|
|
53
|
-
}
|
|
54
|
-
interface AnalystEvidenceResolutionError {
|
|
55
|
-
evidence: EvidenceRef;
|
|
56
|
-
class: string;
|
|
57
|
-
message: string;
|
|
58
|
-
}
|
|
59
|
-
interface AnalystEvidenceResolution {
|
|
60
|
-
checked: number;
|
|
61
|
-
resolved: number;
|
|
62
|
-
unresolvedEvidence: EvidenceRef[];
|
|
63
|
-
errors: AnalystEvidenceResolutionError[];
|
|
64
|
-
/** Null when no citations were checked or any resolution attempt failed. */
|
|
65
|
-
validity: number | null;
|
|
66
|
-
}
|
|
67
|
-
type AnalystEvidenceResolver<TInput = unknown> = (input: {
|
|
68
|
-
caseId: string;
|
|
69
|
-
caseInput: TInput;
|
|
70
|
-
evidence: EvidenceRef;
|
|
71
|
-
signal?: AbortSignal;
|
|
72
|
-
}) => boolean | Promise<boolean>;
|
|
73
|
-
/**
|
|
74
|
-
* Resolve canonical `trace://<trace>/span/<span>` evidence against a trace store.
|
|
75
|
-
* Other evidence kinds and URI schemes require a caller-supplied resolver.
|
|
76
|
-
*/
|
|
77
|
-
declare function traceStoreEvidenceResolver<TInput>(getStore: (input: TInput) => TraceAnalysisStore): AnalystEvidenceResolver<TInput>;
|
|
78
|
-
interface AnalystBenchmarkOutput {
|
|
79
|
-
findings: readonly AnalystFinding[];
|
|
80
|
-
usage?: AnalystUsageReceipt;
|
|
81
|
-
metadata?: Record<string, unknown>;
|
|
82
|
-
/**
|
|
83
|
-
* End-to-end duration measured by an external runner before import.
|
|
84
|
-
* Use null when the source explicitly did not capture duration.
|
|
85
|
-
*/
|
|
86
|
-
observedLatencyMs?: number | null;
|
|
87
|
-
/** Marks a completed transport as a failed analyst run while retaining usage and metadata. */
|
|
88
|
-
error?: AnalystBenchmarkError;
|
|
89
|
-
}
|
|
90
|
-
interface AnalystBenchmarkError {
|
|
91
|
-
class: string;
|
|
92
|
-
message: string;
|
|
93
|
-
code?: string;
|
|
94
|
-
status?: number;
|
|
95
|
-
}
|
|
96
|
-
interface AnalystBenchmarkRunner<TInput = unknown> {
|
|
97
|
-
id: string;
|
|
98
|
-
analyze(input: TInput, context: {
|
|
99
|
-
caseId: string;
|
|
100
|
-
repetition: number;
|
|
101
|
-
signal?: AbortSignal;
|
|
102
|
-
}): AnalystBenchmarkOutput | Promise<AnalystBenchmarkOutput>;
|
|
103
|
-
}
|
|
104
|
-
interface AnalystBenchmarkObservation {
|
|
105
|
-
runnerId: string;
|
|
106
|
-
caseId: string;
|
|
107
|
-
clusterId: string;
|
|
108
|
-
labelState: AnalystBenchmarkLabelState;
|
|
109
|
-
repetition: number;
|
|
110
|
-
executionIndex: number;
|
|
111
|
-
latencyMs: number | null;
|
|
112
|
-
latencySource: 'benchmark-clock' | 'runner-reported' | 'uncaptured';
|
|
113
|
-
findings: readonly AnalystFinding[];
|
|
114
|
-
score: AnalystFindingScore;
|
|
115
|
-
evidenceResolution?: AnalystEvidenceResolution;
|
|
116
|
-
caseTags: readonly string[];
|
|
117
|
-
caseMetadata?: Record<string, unknown>;
|
|
118
|
-
usage?: AnalystUsageReceipt;
|
|
119
|
-
runnerMetadata?: Record<string, unknown>;
|
|
120
|
-
error?: AnalystBenchmarkError;
|
|
121
|
-
}
|
|
122
|
-
interface AnalystLatencyDistribution {
|
|
123
|
-
min: number;
|
|
124
|
-
mean: number;
|
|
125
|
-
p50: number;
|
|
126
|
-
p95: number;
|
|
127
|
-
max: number;
|
|
128
|
-
}
|
|
129
|
-
interface AnalystBenchmarkSummary {
|
|
130
|
-
runnerId: string;
|
|
131
|
-
plannedRuns: number;
|
|
132
|
-
completedRuns: number;
|
|
133
|
-
failedRuns: number;
|
|
134
|
-
issueBearingRuns: number;
|
|
135
|
-
trustedNegativeRuns: number;
|
|
136
|
-
unlabeledRuns: number;
|
|
137
|
-
/** Pooled across all labeled issues and findings. */
|
|
138
|
-
issueRecall: number | null;
|
|
139
|
-
/** Pooled across all labeled issues and findings. */
|
|
140
|
-
findingPrecision: number | null;
|
|
141
|
-
/** Harmonic mean of the pooled precision and recall. */
|
|
142
|
-
f1: number | null;
|
|
143
|
-
/** Mean of per-case recall over issue-bearing runs. */
|
|
144
|
-
macroIssueRecall: number | null;
|
|
145
|
-
/** Mean of per-case precision over issue-bearing runs. */
|
|
146
|
-
macroFindingPrecision: number | null;
|
|
147
|
-
/** Mean of per-case F1 over issue-bearing runs. */
|
|
148
|
-
macroF1: number | null;
|
|
149
|
-
criticalStepAccuracy: number | null;
|
|
150
|
-
citationCoverage: number | null;
|
|
151
|
-
citationExcerptCoverage: number | null;
|
|
152
|
-
citationLabelAgreement: number | null;
|
|
153
|
-
citationResolution: number | null;
|
|
154
|
-
citationResolutionUnknownRuns: number;
|
|
155
|
-
unresolvedCitations: number;
|
|
156
|
-
citationResolutionErrors: number;
|
|
157
|
-
trustedNegativeFalsePositiveRate: number | null;
|
|
158
|
-
trustedNegativeFailureRate: number | null;
|
|
159
|
-
unlabeledPredictionRate: number | null;
|
|
160
|
-
unlabeledFailureRate: number | null;
|
|
161
|
-
/** Primary repeatability measure over complete finding identity and evidence. */
|
|
162
|
-
predictionAgreement: number | null;
|
|
163
|
-
/** Repeated cases contributing equally to predictionAgreement. */
|
|
164
|
-
predictionAgreementCases: number;
|
|
165
|
-
/** Secondary repeatability detail over matched expected labels. */
|
|
166
|
-
matchedLabelAgreement: number | null;
|
|
167
|
-
/** Positive repeated cases contributing equally to matchedLabelAgreement. */
|
|
168
|
-
matchedLabelAgreementCases: number;
|
|
169
|
-
latencyMs: AnalystLatencyDistribution | null;
|
|
170
|
-
benchmarkClockLatencyRuns: number;
|
|
171
|
-
runnerReportedLatencyRuns: number;
|
|
172
|
-
latencyUnknownRuns: number;
|
|
173
|
-
calls: number;
|
|
174
|
-
callsUnknownRuns: number;
|
|
175
|
-
inputTokens: number;
|
|
176
|
-
outputTokens: number;
|
|
177
|
-
reasoningTokens: number;
|
|
178
|
-
cachedTokens: number;
|
|
179
|
-
cacheWriteTokens: number;
|
|
180
|
-
tokenUsageUnknownRuns: number;
|
|
181
|
-
reasoningTokenUsageUnknownRuns: number;
|
|
182
|
-
cachedTokenUsageUnknownRuns: number;
|
|
183
|
-
cacheWriteTokenUsageUnknownRuns: number;
|
|
184
|
-
knownCostUsd: number;
|
|
185
|
-
costUnknownRuns: number;
|
|
186
|
-
}
|
|
187
|
-
interface AnalystBenchmarkDatasetRef {
|
|
188
|
-
id: string;
|
|
189
|
-
revision: string;
|
|
190
|
-
split?: string;
|
|
191
|
-
}
|
|
192
|
-
interface AnalystBenchmarkDescriptor {
|
|
193
|
-
id?: string;
|
|
194
|
-
dataset?: AnalystBenchmarkDatasetRef;
|
|
195
|
-
command?: string;
|
|
196
|
-
environment?: Record<string, string>;
|
|
197
|
-
metadata?: Record<string, unknown>;
|
|
198
|
-
}
|
|
199
|
-
interface AnalystBenchmarkProvenance extends AnalystBenchmarkDescriptor {
|
|
200
|
-
startedAt: string;
|
|
201
|
-
endedAt: string;
|
|
202
|
-
caseCount: number;
|
|
203
|
-
runnerIds: string[];
|
|
204
|
-
repetitions: number;
|
|
205
|
-
maxConcurrency: number;
|
|
206
|
-
runnerOrderSeed: number;
|
|
207
|
-
}
|
|
208
|
-
interface AnalystBenchmarkResult {
|
|
209
|
-
provenance: AnalystBenchmarkProvenance;
|
|
210
|
-
observations: AnalystBenchmarkObservation[];
|
|
211
|
-
summaries: AnalystBenchmarkSummary[];
|
|
212
|
-
}
|
|
213
|
-
interface RunAnalystBenchmarkOptions<TInput> {
|
|
214
|
-
cases: readonly AnalystBenchmarkCase<TInput>[];
|
|
215
|
-
runners: readonly AnalystBenchmarkRunner<TInput>[];
|
|
216
|
-
repetitions?: number;
|
|
217
|
-
maxConcurrency?: number;
|
|
218
|
-
runnerOrderSeed?: number;
|
|
219
|
-
resolveEvidence?: AnalystEvidenceResolver<TInput>;
|
|
220
|
-
benchmark?: AnalystBenchmarkDescriptor;
|
|
221
|
-
/** Previously persisted rows. Exact case, runner, repetition, and execution identities are required. */
|
|
222
|
-
initialObservations?: readonly AnalystBenchmarkObservation[];
|
|
223
|
-
onObservation?: (observation: AnalystBenchmarkObservation) => void | Promise<void>;
|
|
224
|
-
signal?: AbortSignal;
|
|
225
|
-
}
|
|
226
|
-
declare function runAnalystBenchmark<TInput>(options: RunAnalystBenchmarkOptions<TInput>): Promise<AnalystBenchmarkResult>;
|
|
227
|
-
declare function registryBenchmarkRunner(options: {
|
|
228
|
-
id: string;
|
|
229
|
-
registry: AnalystRegistry;
|
|
230
|
-
runOptions?: Omit<RegistryRunOpts, 'signal'>;
|
|
231
|
-
/** Count any selected analyst failure as a failed benchmark run. */
|
|
232
|
-
failOnAnalystFailure?: boolean;
|
|
233
|
-
}): AnalystBenchmarkRunner<AnalystRunInputs>;
|
|
234
|
-
//#endregion
|
|
235
|
-
export { scoreAnalystFindings as C, traceStoreEvidenceResolver as S, AnalystIssueExpectation as _, AnalystBenchmarkLabelState as a, registryBenchmarkRunner as b, AnalystBenchmarkProvenance as c, AnalystBenchmarkSummary as d, AnalystEvidenceExpectation as f, AnalystFindingScore as g, AnalystEvidenceResolver as h, AnalystBenchmarkError as i, AnalystBenchmarkResult as l, AnalystEvidenceResolutionError as m, AnalystBenchmarkDatasetRef as n, AnalystBenchmarkObservation as o, AnalystEvidenceResolution as p, AnalystBenchmarkDescriptor as r, AnalystBenchmarkOutput as s, AnalystBenchmarkCase as t, AnalystBenchmarkRunner as u, AnalystLatencyDistribution as v, runAnalystBenchmark as x, RunAnalystBenchmarkOptions as y };
|
|
236
|
-
//# sourceMappingURL=benchmark-BjLGkfnN.d.ts.map
|
|
@@ -1 +0,0 @@
|
|
|
1
|
-
{"version":3,"file":"benchmark-BjLGkfnN.d.ts","names":[],"sources":["../src/analyst/benchmark-scoring.ts","../src/analyst/benchmark.ts"],"mappings":";;;iBASgB,qBACd,UAAU,KAAK,oEACf,mBAAmB,mBAClB;;;UCIc;EACf;EACA,OAAO;;UAGQ;EACf;EACA;EACA;EACA;EACA,oBAAoB;EACpB;;EAEA,4BAA4B;;KAGlB;UAEK,qBAAqB;EACpC;;EAEA;;EAEA,YAAY;EACZ,OAAO;EACP,yBAAyB;;EAEzB,2BAA2B;EAC3B;EACA,WAAW;;UAGI;EACf;EACA;EACA;EACA;EACA;EACA,mBAAmB;EACnB;EACA;EACA;EACA;;EAEA;;EAEA;;EAEA;EACA;;UAGe;EACf,UAAU;EACV;EACA;;UAGe;EACf;EACA;EACA,oBAAoB;EACpB,QAAQ;;EAER;;KAGU,wBAAwB,qBAAqB;EACvD;EACA,WAAW;EACX,UAAU;EACV,SAAS;gBACK;;;;;iBAMA,2BAA2B,QACzC,WAAW,OAAO,WAAW,qBAC5B,wBAAwB;UAoBV;EACf,mBAAmB;EACnB,QAAQ;EACR,WAAW;;;;;EAKX;;EAEA,QAAQ;;UAGO;EACf;EACA;EACA;EACA;;UAGe,uBAAuB;EACtC;EACA,QACE,OAAO,QACP;IAAW;IAAgB;IAAoB,SAAS;MACvD,yBAAyB,QAAQ;;UAGrB;EACf;EACA;EACA;EACA,YAAY;EACZ;EACA;EACA;EACA;EACA,mBAAmB;EACnB,OAAO;EACP,qBAAqB;EACrB;EACA,eAAe;EACf,QAAQ;EACR,iBAAiB;EACjB,QAAQ;;UAGO;EACf;EACA;EACA;EACA;EACA;;UAGe;EACf;EACA;EACA;EACA;EACA;EACA;EACA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;;EAEA;;EAEA;;EAEA;;EAEA;EACA,WAAW;EACX;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;;UAGe;EACf;EACA;EACA;;UAGe;EACf;EACA,UAAU;EACV;EACA,cAAc;EACd,WAAW;;UAGI,mCAAmC;EAClD;EACA;EACA;EACA;EACA;EACA;EACA;;UAGe;EACf,YAAY;EACZ,cAAc;EACd,WAAW;;UAGI,2BAA2B;EAC1C,gBAAgB,qBAAqB;EACrC,kBAAkB,uBAAuB;EACzC;EACA;EACA;EACA,kBAAkB,wBAAwB;EAC1C,YAAY;;EAEZ,+BAA+B;EAC/B,iBAAiB,aAAa,uCAAuC;EACrE,SAAS;;iBAGW,oBAAoB,QACxC,SAAS,2BAA2B,UACnC,QAAQ;iBAyDK,wBAAwB;EACtC;EACA,UAAU;EACV,aAAa,KAAK;;EAElB;IACE,uBAAuB"}
|