@tangle-network/agent-eval 0.170.0 → 0.172.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +68 -0
- package/README.md +3 -0
- package/dist/adapters/http.d.ts +2 -2
- package/dist/analyst/index.d.ts +5 -5
- package/dist/analyst/index.js +2 -2
- package/dist/{experiment-tracker-Dm8yQMqb.d.ts → attestation-CJBGmMVh.d.ts} +78 -2
- package/dist/attestation-CJBGmMVh.d.ts.map +1 -0
- package/dist/{experiment-tracker-BKEumQug.js → attestation-XSUpbc4o.js} +96 -2
- package/dist/attestation-XSUpbc4o.js.map +1 -0
- package/dist/{benchmark-command-BA7qOdWw.js → benchmark-command-DoFcisuM.js} +2 -2
- package/dist/{benchmark-command-BA7qOdWw.js.map → benchmark-command-DoFcisuM.js.map} +1 -1
- package/dist/benchmarks/index.d.ts +4 -4
- package/dist/benchmarks/index.js +2 -2
- package/dist/bounded-process-CVOC_D3H.js +181 -0
- package/dist/bounded-process-CVOC_D3H.js.map +1 -0
- package/dist/builder-eval/index.js +39 -98
- package/dist/builder-eval/index.js.map +1 -1
- package/dist/campaign/index.d.ts +8 -7
- package/dist/campaign/index.js +4 -4
- package/dist/{campaign-BeCbxFqs.js → campaign-Dp35pBbS.js} +4 -4
- package/dist/{campaign-BeCbxFqs.js.map → campaign-Dp35pBbS.js.map} +1 -1
- package/dist/canonical-CFpojCN5.d.ts +31 -0
- package/dist/canonical-CFpojCN5.d.ts.map +1 -0
- package/dist/{chat-client-DEtybj5i.js → chat-client-DI79OPye.js} +2 -2
- package/dist/{chat-client-DEtybj5i.js.map → chat-client-DI79OPye.js.map} +1 -1
- package/dist/cli.js +1 -1
- package/dist/{client-_Fsa5c2_.d.ts → client-Df7wdslk.d.ts} +2 -2
- package/dist/{client-_Fsa5c2_.d.ts.map → client-Df7wdslk.d.ts.map} +1 -1
- package/dist/contract/index.d.ts +9 -9
- package/dist/contract/index.js +4 -4
- package/dist/{default-registry-ovxrOP0_.d.ts → default-registry-XxedTLwu.d.ts} +3 -3
- package/dist/{default-registry-ovxrOP0_.d.ts.map → default-registry-XxedTLwu.d.ts.map} +1 -1
- package/dist/{define-agent-eval-DVJm8Xlh.d.ts → define-agent-eval-0wW7gFhr.d.ts} +4 -4
- package/dist/{define-agent-eval-DVJm8Xlh.d.ts.map → define-agent-eval-0wW7gFhr.d.ts.map} +1 -1
- package/dist/{define-agent-eval-Clj-8igZ.js → define-agent-eval-jS8xj_Q_.js} +2 -2
- package/dist/{define-agent-eval-Clj-8igZ.js.map → define-agent-eval-jS8xj_Q_.js.map} +1 -1
- package/dist/descriptive-B2iPaT9J.d.ts +89 -0
- package/dist/descriptive-B2iPaT9J.d.ts.map +1 -0
- package/dist/{engine-D12Rb6WB.d.ts → engine-BfRay1qD.d.ts} +2 -2
- package/dist/{engine-D12Rb6WB.d.ts.map → engine-BfRay1qD.d.ts.map} +1 -1
- package/dist/experiment/index.d.ts +3 -54
- package/dist/experiment/index.d.ts.map +1 -1
- package/dist/experiment/index.js +1 -95
- package/dist/experiment/index.js.map +1 -1
- package/dist/{heldout-gate-Bn7_xWCv.d.ts → heldout-gate-JgNRDZwZ.d.ts} +3 -3
- package/dist/{heldout-gate-Bn7_xWCv.d.ts.map → heldout-gate-JgNRDZwZ.d.ts.map} +1 -1
- package/dist/hosted/index.d.ts +1 -1
- package/dist/{index-lfaSeKSD.d.ts → index-D-UdhAmg.d.ts} +3 -31
- package/dist/index-D-UdhAmg.d.ts.map +1 -0
- package/dist/{index-CM-SM00y.d.ts → index-DDAPhUJJ.d.ts} +4 -4
- package/dist/{index-CM-SM00y.d.ts.map → index-DDAPhUJJ.d.ts.map} +1 -1
- package/dist/{index-Bfs5aufo.d.ts → index-DnglhM0A.d.ts} +9 -9
- package/dist/{index-Bfs5aufo.d.ts.map → index-DnglhM0A.d.ts.map} +1 -1
- package/dist/{index-DBbivBNs.d.ts → index-_vPrVMRX.d.ts} +6 -6
- package/dist/{index-DBbivBNs.d.ts.map → index-_vPrVMRX.d.ts.map} +1 -1
- package/dist/index.d.ts +245 -15
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +712 -9
- package/dist/index.js.map +1 -1
- package/dist/{integrity-CyWSSoQS.js → integrity-BWywb34E.js} +34 -11
- package/dist/{integrity-CyWSSoQS.js.map → integrity-BWywb34E.js.map} +1 -1
- package/dist/ledger-core/index.d.ts +2 -1
- package/dist/{llm-judge-DbJdo8Nj.js → llm-judge-aQHIk5_-.js} +20 -11
- package/dist/llm-judge-aQHIk5_-.js.map +1 -0
- package/dist/{matrix-BpI5Trmo.d.ts → matrix-Ch8JO1pG.d.ts} +2 -2
- package/dist/{matrix-BpI5Trmo.d.ts.map → matrix-Ch8JO1pG.d.ts.map} +1 -1
- package/dist/meta-eval/index.d.ts +162 -2
- package/dist/meta-eval/index.d.ts.map +1 -1
- package/dist/meta-eval/index.js +287 -2
- package/dist/meta-eval/index.js.map +1 -1
- package/dist/multishot/golden/index.d.ts +1 -1
- package/dist/multishot/index.d.ts +2 -2
- package/dist/openapi.json +1 -1
- package/dist/{produced-state-CtSIp5cQ.js → produced-state-CxmbFxFd.js} +2 -2
- package/dist/{produced-state-CtSIp5cQ.js.map → produced-state-CxmbFxFd.js.map} +1 -1
- package/dist/{promotion-policy-CkXSgKkF.d.ts → promotion-policy-BBBcz5_3.d.ts} +2 -2
- package/dist/{promotion-policy-CkXSgKkF.d.ts.map → promotion-policy-BBBcz5_3.d.ts.map} +1 -1
- package/dist/{provenance-CIRUardl.d.ts → provenance-Dp-vvyrU.d.ts} +17 -5
- package/dist/provenance-Dp-vvyrU.d.ts.map +1 -0
- package/dist/rl.d.ts +1 -1
- package/dist/sandbox-harness-BlSOu4LX.d.ts.map +1 -1
- package/dist/{skillopt-optimization-method-B2R9C5aG.js → skillopt-optimization-method-LHi02MzH.js} +2 -2
- package/dist/{skillopt-optimization-method-B2R9C5aG.js.map → skillopt-optimization-method-LHi02MzH.js.map} +1 -1
- package/dist/{statistical-heldout-DFS7QGpS.d.ts → statistical-heldout-Yldkntvy.d.ts} +2 -2
- package/dist/{statistical-heldout-DFS7QGpS.d.ts.map → statistical-heldout-Yldkntvy.d.ts.map} +1 -1
- package/dist/{store-tool-spans-B2DJ_82T.d.ts → store-tool-spans-BvdUbeOB.d.ts} +3 -3
- package/dist/{store-tool-spans-B2DJ_82T.d.ts.map → store-tool-spans-BvdUbeOB.d.ts.map} +1 -1
- package/dist/supervisor-run/index.d.ts +35 -2
- package/dist/supervisor-run/index.d.ts.map +1 -1
- package/dist/supervisor-run/index.js +83 -38
- package/dist/supervisor-run/index.js.map +1 -1
- package/dist/{tool-groups-BnXlCJZQ.d.ts → tool-groups-DjwlMBvW.d.ts} +2 -2
- package/dist/tool-groups-DjwlMBvW.d.ts.map +1 -0
- package/dist/trace-repair/index.d.ts +1 -1
- package/dist/traces.d.ts +2 -2
- package/dist/{types-i21ccEkr.d.ts → types-CCZ34qmV.d.ts} +2 -2
- package/dist/{types-i21ccEkr.d.ts.map → types-CCZ34qmV.d.ts.map} +1 -1
- package/dist/{types-DeIUdzNd.d.ts → types-CoPUTiXb.d.ts} +23 -92
- package/dist/types-CoPUTiXb.d.ts.map +1 -0
- package/dist/{types-Dy237wiH.d.ts → types-nokrtr7M.d.ts} +21 -1
- package/dist/types-nokrtr7M.d.ts.map +1 -0
- package/dist/wire/index.d.ts +3 -3
- package/dist/wire/index.d.ts.map +1 -1
- package/docs/campaign-proposers.md +4 -0
- package/docs/eval-surface-map.md +36 -0
- package/docs/insight-report.md +19 -0
- package/docs/plants.md +123 -0
- package/docs/public-api.md +45 -20
- package/package.json +19 -18
- package/dist/experiment-tracker-BKEumQug.js.map +0 -1
- package/dist/experiment-tracker-Dm8yQMqb.d.ts.map +0 -1
- package/dist/index-lfaSeKSD.d.ts.map +0 -1
- package/dist/llm-judge-DbJdo8Nj.js.map +0 -1
- package/dist/provenance-CIRUardl.d.ts.map +0 -1
- package/dist/tool-groups-BnXlCJZQ.d.ts.map +0 -1
- package/dist/types-DeIUdzNd.d.ts.map +0 -1
- package/dist/types-Dy237wiH.d.ts.map +0 -1
package/dist/index.js
CHANGED
|
@@ -1,9 +1,9 @@
|
|
|
1
1
|
import { t as __exportAll } from "./rolldown-runtime-8H4AJuhK.js";
|
|
2
2
|
import { i as JudgeError, o as NotFoundError, r as ConfigError, s as ValidationError, t as AgentEvalError } from "./errors-Dngq5h35.js";
|
|
3
|
-
import { r as canonicalString } from "./canonical-DPyQ_rpt.js";
|
|
4
|
-
import { a as CODING_HARNESSES, c as agentProfileId, d as harnessAxisOf, i as verifyCompletion, l as agentProfileModelId, n as completionVerdict, o as HARNESS_NATIVE_MODEL, r as createLlmCorrectnessChecker, s as agentProfileHash, t as extractProducedState, u as expandProfileAxes } from "./produced-state-
|
|
3
|
+
import { a as hashCanonical, r as canonicalString } from "./canonical-DPyQ_rpt.js";
|
|
4
|
+
import { a as CODING_HARNESSES, c as agentProfileId, d as harnessAxisOf, i as verifyCompletion, l as agentProfileModelId, n as completionVerdict, o as HARNESS_NATIVE_MODEL, r as createLlmCorrectnessChecker, s as agentProfileHash, t as extractProducedState, u as expandProfileAxes } from "./produced-state-CxmbFxFd.js";
|
|
5
5
|
import { n as hashJson, r as manifestContentDigest } from "./pre-registration-D94b7Of5.js";
|
|
6
|
-
import { AGENT_PROFILE_KINDS, agentProfileCellHashMaterial, agentProfileCellKey, buildAgentProfileCell, groupRunsByAgentProfileCell, toAgentProfileJson, verifyAgentProfileCell } from "./profile-cell.js";
|
|
6
|
+
import { AGENT_PROFILE_KINDS, agentProfileCellHashMaterial, agentProfileCellKey, buildAgentProfileCell, groupRunsByAgentProfileCell, toAgentProfileJson, validateAgentProfileCell, verifyAgentProfileCell } from "./profile-cell.js";
|
|
7
7
|
import { t as mulberry32 } from "./random-Dn5fPWkt.js";
|
|
8
8
|
import { a as spearmanR, c as weightedMean, i as ranks, n as partialCredit, o as summarizeNumberSeries, r as pearsonR, s as weightedComposite, t as confidenceInterval } from "./descriptive-1V17A-qa.js";
|
|
9
9
|
import { i as verbosityBias, n as calibrateJudgeContinuous, r as continuousAgreement, t as calibrateJudge } from "./judge-calibration-BnpVKtnb.js";
|
|
@@ -16,12 +16,12 @@ import { t as eProcess } from "./sequential-eprocess-D1jKoihe.js";
|
|
|
16
16
|
import { n as iqr, r as welchsTTest } from "./baseline-BC-eBZ7U.js";
|
|
17
17
|
import { a as isToolSpan, i as isLlmSpan, r as isJudgeSpan, t as FAILURE_CLASSES } from "./schema-CdIX2aHu.js";
|
|
18
18
|
import { d as runsForScenario, r as argHash, s as judgeSpans } from "./query-BPGMVlbM.js";
|
|
19
|
-
import { i as analyzeRuns, o as checkCanaries, r as selfImprove, t as defineAgentEval } from "./define-agent-eval-
|
|
19
|
+
import { i as analyzeRuns, o as checkCanaries, r as selfImprove, t as defineAgentEval } from "./define-agent-eval-jS8xj_Q_.js";
|
|
20
20
|
import { r as observedSplitScore, t as isRealnessGated } from "./reward-nw2xZGZG.js";
|
|
21
21
|
import { a as parseRunRecordSafe, c as validateRunRecord, i as modelHasSnapshot, n as UNKNOWN_MODEL, o as roundTripRunRecord, r as isRunRecord, s as runTaskScore, t as RunRecordValidationError } from "./run-record-DLORoL7t.js";
|
|
22
22
|
import { a as summaryTable, n as gainHistogram, r as paretoChart } from "./summary-report-Bgh8CpNK.js";
|
|
23
23
|
import { n as contentHash, r as fileVerdictCache, t as canonicalJson } from "./verdict-cache-B3eCVQtY.js";
|
|
24
|
-
import { $ as redTeamDataset, C as paretoFrontier, F as surfaceContentHash, K as buildReflectionPrompt, Q as DEFAULT_RED_TEAM_CORPUS, S as dominates, a as aggregateRunScore, ct as BackendIntegrityError, et as redTeamReport, i as transientDispatchFailure, it as runCampaign, lt as assertRealBackend, nt as runCanaries, o as clamp01, q as parseReflectionResponse, t as llmJudge, tt as scoreRedTeamOutput, ut as summarizeBackendIntegrity } from "./llm-judge-
|
|
24
|
+
import { $ as redTeamDataset, C as paretoFrontier, F as surfaceContentHash, K as buildReflectionPrompt, Q as DEFAULT_RED_TEAM_CORPUS, S as dominates, a as aggregateRunScore, ct as BackendIntegrityError, et as redTeamReport, i as transientDispatchFailure, it as runCampaign, lt as assertRealBackend, nt as runCanaries, o as clamp01, q as parseReflectionResponse, t as llmJudge, tt as scoreRedTeamOutput, ut as summarizeBackendIntegrity } from "./llm-judge-aQHIk5_-.js";
|
|
25
25
|
import { a as resolveModelPricing, i as isModelPriced, n as estimateCost, r as estimateTokens, t as MODEL_PRICING } from "./metrics-Qv-cpptD.js";
|
|
26
26
|
import { a as CostLedgerPersistenceError, c as costForTokenPricing, i as CostLedger, l as costForUsage, n as CostCallConflictError, o as CostReceiptCaptureError, r as CostCeilingReachedError, s as CostReservationExceededError, t as CostAccountingIncompleteError, u as modelPriceKey } from "./cost-ledger-B1qx30B4.js";
|
|
27
27
|
import { n as REDACTION_VERSION, r as redactString, t as DEFAULT_REDACTION_RULES } from "./redact-7Aq1ukl-.js";
|
|
@@ -31,8 +31,8 @@ import { n as equivalenceVerdict, t as certificationEvidenceDigest } from "./ver
|
|
|
31
31
|
import { t as TraceEmitter } from "./emitter-DeQHiDMm.js";
|
|
32
32
|
import { t as buildTrajectory } from "./trajectory-D_7rLrvE.js";
|
|
33
33
|
import { t as runCounterfactual } from "./counterfactual-Bjq1mlUu.js";
|
|
34
|
-
import { c as createBoundedTraceAnalysisStore, f as DEFAULT_TRACE_ANALYST_BUDGETS, p as TRACE_ANALYST_TRUNCATION_MARKER_PREFIX, t as createTraceAnalyst } from "./kind-factory-DMeEoMQZ.js";
|
|
35
|
-
import { _ as assertUniqueFindingIds, a as DEFAULT_TRACE_ANALYST_KINDS, b as snapshotAnalystRun, g as analystRunDigest, h as analystFindingDigest, l as FAILURE_MODE_KIND_SPEC, n as buildDefaultAnalystRegistry, r as AnalystRegistry, t as createChatClient, v as completedAnalystReviewQuality, x as validateAnalystReviewDecisions, y as readAnalystReview } from "./chat-client-
|
|
34
|
+
import { N as parseFindingSubject, c as createBoundedTraceAnalysisStore, f as DEFAULT_TRACE_ANALYST_BUDGETS, p as TRACE_ANALYST_TRUNCATION_MARKER_PREFIX, t as createTraceAnalyst } from "./kind-factory-DMeEoMQZ.js";
|
|
35
|
+
import { _ as assertUniqueFindingIds, a as DEFAULT_TRACE_ANALYST_KINDS, b as snapshotAnalystRun, g as analystRunDigest, h as analystFindingDigest, l as FAILURE_MODE_KIND_SPEC, n as buildDefaultAnalystRegistry, r as AnalystRegistry, t as createChatClient, v as completedAnalystReviewQuality, x as validateAnalystReviewDecisions, y as readAnalystReview } from "./chat-client-DI79OPye.js";
|
|
36
36
|
import { OUTPUT_VALUE } from "./trace-attributes.js";
|
|
37
37
|
import { r as extractUsageFromSse, t as extractUsage } from "./extract-usage-BrQ8mCLX.js";
|
|
38
38
|
import { n as InMemoryRawProviderSink, r as NoopRawProviderSink, t as FileSystemRawProviderSink } from "./raw-provider-sink-BQd7mzyT.js";
|
|
@@ -40,8 +40,9 @@ import { d as inferDomainKeywords, h as analyzeTraces, l as describeTraceInsight
|
|
|
40
40
|
import { n as assertRunCaptured, t as RunIntegrityError } from "./integrity-Cy9WHAtb.js";
|
|
41
41
|
import { n as InMemoryTraceStore, t as FileSystemTraceStore } from "./store-DNe_Uv1Q.js";
|
|
42
42
|
import { t as packageVersion$1 } from "./package-version-D7lQHt_-.js";
|
|
43
|
+
import { t as runBoundedProcess } from "./bounded-process-CVOC_D3H.js";
|
|
43
44
|
import { t as runEvalCampaign } from "./eval-campaign-JDTeE6Pl.js";
|
|
44
|
-
import {
|
|
45
|
+
import { a as computeExperimentStats, n as attest, r as verifyAttestation, s as improvementVerdict, t as ATTESTATION_ALGORITHM } from "./attestation-XSUpbc4o.js";
|
|
45
46
|
import "./rollout-Crypdx8s.js";
|
|
46
47
|
import { t as mintRolloutRows } from "./mint-DjfDUMHr.js";
|
|
47
48
|
import { n as pairedEvalueSequence, t as evaluateInterimReleaseConfidence } from "./sequential-CzK5DarL.js";
|
|
@@ -3279,6 +3280,708 @@ function controlFailureClassFromVerification(verification) {
|
|
|
3279
3280
|
return verification.failingLayers?.length ? "instruction_following" : "unknown";
|
|
3280
3281
|
}
|
|
3281
3282
|
//#endregion
|
|
3283
|
+
//#region src/analyst/steer-firewall.ts
|
|
3284
|
+
/** True iff the finding is a JUDGE VERDICT (an acceptance score lifted into a
|
|
3285
|
+
* finding), identified by provenance set at the lift site — independent of
|
|
3286
|
+
* whatever evidence it cites. */
|
|
3287
|
+
function isJudgeVerdict(finding) {
|
|
3288
|
+
return finding.derived_from_judge === true;
|
|
3289
|
+
}
|
|
3290
|
+
/**
|
|
3291
|
+
* THE steer firewall. Fail-loud guard for any path that admits analyst findings
|
|
3292
|
+
* as STEERING input (the `f(trace)` role): rejects — naming the offenders — any
|
|
3293
|
+
* finding whose provenance is a judge verdict, rather than let `J` leak into the
|
|
3294
|
+
* loop. Returns the findings unchanged for chaining.
|
|
3295
|
+
*
|
|
3296
|
+
* Call this at the chokepoint where a detector that ALSO scores/gates has its
|
|
3297
|
+
* findings turned into a steer (the judge-and-steer dual-role case). It keys on
|
|
3298
|
+
* provenance, so it correctly admits evidence-less trace-analyst observations and
|
|
3299
|
+
* correctly rejects an artifact-citing judge verdict — the cases an evidence
|
|
3300
|
+
* check gets backwards.
|
|
3301
|
+
*
|
|
3302
|
+
* It is necessary, not sufficient: it stops PROVENANCE-tagged verdicts. A judge
|
|
3303
|
+
* whose output is laundered through a hand-built finding with no provenance flag
|
|
3304
|
+
* is out of its reach — provenance must be honestly set at every judge→finding
|
|
3305
|
+
* lift (today: createJudgeAdapter). That is why the integrity rule lives at the
|
|
3306
|
+
* lift site, and why ProposeContext.judgeScores?: never is the complementary
|
|
3307
|
+
* compile-time tripwire on the obvious direct channel.
|
|
3308
|
+
*/
|
|
3309
|
+
function assertNoJudgeVerdict(findings, context = "steer") {
|
|
3310
|
+
const leaks = findings.filter(isJudgeVerdict);
|
|
3311
|
+
if (leaks.length > 0) throw new Error(`${context}: a judge verdict cannot be admitted as steering input — that is the held-out judge leaking into the loop. Offending judge-derived findings: [${leaks.map((f) => f.finding_id).join(", ")}]. Steering consumes observations of behavior, never acceptance verdicts.`);
|
|
3312
|
+
return findings;
|
|
3313
|
+
}
|
|
3314
|
+
//#endregion
|
|
3315
|
+
//#region src/analyst/policy-edit.ts
|
|
3316
|
+
const POLICY_EDIT_AXES = [
|
|
3317
|
+
"carrier",
|
|
3318
|
+
"representation",
|
|
3319
|
+
"budget",
|
|
3320
|
+
"sampling",
|
|
3321
|
+
"output_contract",
|
|
3322
|
+
"tool_contract",
|
|
3323
|
+
"routing",
|
|
3324
|
+
"memory",
|
|
3325
|
+
"agent_profile",
|
|
3326
|
+
"deployment_target"
|
|
3327
|
+
];
|
|
3328
|
+
const POLICY_EDIT_TARGET_SURFACES = [
|
|
3329
|
+
"prompt",
|
|
3330
|
+
"tool-contract",
|
|
3331
|
+
"runtime-config",
|
|
3332
|
+
"memory",
|
|
3333
|
+
"agent-profile",
|
|
3334
|
+
"code",
|
|
3335
|
+
"deployment"
|
|
3336
|
+
];
|
|
3337
|
+
const POLICY_EDIT_CANDIDATE_RECORD_SCHEMA = "tangle.policy-edit-candidate.v1";
|
|
3338
|
+
var PolicyEditValidationError = class extends ValidationError {
|
|
3339
|
+
path;
|
|
3340
|
+
constructor(message, path = "") {
|
|
3341
|
+
super(path ? `${message} (at ${path})` : message);
|
|
3342
|
+
this.path = path;
|
|
3343
|
+
}
|
|
3344
|
+
};
|
|
3345
|
+
const DEFAULT_MIN_SCORE = .7;
|
|
3346
|
+
const DEFAULT_MIN_EXPECTED_GAIN = .01;
|
|
3347
|
+
const POLICY_EDIT_ID = /^policy-edit:sha256:[0-9a-f]{64}$/;
|
|
3348
|
+
function makePolicyEdit(init) {
|
|
3349
|
+
const normalized = normalizePolicyEdit({
|
|
3350
|
+
schemaVersion: "policy-edit/v1",
|
|
3351
|
+
...init,
|
|
3352
|
+
source: normalizeSource(init.source)
|
|
3353
|
+
});
|
|
3354
|
+
return validatePolicyEdit({
|
|
3355
|
+
...normalized,
|
|
3356
|
+
editId: init.editId ?? computePolicyEditId(normalized)
|
|
3357
|
+
});
|
|
3358
|
+
}
|
|
3359
|
+
function computePolicyEditId(edit) {
|
|
3360
|
+
const { editId: _editId, schemaVersion, ...material } = edit;
|
|
3361
|
+
return `policy-edit:${hashCanonical({
|
|
3362
|
+
schemaVersion,
|
|
3363
|
+
...material
|
|
3364
|
+
})}`;
|
|
3365
|
+
}
|
|
3366
|
+
function validatePolicyEdit(input) {
|
|
3367
|
+
if (input === null || typeof input !== "object") throw new PolicyEditValidationError("expected object");
|
|
3368
|
+
const obj = input;
|
|
3369
|
+
expectLiteral(obj.schemaVersion, "policy-edit/v1", "schemaVersion");
|
|
3370
|
+
expectString$1(obj.editId, "editId");
|
|
3371
|
+
if (!POLICY_EDIT_ID.test(obj.editId)) throw new PolicyEditValidationError("editId must match policy-edit:sha256:<64 lowercase hex chars>", "editId");
|
|
3372
|
+
expectOneOf(obj.axis, POLICY_EDIT_AXES, "axis");
|
|
3373
|
+
validateTarget(obj.target);
|
|
3374
|
+
validateChange(obj.change);
|
|
3375
|
+
expectString$1(obj.claim, "claim");
|
|
3376
|
+
validateExpectedGain(obj.expectedGain);
|
|
3377
|
+
expectConfidence(obj.confidence, "confidence");
|
|
3378
|
+
expectOneOf(obj.risk, [
|
|
3379
|
+
"low",
|
|
3380
|
+
"medium",
|
|
3381
|
+
"high",
|
|
3382
|
+
"unknown"
|
|
3383
|
+
], "risk");
|
|
3384
|
+
validateSource(obj.source);
|
|
3385
|
+
if (obj.rationale !== void 0) expectString$1(obj.rationale, "rationale");
|
|
3386
|
+
if (obj.validationPlan !== void 0) expectString$1(obj.validationPlan, "validationPlan");
|
|
3387
|
+
if (obj.metadata !== void 0 && (obj.metadata === null || typeof obj.metadata !== "object")) throw new PolicyEditValidationError("expected object", "metadata");
|
|
3388
|
+
const expectedId = computePolicyEditId(obj);
|
|
3389
|
+
if (obj.editId !== expectedId) throw new PolicyEditValidationError("editId does not match policy edit content", "editId");
|
|
3390
|
+
return obj;
|
|
3391
|
+
}
|
|
3392
|
+
function makePolicyEditCandidateRecord(edit) {
|
|
3393
|
+
return validatePolicyEditCandidateRecord({
|
|
3394
|
+
schema: POLICY_EDIT_CANDIDATE_RECORD_SCHEMA,
|
|
3395
|
+
policyEdit: edit
|
|
3396
|
+
});
|
|
3397
|
+
}
|
|
3398
|
+
function validatePolicyEditCandidateRecord(input) {
|
|
3399
|
+
if (input === null || typeof input !== "object" || Array.isArray(input)) throw new PolicyEditValidationError("expected object", "candidateRecord");
|
|
3400
|
+
const obj = input;
|
|
3401
|
+
const keys = Object.keys(obj).sort();
|
|
3402
|
+
if (keys.length !== 2 || keys[0] !== "policyEdit" || keys[1] !== "schema") throw new PolicyEditValidationError("expected exactly schema and policyEdit", "candidateRecord");
|
|
3403
|
+
expectLiteral(obj.schema, POLICY_EDIT_CANDIDATE_RECORD_SCHEMA, "candidateRecord.schema");
|
|
3404
|
+
const policyEdit = validatePolicyEdit(obj.policyEdit);
|
|
3405
|
+
assertJsonSafe(policyEdit, "candidateRecord.policyEdit");
|
|
3406
|
+
const snapshot = JSON.parse(JSON.stringify(policyEdit));
|
|
3407
|
+
return {
|
|
3408
|
+
schema: POLICY_EDIT_CANDIDATE_RECORD_SCHEMA,
|
|
3409
|
+
policyEdit: validatePolicyEdit(snapshot)
|
|
3410
|
+
};
|
|
3411
|
+
}
|
|
3412
|
+
function isPolicyEdit(input) {
|
|
3413
|
+
try {
|
|
3414
|
+
validatePolicyEdit(input);
|
|
3415
|
+
return true;
|
|
3416
|
+
} catch {
|
|
3417
|
+
return false;
|
|
3418
|
+
}
|
|
3419
|
+
}
|
|
3420
|
+
function policyEditsFromFindings(findings, opts = {}) {
|
|
3421
|
+
assertNoJudgeVerdict(findings, "policyEditsFromFindings");
|
|
3422
|
+
const edits = [];
|
|
3423
|
+
for (const finding of findings) {
|
|
3424
|
+
const edit = policyEditFromFinding(finding, opts);
|
|
3425
|
+
if (edit) edits.push(edit);
|
|
3426
|
+
}
|
|
3427
|
+
return edits;
|
|
3428
|
+
}
|
|
3429
|
+
function policyEditFromFinding(finding, opts = {}) {
|
|
3430
|
+
assertNoJudgeVerdict([finding], "policyEditFromFinding");
|
|
3431
|
+
if (!finding.recommended_action?.trim()) return null;
|
|
3432
|
+
const expectedGain = resolveExpectedGain(finding, opts);
|
|
3433
|
+
if (!expectedGain) return null;
|
|
3434
|
+
if (typeof finding.confidence !== "number" || !Number.isFinite(finding.confidence) || finding.confidence < 0 || finding.confidence > 1) return null;
|
|
3435
|
+
const routed = routeFindingSubject(finding.subject, opts);
|
|
3436
|
+
const risk = resolveRisk(finding, opts);
|
|
3437
|
+
return makePolicyEdit({
|
|
3438
|
+
axis: routed.axis,
|
|
3439
|
+
target: routed.target,
|
|
3440
|
+
change: {
|
|
3441
|
+
kind: "text",
|
|
3442
|
+
mode: "append",
|
|
3443
|
+
value: finding.recommended_action.trim()
|
|
3444
|
+
},
|
|
3445
|
+
claim: finding.claim,
|
|
3446
|
+
rationale: finding.rationale,
|
|
3447
|
+
expectedGain,
|
|
3448
|
+
confidence: finding.confidence,
|
|
3449
|
+
risk,
|
|
3450
|
+
validationPlan: finding.validation_plan,
|
|
3451
|
+
source: {
|
|
3452
|
+
findingIds: [finding.finding_id],
|
|
3453
|
+
analystIds: [finding.analyst_id],
|
|
3454
|
+
evidenceRefs: finding.evidence_refs,
|
|
3455
|
+
derivedFromJudge: finding.derived_from_judge
|
|
3456
|
+
}
|
|
3457
|
+
});
|
|
3458
|
+
}
|
|
3459
|
+
function scorePolicyEditReadiness(edit, opts = {}) {
|
|
3460
|
+
validatePolicyEdit(edit);
|
|
3461
|
+
const minExpectedGain = opts.minExpectedGain ?? DEFAULT_MIN_EXPECTED_GAIN;
|
|
3462
|
+
const requireEvidence = opts.requireEvidence ?? false;
|
|
3463
|
+
const evidenceScore = Math.min(1, edit.source.evidenceRefs.length / 2);
|
|
3464
|
+
const confidenceScore = clamp01$3(edit.confidence);
|
|
3465
|
+
const gainScore = clamp01$3(Math.abs(edit.expectedGain.amount) / Math.max(minExpectedGain * 5, .001));
|
|
3466
|
+
const targetScore = targetSpecificityScore(edit);
|
|
3467
|
+
const riskPenalty = edit.risk === "high" && opts.allowHighRisk !== true ? .35 : edit.risk === "unknown" ? .2 : 0;
|
|
3468
|
+
return clamp01$3((requireEvidence ? .3 * evidenceScore + .25 * confidenceScore + .25 * gainScore + .2 * targetScore : .35 * confidenceScore + .35 * gainScore + .3 * targetScore + .1 * evidenceScore) - riskPenalty);
|
|
3469
|
+
}
|
|
3470
|
+
function admitPolicyEdit(edit, opts = {}) {
|
|
3471
|
+
const validated = validatePolicyEdit(edit);
|
|
3472
|
+
const score = scorePolicyEditReadiness(validated, opts);
|
|
3473
|
+
const reasons = [];
|
|
3474
|
+
const minExpectedGain = opts.minExpectedGain ?? DEFAULT_MIN_EXPECTED_GAIN;
|
|
3475
|
+
const requireEvidence = opts.requireEvidence ?? false;
|
|
3476
|
+
if (validated.source.derivedFromJudge) reasons.push("source is judge-derived; judge verdicts cannot steer policy edits");
|
|
3477
|
+
if (requireEvidence && validated.source.evidenceRefs.length === 0) reasons.push("missing evidence refs");
|
|
3478
|
+
if (Math.abs(validated.expectedGain.amount) < minExpectedGain) reasons.push(`expected gain below ${minExpectedGain}`);
|
|
3479
|
+
if (validated.risk === "high" && opts.allowHighRisk !== true) reasons.push("high-risk edit requires explicit allowHighRisk");
|
|
3480
|
+
if (score < (opts.minScore ?? DEFAULT_MIN_SCORE)) reasons.push(`readiness score ${score.toFixed(3)} below ${(opts.minScore ?? DEFAULT_MIN_SCORE).toFixed(3)}`);
|
|
3481
|
+
return {
|
|
3482
|
+
edit: validated,
|
|
3483
|
+
decision: reasons.length === 0 ? "admit" : "reject",
|
|
3484
|
+
score,
|
|
3485
|
+
reasons
|
|
3486
|
+
};
|
|
3487
|
+
}
|
|
3488
|
+
function applyPolicyEditToSurface(surface, edit) {
|
|
3489
|
+
const validated = validatePolicyEdit(edit);
|
|
3490
|
+
if (validated.change.kind === "text") return applyTextChange(surface, validated.change);
|
|
3491
|
+
return applyJsonChange(surface, validated.change);
|
|
3492
|
+
}
|
|
3493
|
+
function routeFindingSubject(subject, opts) {
|
|
3494
|
+
const parsed = parseFindingSubject(subject);
|
|
3495
|
+
if (!parsed) return {
|
|
3496
|
+
axis: opts.defaultAxis ?? "representation",
|
|
3497
|
+
target: { surface: opts.defaultTargetSurface ?? "prompt" }
|
|
3498
|
+
};
|
|
3499
|
+
return routeParsedSubject(parsed);
|
|
3500
|
+
}
|
|
3501
|
+
function routeParsedSubject(subject) {
|
|
3502
|
+
switch (subject.kind) {
|
|
3503
|
+
case "system-prompt": return {
|
|
3504
|
+
axis: "representation",
|
|
3505
|
+
target: {
|
|
3506
|
+
surface: "prompt",
|
|
3507
|
+
path: `system-prompt:${subject.section}`
|
|
3508
|
+
}
|
|
3509
|
+
};
|
|
3510
|
+
case "skill": return {
|
|
3511
|
+
axis: "agent_profile",
|
|
3512
|
+
target: {
|
|
3513
|
+
surface: "agent-profile",
|
|
3514
|
+
path: `skill:${subject.name}`
|
|
3515
|
+
}
|
|
3516
|
+
};
|
|
3517
|
+
case "tool-doc": return {
|
|
3518
|
+
axis: "tool_contract",
|
|
3519
|
+
target: {
|
|
3520
|
+
surface: "tool-contract",
|
|
3521
|
+
path: subject.aspect ? `tool-doc:${subject.tool}:${subject.aspect}` : `tool-doc:${subject.tool}`
|
|
3522
|
+
}
|
|
3523
|
+
};
|
|
3524
|
+
case "new-tool": return {
|
|
3525
|
+
axis: "tool_contract",
|
|
3526
|
+
target: {
|
|
3527
|
+
surface: "tool-contract",
|
|
3528
|
+
path: `new-tool:${subject.name}`
|
|
3529
|
+
}
|
|
3530
|
+
};
|
|
3531
|
+
case "mcp": return {
|
|
3532
|
+
axis: "tool_contract",
|
|
3533
|
+
target: {
|
|
3534
|
+
surface: "agent-profile",
|
|
3535
|
+
path: subject.tool ? `mcp:${subject.server}:${subject.tool}` : `mcp:${subject.server}`
|
|
3536
|
+
}
|
|
3537
|
+
};
|
|
3538
|
+
case "hook": return {
|
|
3539
|
+
axis: "agent_profile",
|
|
3540
|
+
target: {
|
|
3541
|
+
surface: "agent-profile",
|
|
3542
|
+
path: `hook:${subject.name}`
|
|
3543
|
+
}
|
|
3544
|
+
};
|
|
3545
|
+
case "subagent": return {
|
|
3546
|
+
axis: "routing",
|
|
3547
|
+
target: {
|
|
3548
|
+
surface: "agent-profile",
|
|
3549
|
+
path: `subagent:${subject.name}`
|
|
3550
|
+
}
|
|
3551
|
+
};
|
|
3552
|
+
case "workflow": return {
|
|
3553
|
+
axis: "routing",
|
|
3554
|
+
target: {
|
|
3555
|
+
surface: "runtime-config",
|
|
3556
|
+
path: `workflow:${subject.name}`
|
|
3557
|
+
}
|
|
3558
|
+
};
|
|
3559
|
+
case "rollout-policy": return {
|
|
3560
|
+
axis: rolloutPolicyAxis(subject.field),
|
|
3561
|
+
target: {
|
|
3562
|
+
surface: "runtime-config",
|
|
3563
|
+
path: `rollout-policy:${subject.field}`
|
|
3564
|
+
}
|
|
3565
|
+
};
|
|
3566
|
+
case "agent-profile": return {
|
|
3567
|
+
axis: "agent_profile",
|
|
3568
|
+
target: {
|
|
3569
|
+
surface: "agent-profile",
|
|
3570
|
+
path: `agent-profile:${subject.field}`
|
|
3571
|
+
}
|
|
3572
|
+
};
|
|
3573
|
+
case "code": return {
|
|
3574
|
+
axis: "representation",
|
|
3575
|
+
target: {
|
|
3576
|
+
surface: "code",
|
|
3577
|
+
path: `code:${subject.path}`
|
|
3578
|
+
}
|
|
3579
|
+
};
|
|
3580
|
+
case "rag": return {
|
|
3581
|
+
axis: "memory",
|
|
3582
|
+
target: {
|
|
3583
|
+
surface: "memory",
|
|
3584
|
+
path: `rag:${subject.corpus}:${subject.docId}`
|
|
3585
|
+
}
|
|
3586
|
+
};
|
|
3587
|
+
case "memory": return {
|
|
3588
|
+
axis: "memory",
|
|
3589
|
+
target: {
|
|
3590
|
+
surface: "memory",
|
|
3591
|
+
path: `memory:${subject.key}`
|
|
3592
|
+
}
|
|
3593
|
+
};
|
|
3594
|
+
case "scaffolding": return {
|
|
3595
|
+
axis: "routing",
|
|
3596
|
+
target: {
|
|
3597
|
+
surface: "runtime-config",
|
|
3598
|
+
path: `scaffolding:${subject.concern}`
|
|
3599
|
+
}
|
|
3600
|
+
};
|
|
3601
|
+
case "output-schema": return {
|
|
3602
|
+
axis: "output_contract",
|
|
3603
|
+
target: {
|
|
3604
|
+
surface: "runtime-config",
|
|
3605
|
+
path: `output-schema:${subject.field}`
|
|
3606
|
+
}
|
|
3607
|
+
};
|
|
3608
|
+
case "knowledge.wiki": return {
|
|
3609
|
+
axis: "memory",
|
|
3610
|
+
target: {
|
|
3611
|
+
surface: "memory",
|
|
3612
|
+
path: `agent-knowledge:wiki:${subject.slug}${subject.heading ? `#${subject.heading}` : ""}`
|
|
3613
|
+
}
|
|
3614
|
+
};
|
|
3615
|
+
case "knowledge.claim": return {
|
|
3616
|
+
axis: "memory",
|
|
3617
|
+
target: {
|
|
3618
|
+
surface: "memory",
|
|
3619
|
+
path: `agent-knowledge:claim:${subject.topic}`
|
|
3620
|
+
}
|
|
3621
|
+
};
|
|
3622
|
+
case "knowledge.raw": return {
|
|
3623
|
+
axis: "memory",
|
|
3624
|
+
target: {
|
|
3625
|
+
surface: "memory",
|
|
3626
|
+
path: `agent-knowledge:raw:${subject.sourceId}`
|
|
3627
|
+
}
|
|
3628
|
+
};
|
|
3629
|
+
case "knowledge.stale": return {
|
|
3630
|
+
axis: "memory",
|
|
3631
|
+
target: {
|
|
3632
|
+
surface: "memory",
|
|
3633
|
+
path: `agent-knowledge:stale:${subject.slug}`
|
|
3634
|
+
}
|
|
3635
|
+
};
|
|
3636
|
+
case "websearch.outdated": return {
|
|
3637
|
+
axis: "memory",
|
|
3638
|
+
target: {
|
|
3639
|
+
surface: "memory",
|
|
3640
|
+
path: `websearch:outdated:${subject.topic}`
|
|
3641
|
+
}
|
|
3642
|
+
};
|
|
3643
|
+
case "prior-run-summary": return {
|
|
3644
|
+
axis: "memory",
|
|
3645
|
+
target: {
|
|
3646
|
+
surface: "memory",
|
|
3647
|
+
path: `prior-run-summary:${subject.topic}`
|
|
3648
|
+
}
|
|
3649
|
+
};
|
|
3650
|
+
case "cluster": return {
|
|
3651
|
+
axis: "representation",
|
|
3652
|
+
target: {
|
|
3653
|
+
surface: "prompt",
|
|
3654
|
+
path: subject.label
|
|
3655
|
+
}
|
|
3656
|
+
};
|
|
3657
|
+
}
|
|
3658
|
+
}
|
|
3659
|
+
function rolloutPolicyAxis(field) {
|
|
3660
|
+
const normalized = field.toLowerCase();
|
|
3661
|
+
if (/budget|max(?:imum)?[-_. ]?(?:turns?|tokens?|cost)|timeout|deadline/.test(normalized)) return "budget";
|
|
3662
|
+
if (/temperature|top[-_. ]?p|sampling|seed|shots?|parallel|concurrency/.test(normalized)) return "sampling";
|
|
3663
|
+
if (/output|schema|format/.test(normalized)) return "output_contract";
|
|
3664
|
+
return "routing";
|
|
3665
|
+
}
|
|
3666
|
+
function resolveExpectedGain(finding, opts) {
|
|
3667
|
+
if (typeof opts.expectedGain === "function") return opts.expectedGain(finding) ?? null;
|
|
3668
|
+
if (opts.expectedGain) return opts.expectedGain;
|
|
3669
|
+
return readExpectedGainFromMetadata(finding.metadata);
|
|
3670
|
+
}
|
|
3671
|
+
function readExpectedGainFromMetadata(metadata) {
|
|
3672
|
+
const raw = readPolicyEditMetadata(metadata)?.expectedGain ?? readPolicyEditMetadata(metadata)?.expected_gain;
|
|
3673
|
+
if (!raw || typeof raw !== "object") return null;
|
|
3674
|
+
const obj = raw;
|
|
3675
|
+
if (typeof obj.metric !== "string" || obj.direction !== "increase" && obj.direction !== "decrease" || typeof obj.amount !== "number" || !Number.isFinite(obj.amount) || obj.amount <= 0) return null;
|
|
3676
|
+
const out = {
|
|
3677
|
+
metric: obj.metric,
|
|
3678
|
+
direction: obj.direction,
|
|
3679
|
+
amount: obj.amount
|
|
3680
|
+
};
|
|
3681
|
+
if (obj.unit === "absolute" || obj.unit === "relative" || obj.unit === "percent" || obj.unit === "score") out.unit = obj.unit;
|
|
3682
|
+
if (typeof obj.rationale === "string") out.rationale = obj.rationale;
|
|
3683
|
+
return out;
|
|
3684
|
+
}
|
|
3685
|
+
function readPolicyEditMetadata(metadata) {
|
|
3686
|
+
const raw = metadata?.policyEdit ?? metadata?.policy_edit;
|
|
3687
|
+
return raw && typeof raw === "object" ? raw : null;
|
|
3688
|
+
}
|
|
3689
|
+
function resolveRisk(finding, opts) {
|
|
3690
|
+
if (typeof opts.risk === "function") return opts.risk(finding);
|
|
3691
|
+
if (opts.risk) return opts.risk;
|
|
3692
|
+
const raw = readPolicyEditMetadata(finding.metadata)?.risk;
|
|
3693
|
+
if (raw === "low" || raw === "medium" || raw === "high" || raw === "unknown") return raw;
|
|
3694
|
+
if (finding.severity === "critical" || finding.severity === "high") return "medium";
|
|
3695
|
+
return "low";
|
|
3696
|
+
}
|
|
3697
|
+
function applyTextChange(surface, change) {
|
|
3698
|
+
if (typeof surface !== "string") throw new PolicyEditValidationError("text policy edits require a string surface", "change");
|
|
3699
|
+
if (change.mode === "append") {
|
|
3700
|
+
if (hasExactTextBlock(surface, change.value)) return surface;
|
|
3701
|
+
return `${surface.trimEnd()}\n\n${change.value}`.trimStart();
|
|
3702
|
+
}
|
|
3703
|
+
if (change.mode === "prepend") {
|
|
3704
|
+
if (hasExactTextBlock(surface, change.value)) return surface;
|
|
3705
|
+
return `${change.value}\n\n${surface.trimStart()}`.trimEnd();
|
|
3706
|
+
}
|
|
3707
|
+
const find = expectNonEmpty(change.find, "change.find");
|
|
3708
|
+
if (!surface.includes(find)) throw new PolicyEditValidationError("replace target not found in surface", "change.find");
|
|
3709
|
+
return surface.replace(find, change.value);
|
|
3710
|
+
}
|
|
3711
|
+
function applyJsonChange(surface, change) {
|
|
3712
|
+
const root = parseJsonSurface(surface);
|
|
3713
|
+
const path = splitPath(change.path);
|
|
3714
|
+
if (change.mode === "remove") return setJsonAtPath(root, path, void 0, "remove");
|
|
3715
|
+
if (change.mode === "set") return setJsonAtPath(root, path, change.value ?? null, "set");
|
|
3716
|
+
const prior = readJsonAtPath(root, path);
|
|
3717
|
+
return setJsonAtPath(root, path, prior && typeof prior === "object" && !Array.isArray(prior) && change.value && typeof change.value === "object" && !Array.isArray(change.value) ? {
|
|
3718
|
+
...prior,
|
|
3719
|
+
...change.value
|
|
3720
|
+
} : change.value ?? null, "set");
|
|
3721
|
+
}
|
|
3722
|
+
function parseJsonSurface(surface) {
|
|
3723
|
+
if (typeof surface === "string") try {
|
|
3724
|
+
return JSON.parse(surface);
|
|
3725
|
+
} catch {
|
|
3726
|
+
throw new PolicyEditValidationError("json policy edits require a JSON string surface", "change");
|
|
3727
|
+
}
|
|
3728
|
+
assertJson(surface, "surface");
|
|
3729
|
+
return surface;
|
|
3730
|
+
}
|
|
3731
|
+
function readJsonAtPath(root, path) {
|
|
3732
|
+
let cursor = root;
|
|
3733
|
+
for (const part of path) {
|
|
3734
|
+
if (!cursor || typeof cursor !== "object" || Array.isArray(cursor)) return void 0;
|
|
3735
|
+
cursor = cursor[part];
|
|
3736
|
+
}
|
|
3737
|
+
return cursor;
|
|
3738
|
+
}
|
|
3739
|
+
function setJsonAtPath(root, path, value, mode) {
|
|
3740
|
+
if (path.length === 0) {
|
|
3741
|
+
if (mode === "remove") return null;
|
|
3742
|
+
return value ?? null;
|
|
3743
|
+
}
|
|
3744
|
+
if (root === null || typeof root !== "object" || Array.isArray(root)) throw new PolicyEditValidationError("json edit root must be an object", "change.path");
|
|
3745
|
+
const out = { ...root };
|
|
3746
|
+
let cursor = out;
|
|
3747
|
+
for (let i = 0; i < path.length - 1; i++) {
|
|
3748
|
+
const key = path[i];
|
|
3749
|
+
const existing = cursor[key];
|
|
3750
|
+
if (mode === "remove" && (!existing || typeof existing !== "object" || Array.isArray(existing))) return out;
|
|
3751
|
+
const next = existing && typeof existing === "object" && !Array.isArray(existing) ? { ...existing } : {};
|
|
3752
|
+
cursor[key] = next;
|
|
3753
|
+
cursor = next;
|
|
3754
|
+
}
|
|
3755
|
+
const leaf = path[path.length - 1];
|
|
3756
|
+
if (mode === "remove") delete cursor[leaf];
|
|
3757
|
+
else cursor[leaf] = value ?? null;
|
|
3758
|
+
return out;
|
|
3759
|
+
}
|
|
3760
|
+
function normalizePolicyEdit(input) {
|
|
3761
|
+
const out = {
|
|
3762
|
+
schemaVersion: "policy-edit/v1",
|
|
3763
|
+
axis: input.axis,
|
|
3764
|
+
target: normalizeTarget(input.target),
|
|
3765
|
+
change: normalizeChange(input.change),
|
|
3766
|
+
claim: input.claim.trim(),
|
|
3767
|
+
expectedGain: normalizeExpectedGain(input.expectedGain),
|
|
3768
|
+
confidence: input.confidence,
|
|
3769
|
+
risk: input.risk,
|
|
3770
|
+
source: normalizeSource(input.source)
|
|
3771
|
+
};
|
|
3772
|
+
if (input.rationale?.trim()) out.rationale = input.rationale.trim();
|
|
3773
|
+
if (input.validationPlan?.trim()) out.validationPlan = input.validationPlan.trim();
|
|
3774
|
+
if (input.metadata) out.metadata = input.metadata;
|
|
3775
|
+
return out;
|
|
3776
|
+
}
|
|
3777
|
+
function assertJsonSafe(value, path, ancestors = /* @__PURE__ */ new WeakSet()) {
|
|
3778
|
+
if (value === null || typeof value === "string" || typeof value === "boolean") return;
|
|
3779
|
+
if (typeof value === "number") {
|
|
3780
|
+
if (Number.isFinite(value)) return;
|
|
3781
|
+
throw new PolicyEditValidationError("expected finite JSON number", path);
|
|
3782
|
+
}
|
|
3783
|
+
if (typeof value !== "object") throw new PolicyEditValidationError("expected JSON-safe value", path);
|
|
3784
|
+
if (ancestors.has(value)) throw new PolicyEditValidationError("cyclic value is not JSON-safe", path);
|
|
3785
|
+
ancestors.add(value);
|
|
3786
|
+
if (Array.isArray(value)) for (let i = 0; i < value.length; i++) {
|
|
3787
|
+
if (!(i in value)) throw new PolicyEditValidationError("sparse array is not JSON-safe", `${path}.${i}`);
|
|
3788
|
+
assertJsonSafe(value[i], `${path}.${i}`, ancestors);
|
|
3789
|
+
}
|
|
3790
|
+
else {
|
|
3791
|
+
const prototype = Object.getPrototypeOf(value);
|
|
3792
|
+
if (prototype !== Object.prototype && prototype !== null) throw new PolicyEditValidationError("expected plain JSON object", path);
|
|
3793
|
+
if (Object.getOwnPropertySymbols(value).length > 0) throw new PolicyEditValidationError("symbol keys are not JSON-safe", path);
|
|
3794
|
+
for (const [key, child] of Object.entries(value)) assertJsonSafe(child, `${path}.${key}`, ancestors);
|
|
3795
|
+
}
|
|
3796
|
+
ancestors.delete(value);
|
|
3797
|
+
}
|
|
3798
|
+
function normalizeTarget(target) {
|
|
3799
|
+
const out = { surface: target.surface };
|
|
3800
|
+
if (target.path?.trim()) out.path = target.path.trim();
|
|
3801
|
+
if (target.agentProfileCell) out.agentProfileCell = validateAgentProfileCell(target.agentProfileCell);
|
|
3802
|
+
if (target.label?.trim()) out.label = target.label.trim();
|
|
3803
|
+
return out;
|
|
3804
|
+
}
|
|
3805
|
+
function normalizeChange(change) {
|
|
3806
|
+
if (change.kind === "text") {
|
|
3807
|
+
const out = {
|
|
3808
|
+
kind: "text",
|
|
3809
|
+
mode: change.mode,
|
|
3810
|
+
value: change.value.trim()
|
|
3811
|
+
};
|
|
3812
|
+
if (change.find?.trim()) out.find = change.find.trim();
|
|
3813
|
+
return out;
|
|
3814
|
+
}
|
|
3815
|
+
const out = {
|
|
3816
|
+
kind: "json",
|
|
3817
|
+
mode: change.mode,
|
|
3818
|
+
path: change.path.trim()
|
|
3819
|
+
};
|
|
3820
|
+
if (change.value !== void 0) out.value = change.value;
|
|
3821
|
+
return out;
|
|
3822
|
+
}
|
|
3823
|
+
function normalizeExpectedGain(gain) {
|
|
3824
|
+
const out = {
|
|
3825
|
+
metric: gain.metric.trim(),
|
|
3826
|
+
direction: gain.direction,
|
|
3827
|
+
amount: gain.amount
|
|
3828
|
+
};
|
|
3829
|
+
if (gain.unit) out.unit = gain.unit;
|
|
3830
|
+
if (gain.rationale?.trim()) out.rationale = gain.rationale.trim();
|
|
3831
|
+
return out;
|
|
3832
|
+
}
|
|
3833
|
+
function normalizeSource(source) {
|
|
3834
|
+
const out = {
|
|
3835
|
+
findingIds: uniqueSorted(source.findingIds.map((s) => s.trim()).filter(Boolean)),
|
|
3836
|
+
analystIds: uniqueSorted(source.analystIds.map((s) => s.trim()).filter(Boolean)),
|
|
3837
|
+
evidenceRefs: source.evidenceRefs
|
|
3838
|
+
};
|
|
3839
|
+
if (source.derivedFromJudge) out.derivedFromJudge = true;
|
|
3840
|
+
return out;
|
|
3841
|
+
}
|
|
3842
|
+
function validateTarget(target) {
|
|
3843
|
+
if (!target || typeof target !== "object") throw new PolicyEditValidationError("expected object", "target");
|
|
3844
|
+
const obj = target;
|
|
3845
|
+
expectOneOf(obj.surface, POLICY_EDIT_TARGET_SURFACES, "target.surface");
|
|
3846
|
+
if (obj.path !== void 0) expectString$1(obj.path, "target.path");
|
|
3847
|
+
if (obj.label !== void 0) expectString$1(obj.label, "target.label");
|
|
3848
|
+
if (obj.agentProfileCell !== void 0) validateAgentProfileCell(obj.agentProfileCell);
|
|
3849
|
+
}
|
|
3850
|
+
function validateChange(change) {
|
|
3851
|
+
if (!change || typeof change !== "object") throw new PolicyEditValidationError("expected object", "change");
|
|
3852
|
+
const obj = change;
|
|
3853
|
+
if (obj.kind !== "text" && obj.kind !== "json") throw new PolicyEditValidationError("kind must be text or json", "change.kind");
|
|
3854
|
+
if (obj.kind === "text") {
|
|
3855
|
+
expectOneOf(obj.mode, [
|
|
3856
|
+
"append",
|
|
3857
|
+
"prepend",
|
|
3858
|
+
"replace"
|
|
3859
|
+
], "change.mode");
|
|
3860
|
+
expectString$1(obj.value, "change.value");
|
|
3861
|
+
if (obj.mode === "replace") expectString$1(obj.find, "change.find");
|
|
3862
|
+
return;
|
|
3863
|
+
}
|
|
3864
|
+
expectOneOf(obj.mode, [
|
|
3865
|
+
"set",
|
|
3866
|
+
"merge",
|
|
3867
|
+
"remove"
|
|
3868
|
+
], "change.mode");
|
|
3869
|
+
expectString$1(obj.path, "change.path");
|
|
3870
|
+
splitPath(obj.path);
|
|
3871
|
+
if (obj.value !== void 0) assertJson(obj.value, "change.value");
|
|
3872
|
+
}
|
|
3873
|
+
function validateExpectedGain(gain) {
|
|
3874
|
+
if (!gain || typeof gain !== "object") throw new PolicyEditValidationError("expected object", "expectedGain");
|
|
3875
|
+
const obj = gain;
|
|
3876
|
+
expectString$1(obj.metric, "expectedGain.metric");
|
|
3877
|
+
expectOneOf(obj.direction, ["increase", "decrease"], "expectedGain.direction");
|
|
3878
|
+
if (!Number.isFinite(obj.amount) || obj.amount <= 0) throw new PolicyEditValidationError("amount must be a positive finite number", "expectedGain.amount");
|
|
3879
|
+
if (obj.unit !== void 0) expectOneOf(obj.unit, [
|
|
3880
|
+
"absolute",
|
|
3881
|
+
"relative",
|
|
3882
|
+
"percent",
|
|
3883
|
+
"score"
|
|
3884
|
+
], "expectedGain.unit");
|
|
3885
|
+
if (obj.rationale !== void 0) expectString$1(obj.rationale, "expectedGain.rationale");
|
|
3886
|
+
}
|
|
3887
|
+
function validateSource(source) {
|
|
3888
|
+
if (!source || typeof source !== "object") throw new PolicyEditValidationError("expected object", "source");
|
|
3889
|
+
const obj = source;
|
|
3890
|
+
expectNonEmptyStringArray(obj.findingIds, "source.findingIds");
|
|
3891
|
+
expectNonEmptyStringArray(obj.analystIds, "source.analystIds");
|
|
3892
|
+
if (!Array.isArray(obj.evidenceRefs)) throw new PolicyEditValidationError("expected array", "source.evidenceRefs");
|
|
3893
|
+
for (const [i, ref] of obj.evidenceRefs.entries()) validateEvidenceRef(ref, `source.evidenceRefs.${i}`);
|
|
3894
|
+
if (obj.derivedFromJudge !== void 0 && typeof obj.derivedFromJudge !== "boolean") throw new PolicyEditValidationError("expected boolean", "source.derivedFromJudge");
|
|
3895
|
+
}
|
|
3896
|
+
function validateEvidenceRef(ref, path) {
|
|
3897
|
+
if (!ref || typeof ref !== "object") throw new PolicyEditValidationError("expected object", path);
|
|
3898
|
+
const obj = ref;
|
|
3899
|
+
expectOneOf(obj.kind, [
|
|
3900
|
+
"span",
|
|
3901
|
+
"event",
|
|
3902
|
+
"artifact",
|
|
3903
|
+
"finding",
|
|
3904
|
+
"metric"
|
|
3905
|
+
], `${path}.kind`);
|
|
3906
|
+
expectString$1(obj.uri, `${path}.uri`);
|
|
3907
|
+
if (obj.excerpt !== void 0) expectString$1(obj.excerpt, `${path}.excerpt`);
|
|
3908
|
+
}
|
|
3909
|
+
function assertJson(value, path) {
|
|
3910
|
+
if (value === null || typeof value === "string" || typeof value === "boolean" || typeof value === "number" && Number.isFinite(value)) return;
|
|
3911
|
+
if (Array.isArray(value)) {
|
|
3912
|
+
for (const [i, item] of value.entries()) assertJson(item, `${path}.${i}`);
|
|
3913
|
+
return;
|
|
3914
|
+
}
|
|
3915
|
+
if (typeof value === "object") {
|
|
3916
|
+
for (const [key, item] of Object.entries(value)) {
|
|
3917
|
+
if (!key) throw new PolicyEditValidationError("empty object key", path);
|
|
3918
|
+
assertJson(item, `${path}.${key}`);
|
|
3919
|
+
}
|
|
3920
|
+
return;
|
|
3921
|
+
}
|
|
3922
|
+
throw new PolicyEditValidationError("expected JSON-compatible value", path);
|
|
3923
|
+
}
|
|
3924
|
+
function targetSpecificityScore(edit) {
|
|
3925
|
+
let score = .4;
|
|
3926
|
+
if (edit.target.path) score += .25;
|
|
3927
|
+
if (edit.target.agentProfileCell) score += .15;
|
|
3928
|
+
if (edit.change.kind === "json" || edit.change.mode === "replace") score += .2;
|
|
3929
|
+
else if (edit.change.value.length > 0) score += .1;
|
|
3930
|
+
return clamp01$3(score);
|
|
3931
|
+
}
|
|
3932
|
+
const FORBIDDEN_PATH_KEYS = /* @__PURE__ */ new Set([
|
|
3933
|
+
"__proto__",
|
|
3934
|
+
"constructor",
|
|
3935
|
+
"prototype"
|
|
3936
|
+
]);
|
|
3937
|
+
function splitPath(path) {
|
|
3938
|
+
const parts = path.split(".").map((p) => p.trim()).filter(Boolean);
|
|
3939
|
+
if (parts.length === 0) throw new PolicyEditValidationError("path must not be empty", "change.path");
|
|
3940
|
+
for (const part of parts) if (FORBIDDEN_PATH_KEYS.has(part)) throw new PolicyEditValidationError(`path segment "${part}" would write through the prototype chain`, "change.path");
|
|
3941
|
+
return parts;
|
|
3942
|
+
}
|
|
3943
|
+
function expectLiteral(value, expected, path) {
|
|
3944
|
+
if (value !== expected) throw new PolicyEditValidationError(`expected ${expected}`, path);
|
|
3945
|
+
}
|
|
3946
|
+
function expectString$1(value, path) {
|
|
3947
|
+
if (typeof value !== "string" || value.trim().length === 0) throw new PolicyEditValidationError("expected non-empty string", path);
|
|
3948
|
+
}
|
|
3949
|
+
function expectNonEmpty(value, path) {
|
|
3950
|
+
expectString$1(value, path);
|
|
3951
|
+
return value;
|
|
3952
|
+
}
|
|
3953
|
+
function expectConfidence(value, path) {
|
|
3954
|
+
if (typeof value !== "number" || !Number.isFinite(value) || value < 0 || value > 1) throw new PolicyEditValidationError("expected finite number in [0,1]", path);
|
|
3955
|
+
}
|
|
3956
|
+
function hasExactTextBlock(surface, value) {
|
|
3957
|
+
const needle = normalizeTextBlock(value);
|
|
3958
|
+
const normalizedSurface = surface.replace(/\r\n/g, "\n");
|
|
3959
|
+
return [...normalizedSurface.split(/\n{2,}/), ...normalizedSurface.split("\n")].some((block) => normalizeTextBlock(block) === needle);
|
|
3960
|
+
}
|
|
3961
|
+
function normalizeTextBlock(value) {
|
|
3962
|
+
return value.replace(/\r\n/g, "\n").trim();
|
|
3963
|
+
}
|
|
3964
|
+
function expectOneOf(value, allowed, path) {
|
|
3965
|
+
if (typeof value !== "string" || !allowed.includes(value)) throw new PolicyEditValidationError(`expected one of ${allowed.join(", ")}`, path);
|
|
3966
|
+
}
|
|
3967
|
+
function expectStringArray$1(value, path) {
|
|
3968
|
+
if (!Array.isArray(value)) throw new PolicyEditValidationError("expected array", path);
|
|
3969
|
+
for (const [i, item] of value.entries()) expectString$1(item, `${path}.${i}`);
|
|
3970
|
+
}
|
|
3971
|
+
function expectNonEmptyStringArray(value, path) {
|
|
3972
|
+
expectStringArray$1(value, path);
|
|
3973
|
+
if (value.length === 0) throw new PolicyEditValidationError("expected non-empty array", path);
|
|
3974
|
+
}
|
|
3975
|
+
function uniqueSorted(values) {
|
|
3976
|
+
return [...new Set(values)].sort();
|
|
3977
|
+
}
|
|
3978
|
+
function clamp01$3(n) {
|
|
3979
|
+
if (!Number.isFinite(n)) return 0;
|
|
3980
|
+
if (n < 0) return 0;
|
|
3981
|
+
if (n > 1) return 1;
|
|
3982
|
+
return n;
|
|
3983
|
+
}
|
|
3984
|
+
//#endregion
|
|
3282
3985
|
//#region src/capability-headroom.ts
|
|
3283
3986
|
/**
|
|
3284
3987
|
* Capability-headroom gate — "can this task set even SEE the capability
|
|
@@ -7240,6 +7943,6 @@ function rankRows(rows, weights) {
|
|
|
7240
7943
|
})).sort((a, b) => b.mean - a.mean);
|
|
7241
7944
|
}
|
|
7242
7945
|
//#endregion
|
|
7243
|
-
export { AGENT_PROFILE_KINDS, AgentEvalError, AnalystRegistry, BOOTSTRAP_GATE_MIN_N, BackendIntegrityError, BudgetBreachError, BudgetGuard, CODING_HARNESSES, ConfigError, CostAccountingIncompleteError, CostCallConflictError, CostCeilingReachedError, CostLedger, CostLedgerPersistenceError, CostReceiptCaptureError, CostReservationExceededError, CostTracker, CrossFamilyError, DECISION_PAIRED_DELTA_STATISTIC, DEFAULT_PERMUTATIONS, DEFAULT_REDACTION_RULES, DEFAULT_RED_TEAM_CORPUS, DEFAULT_TRACE_ANALYST_BUDGETS, DEFAULT_TRACE_ANALYST_KINDS, ERROR_COUNT_PATTERNS, EquivalenceProtocolError, FAILURE_CLASSES, FAILURE_MODE_KIND_SPEC, FileSystemFeedbackTrajectoryStore, FileSystemRawProviderSink, FileSystemTraceStore, FindingsStore, HARNESS_NATIVE_MODEL, HeldOutGate, InMemoryFeedbackTrajectoryStore, InMemoryRawProviderSink, InMemoryTraceStore, JudgeError, LlmCallError, LlmResponseError, MANN_WHITNEY_EXACT_MAX_STATES, MANN_WHITNEY_EXACT_MAX_WORK, MODEL_PRICING, ModelSubstitutionError, MultiLayerVerifier, NoopRawProviderSink, NotFoundError, OUTPUT_VALUE, OtlpFileTraceStore, PairwiseSteeringOptimizer, ProductClient, PromptRegistry, REDACTION_VERSION, RunIntegrityError, RunRecordValidationError, SEMANTIC_CONCEPT_JUDGE_VERSION, ServedCrossFamilyError, TRACE_ANALYST_TRUNCATION_MARKER_PREFIX, TraceEmitter, UNKNOWN_MODEL, VERIFICATION_STRATEGIES, VERIFICATION_STRATEGY_SOURCES, ValidationError, WILCOXON_EXACT_MAX_N, acquisitionPlansForKnowledgeGaps, agentProfileCellHashMaterial, agentProfileCellKey, agentProfileHash, agentProfileId, aggregateJudgeVerdicts, aggregateRunScore, analystFindingDigest, analystRunDigest, analystRunToFeedbackTrajectory, analystRunToReviewRequests, analyzeAntiSlop, analyzeRuns, analyzeSeries, analyzeTraces, argHash, assertCapabilityHeadroom, assertCrossFamily, assertCrossFamilyServed, assertNoHiddenLeak, assertProductBenchmarkRun, assertRealBackend, assertRunCaptured, assertServedModel, assertServedModels, assertSingleBackend, assignFeedbackSplit, benjaminiHochberg, blendHeldout, blockingKnowledgeEval, bonferroni, bootstrapCi, budgetBreachView, buildAgentProfileCell, buildDefaultAnalystRegistry, buildEquivalenceRecord, buildReflectionPrompt, buildTraceInsightContext, buildTraceInsightPrompt, buildTrajectory, calibrateJudge, calibrateJudgeContinuous, canonicalJson, capabilityHeadroom, captureFetchToRawSink, certificationEvidenceDigest, checkCanaries, checkServedModel, checkTraceContracts, clamp01, classifyFailure, cliffsDelta, cohensD, comparePairedArms, completionVerdict, computeExperimentStats, computeFindingId, computeToolUseMetrics, confidenceInterval, contentHash, continuousAgreement, controlRunToFeedbackTrajectory, corpusInterRaterAgreement, corpusInterRaterAgreementFromJudgeScores, costForTokenPricing, costForUsage, costReceiptFromLlm, costReceiptFromLlmError, createAntiSlopJudge, createBoundedTraceAnalysisStore, createChatClient, createDspyRlmTraceEngine, createFeedbackTrajectory, createLlmCorrectnessChecker, createLlmReviewer, createTraceAnalyst, decidePairedPromotion, defaultBlendWeights, defineAgentEval, defineEquivalenceCheck, deployGateLayer, describeTraceInsightScope, diffFindings, diffScorecard, discoverPersonas, domainEvidencePattern, dominates, eProcess, ensembleJudge, equivalenceVerdict, errorStreakDetector, estimateCost, estimateTokens, evaluateActionPolicy, evaluateInterimReleaseConfidence, evaluateOracles, evaluateReleaseConfidence, expandProfileAxes, exportProductBenchmark, exportProductBenchmarkRuns, exportRunAsOtlp, extractErrorCount, extractProducedState, extractUsage, extractUsageFromSse, failureClusterView, feedbackTrajectoriesToDatasetScenarios, feedbackTrajectoriesToOptimizerRows, feedbackTrajectoryToOptimizerRow, fileVerdictCache, formatScorecardDiff, gainHistogram, gateTreatmentApplied, gradeOnHidden, gradeSemanticStatus, groupRunsByAgentProfileCell, harnessAxisOf, hashContent, hashJson, hiddenGrade, holm, improvementVerdict, inMemoryReviewStore, inferDomainKeywords, interRaterReliability, interpretCliffs, iqr, isBinaryOutcomeVector, isJudgeSpan, isLlmSpan, isModelPriced, isRunRecord, isToolSpan, isTransientLlmError, jsonShape, jsonlReviewStore, jsonlRunRecordBackend, judgeAgreementView, judgeFamily, judgeSpans, knowledgeReadinessTracePayload, leaderboard, llmJudge, loadScorecard, localCommandRunner, makeFinding, makeProposalFinding, manifestContentDigest, mannWhitneyU, maximumChargeForLlmRequest, mcnemar, mcnemarPower, mcnemarRequiredN, minimumPairsForPairedDeltaTest, mintRolloutRows, modelHasSnapshot, modelPriceKey, mulberry32, notBlocked, objectiveEval, observeAll, otlpTextToTraceAnalysisStore, pairArms, pairRunRecords, pairedBinaryScale, pairedBootstrap, pairedCohensDz, pairedDeltaTest, pairedDeltaTieFraction, pairedEvalueSequence, pairedMde, pairedRiskDifference, pairedRiskDifferenceExact, pairedRiskDifferenceScore, pairedSignTest, pairedTTest, paretoChart, paretoFrontier, parseReflectionResponse, parseRunRecordSafe, partialCredit, partitionHeldOut, passAtK, pearsonR, preflightModels, productBenchmarkRepoIdentity, profile_exports as profile, projectRuntimeTrajectoryEvidence, proposeSynthesisTargets, ranks, readProductBenchmarkManifest, recordRuns, recordRunsToScorecard, redTeamDataset, redTeamReport, redactString, regexMatches, renderPreferenceMemoryMarkdown, repeatedActionDetector, requiredPairedSampleSize, requiredSampleSize, resolveModelPricing, resolveSeat, roundTripRunRecord, routeFields, runAgentControlLoop, runCampaign, runCanaries, runCounterfactual, runEquivalenceCheck, runEvalCampaign, runIntentMatchJudge, runKeywordCoverageJudge, runKeywordCoverageJudgeUrl, runProposeReview, runProposeReviewAsControlLoop, runSemanticConceptJudge, runTaskScore, runsForScenario, scoreKnowledgeReadiness, scoreRedTeamOutput, scoreTraceInsightReadiness, seatPresets, selfImprove, servedModelAcceptable, spearmanR, stripFencedJson, subjectiveEval, summarizeBackendIntegrity, summarizeNumberSeries, summarizePreferenceMemory, summaryTable, textInSnapshot, toAgentProfileJson, tokenizeDomainWords, toolSpansToTraceAnalysisStore, toolWasteView, traceContract, transientDispatchFailure, urlContains, userQuestionsForKnowledgeGaps, validateRunRecord, verbosityBias, verifyAgentProfileCell, verifyCompletion, viteDeployRunner, weightedComposite, weightedMean, wilcoxonSignedRank, wilson, withAssignedFeedbackSplit, withHeldoutBlend, withJudgeRetry, wranglerDeployRunner };
|
|
7946
|
+
export { AGENT_PROFILE_KINDS, ATTESTATION_ALGORITHM, AgentEvalError, AnalystRegistry, BOOTSTRAP_GATE_MIN_N, BackendIntegrityError, BudgetBreachError, BudgetGuard, CODING_HARNESSES, ConfigError, CostAccountingIncompleteError, CostCallConflictError, CostCeilingReachedError, CostLedger, CostLedgerPersistenceError, CostReceiptCaptureError, CostReservationExceededError, CostTracker, CrossFamilyError, DECISION_PAIRED_DELTA_STATISTIC, DEFAULT_PERMUTATIONS, DEFAULT_REDACTION_RULES, DEFAULT_RED_TEAM_CORPUS, DEFAULT_TRACE_ANALYST_BUDGETS, DEFAULT_TRACE_ANALYST_KINDS, ERROR_COUNT_PATTERNS, EquivalenceProtocolError, FAILURE_CLASSES, FAILURE_MODE_KIND_SPEC, FileSystemFeedbackTrajectoryStore, FileSystemRawProviderSink, FileSystemTraceStore, FindingsStore, HARNESS_NATIVE_MODEL, HeldOutGate, InMemoryFeedbackTrajectoryStore, InMemoryRawProviderSink, InMemoryTraceStore, JudgeError, LlmCallError, LlmResponseError, MANN_WHITNEY_EXACT_MAX_STATES, MANN_WHITNEY_EXACT_MAX_WORK, MODEL_PRICING, ModelSubstitutionError, MultiLayerVerifier, NoopRawProviderSink, NotFoundError, OUTPUT_VALUE, OtlpFileTraceStore, POLICY_EDIT_AXES, POLICY_EDIT_CANDIDATE_RECORD_SCHEMA, POLICY_EDIT_TARGET_SURFACES, PairwiseSteeringOptimizer, PolicyEditValidationError, ProductClient, PromptRegistry, REDACTION_VERSION, RunIntegrityError, RunRecordValidationError, SEMANTIC_CONCEPT_JUDGE_VERSION, ServedCrossFamilyError, TRACE_ANALYST_TRUNCATION_MARKER_PREFIX, TraceEmitter, UNKNOWN_MODEL, VERIFICATION_STRATEGIES, VERIFICATION_STRATEGY_SOURCES, ValidationError, WILCOXON_EXACT_MAX_N, acquisitionPlansForKnowledgeGaps, admitPolicyEdit, agentProfileCellHashMaterial, agentProfileCellKey, agentProfileHash, agentProfileId, aggregateJudgeVerdicts, aggregateRunScore, analystFindingDigest, analystRunDigest, analystRunToFeedbackTrajectory, analystRunToReviewRequests, analyzeAntiSlop, analyzeRuns, analyzeSeries, analyzeTraces, applyPolicyEditToSurface, argHash, assertCapabilityHeadroom, assertCrossFamily, assertCrossFamilyServed, assertNoHiddenLeak, assertNoJudgeVerdict, assertProductBenchmarkRun, assertRealBackend, assertRunCaptured, assertServedModel, assertServedModels, assertSingleBackend, assignFeedbackSplit, attest, benjaminiHochberg, blendHeldout, blockingKnowledgeEval, bonferroni, bootstrapCi, budgetBreachView, buildAgentProfileCell, buildDefaultAnalystRegistry, buildEquivalenceRecord, buildReflectionPrompt, buildTraceInsightContext, buildTraceInsightPrompt, buildTrajectory, calibrateJudge, calibrateJudgeContinuous, canonicalJson, capabilityHeadroom, captureFetchToRawSink, certificationEvidenceDigest, checkCanaries, checkServedModel, checkTraceContracts, clamp01, classifyFailure, cliffsDelta, cohensD, comparePairedArms, completionVerdict, computeExperimentStats, computeFindingId, computePolicyEditId, computeToolUseMetrics, confidenceInterval, contentHash, continuousAgreement, controlRunToFeedbackTrajectory, corpusInterRaterAgreement, corpusInterRaterAgreementFromJudgeScores, costForTokenPricing, costForUsage, costReceiptFromLlm, costReceiptFromLlmError, createAntiSlopJudge, createBoundedTraceAnalysisStore, createChatClient, createDspyRlmTraceEngine, createFeedbackTrajectory, createLlmCorrectnessChecker, createLlmReviewer, createTraceAnalyst, decidePairedPromotion, defaultBlendWeights, defineAgentEval, defineEquivalenceCheck, deployGateLayer, describeTraceInsightScope, diffFindings, diffScorecard, discoverPersonas, domainEvidencePattern, dominates, eProcess, ensembleJudge, equivalenceVerdict, errorStreakDetector, estimateCost, estimateTokens, evaluateActionPolicy, evaluateInterimReleaseConfidence, evaluateOracles, evaluateReleaseConfidence, expandProfileAxes, exportProductBenchmark, exportProductBenchmarkRuns, exportRunAsOtlp, extractErrorCount, extractProducedState, extractUsage, extractUsageFromSse, failureClusterView, feedbackTrajectoriesToDatasetScenarios, feedbackTrajectoriesToOptimizerRows, feedbackTrajectoryToOptimizerRow, fileVerdictCache, formatScorecardDiff, gainHistogram, gateTreatmentApplied, gradeOnHidden, gradeSemanticStatus, groupRunsByAgentProfileCell, harnessAxisOf, hashContent, hashJson, hiddenGrade, holm, improvementVerdict, inMemoryReviewStore, inferDomainKeywords, interRaterReliability, interpretCliffs, iqr, isBinaryOutcomeVector, isJudgeSpan, isLlmSpan, isModelPriced, isPolicyEdit, isRunRecord, isToolSpan, isTransientLlmError, jsonShape, jsonlReviewStore, jsonlRunRecordBackend, judgeAgreementView, judgeFamily, judgeSpans, knowledgeReadinessTracePayload, leaderboard, llmJudge, loadScorecard, localCommandRunner, makeFinding, makePolicyEdit, makePolicyEditCandidateRecord, makeProposalFinding, manifestContentDigest, mannWhitneyU, maximumChargeForLlmRequest, mcnemar, mcnemarPower, mcnemarRequiredN, minimumPairsForPairedDeltaTest, mintRolloutRows, modelHasSnapshot, modelPriceKey, mulberry32, notBlocked, objectiveEval, observeAll, otlpTextToTraceAnalysisStore, pairArms, pairRunRecords, pairedBinaryScale, pairedBootstrap, pairedCohensDz, pairedDeltaTest, pairedDeltaTieFraction, pairedEvalueSequence, pairedMde, pairedRiskDifference, pairedRiskDifferenceExact, pairedRiskDifferenceScore, pairedSignTest, pairedTTest, paretoChart, paretoFrontier, parseReflectionResponse, parseRunRecordSafe, partialCredit, partitionHeldOut, passAtK, pearsonR, policyEditFromFinding, policyEditsFromFindings, preflightModels, productBenchmarkRepoIdentity, profile_exports as profile, projectRuntimeTrajectoryEvidence, proposeSynthesisTargets, ranks, readProductBenchmarkManifest, recordRuns, recordRunsToScorecard, redTeamDataset, redTeamReport, redactString, regexMatches, renderPreferenceMemoryMarkdown, repeatedActionDetector, requiredPairedSampleSize, requiredSampleSize, resolveModelPricing, resolveSeat, roundTripRunRecord, routeFields, runAgentControlLoop, runBoundedProcess, runCampaign, runCanaries, runCounterfactual, runEquivalenceCheck, runEvalCampaign, runIntentMatchJudge, runKeywordCoverageJudge, runKeywordCoverageJudgeUrl, runProposeReview, runProposeReviewAsControlLoop, runSemanticConceptJudge, runTaskScore, runsForScenario, scoreKnowledgeReadiness, scorePolicyEditReadiness, scoreRedTeamOutput, scoreTraceInsightReadiness, seatPresets, selfImprove, servedModelAcceptable, spearmanR, stripFencedJson, subjectiveEval, summarizeBackendIntegrity, summarizeNumberSeries, summarizePreferenceMemory, summaryTable, textInSnapshot, toAgentProfileJson, tokenizeDomainWords, toolSpansToTraceAnalysisStore, toolWasteView, traceContract, transientDispatchFailure, urlContains, userQuestionsForKnowledgeGaps, validatePolicyEdit, validatePolicyEditCandidateRecord, validateRunRecord, verbosityBias, verifyAgentProfileCell, verifyAttestation, verifyCompletion, viteDeployRunner, weightedComposite, weightedMean, wilcoxonSignedRank, wilson, withAssignedFeedbackSplit, withHeldoutBlend, withJudgeRetry, wranglerDeployRunner };
|
|
7244
7947
|
|
|
7245
7948
|
//# sourceMappingURL=index.js.map
|