@tangle-network/agent-eval 0.170.0 → 0.171.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (73) hide show
  1. package/CHANGELOG.md +13 -0
  2. package/dist/adapters/http.d.ts +2 -2
  3. package/dist/analyst/index.d.ts +5 -5
  4. package/dist/analyst/index.js +1 -1
  5. package/dist/{benchmark-command-BA7qOdWw.js → benchmark-command-D8k3Gf0J.js} +2 -2
  6. package/dist/{benchmark-command-BA7qOdWw.js.map → benchmark-command-D8k3Gf0J.js.map} +1 -1
  7. package/dist/benchmarks/index.d.ts +4 -4
  8. package/dist/benchmarks/index.js +2 -2
  9. package/dist/campaign/index.d.ts +6 -6
  10. package/dist/campaign/index.js +3 -3
  11. package/dist/{campaign-BeCbxFqs.js → campaign-B72njjHj.js} +4 -4
  12. package/dist/{campaign-BeCbxFqs.js.map → campaign-B72njjHj.js.map} +1 -1
  13. package/dist/cli.js +1 -1
  14. package/dist/{client-_Fsa5c2_.d.ts → client-CDtcZ3p9.d.ts} +2 -2
  15. package/dist/{client-_Fsa5c2_.d.ts.map → client-CDtcZ3p9.d.ts.map} +1 -1
  16. package/dist/contract/index.d.ts +9 -9
  17. package/dist/contract/index.js +3 -3
  18. package/dist/{default-registry-ovxrOP0_.d.ts → default-registry-B0s2zU-s.d.ts} +3 -3
  19. package/dist/{default-registry-ovxrOP0_.d.ts.map → default-registry-B0s2zU-s.d.ts.map} +1 -1
  20. package/dist/{define-agent-eval-DVJm8Xlh.d.ts → define-agent-eval-Cjy2yhqP.d.ts} +4 -4
  21. package/dist/{define-agent-eval-DVJm8Xlh.d.ts.map → define-agent-eval-Cjy2yhqP.d.ts.map} +1 -1
  22. package/dist/{define-agent-eval-Clj-8igZ.js → define-agent-eval-Dy8QgxAI.js} +2 -2
  23. package/dist/{define-agent-eval-Clj-8igZ.js.map → define-agent-eval-Dy8QgxAI.js.map} +1 -1
  24. package/dist/{engine-D12Rb6WB.d.ts → engine-CAmTUk52.d.ts} +2 -2
  25. package/dist/{engine-D12Rb6WB.d.ts.map → engine-CAmTUk52.d.ts.map} +1 -1
  26. package/dist/experiment/index.d.ts +2 -2
  27. package/dist/{heldout-gate-Bn7_xWCv.d.ts → heldout-gate-Dh2b62w8.d.ts} +3 -3
  28. package/dist/{heldout-gate-Bn7_xWCv.d.ts.map → heldout-gate-Dh2b62w8.d.ts.map} +1 -1
  29. package/dist/hosted/index.d.ts +1 -1
  30. package/dist/{index-DBbivBNs.d.ts → index-8VIogTyS.d.ts} +6 -6
  31. package/dist/{index-DBbivBNs.d.ts.map → index-8VIogTyS.d.ts.map} +1 -1
  32. package/dist/{index-CM-SM00y.d.ts → index-DT73JraI.d.ts} +3 -3
  33. package/dist/{index-CM-SM00y.d.ts.map → index-DT73JraI.d.ts.map} +1 -1
  34. package/dist/{index-Bfs5aufo.d.ts → index-fNXZMCzX.d.ts} +9 -9
  35. package/dist/{index-Bfs5aufo.d.ts.map → index-fNXZMCzX.d.ts.map} +1 -1
  36. package/dist/index.d.ts +138 -9
  37. package/dist/index.d.ts.map +1 -1
  38. package/dist/index.js +709 -7
  39. package/dist/index.js.map +1 -1
  40. package/dist/{llm-judge-DbJdo8Nj.js → llm-judge-BtJ2Sfk_.js} +5 -2
  41. package/dist/{llm-judge-DbJdo8Nj.js.map → llm-judge-BtJ2Sfk_.js.map} +1 -1
  42. package/dist/{matrix-BpI5Trmo.d.ts → matrix-DiHmUobV.d.ts} +2 -2
  43. package/dist/{matrix-BpI5Trmo.d.ts.map → matrix-DiHmUobV.d.ts.map} +1 -1
  44. package/dist/multishot/golden/index.d.ts +1 -1
  45. package/dist/multishot/index.d.ts +2 -2
  46. package/dist/openapi.json +1 -1
  47. package/dist/{produced-state-CtSIp5cQ.js → produced-state-Cm6DU_Ao.js} +2 -2
  48. package/dist/{produced-state-CtSIp5cQ.js.map → produced-state-Cm6DU_Ao.js.map} +1 -1
  49. package/dist/{promotion-policy-CkXSgKkF.d.ts → promotion-policy-WSXtBgBb.d.ts} +2 -2
  50. package/dist/{promotion-policy-CkXSgKkF.d.ts.map → promotion-policy-WSXtBgBb.d.ts.map} +1 -1
  51. package/dist/{provenance-CIRUardl.d.ts → provenance-CafMdZKM.d.ts} +7 -3
  52. package/dist/{provenance-CIRUardl.d.ts.map → provenance-CafMdZKM.d.ts.map} +1 -1
  53. package/dist/rl.d.ts +1 -1
  54. package/dist/{skillopt-optimization-method-B2R9C5aG.js → skillopt-optimization-method-DzlF2RM7.js} +2 -2
  55. package/dist/{skillopt-optimization-method-B2R9C5aG.js.map → skillopt-optimization-method-DzlF2RM7.js.map} +1 -1
  56. package/dist/{statistical-heldout-DFS7QGpS.d.ts → statistical-heldout-UhiexnjU.d.ts} +2 -2
  57. package/dist/{statistical-heldout-DFS7QGpS.d.ts.map → statistical-heldout-UhiexnjU.d.ts.map} +1 -1
  58. package/dist/{store-tool-spans-B2DJ_82T.d.ts → store-tool-spans-D_qMl2__.d.ts} +3 -3
  59. package/dist/{store-tool-spans-B2DJ_82T.d.ts.map → store-tool-spans-D_qMl2__.d.ts.map} +1 -1
  60. package/dist/{tool-groups-BnXlCJZQ.d.ts → tool-groups-RGYfVWpc.d.ts} +2 -2
  61. package/dist/tool-groups-RGYfVWpc.d.ts.map +1 -0
  62. package/dist/trace-repair/index.d.ts +1 -1
  63. package/dist/traces.d.ts +2 -2
  64. package/dist/{types-i21ccEkr.d.ts → types-CMyW4GnH.d.ts} +2 -2
  65. package/dist/{types-i21ccEkr.d.ts.map → types-CMyW4GnH.d.ts.map} +1 -1
  66. package/dist/{types-Dy237wiH.d.ts → types-JHMOqZI4.d.ts} +11 -1
  67. package/dist/types-JHMOqZI4.d.ts.map +1 -0
  68. package/dist/wire/index.d.ts +3 -3
  69. package/dist/wire/index.d.ts.map +1 -1
  70. package/docs/campaign-proposers.md +4 -0
  71. package/package.json +19 -18
  72. package/dist/tool-groups-BnXlCJZQ.d.ts.map +0 -1
  73. package/dist/types-Dy237wiH.d.ts.map +0 -1
package/dist/index.js CHANGED
@@ -1,9 +1,9 @@
1
1
  import { t as __exportAll } from "./rolldown-runtime-8H4AJuhK.js";
2
2
  import { i as JudgeError, o as NotFoundError, r as ConfigError, s as ValidationError, t as AgentEvalError } from "./errors-Dngq5h35.js";
3
- import { r as canonicalString } from "./canonical-DPyQ_rpt.js";
4
- import { a as CODING_HARNESSES, c as agentProfileId, d as harnessAxisOf, i as verifyCompletion, l as agentProfileModelId, n as completionVerdict, o as HARNESS_NATIVE_MODEL, r as createLlmCorrectnessChecker, s as agentProfileHash, t as extractProducedState, u as expandProfileAxes } from "./produced-state-CtSIp5cQ.js";
3
+ import { a as hashCanonical, r as canonicalString } from "./canonical-DPyQ_rpt.js";
4
+ import { a as CODING_HARNESSES, c as agentProfileId, d as harnessAxisOf, i as verifyCompletion, l as agentProfileModelId, n as completionVerdict, o as HARNESS_NATIVE_MODEL, r as createLlmCorrectnessChecker, s as agentProfileHash, t as extractProducedState, u as expandProfileAxes } from "./produced-state-Cm6DU_Ao.js";
5
5
  import { n as hashJson, r as manifestContentDigest } from "./pre-registration-D94b7Of5.js";
6
- import { AGENT_PROFILE_KINDS, agentProfileCellHashMaterial, agentProfileCellKey, buildAgentProfileCell, groupRunsByAgentProfileCell, toAgentProfileJson, verifyAgentProfileCell } from "./profile-cell.js";
6
+ import { AGENT_PROFILE_KINDS, agentProfileCellHashMaterial, agentProfileCellKey, buildAgentProfileCell, groupRunsByAgentProfileCell, toAgentProfileJson, validateAgentProfileCell, verifyAgentProfileCell } from "./profile-cell.js";
7
7
  import { t as mulberry32 } from "./random-Dn5fPWkt.js";
8
8
  import { a as spearmanR, c as weightedMean, i as ranks, n as partialCredit, o as summarizeNumberSeries, r as pearsonR, s as weightedComposite, t as confidenceInterval } from "./descriptive-1V17A-qa.js";
9
9
  import { i as verbosityBias, n as calibrateJudgeContinuous, r as continuousAgreement, t as calibrateJudge } from "./judge-calibration-BnpVKtnb.js";
@@ -16,12 +16,12 @@ import { t as eProcess } from "./sequential-eprocess-D1jKoihe.js";
16
16
  import { n as iqr, r as welchsTTest } from "./baseline-BC-eBZ7U.js";
17
17
  import { a as isToolSpan, i as isLlmSpan, r as isJudgeSpan, t as FAILURE_CLASSES } from "./schema-CdIX2aHu.js";
18
18
  import { d as runsForScenario, r as argHash, s as judgeSpans } from "./query-BPGMVlbM.js";
19
- import { i as analyzeRuns, o as checkCanaries, r as selfImprove, t as defineAgentEval } from "./define-agent-eval-Clj-8igZ.js";
19
+ import { i as analyzeRuns, o as checkCanaries, r as selfImprove, t as defineAgentEval } from "./define-agent-eval-Dy8QgxAI.js";
20
20
  import { r as observedSplitScore, t as isRealnessGated } from "./reward-nw2xZGZG.js";
21
21
  import { a as parseRunRecordSafe, c as validateRunRecord, i as modelHasSnapshot, n as UNKNOWN_MODEL, o as roundTripRunRecord, r as isRunRecord, s as runTaskScore, t as RunRecordValidationError } from "./run-record-DLORoL7t.js";
22
22
  import { a as summaryTable, n as gainHistogram, r as paretoChart } from "./summary-report-Bgh8CpNK.js";
23
23
  import { n as contentHash, r as fileVerdictCache, t as canonicalJson } from "./verdict-cache-B3eCVQtY.js";
24
- import { $ as redTeamDataset, C as paretoFrontier, F as surfaceContentHash, K as buildReflectionPrompt, Q as DEFAULT_RED_TEAM_CORPUS, S as dominates, a as aggregateRunScore, ct as BackendIntegrityError, et as redTeamReport, i as transientDispatchFailure, it as runCampaign, lt as assertRealBackend, nt as runCanaries, o as clamp01, q as parseReflectionResponse, t as llmJudge, tt as scoreRedTeamOutput, ut as summarizeBackendIntegrity } from "./llm-judge-DbJdo8Nj.js";
24
+ import { $ as redTeamDataset, C as paretoFrontier, F as surfaceContentHash, K as buildReflectionPrompt, Q as DEFAULT_RED_TEAM_CORPUS, S as dominates, a as aggregateRunScore, ct as BackendIntegrityError, et as redTeamReport, i as transientDispatchFailure, it as runCampaign, lt as assertRealBackend, nt as runCanaries, o as clamp01, q as parseReflectionResponse, t as llmJudge, tt as scoreRedTeamOutput, ut as summarizeBackendIntegrity } from "./llm-judge-BtJ2Sfk_.js";
25
25
  import { a as resolveModelPricing, i as isModelPriced, n as estimateCost, r as estimateTokens, t as MODEL_PRICING } from "./metrics-Qv-cpptD.js";
26
26
  import { a as CostLedgerPersistenceError, c as costForTokenPricing, i as CostLedger, l as costForUsage, n as CostCallConflictError, o as CostReceiptCaptureError, r as CostCeilingReachedError, s as CostReservationExceededError, t as CostAccountingIncompleteError, u as modelPriceKey } from "./cost-ledger-B1qx30B4.js";
27
27
  import { n as REDACTION_VERSION, r as redactString, t as DEFAULT_REDACTION_RULES } from "./redact-7Aq1ukl-.js";
@@ -31,7 +31,7 @@ import { n as equivalenceVerdict, t as certificationEvidenceDigest } from "./ver
31
31
  import { t as TraceEmitter } from "./emitter-DeQHiDMm.js";
32
32
  import { t as buildTrajectory } from "./trajectory-D_7rLrvE.js";
33
33
  import { t as runCounterfactual } from "./counterfactual-Bjq1mlUu.js";
34
- import { c as createBoundedTraceAnalysisStore, f as DEFAULT_TRACE_ANALYST_BUDGETS, p as TRACE_ANALYST_TRUNCATION_MARKER_PREFIX, t as createTraceAnalyst } from "./kind-factory-DMeEoMQZ.js";
34
+ import { N as parseFindingSubject, c as createBoundedTraceAnalysisStore, f as DEFAULT_TRACE_ANALYST_BUDGETS, p as TRACE_ANALYST_TRUNCATION_MARKER_PREFIX, t as createTraceAnalyst } from "./kind-factory-DMeEoMQZ.js";
35
35
  import { _ as assertUniqueFindingIds, a as DEFAULT_TRACE_ANALYST_KINDS, b as snapshotAnalystRun, g as analystRunDigest, h as analystFindingDigest, l as FAILURE_MODE_KIND_SPEC, n as buildDefaultAnalystRegistry, r as AnalystRegistry, t as createChatClient, v as completedAnalystReviewQuality, x as validateAnalystReviewDecisions, y as readAnalystReview } from "./chat-client-DEtybj5i.js";
36
36
  import { OUTPUT_VALUE } from "./trace-attributes.js";
37
37
  import { r as extractUsageFromSse, t as extractUsage } from "./extract-usage-BrQ8mCLX.js";
@@ -3279,6 +3279,708 @@ function controlFailureClassFromVerification(verification) {
3279
3279
  return verification.failingLayers?.length ? "instruction_following" : "unknown";
3280
3280
  }
3281
3281
  //#endregion
3282
+ //#region src/analyst/steer-firewall.ts
3283
+ /** True iff the finding is a JUDGE VERDICT (an acceptance score lifted into a
3284
+ * finding), identified by provenance set at the lift site — independent of
3285
+ * whatever evidence it cites. */
3286
+ function isJudgeVerdict(finding) {
3287
+ return finding.derived_from_judge === true;
3288
+ }
3289
+ /**
3290
+ * THE steer firewall. Fail-loud guard for any path that admits analyst findings
3291
+ * as STEERING input (the `f(trace)` role): rejects — naming the offenders — any
3292
+ * finding whose provenance is a judge verdict, rather than let `J` leak into the
3293
+ * loop. Returns the findings unchanged for chaining.
3294
+ *
3295
+ * Call this at the chokepoint where a detector that ALSO scores/gates has its
3296
+ * findings turned into a steer (the judge-and-steer dual-role case). It keys on
3297
+ * provenance, so it correctly admits evidence-less trace-analyst observations and
3298
+ * correctly rejects an artifact-citing judge verdict — the cases an evidence
3299
+ * check gets backwards.
3300
+ *
3301
+ * It is necessary, not sufficient: it stops PROVENANCE-tagged verdicts. A judge
3302
+ * whose output is laundered through a hand-built finding with no provenance flag
3303
+ * is out of its reach — provenance must be honestly set at every judge→finding
3304
+ * lift (today: createJudgeAdapter). That is why the integrity rule lives at the
3305
+ * lift site, and why ProposeContext.judgeScores?: never is the complementary
3306
+ * compile-time tripwire on the obvious direct channel.
3307
+ */
3308
+ function assertNoJudgeVerdict(findings, context = "steer") {
3309
+ const leaks = findings.filter(isJudgeVerdict);
3310
+ if (leaks.length > 0) throw new Error(`${context}: a judge verdict cannot be admitted as steering input — that is the held-out judge leaking into the loop. Offending judge-derived findings: [${leaks.map((f) => f.finding_id).join(", ")}]. Steering consumes observations of behavior, never acceptance verdicts.`);
3311
+ return findings;
3312
+ }
3313
+ //#endregion
3314
+ //#region src/analyst/policy-edit.ts
3315
+ const POLICY_EDIT_AXES = [
3316
+ "carrier",
3317
+ "representation",
3318
+ "budget",
3319
+ "sampling",
3320
+ "output_contract",
3321
+ "tool_contract",
3322
+ "routing",
3323
+ "memory",
3324
+ "agent_profile",
3325
+ "deployment_target"
3326
+ ];
3327
+ const POLICY_EDIT_TARGET_SURFACES = [
3328
+ "prompt",
3329
+ "tool-contract",
3330
+ "runtime-config",
3331
+ "memory",
3332
+ "agent-profile",
3333
+ "code",
3334
+ "deployment"
3335
+ ];
3336
+ const POLICY_EDIT_CANDIDATE_RECORD_SCHEMA = "tangle.policy-edit-candidate.v1";
3337
+ var PolicyEditValidationError = class extends ValidationError {
3338
+ path;
3339
+ constructor(message, path = "") {
3340
+ super(path ? `${message} (at ${path})` : message);
3341
+ this.path = path;
3342
+ }
3343
+ };
3344
+ const DEFAULT_MIN_SCORE = .7;
3345
+ const DEFAULT_MIN_EXPECTED_GAIN = .01;
3346
+ const POLICY_EDIT_ID = /^policy-edit:sha256:[0-9a-f]{64}$/;
3347
+ function makePolicyEdit(init) {
3348
+ const normalized = normalizePolicyEdit({
3349
+ schemaVersion: "policy-edit/v1",
3350
+ ...init,
3351
+ source: normalizeSource(init.source)
3352
+ });
3353
+ return validatePolicyEdit({
3354
+ ...normalized,
3355
+ editId: init.editId ?? computePolicyEditId(normalized)
3356
+ });
3357
+ }
3358
+ function computePolicyEditId(edit) {
3359
+ const { editId: _editId, schemaVersion, ...material } = edit;
3360
+ return `policy-edit:${hashCanonical({
3361
+ schemaVersion,
3362
+ ...material
3363
+ })}`;
3364
+ }
3365
+ function validatePolicyEdit(input) {
3366
+ if (input === null || typeof input !== "object") throw new PolicyEditValidationError("expected object");
3367
+ const obj = input;
3368
+ expectLiteral(obj.schemaVersion, "policy-edit/v1", "schemaVersion");
3369
+ expectString$1(obj.editId, "editId");
3370
+ if (!POLICY_EDIT_ID.test(obj.editId)) throw new PolicyEditValidationError("editId must match policy-edit:sha256:<64 lowercase hex chars>", "editId");
3371
+ expectOneOf(obj.axis, POLICY_EDIT_AXES, "axis");
3372
+ validateTarget(obj.target);
3373
+ validateChange(obj.change);
3374
+ expectString$1(obj.claim, "claim");
3375
+ validateExpectedGain(obj.expectedGain);
3376
+ expectConfidence(obj.confidence, "confidence");
3377
+ expectOneOf(obj.risk, [
3378
+ "low",
3379
+ "medium",
3380
+ "high",
3381
+ "unknown"
3382
+ ], "risk");
3383
+ validateSource(obj.source);
3384
+ if (obj.rationale !== void 0) expectString$1(obj.rationale, "rationale");
3385
+ if (obj.validationPlan !== void 0) expectString$1(obj.validationPlan, "validationPlan");
3386
+ if (obj.metadata !== void 0 && (obj.metadata === null || typeof obj.metadata !== "object")) throw new PolicyEditValidationError("expected object", "metadata");
3387
+ const expectedId = computePolicyEditId(obj);
3388
+ if (obj.editId !== expectedId) throw new PolicyEditValidationError("editId does not match policy edit content", "editId");
3389
+ return obj;
3390
+ }
3391
+ function makePolicyEditCandidateRecord(edit) {
3392
+ return validatePolicyEditCandidateRecord({
3393
+ schema: POLICY_EDIT_CANDIDATE_RECORD_SCHEMA,
3394
+ policyEdit: edit
3395
+ });
3396
+ }
3397
+ function validatePolicyEditCandidateRecord(input) {
3398
+ if (input === null || typeof input !== "object" || Array.isArray(input)) throw new PolicyEditValidationError("expected object", "candidateRecord");
3399
+ const obj = input;
3400
+ const keys = Object.keys(obj).sort();
3401
+ if (keys.length !== 2 || keys[0] !== "policyEdit" || keys[1] !== "schema") throw new PolicyEditValidationError("expected exactly schema and policyEdit", "candidateRecord");
3402
+ expectLiteral(obj.schema, POLICY_EDIT_CANDIDATE_RECORD_SCHEMA, "candidateRecord.schema");
3403
+ const policyEdit = validatePolicyEdit(obj.policyEdit);
3404
+ assertJsonSafe(policyEdit, "candidateRecord.policyEdit");
3405
+ const snapshot = JSON.parse(JSON.stringify(policyEdit));
3406
+ return {
3407
+ schema: POLICY_EDIT_CANDIDATE_RECORD_SCHEMA,
3408
+ policyEdit: validatePolicyEdit(snapshot)
3409
+ };
3410
+ }
3411
+ function isPolicyEdit(input) {
3412
+ try {
3413
+ validatePolicyEdit(input);
3414
+ return true;
3415
+ } catch {
3416
+ return false;
3417
+ }
3418
+ }
3419
+ function policyEditsFromFindings(findings, opts = {}) {
3420
+ assertNoJudgeVerdict(findings, "policyEditsFromFindings");
3421
+ const edits = [];
3422
+ for (const finding of findings) {
3423
+ const edit = policyEditFromFinding(finding, opts);
3424
+ if (edit) edits.push(edit);
3425
+ }
3426
+ return edits;
3427
+ }
3428
+ function policyEditFromFinding(finding, opts = {}) {
3429
+ assertNoJudgeVerdict([finding], "policyEditFromFinding");
3430
+ if (!finding.recommended_action?.trim()) return null;
3431
+ const expectedGain = resolveExpectedGain(finding, opts);
3432
+ if (!expectedGain) return null;
3433
+ if (typeof finding.confidence !== "number" || !Number.isFinite(finding.confidence) || finding.confidence < 0 || finding.confidence > 1) return null;
3434
+ const routed = routeFindingSubject(finding.subject, opts);
3435
+ const risk = resolveRisk(finding, opts);
3436
+ return makePolicyEdit({
3437
+ axis: routed.axis,
3438
+ target: routed.target,
3439
+ change: {
3440
+ kind: "text",
3441
+ mode: "append",
3442
+ value: finding.recommended_action.trim()
3443
+ },
3444
+ claim: finding.claim,
3445
+ rationale: finding.rationale,
3446
+ expectedGain,
3447
+ confidence: finding.confidence,
3448
+ risk,
3449
+ validationPlan: finding.validation_plan,
3450
+ source: {
3451
+ findingIds: [finding.finding_id],
3452
+ analystIds: [finding.analyst_id],
3453
+ evidenceRefs: finding.evidence_refs,
3454
+ derivedFromJudge: finding.derived_from_judge
3455
+ }
3456
+ });
3457
+ }
3458
+ function scorePolicyEditReadiness(edit, opts = {}) {
3459
+ validatePolicyEdit(edit);
3460
+ const minExpectedGain = opts.minExpectedGain ?? DEFAULT_MIN_EXPECTED_GAIN;
3461
+ const requireEvidence = opts.requireEvidence ?? false;
3462
+ const evidenceScore = Math.min(1, edit.source.evidenceRefs.length / 2);
3463
+ const confidenceScore = clamp01$3(edit.confidence);
3464
+ const gainScore = clamp01$3(Math.abs(edit.expectedGain.amount) / Math.max(minExpectedGain * 5, .001));
3465
+ const targetScore = targetSpecificityScore(edit);
3466
+ const riskPenalty = edit.risk === "high" && opts.allowHighRisk !== true ? .35 : edit.risk === "unknown" ? .2 : 0;
3467
+ return clamp01$3((requireEvidence ? .3 * evidenceScore + .25 * confidenceScore + .25 * gainScore + .2 * targetScore : .35 * confidenceScore + .35 * gainScore + .3 * targetScore + .1 * evidenceScore) - riskPenalty);
3468
+ }
3469
+ function admitPolicyEdit(edit, opts = {}) {
3470
+ const validated = validatePolicyEdit(edit);
3471
+ const score = scorePolicyEditReadiness(validated, opts);
3472
+ const reasons = [];
3473
+ const minExpectedGain = opts.minExpectedGain ?? DEFAULT_MIN_EXPECTED_GAIN;
3474
+ const requireEvidence = opts.requireEvidence ?? false;
3475
+ if (validated.source.derivedFromJudge) reasons.push("source is judge-derived; judge verdicts cannot steer policy edits");
3476
+ if (requireEvidence && validated.source.evidenceRefs.length === 0) reasons.push("missing evidence refs");
3477
+ if (Math.abs(validated.expectedGain.amount) < minExpectedGain) reasons.push(`expected gain below ${minExpectedGain}`);
3478
+ if (validated.risk === "high" && opts.allowHighRisk !== true) reasons.push("high-risk edit requires explicit allowHighRisk");
3479
+ if (score < (opts.minScore ?? DEFAULT_MIN_SCORE)) reasons.push(`readiness score ${score.toFixed(3)} below ${(opts.minScore ?? DEFAULT_MIN_SCORE).toFixed(3)}`);
3480
+ return {
3481
+ edit: validated,
3482
+ decision: reasons.length === 0 ? "admit" : "reject",
3483
+ score,
3484
+ reasons
3485
+ };
3486
+ }
3487
+ function applyPolicyEditToSurface(surface, edit) {
3488
+ const validated = validatePolicyEdit(edit);
3489
+ if (validated.change.kind === "text") return applyTextChange(surface, validated.change);
3490
+ return applyJsonChange(surface, validated.change);
3491
+ }
3492
+ function routeFindingSubject(subject, opts) {
3493
+ const parsed = parseFindingSubject(subject);
3494
+ if (!parsed) return {
3495
+ axis: opts.defaultAxis ?? "representation",
3496
+ target: { surface: opts.defaultTargetSurface ?? "prompt" }
3497
+ };
3498
+ return routeParsedSubject(parsed);
3499
+ }
3500
+ function routeParsedSubject(subject) {
3501
+ switch (subject.kind) {
3502
+ case "system-prompt": return {
3503
+ axis: "representation",
3504
+ target: {
3505
+ surface: "prompt",
3506
+ path: `system-prompt:${subject.section}`
3507
+ }
3508
+ };
3509
+ case "skill": return {
3510
+ axis: "agent_profile",
3511
+ target: {
3512
+ surface: "agent-profile",
3513
+ path: `skill:${subject.name}`
3514
+ }
3515
+ };
3516
+ case "tool-doc": return {
3517
+ axis: "tool_contract",
3518
+ target: {
3519
+ surface: "tool-contract",
3520
+ path: subject.aspect ? `tool-doc:${subject.tool}:${subject.aspect}` : `tool-doc:${subject.tool}`
3521
+ }
3522
+ };
3523
+ case "new-tool": return {
3524
+ axis: "tool_contract",
3525
+ target: {
3526
+ surface: "tool-contract",
3527
+ path: `new-tool:${subject.name}`
3528
+ }
3529
+ };
3530
+ case "mcp": return {
3531
+ axis: "tool_contract",
3532
+ target: {
3533
+ surface: "agent-profile",
3534
+ path: subject.tool ? `mcp:${subject.server}:${subject.tool}` : `mcp:${subject.server}`
3535
+ }
3536
+ };
3537
+ case "hook": return {
3538
+ axis: "agent_profile",
3539
+ target: {
3540
+ surface: "agent-profile",
3541
+ path: `hook:${subject.name}`
3542
+ }
3543
+ };
3544
+ case "subagent": return {
3545
+ axis: "routing",
3546
+ target: {
3547
+ surface: "agent-profile",
3548
+ path: `subagent:${subject.name}`
3549
+ }
3550
+ };
3551
+ case "workflow": return {
3552
+ axis: "routing",
3553
+ target: {
3554
+ surface: "runtime-config",
3555
+ path: `workflow:${subject.name}`
3556
+ }
3557
+ };
3558
+ case "rollout-policy": return {
3559
+ axis: rolloutPolicyAxis(subject.field),
3560
+ target: {
3561
+ surface: "runtime-config",
3562
+ path: `rollout-policy:${subject.field}`
3563
+ }
3564
+ };
3565
+ case "agent-profile": return {
3566
+ axis: "agent_profile",
3567
+ target: {
3568
+ surface: "agent-profile",
3569
+ path: `agent-profile:${subject.field}`
3570
+ }
3571
+ };
3572
+ case "code": return {
3573
+ axis: "representation",
3574
+ target: {
3575
+ surface: "code",
3576
+ path: `code:${subject.path}`
3577
+ }
3578
+ };
3579
+ case "rag": return {
3580
+ axis: "memory",
3581
+ target: {
3582
+ surface: "memory",
3583
+ path: `rag:${subject.corpus}:${subject.docId}`
3584
+ }
3585
+ };
3586
+ case "memory": return {
3587
+ axis: "memory",
3588
+ target: {
3589
+ surface: "memory",
3590
+ path: `memory:${subject.key}`
3591
+ }
3592
+ };
3593
+ case "scaffolding": return {
3594
+ axis: "routing",
3595
+ target: {
3596
+ surface: "runtime-config",
3597
+ path: `scaffolding:${subject.concern}`
3598
+ }
3599
+ };
3600
+ case "output-schema": return {
3601
+ axis: "output_contract",
3602
+ target: {
3603
+ surface: "runtime-config",
3604
+ path: `output-schema:${subject.field}`
3605
+ }
3606
+ };
3607
+ case "knowledge.wiki": return {
3608
+ axis: "memory",
3609
+ target: {
3610
+ surface: "memory",
3611
+ path: `agent-knowledge:wiki:${subject.slug}${subject.heading ? `#${subject.heading}` : ""}`
3612
+ }
3613
+ };
3614
+ case "knowledge.claim": return {
3615
+ axis: "memory",
3616
+ target: {
3617
+ surface: "memory",
3618
+ path: `agent-knowledge:claim:${subject.topic}`
3619
+ }
3620
+ };
3621
+ case "knowledge.raw": return {
3622
+ axis: "memory",
3623
+ target: {
3624
+ surface: "memory",
3625
+ path: `agent-knowledge:raw:${subject.sourceId}`
3626
+ }
3627
+ };
3628
+ case "knowledge.stale": return {
3629
+ axis: "memory",
3630
+ target: {
3631
+ surface: "memory",
3632
+ path: `agent-knowledge:stale:${subject.slug}`
3633
+ }
3634
+ };
3635
+ case "websearch.outdated": return {
3636
+ axis: "memory",
3637
+ target: {
3638
+ surface: "memory",
3639
+ path: `websearch:outdated:${subject.topic}`
3640
+ }
3641
+ };
3642
+ case "prior-run-summary": return {
3643
+ axis: "memory",
3644
+ target: {
3645
+ surface: "memory",
3646
+ path: `prior-run-summary:${subject.topic}`
3647
+ }
3648
+ };
3649
+ case "cluster": return {
3650
+ axis: "representation",
3651
+ target: {
3652
+ surface: "prompt",
3653
+ path: subject.label
3654
+ }
3655
+ };
3656
+ }
3657
+ }
3658
+ function rolloutPolicyAxis(field) {
3659
+ const normalized = field.toLowerCase();
3660
+ if (/budget|max(?:imum)?[-_. ]?(?:turns?|tokens?|cost)|timeout|deadline/.test(normalized)) return "budget";
3661
+ if (/temperature|top[-_. ]?p|sampling|seed|shots?|parallel|concurrency/.test(normalized)) return "sampling";
3662
+ if (/output|schema|format/.test(normalized)) return "output_contract";
3663
+ return "routing";
3664
+ }
3665
+ function resolveExpectedGain(finding, opts) {
3666
+ if (typeof opts.expectedGain === "function") return opts.expectedGain(finding) ?? null;
3667
+ if (opts.expectedGain) return opts.expectedGain;
3668
+ return readExpectedGainFromMetadata(finding.metadata);
3669
+ }
3670
+ function readExpectedGainFromMetadata(metadata) {
3671
+ const raw = readPolicyEditMetadata(metadata)?.expectedGain ?? readPolicyEditMetadata(metadata)?.expected_gain;
3672
+ if (!raw || typeof raw !== "object") return null;
3673
+ const obj = raw;
3674
+ if (typeof obj.metric !== "string" || obj.direction !== "increase" && obj.direction !== "decrease" || typeof obj.amount !== "number" || !Number.isFinite(obj.amount) || obj.amount <= 0) return null;
3675
+ const out = {
3676
+ metric: obj.metric,
3677
+ direction: obj.direction,
3678
+ amount: obj.amount
3679
+ };
3680
+ if (obj.unit === "absolute" || obj.unit === "relative" || obj.unit === "percent" || obj.unit === "score") out.unit = obj.unit;
3681
+ if (typeof obj.rationale === "string") out.rationale = obj.rationale;
3682
+ return out;
3683
+ }
3684
+ function readPolicyEditMetadata(metadata) {
3685
+ const raw = metadata?.policyEdit ?? metadata?.policy_edit;
3686
+ return raw && typeof raw === "object" ? raw : null;
3687
+ }
3688
+ function resolveRisk(finding, opts) {
3689
+ if (typeof opts.risk === "function") return opts.risk(finding);
3690
+ if (opts.risk) return opts.risk;
3691
+ const raw = readPolicyEditMetadata(finding.metadata)?.risk;
3692
+ if (raw === "low" || raw === "medium" || raw === "high" || raw === "unknown") return raw;
3693
+ if (finding.severity === "critical" || finding.severity === "high") return "medium";
3694
+ return "low";
3695
+ }
3696
+ function applyTextChange(surface, change) {
3697
+ if (typeof surface !== "string") throw new PolicyEditValidationError("text policy edits require a string surface", "change");
3698
+ if (change.mode === "append") {
3699
+ if (hasExactTextBlock(surface, change.value)) return surface;
3700
+ return `${surface.trimEnd()}\n\n${change.value}`.trimStart();
3701
+ }
3702
+ if (change.mode === "prepend") {
3703
+ if (hasExactTextBlock(surface, change.value)) return surface;
3704
+ return `${change.value}\n\n${surface.trimStart()}`.trimEnd();
3705
+ }
3706
+ const find = expectNonEmpty(change.find, "change.find");
3707
+ if (!surface.includes(find)) throw new PolicyEditValidationError("replace target not found in surface", "change.find");
3708
+ return surface.replace(find, change.value);
3709
+ }
3710
+ function applyJsonChange(surface, change) {
3711
+ const root = parseJsonSurface(surface);
3712
+ const path = splitPath(change.path);
3713
+ if (change.mode === "remove") return setJsonAtPath(root, path, void 0, "remove");
3714
+ if (change.mode === "set") return setJsonAtPath(root, path, change.value ?? null, "set");
3715
+ const prior = readJsonAtPath(root, path);
3716
+ return setJsonAtPath(root, path, prior && typeof prior === "object" && !Array.isArray(prior) && change.value && typeof change.value === "object" && !Array.isArray(change.value) ? {
3717
+ ...prior,
3718
+ ...change.value
3719
+ } : change.value ?? null, "set");
3720
+ }
3721
+ function parseJsonSurface(surface) {
3722
+ if (typeof surface === "string") try {
3723
+ return JSON.parse(surface);
3724
+ } catch {
3725
+ throw new PolicyEditValidationError("json policy edits require a JSON string surface", "change");
3726
+ }
3727
+ assertJson(surface, "surface");
3728
+ return surface;
3729
+ }
3730
+ function readJsonAtPath(root, path) {
3731
+ let cursor = root;
3732
+ for (const part of path) {
3733
+ if (!cursor || typeof cursor !== "object" || Array.isArray(cursor)) return void 0;
3734
+ cursor = cursor[part];
3735
+ }
3736
+ return cursor;
3737
+ }
3738
+ function setJsonAtPath(root, path, value, mode) {
3739
+ if (path.length === 0) {
3740
+ if (mode === "remove") return null;
3741
+ return value ?? null;
3742
+ }
3743
+ if (root === null || typeof root !== "object" || Array.isArray(root)) throw new PolicyEditValidationError("json edit root must be an object", "change.path");
3744
+ const out = { ...root };
3745
+ let cursor = out;
3746
+ for (let i = 0; i < path.length - 1; i++) {
3747
+ const key = path[i];
3748
+ const existing = cursor[key];
3749
+ if (mode === "remove" && (!existing || typeof existing !== "object" || Array.isArray(existing))) return out;
3750
+ const next = existing && typeof existing === "object" && !Array.isArray(existing) ? { ...existing } : {};
3751
+ cursor[key] = next;
3752
+ cursor = next;
3753
+ }
3754
+ const leaf = path[path.length - 1];
3755
+ if (mode === "remove") delete cursor[leaf];
3756
+ else cursor[leaf] = value ?? null;
3757
+ return out;
3758
+ }
3759
+ function normalizePolicyEdit(input) {
3760
+ const out = {
3761
+ schemaVersion: "policy-edit/v1",
3762
+ axis: input.axis,
3763
+ target: normalizeTarget(input.target),
3764
+ change: normalizeChange(input.change),
3765
+ claim: input.claim.trim(),
3766
+ expectedGain: normalizeExpectedGain(input.expectedGain),
3767
+ confidence: input.confidence,
3768
+ risk: input.risk,
3769
+ source: normalizeSource(input.source)
3770
+ };
3771
+ if (input.rationale?.trim()) out.rationale = input.rationale.trim();
3772
+ if (input.validationPlan?.trim()) out.validationPlan = input.validationPlan.trim();
3773
+ if (input.metadata) out.metadata = input.metadata;
3774
+ return out;
3775
+ }
3776
+ function assertJsonSafe(value, path, ancestors = /* @__PURE__ */ new WeakSet()) {
3777
+ if (value === null || typeof value === "string" || typeof value === "boolean") return;
3778
+ if (typeof value === "number") {
3779
+ if (Number.isFinite(value)) return;
3780
+ throw new PolicyEditValidationError("expected finite JSON number", path);
3781
+ }
3782
+ if (typeof value !== "object") throw new PolicyEditValidationError("expected JSON-safe value", path);
3783
+ if (ancestors.has(value)) throw new PolicyEditValidationError("cyclic value is not JSON-safe", path);
3784
+ ancestors.add(value);
3785
+ if (Array.isArray(value)) for (let i = 0; i < value.length; i++) {
3786
+ if (!(i in value)) throw new PolicyEditValidationError("sparse array is not JSON-safe", `${path}.${i}`);
3787
+ assertJsonSafe(value[i], `${path}.${i}`, ancestors);
3788
+ }
3789
+ else {
3790
+ const prototype = Object.getPrototypeOf(value);
3791
+ if (prototype !== Object.prototype && prototype !== null) throw new PolicyEditValidationError("expected plain JSON object", path);
3792
+ if (Object.getOwnPropertySymbols(value).length > 0) throw new PolicyEditValidationError("symbol keys are not JSON-safe", path);
3793
+ for (const [key, child] of Object.entries(value)) assertJsonSafe(child, `${path}.${key}`, ancestors);
3794
+ }
3795
+ ancestors.delete(value);
3796
+ }
3797
+ function normalizeTarget(target) {
3798
+ const out = { surface: target.surface };
3799
+ if (target.path?.trim()) out.path = target.path.trim();
3800
+ if (target.agentProfileCell) out.agentProfileCell = validateAgentProfileCell(target.agentProfileCell);
3801
+ if (target.label?.trim()) out.label = target.label.trim();
3802
+ return out;
3803
+ }
3804
+ function normalizeChange(change) {
3805
+ if (change.kind === "text") {
3806
+ const out = {
3807
+ kind: "text",
3808
+ mode: change.mode,
3809
+ value: change.value.trim()
3810
+ };
3811
+ if (change.find?.trim()) out.find = change.find.trim();
3812
+ return out;
3813
+ }
3814
+ const out = {
3815
+ kind: "json",
3816
+ mode: change.mode,
3817
+ path: change.path.trim()
3818
+ };
3819
+ if (change.value !== void 0) out.value = change.value;
3820
+ return out;
3821
+ }
3822
+ function normalizeExpectedGain(gain) {
3823
+ const out = {
3824
+ metric: gain.metric.trim(),
3825
+ direction: gain.direction,
3826
+ amount: gain.amount
3827
+ };
3828
+ if (gain.unit) out.unit = gain.unit;
3829
+ if (gain.rationale?.trim()) out.rationale = gain.rationale.trim();
3830
+ return out;
3831
+ }
3832
+ function normalizeSource(source) {
3833
+ const out = {
3834
+ findingIds: uniqueSorted(source.findingIds.map((s) => s.trim()).filter(Boolean)),
3835
+ analystIds: uniqueSorted(source.analystIds.map((s) => s.trim()).filter(Boolean)),
3836
+ evidenceRefs: source.evidenceRefs
3837
+ };
3838
+ if (source.derivedFromJudge) out.derivedFromJudge = true;
3839
+ return out;
3840
+ }
3841
+ function validateTarget(target) {
3842
+ if (!target || typeof target !== "object") throw new PolicyEditValidationError("expected object", "target");
3843
+ const obj = target;
3844
+ expectOneOf(obj.surface, POLICY_EDIT_TARGET_SURFACES, "target.surface");
3845
+ if (obj.path !== void 0) expectString$1(obj.path, "target.path");
3846
+ if (obj.label !== void 0) expectString$1(obj.label, "target.label");
3847
+ if (obj.agentProfileCell !== void 0) validateAgentProfileCell(obj.agentProfileCell);
3848
+ }
3849
+ function validateChange(change) {
3850
+ if (!change || typeof change !== "object") throw new PolicyEditValidationError("expected object", "change");
3851
+ const obj = change;
3852
+ if (obj.kind !== "text" && obj.kind !== "json") throw new PolicyEditValidationError("kind must be text or json", "change.kind");
3853
+ if (obj.kind === "text") {
3854
+ expectOneOf(obj.mode, [
3855
+ "append",
3856
+ "prepend",
3857
+ "replace"
3858
+ ], "change.mode");
3859
+ expectString$1(obj.value, "change.value");
3860
+ if (obj.mode === "replace") expectString$1(obj.find, "change.find");
3861
+ return;
3862
+ }
3863
+ expectOneOf(obj.mode, [
3864
+ "set",
3865
+ "merge",
3866
+ "remove"
3867
+ ], "change.mode");
3868
+ expectString$1(obj.path, "change.path");
3869
+ splitPath(obj.path);
3870
+ if (obj.value !== void 0) assertJson(obj.value, "change.value");
3871
+ }
3872
+ function validateExpectedGain(gain) {
3873
+ if (!gain || typeof gain !== "object") throw new PolicyEditValidationError("expected object", "expectedGain");
3874
+ const obj = gain;
3875
+ expectString$1(obj.metric, "expectedGain.metric");
3876
+ expectOneOf(obj.direction, ["increase", "decrease"], "expectedGain.direction");
3877
+ if (!Number.isFinite(obj.amount) || obj.amount <= 0) throw new PolicyEditValidationError("amount must be a positive finite number", "expectedGain.amount");
3878
+ if (obj.unit !== void 0) expectOneOf(obj.unit, [
3879
+ "absolute",
3880
+ "relative",
3881
+ "percent",
3882
+ "score"
3883
+ ], "expectedGain.unit");
3884
+ if (obj.rationale !== void 0) expectString$1(obj.rationale, "expectedGain.rationale");
3885
+ }
3886
+ function validateSource(source) {
3887
+ if (!source || typeof source !== "object") throw new PolicyEditValidationError("expected object", "source");
3888
+ const obj = source;
3889
+ expectNonEmptyStringArray(obj.findingIds, "source.findingIds");
3890
+ expectNonEmptyStringArray(obj.analystIds, "source.analystIds");
3891
+ if (!Array.isArray(obj.evidenceRefs)) throw new PolicyEditValidationError("expected array", "source.evidenceRefs");
3892
+ for (const [i, ref] of obj.evidenceRefs.entries()) validateEvidenceRef(ref, `source.evidenceRefs.${i}`);
3893
+ if (obj.derivedFromJudge !== void 0 && typeof obj.derivedFromJudge !== "boolean") throw new PolicyEditValidationError("expected boolean", "source.derivedFromJudge");
3894
+ }
3895
+ function validateEvidenceRef(ref, path) {
3896
+ if (!ref || typeof ref !== "object") throw new PolicyEditValidationError("expected object", path);
3897
+ const obj = ref;
3898
+ expectOneOf(obj.kind, [
3899
+ "span",
3900
+ "event",
3901
+ "artifact",
3902
+ "finding",
3903
+ "metric"
3904
+ ], `${path}.kind`);
3905
+ expectString$1(obj.uri, `${path}.uri`);
3906
+ if (obj.excerpt !== void 0) expectString$1(obj.excerpt, `${path}.excerpt`);
3907
+ }
3908
+ function assertJson(value, path) {
3909
+ if (value === null || typeof value === "string" || typeof value === "boolean" || typeof value === "number" && Number.isFinite(value)) return;
3910
+ if (Array.isArray(value)) {
3911
+ for (const [i, item] of value.entries()) assertJson(item, `${path}.${i}`);
3912
+ return;
3913
+ }
3914
+ if (typeof value === "object") {
3915
+ for (const [key, item] of Object.entries(value)) {
3916
+ if (!key) throw new PolicyEditValidationError("empty object key", path);
3917
+ assertJson(item, `${path}.${key}`);
3918
+ }
3919
+ return;
3920
+ }
3921
+ throw new PolicyEditValidationError("expected JSON-compatible value", path);
3922
+ }
3923
+ function targetSpecificityScore(edit) {
3924
+ let score = .4;
3925
+ if (edit.target.path) score += .25;
3926
+ if (edit.target.agentProfileCell) score += .15;
3927
+ if (edit.change.kind === "json" || edit.change.mode === "replace") score += .2;
3928
+ else if (edit.change.value.length > 0) score += .1;
3929
+ return clamp01$3(score);
3930
+ }
3931
+ const FORBIDDEN_PATH_KEYS = /* @__PURE__ */ new Set([
3932
+ "__proto__",
3933
+ "constructor",
3934
+ "prototype"
3935
+ ]);
3936
+ function splitPath(path) {
3937
+ const parts = path.split(".").map((p) => p.trim()).filter(Boolean);
3938
+ if (parts.length === 0) throw new PolicyEditValidationError("path must not be empty", "change.path");
3939
+ for (const part of parts) if (FORBIDDEN_PATH_KEYS.has(part)) throw new PolicyEditValidationError(`path segment "${part}" would write through the prototype chain`, "change.path");
3940
+ return parts;
3941
+ }
3942
+ function expectLiteral(value, expected, path) {
3943
+ if (value !== expected) throw new PolicyEditValidationError(`expected ${expected}`, path);
3944
+ }
3945
+ function expectString$1(value, path) {
3946
+ if (typeof value !== "string" || value.trim().length === 0) throw new PolicyEditValidationError("expected non-empty string", path);
3947
+ }
3948
+ function expectNonEmpty(value, path) {
3949
+ expectString$1(value, path);
3950
+ return value;
3951
+ }
3952
+ function expectConfidence(value, path) {
3953
+ if (typeof value !== "number" || !Number.isFinite(value) || value < 0 || value > 1) throw new PolicyEditValidationError("expected finite number in [0,1]", path);
3954
+ }
3955
+ function hasExactTextBlock(surface, value) {
3956
+ const needle = normalizeTextBlock(value);
3957
+ const normalizedSurface = surface.replace(/\r\n/g, "\n");
3958
+ return [...normalizedSurface.split(/\n{2,}/), ...normalizedSurface.split("\n")].some((block) => normalizeTextBlock(block) === needle);
3959
+ }
3960
+ function normalizeTextBlock(value) {
3961
+ return value.replace(/\r\n/g, "\n").trim();
3962
+ }
3963
+ function expectOneOf(value, allowed, path) {
3964
+ if (typeof value !== "string" || !allowed.includes(value)) throw new PolicyEditValidationError(`expected one of ${allowed.join(", ")}`, path);
3965
+ }
3966
+ function expectStringArray$1(value, path) {
3967
+ if (!Array.isArray(value)) throw new PolicyEditValidationError("expected array", path);
3968
+ for (const [i, item] of value.entries()) expectString$1(item, `${path}.${i}`);
3969
+ }
3970
+ function expectNonEmptyStringArray(value, path) {
3971
+ expectStringArray$1(value, path);
3972
+ if (value.length === 0) throw new PolicyEditValidationError("expected non-empty array", path);
3973
+ }
3974
+ function uniqueSorted(values) {
3975
+ return [...new Set(values)].sort();
3976
+ }
3977
+ function clamp01$3(n) {
3978
+ if (!Number.isFinite(n)) return 0;
3979
+ if (n < 0) return 0;
3980
+ if (n > 1) return 1;
3981
+ return n;
3982
+ }
3983
+ //#endregion
3282
3984
  //#region src/capability-headroom.ts
3283
3985
  /**
3284
3986
  * Capability-headroom gate — "can this task set even SEE the capability
@@ -7240,6 +7942,6 @@ function rankRows(rows, weights) {
7240
7942
  })).sort((a, b) => b.mean - a.mean);
7241
7943
  }
7242
7944
  //#endregion
7243
- export { AGENT_PROFILE_KINDS, AgentEvalError, AnalystRegistry, BOOTSTRAP_GATE_MIN_N, BackendIntegrityError, BudgetBreachError, BudgetGuard, CODING_HARNESSES, ConfigError, CostAccountingIncompleteError, CostCallConflictError, CostCeilingReachedError, CostLedger, CostLedgerPersistenceError, CostReceiptCaptureError, CostReservationExceededError, CostTracker, CrossFamilyError, DECISION_PAIRED_DELTA_STATISTIC, DEFAULT_PERMUTATIONS, DEFAULT_REDACTION_RULES, DEFAULT_RED_TEAM_CORPUS, DEFAULT_TRACE_ANALYST_BUDGETS, DEFAULT_TRACE_ANALYST_KINDS, ERROR_COUNT_PATTERNS, EquivalenceProtocolError, FAILURE_CLASSES, FAILURE_MODE_KIND_SPEC, FileSystemFeedbackTrajectoryStore, FileSystemRawProviderSink, FileSystemTraceStore, FindingsStore, HARNESS_NATIVE_MODEL, HeldOutGate, InMemoryFeedbackTrajectoryStore, InMemoryRawProviderSink, InMemoryTraceStore, JudgeError, LlmCallError, LlmResponseError, MANN_WHITNEY_EXACT_MAX_STATES, MANN_WHITNEY_EXACT_MAX_WORK, MODEL_PRICING, ModelSubstitutionError, MultiLayerVerifier, NoopRawProviderSink, NotFoundError, OUTPUT_VALUE, OtlpFileTraceStore, PairwiseSteeringOptimizer, ProductClient, PromptRegistry, REDACTION_VERSION, RunIntegrityError, RunRecordValidationError, SEMANTIC_CONCEPT_JUDGE_VERSION, ServedCrossFamilyError, TRACE_ANALYST_TRUNCATION_MARKER_PREFIX, TraceEmitter, UNKNOWN_MODEL, VERIFICATION_STRATEGIES, VERIFICATION_STRATEGY_SOURCES, ValidationError, WILCOXON_EXACT_MAX_N, acquisitionPlansForKnowledgeGaps, agentProfileCellHashMaterial, agentProfileCellKey, agentProfileHash, agentProfileId, aggregateJudgeVerdicts, aggregateRunScore, analystFindingDigest, analystRunDigest, analystRunToFeedbackTrajectory, analystRunToReviewRequests, analyzeAntiSlop, analyzeRuns, analyzeSeries, analyzeTraces, argHash, assertCapabilityHeadroom, assertCrossFamily, assertCrossFamilyServed, assertNoHiddenLeak, assertProductBenchmarkRun, assertRealBackend, assertRunCaptured, assertServedModel, assertServedModels, assertSingleBackend, assignFeedbackSplit, benjaminiHochberg, blendHeldout, blockingKnowledgeEval, bonferroni, bootstrapCi, budgetBreachView, buildAgentProfileCell, buildDefaultAnalystRegistry, buildEquivalenceRecord, buildReflectionPrompt, buildTraceInsightContext, buildTraceInsightPrompt, buildTrajectory, calibrateJudge, calibrateJudgeContinuous, canonicalJson, capabilityHeadroom, captureFetchToRawSink, certificationEvidenceDigest, checkCanaries, checkServedModel, checkTraceContracts, clamp01, classifyFailure, cliffsDelta, cohensD, comparePairedArms, completionVerdict, computeExperimentStats, computeFindingId, computeToolUseMetrics, confidenceInterval, contentHash, continuousAgreement, controlRunToFeedbackTrajectory, corpusInterRaterAgreement, corpusInterRaterAgreementFromJudgeScores, costForTokenPricing, costForUsage, costReceiptFromLlm, costReceiptFromLlmError, createAntiSlopJudge, createBoundedTraceAnalysisStore, createChatClient, createDspyRlmTraceEngine, createFeedbackTrajectory, createLlmCorrectnessChecker, createLlmReviewer, createTraceAnalyst, decidePairedPromotion, defaultBlendWeights, defineAgentEval, defineEquivalenceCheck, deployGateLayer, describeTraceInsightScope, diffFindings, diffScorecard, discoverPersonas, domainEvidencePattern, dominates, eProcess, ensembleJudge, equivalenceVerdict, errorStreakDetector, estimateCost, estimateTokens, evaluateActionPolicy, evaluateInterimReleaseConfidence, evaluateOracles, evaluateReleaseConfidence, expandProfileAxes, exportProductBenchmark, exportProductBenchmarkRuns, exportRunAsOtlp, extractErrorCount, extractProducedState, extractUsage, extractUsageFromSse, failureClusterView, feedbackTrajectoriesToDatasetScenarios, feedbackTrajectoriesToOptimizerRows, feedbackTrajectoryToOptimizerRow, fileVerdictCache, formatScorecardDiff, gainHistogram, gateTreatmentApplied, gradeOnHidden, gradeSemanticStatus, groupRunsByAgentProfileCell, harnessAxisOf, hashContent, hashJson, hiddenGrade, holm, improvementVerdict, inMemoryReviewStore, inferDomainKeywords, interRaterReliability, interpretCliffs, iqr, isBinaryOutcomeVector, isJudgeSpan, isLlmSpan, isModelPriced, isRunRecord, isToolSpan, isTransientLlmError, jsonShape, jsonlReviewStore, jsonlRunRecordBackend, judgeAgreementView, judgeFamily, judgeSpans, knowledgeReadinessTracePayload, leaderboard, llmJudge, loadScorecard, localCommandRunner, makeFinding, makeProposalFinding, manifestContentDigest, mannWhitneyU, maximumChargeForLlmRequest, mcnemar, mcnemarPower, mcnemarRequiredN, minimumPairsForPairedDeltaTest, mintRolloutRows, modelHasSnapshot, modelPriceKey, mulberry32, notBlocked, objectiveEval, observeAll, otlpTextToTraceAnalysisStore, pairArms, pairRunRecords, pairedBinaryScale, pairedBootstrap, pairedCohensDz, pairedDeltaTest, pairedDeltaTieFraction, pairedEvalueSequence, pairedMde, pairedRiskDifference, pairedRiskDifferenceExact, pairedRiskDifferenceScore, pairedSignTest, pairedTTest, paretoChart, paretoFrontier, parseReflectionResponse, parseRunRecordSafe, partialCredit, partitionHeldOut, passAtK, pearsonR, preflightModels, productBenchmarkRepoIdentity, profile_exports as profile, projectRuntimeTrajectoryEvidence, proposeSynthesisTargets, ranks, readProductBenchmarkManifest, recordRuns, recordRunsToScorecard, redTeamDataset, redTeamReport, redactString, regexMatches, renderPreferenceMemoryMarkdown, repeatedActionDetector, requiredPairedSampleSize, requiredSampleSize, resolveModelPricing, resolveSeat, roundTripRunRecord, routeFields, runAgentControlLoop, runCampaign, runCanaries, runCounterfactual, runEquivalenceCheck, runEvalCampaign, runIntentMatchJudge, runKeywordCoverageJudge, runKeywordCoverageJudgeUrl, runProposeReview, runProposeReviewAsControlLoop, runSemanticConceptJudge, runTaskScore, runsForScenario, scoreKnowledgeReadiness, scoreRedTeamOutput, scoreTraceInsightReadiness, seatPresets, selfImprove, servedModelAcceptable, spearmanR, stripFencedJson, subjectiveEval, summarizeBackendIntegrity, summarizeNumberSeries, summarizePreferenceMemory, summaryTable, textInSnapshot, toAgentProfileJson, tokenizeDomainWords, toolSpansToTraceAnalysisStore, toolWasteView, traceContract, transientDispatchFailure, urlContains, userQuestionsForKnowledgeGaps, validateRunRecord, verbosityBias, verifyAgentProfileCell, verifyCompletion, viteDeployRunner, weightedComposite, weightedMean, wilcoxonSignedRank, wilson, withAssignedFeedbackSplit, withHeldoutBlend, withJudgeRetry, wranglerDeployRunner };
7945
+ export { AGENT_PROFILE_KINDS, AgentEvalError, AnalystRegistry, BOOTSTRAP_GATE_MIN_N, BackendIntegrityError, BudgetBreachError, BudgetGuard, CODING_HARNESSES, ConfigError, CostAccountingIncompleteError, CostCallConflictError, CostCeilingReachedError, CostLedger, CostLedgerPersistenceError, CostReceiptCaptureError, CostReservationExceededError, CostTracker, CrossFamilyError, DECISION_PAIRED_DELTA_STATISTIC, DEFAULT_PERMUTATIONS, DEFAULT_REDACTION_RULES, DEFAULT_RED_TEAM_CORPUS, DEFAULT_TRACE_ANALYST_BUDGETS, DEFAULT_TRACE_ANALYST_KINDS, ERROR_COUNT_PATTERNS, EquivalenceProtocolError, FAILURE_CLASSES, FAILURE_MODE_KIND_SPEC, FileSystemFeedbackTrajectoryStore, FileSystemRawProviderSink, FileSystemTraceStore, FindingsStore, HARNESS_NATIVE_MODEL, HeldOutGate, InMemoryFeedbackTrajectoryStore, InMemoryRawProviderSink, InMemoryTraceStore, JudgeError, LlmCallError, LlmResponseError, MANN_WHITNEY_EXACT_MAX_STATES, MANN_WHITNEY_EXACT_MAX_WORK, MODEL_PRICING, ModelSubstitutionError, MultiLayerVerifier, NoopRawProviderSink, NotFoundError, OUTPUT_VALUE, OtlpFileTraceStore, POLICY_EDIT_AXES, POLICY_EDIT_CANDIDATE_RECORD_SCHEMA, POLICY_EDIT_TARGET_SURFACES, PairwiseSteeringOptimizer, PolicyEditValidationError, ProductClient, PromptRegistry, REDACTION_VERSION, RunIntegrityError, RunRecordValidationError, SEMANTIC_CONCEPT_JUDGE_VERSION, ServedCrossFamilyError, TRACE_ANALYST_TRUNCATION_MARKER_PREFIX, TraceEmitter, UNKNOWN_MODEL, VERIFICATION_STRATEGIES, VERIFICATION_STRATEGY_SOURCES, ValidationError, WILCOXON_EXACT_MAX_N, acquisitionPlansForKnowledgeGaps, admitPolicyEdit, agentProfileCellHashMaterial, agentProfileCellKey, agentProfileHash, agentProfileId, aggregateJudgeVerdicts, aggregateRunScore, analystFindingDigest, analystRunDigest, analystRunToFeedbackTrajectory, analystRunToReviewRequests, analyzeAntiSlop, analyzeRuns, analyzeSeries, analyzeTraces, applyPolicyEditToSurface, argHash, assertCapabilityHeadroom, assertCrossFamily, assertCrossFamilyServed, assertNoHiddenLeak, assertNoJudgeVerdict, assertProductBenchmarkRun, assertRealBackend, assertRunCaptured, assertServedModel, assertServedModels, assertSingleBackend, assignFeedbackSplit, benjaminiHochberg, blendHeldout, blockingKnowledgeEval, bonferroni, bootstrapCi, budgetBreachView, buildAgentProfileCell, buildDefaultAnalystRegistry, buildEquivalenceRecord, buildReflectionPrompt, buildTraceInsightContext, buildTraceInsightPrompt, buildTrajectory, calibrateJudge, calibrateJudgeContinuous, canonicalJson, capabilityHeadroom, captureFetchToRawSink, certificationEvidenceDigest, checkCanaries, checkServedModel, checkTraceContracts, clamp01, classifyFailure, cliffsDelta, cohensD, comparePairedArms, completionVerdict, computeExperimentStats, computeFindingId, computePolicyEditId, computeToolUseMetrics, confidenceInterval, contentHash, continuousAgreement, controlRunToFeedbackTrajectory, corpusInterRaterAgreement, corpusInterRaterAgreementFromJudgeScores, costForTokenPricing, costForUsage, costReceiptFromLlm, costReceiptFromLlmError, createAntiSlopJudge, createBoundedTraceAnalysisStore, createChatClient, createDspyRlmTraceEngine, createFeedbackTrajectory, createLlmCorrectnessChecker, createLlmReviewer, createTraceAnalyst, decidePairedPromotion, defaultBlendWeights, defineAgentEval, defineEquivalenceCheck, deployGateLayer, describeTraceInsightScope, diffFindings, diffScorecard, discoverPersonas, domainEvidencePattern, dominates, eProcess, ensembleJudge, equivalenceVerdict, errorStreakDetector, estimateCost, estimateTokens, evaluateActionPolicy, evaluateInterimReleaseConfidence, evaluateOracles, evaluateReleaseConfidence, expandProfileAxes, exportProductBenchmark, exportProductBenchmarkRuns, exportRunAsOtlp, extractErrorCount, extractProducedState, extractUsage, extractUsageFromSse, failureClusterView, feedbackTrajectoriesToDatasetScenarios, feedbackTrajectoriesToOptimizerRows, feedbackTrajectoryToOptimizerRow, fileVerdictCache, formatScorecardDiff, gainHistogram, gateTreatmentApplied, gradeOnHidden, gradeSemanticStatus, groupRunsByAgentProfileCell, harnessAxisOf, hashContent, hashJson, hiddenGrade, holm, improvementVerdict, inMemoryReviewStore, inferDomainKeywords, interRaterReliability, interpretCliffs, iqr, isBinaryOutcomeVector, isJudgeSpan, isLlmSpan, isModelPriced, isPolicyEdit, isRunRecord, isToolSpan, isTransientLlmError, jsonShape, jsonlReviewStore, jsonlRunRecordBackend, judgeAgreementView, judgeFamily, judgeSpans, knowledgeReadinessTracePayload, leaderboard, llmJudge, loadScorecard, localCommandRunner, makeFinding, makePolicyEdit, makePolicyEditCandidateRecord, makeProposalFinding, manifestContentDigest, mannWhitneyU, maximumChargeForLlmRequest, mcnemar, mcnemarPower, mcnemarRequiredN, minimumPairsForPairedDeltaTest, mintRolloutRows, modelHasSnapshot, modelPriceKey, mulberry32, notBlocked, objectiveEval, observeAll, otlpTextToTraceAnalysisStore, pairArms, pairRunRecords, pairedBinaryScale, pairedBootstrap, pairedCohensDz, pairedDeltaTest, pairedDeltaTieFraction, pairedEvalueSequence, pairedMde, pairedRiskDifference, pairedRiskDifferenceExact, pairedRiskDifferenceScore, pairedSignTest, pairedTTest, paretoChart, paretoFrontier, parseReflectionResponse, parseRunRecordSafe, partialCredit, partitionHeldOut, passAtK, pearsonR, policyEditFromFinding, policyEditsFromFindings, preflightModels, productBenchmarkRepoIdentity, profile_exports as profile, projectRuntimeTrajectoryEvidence, proposeSynthesisTargets, ranks, readProductBenchmarkManifest, recordRuns, recordRunsToScorecard, redTeamDataset, redTeamReport, redactString, regexMatches, renderPreferenceMemoryMarkdown, repeatedActionDetector, requiredPairedSampleSize, requiredSampleSize, resolveModelPricing, resolveSeat, roundTripRunRecord, routeFields, runAgentControlLoop, runCampaign, runCanaries, runCounterfactual, runEquivalenceCheck, runEvalCampaign, runIntentMatchJudge, runKeywordCoverageJudge, runKeywordCoverageJudgeUrl, runProposeReview, runProposeReviewAsControlLoop, runSemanticConceptJudge, runTaskScore, runsForScenario, scoreKnowledgeReadiness, scorePolicyEditReadiness, scoreRedTeamOutput, scoreTraceInsightReadiness, seatPresets, selfImprove, servedModelAcceptable, spearmanR, stripFencedJson, subjectiveEval, summarizeBackendIntegrity, summarizeNumberSeries, summarizePreferenceMemory, summaryTable, textInSnapshot, toAgentProfileJson, tokenizeDomainWords, toolSpansToTraceAnalysisStore, toolWasteView, traceContract, transientDispatchFailure, urlContains, userQuestionsForKnowledgeGaps, validatePolicyEdit, validatePolicyEditCandidateRecord, validateRunRecord, verbosityBias, verifyAgentProfileCell, verifyCompletion, viteDeployRunner, weightedComposite, weightedMean, wilcoxonSignedRank, wilson, withAssignedFeedbackSplit, withHeldoutBlend, withJudgeRetry, wranglerDeployRunner };
7244
7946
 
7245
7947
  //# sourceMappingURL=index.js.map