@tangle-network/agent-eval 0.106.3 → 0.107.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,4 +1,4 @@
1
- import { S as Scenario, D as DispatchFn, b as DispatchContext } from '../types-Ctf7XIAL.js';
1
+ import { S as Scenario, D as DispatchFn, b as DispatchContext } from '../types-CNZ0tHET.js';
2
2
  import '../run-record-I-Z3JNvO.js';
3
3
  import '@tangle-network/agent-interface';
4
4
  import '../errors-oeQrLqXC.js';
@@ -1,4 +1,4 @@
1
- import { S as Scenario, J as JudgeScore, D as DispatchFn, a as JudgeConfig } from '../types-Ctf7XIAL.js';
1
+ import { S as Scenario, J as JudgeScore, D as DispatchFn, a as JudgeConfig } from '../types-CNZ0tHET.js';
2
2
  import '../run-record-I-Z3JNvO.js';
3
3
  import '@tangle-network/agent-interface';
4
4
  import '../errors-oeQrLqXC.js';
@@ -1,5 +1,5 @@
1
1
  import { TraceSpanEvent, HostedClient } from '../hosted/index.js';
2
- import '../types-Ctf7XIAL.js';
2
+ import '../types-CNZ0tHET.js';
3
3
  import '../run-record-I-Z3JNvO.js';
4
4
  import '@tangle-network/agent-interface';
5
5
  import '../errors-oeQrLqXC.js';
@@ -1,11 +1,11 @@
1
- import { S as SignedManifest, B as BackendIntegrityReport, C as CompletionRequirement, R as RuntimeEventLike, a as CompletionVerdict, P as ProducedState, b as CorrectnessChecker } from '../pre-registration-BkkTmtQG.js';
2
- export { L as LlmJudgeDimension, c as LlmJudgeOptions, l as llmJudge } from '../pre-registration-BkkTmtQG.js';
1
+ import { S as SignedManifest, B as BackendIntegrityReport, C as CompletionRequirement, R as RuntimeEventLike, a as CompletionVerdict, P as ProducedState, b as CorrectnessChecker } from '../pre-registration-x3f0VpxT.js';
2
+ export { L as LlmJudgeDimension, c as LlmJudgeOptions, l as llmJudge } from '../pre-registration-x3f0VpxT.js';
3
3
  import { A as AnalyzeTracesOptions, a as AnalyzeTracesInput, b as AnalyzeTracesResult } from '../analyst-C8HHvfJp.js';
4
- import { S as Scenario, M as MutableSurface, b as DispatchContext, a as JudgeConfig, e as GenerationRecord, g as Gate, J as JudgeScore, L as LabeledScenarioStore, s as LabeledScenarioWrite, t as LabeledScenarioSampleArgs, u as LabeledScenarioRecord, v as LabelTrust, f as SurfaceProposer, w as ProposedCandidate, x as ProposeContext, y as LabeledScenarioSource, C as CampaignResult, m as CodeSurface } from '../types-Ctf7XIAL.js';
5
- export { i as CampaignAggregates, j as CampaignArtifactWriter, k as CampaignCellResult, l as CampaignCostMeter, z as CampaignTokenUsage, d as CampaignTraceWriter, D as DispatchFn, n as GateContext, h as GateDecision, G as GateResult, o as GenerationCandidate, A as JudgeAggregate, c as JudgeDimension, p as Mutator, O as OptimizationProposer, q as OptimizerConfig, P as ParetoParent, R as RedactionStatus, B as ScenarioAggregate, r as SessionScript, T as TraceSpan, E as isProposedCandidate, F as labelTrustRank } from '../types-Ctf7XIAL.js';
6
- import { e as CampaignRunPlan, P as PlanCampaignRunOptions, C as CampaignStorage, b as RunCampaignOptions, c as RunImprovementLoopOptions } from '../gepa-DQ3ruj18.js';
7
- export { h as CampaignRunPlanCell, j as GepaProposerConstraints, G as GepaProposerOptions, O as OpenAutoPrOptions, k as OpenAutoPrResult, a as RunImprovementLoopResult, R as RunOptimizationOptions, l as RunOptimizationResult, m as countSentenceEdits, n as defaultRenderDiff, o as extractH2Sections, f as fsCampaignStorage, g as gepaProposer, i as inMemoryCampaignStorage, p as openAutoPr, q as planCampaignRun, r as runCampaign, d as runImprovementLoop, s as runOptimization, t as surfaceHash } from '../gepa-DQ3ruj18.js';
8
- export { A as AxisEvidence, a as AxisVerdict, B as BuildEvidenceVectorOptions, l as BuildLoopProvenanceArgs, D as DefaultProductionGateOptions, m as EmitLoopProvenanceArgs, n as EmitLoopProvenanceResult, E as EvidenceVector, b as EvolutionaryProposerOptions, H as HeldOutGateOptions, o as LoopProvenanceBackend, q as LoopProvenanceCandidate, L as LoopProvenanceRecord, O as ObjectiveSource, c as ParetoSignificanceGateOptions, P as PowerPreflight, s as PowerPreflightOptions, d as PromotionObjective, e as PromotionPolicy, R as RunEvalOptions, f as buildEvidenceVector, t as buildLoopProvenanceRecord, g as composeGate, h as defaultProductionGate, u as emitLoopProvenance, i as evolutionaryProposer, j as heldOutGate, v as loopProvenanceSpans, p as paretoPolicy, k as paretoSignificanceGate, w as powerPreflight, x as provenanceRecordPath, y as provenanceSpansPath, r as runEval, z as surfaceContentHash } from '../provenance-Bsyjc67Z.js';
4
+ import { S as Scenario, M as MutableSurface, b as DispatchContext, a as JudgeConfig, g as Gate, e as GenerationRecord, J as JudgeScore, L as LabeledScenarioStore, s as LabeledScenarioWrite, t as LabeledScenarioSampleArgs, u as LabeledScenarioRecord, v as LabelTrust, f as SurfaceProposer, w as ProposedCandidate, x as ProposeContext, y as LabeledScenarioSource, C as CampaignResult, m as CodeSurface } from '../types-CNZ0tHET.js';
5
+ export { i as CampaignAggregates, j as CampaignArtifactWriter, k as CampaignCellResult, l as CampaignCostMeter, z as CampaignTokenUsage, d as CampaignTraceWriter, D as DispatchFn, n as GateContext, h as GateDecision, G as GateResult, o as GenerationCandidate, A as JudgeAggregate, c as JudgeDimension, p as Mutator, O as OptimizationProposer, q as OptimizerConfig, P as ParetoParent, R as RedactionStatus, B as ScenarioAggregate, r as SessionScript, T as TraceSpan, E as isProposedCandidate, F as labelTrustRank } from '../types-CNZ0tHET.js';
6
+ import { e as CampaignRunPlan, P as PlanCampaignRunOptions, C as CampaignStorage, b as RunCampaignOptions, c as RunImprovementLoopOptions } from '../gepa-BmqTrdtg.js';
7
+ export { h as CampaignRunPlanCell, j as GepaProposerConstraints, G as GepaProposerOptions, O as OpenAutoPrOptions, k as OpenAutoPrResult, a as RunImprovementLoopResult, R as RunOptimizationOptions, l as RunOptimizationResult, m as countSentenceEdits, n as defaultRenderDiff, o as extractH2Sections, f as fsCampaignStorage, g as gepaProposer, i as inMemoryCampaignStorage, p as openAutoPr, q as planCampaignRun, r as runCampaign, d as runImprovementLoop, s as runOptimization, t as surfaceHash } from '../gepa-BmqTrdtg.js';
8
+ export { A as AxisEvidence, a as AxisVerdict, B as BuildEvidenceVectorOptions, l as BuildLoopProvenanceArgs, D as DefaultProductionGateOptions, m as EmitLoopProvenanceArgs, n as EmitLoopProvenanceResult, E as EvidenceVector, b as EvolutionaryProposerOptions, H as HeldOutGateOptions, o as LoopProvenanceBackend, q as LoopProvenanceCandidate, L as LoopProvenanceRecord, O as ObjectiveSource, c as ParetoSignificanceGateOptions, P as PowerPreflight, s as PowerPreflightOptions, d as PromotionObjective, e as PromotionPolicy, R as RunEvalOptions, f as buildEvidenceVector, t as buildLoopProvenanceRecord, g as composeGate, h as defaultProductionGate, u as emitLoopProvenance, i as evolutionaryProposer, j as heldOutGate, v as loopProvenanceSpans, p as paretoPolicy, k as paretoSignificanceGate, w as powerPreflight, x as provenanceRecordPath, y as provenanceSpansPath, r as runEval, z as surfaceContentHash } from '../provenance-CIjRcI6m.js';
9
9
  import { E as EProcessState, a as PairedBootstrapResult } from '../statistics-D88peojY.js';
10
10
  import { L as LlmClientOptions } from '../llm-client-DyqEH4jH.js';
11
11
  import { AgentProfile } from '@tangle-network/agent-interface';
@@ -435,6 +435,43 @@ declare function loadEvalFixtureScenarios(evalsDir: string, options?: LoadEvalFi
435
435
  */
436
436
  declare function planEvalFixtureRun<TArtifact = unknown>(options: PlanEvalFixtureRunOptions<TArtifact>): EvalFixtureRunPlan;
437
437
 
438
+ /**
439
+ * @module
440
+ * Composable placebo / neutralization promotion gate.
441
+ *
442
+ * A held-out gate proves a candidate beat baseline. It CANNOT prove the lift came
443
+ * from the candidate's CONTENT rather than from the prompt/mount FOOTPRINT the
444
+ * content happened to add (more bytes, a longer prompt). This gate closes that
445
+ * hole: it compares the candidate's held-out lift against the lift of a
446
+ * FOOTPRINT-MATCHED neutralized variant (same layout + length, zero content, via
447
+ * `neutralizeText`). If the neutralized variant reproduces more than
448
+ * `maxDecorativeFraction` of the candidate's lift, the lift is decorative — it
449
+ * survives blanking the content — and the candidate is HELD regardless of how
450
+ * large or significant its raw lift is.
451
+ *
452
+ * Compose it AFTER the significance gate — significance says the lift is real,
453
+ * this says the lift is CAUSED BY THE CONTENT:
454
+ * composeGate(heldOutGate({ ... }), neutralizationGate({ ... }))
455
+ *
456
+ * Requires `ctx.neutralizedJudgeScores`, populated by `runImprovementLoop` when it
457
+ * is given a `neutralize` function. A gate composed without that wiring fails
458
+ * loud rather than silently passing an unproven candidate.
459
+ */
460
+
461
+ interface NeutralizationGateOptions<TScenario extends Scenario = Scenario> {
462
+ scenarios: TScenario[];
463
+ /** Reject when the neutralized (content-blanked, footprint-matched) variant
464
+ * reproduces at least this fraction of the candidate's held-out lift. Default
465
+ * 0.5 — if blanking the content keeps half the lift, the content is decorative.
466
+ * Equality rejects: a neutralized lift == threshold·candidateLift is decorative. */
467
+ maxDecorativeFraction?: number;
468
+ }
469
+ /**
470
+ * Composable placebo gate: ships only when the candidate's held-out lift is NOT
471
+ * mostly reproduced by a footprint-matched neutralized variant.
472
+ */
473
+ declare function neutralizationGate<TArtifact, TScenario extends Scenario>(options: NeutralizationGateOptions<TScenario>): Gate<TArtifact, TScenario>;
474
+
438
475
  /**
439
476
  * Anytime-valid sequential promotion gate — an e-process (betting
440
477
  * test-martingale, see `eProcess` in `statistics.ts`) over paired
@@ -751,6 +788,34 @@ declare class FsLabeledScenarioStore implements LabeledScenarioStore {
751
788
  private pathForSource;
752
789
  }
753
790
 
791
+ /**
792
+ * @module
793
+ * Footprint-matched neutralization — the placebo control for content-vs-footprint
794
+ * attribution in a promotion gate.
795
+ *
796
+ * A promoted surface can raise a held-out score two different ways:
797
+ * 1. its CONTENT is informative (the thing we want to promote), or
798
+ * 2. it merely added prompt/mount FOOTPRINT — more bytes, more lines, a longer
799
+ * more authoritative-looking prompt — that the model spends attention on
800
+ * regardless of what the bytes say.
801
+ *
802
+ * A held-out gate proves the candidate beat baseline; it cannot separate (1) from
803
+ * (2). `neutralizeText` produces a variant that keeps the input's layout and
804
+ * length while carrying ZERO information, so scoring it isolates the footprint
805
+ * contribution (2). Feed the neutralized variant's scores to `neutralizationGate`:
806
+ * any lift it still holds over baseline is decorative, and a candidate whose lift
807
+ * survives neutralization is rejected however large its raw lift.
808
+ */
809
+ /**
810
+ * Blank every non-whitespace character to a 1-byte filler while preserving all
811
+ * whitespace. Line count, indentation, and word/line lengths are unchanged — so
812
+ * the neutralized variant has the same layout and (for ASCII) the same byte
813
+ * footprint as the input, but no readable content. Whitespace is preserved
814
+ * deliberately: collapsing it would change the token structure and stop the
815
+ * variant from being a true footprint match.
816
+ */
817
+ declare function neutralizeText(content: string): string;
818
+
754
819
  /**
755
820
  * FAPO (Fully Autonomous Prompt Optimization) is an orchestration policy, not
756
821
  * a new prompt mutation primitive. The paper's loop evaluates an inspectable
@@ -2007,4 +2072,4 @@ declare function gitWorktreeAdapter(opts: GitWorktreeAdapterOptions): WorktreeAd
2007
2072
  * as a ref under the adapter's worktree dir. */
2008
2073
  declare function resolveWorktreePath(surface: CodeSurface, worktreeDir?: string): string;
2009
2074
 
2010
- export { type AcceptedEdit, type AceProposerOptions, type AnalystArtifact, type AnalystScenario, type ApplySkillPatchResult, type BuildAnalystSurfaceDispatchOptions, type CampaignBreakdown, CampaignResult, CampaignRunPlan, CampaignStorage, CodeSurface, type CompareProposersOptions, type DimensionRegression, type DiscriminationScore, DispatchContext, type EvalFixture, type EvalFixtureFile, type EvalFixtureLoadOptions, type EvalFixtureRunPlan, type EvalFixtureScenario, type EvalFixtureValidationMode, type FailureModeRecallJudgeOptions, type FapoAttributionSignals, type FapoEntryConfig, type FapoFailureCluster, type FapoOptimizationLevel, type FapoProposerOptions, type FapoReviewInput, type FapoReviewIssue, type FapoReviewResult, type FapoScopeContract, FsLabeledScenarioStore, type FsLabeledScenarioStoreOptions, Gate, GenerationRecord, type GitWorktreeAdapterOptions, type Governor, type GovernorContext, type GovernorOp, type HaloProposerOptions, type HeldoutSignificance, type HeldoutSignificanceOptions, type HeuristicGovernorOptions, type JsonPrimitive, type JsonValue, JudgeConfig, JudgeScore, LabelTrust, LabeledScenarioRecord, LabeledScenarioSampleArgs, LabeledScenarioSource, LabeledScenarioStore, LabeledScenarioStoreError, LabeledScenarioWrite, Lineage, type LineageEdge, type LineageGraph, type LineageNode, type LineageNodeInput, type LineageStore, type LoadEvalFixtureScenariosOptions, type MemoryCurationProposerOptions, MutableSurface, type OptimizerEntryConfig, type PairedHoldout, type ParameterCandidate, type ParameterChange, type ParameterSweepProposerOptions, PlanCampaignRunOptions, type PlanEvalFixtureRunOptions, type PlaybackContext, type PlaybackDriver, type PlaybackStep, type PolicyEditProposerOptions, type ProfileDispatchFn, ProfileMatrixError, type ProfileSummary, ProposeContext, type ProposePatchesArgs, ProposedCandidate, type ProposerComparison, type ProposerEntry, type ProposerPairwise, type ProposerScore, type RejectedEdit, RunCampaignOptions, RunImprovementLoopOptions, type RunLineageLoopOptions, type RunLineageLoopResult, type RunLineageLoopSeed, type RunLineageOptions, type RunLineageResult, type RunLineageSeed, type RunLineageStepResult, type RunProfileMatrixOptions, type RunProfileMatrixResult, type RunSkillOptOptions, type RunSkillOptResult, Scenario, type ScenarioRollup, type ScenarioSignal, type ScoreboardRenderOptions, type ScoreboardRow, type ScoreboardSummary, type SequentialDecideFn, type SequentialDecideOptions, type SequentialDecision, type SequentialObservation, type SequentialPairedGate, type SequentialPairedGateOptions, type SkillOptEpochRecord, type SkillOptEvidence, type SkillOptProposer, type SkillOptProposerOptions, type SkillPatch, type SkillPatchOp, SkillPatchParseError, type SkillPatchRejection, SurfaceProposer, type SurfaceScore, type TraceAnalystProposerOptions, type UserStory, type UserStoryVerdict, type Worktree, type WorktreeAdapter, WorktreeAdapterError, aceProposer, applySkillPatch, buildAnalystSurfaceDispatch, callbackGovernor, campaignBreakdown, campaignMeanComposite, compareProposers, detectScale, dimensionRegressions, discoverEvalFixtures, extractFapoAttributionSignals, failureModeRecallJudge, fapoEscalationEntry, fapoProposer, fsLineageStore, gepaParetoEntry, gepaReflectionEntry, gitWorktreeAdapter, haloProposer, heldoutSignificance, heuristicGovernor, lineageNodeId, loadEvalFixture, loadEvalFixtureScenarios, makePlaybackDispatch, memLineageStore, memoryCurationProposer, pairHoldout, parameterSweepProposer, parseSkillPatchResponse, patchEditCount, planEvalFixtureRun, policyEditProposer, renderScoreboardMarkdown, resolveRunDir, resolveWorktreePath, runLineage, runLineageLoop, runProfileMatrix, runSkillOpt, scoreDiscrimination, scoreUserStory, scoreboardSummary, selectDiscriminative, sequentialDecide, sequentialPairedGate, skillOptEntry, skillOptProposer, tangleTracesRoot, traceAnalystProposer, userStoryScoreboard };
2075
+ export { type AcceptedEdit, type AceProposerOptions, type AnalystArtifact, type AnalystScenario, type ApplySkillPatchResult, type BuildAnalystSurfaceDispatchOptions, type CampaignBreakdown, CampaignResult, CampaignRunPlan, CampaignStorage, CodeSurface, type CompareProposersOptions, type DimensionRegression, type DiscriminationScore, DispatchContext, type EvalFixture, type EvalFixtureFile, type EvalFixtureLoadOptions, type EvalFixtureRunPlan, type EvalFixtureScenario, type EvalFixtureValidationMode, type FailureModeRecallJudgeOptions, type FapoAttributionSignals, type FapoEntryConfig, type FapoFailureCluster, type FapoOptimizationLevel, type FapoProposerOptions, type FapoReviewInput, type FapoReviewIssue, type FapoReviewResult, type FapoScopeContract, FsLabeledScenarioStore, type FsLabeledScenarioStoreOptions, Gate, GenerationRecord, type GitWorktreeAdapterOptions, type Governor, type GovernorContext, type GovernorOp, type HaloProposerOptions, type HeldoutSignificance, type HeldoutSignificanceOptions, type HeuristicGovernorOptions, type JsonPrimitive, type JsonValue, JudgeConfig, JudgeScore, LabelTrust, LabeledScenarioRecord, LabeledScenarioSampleArgs, LabeledScenarioSource, LabeledScenarioStore, LabeledScenarioStoreError, LabeledScenarioWrite, Lineage, type LineageEdge, type LineageGraph, type LineageNode, type LineageNodeInput, type LineageStore, type LoadEvalFixtureScenariosOptions, type MemoryCurationProposerOptions, MutableSurface, type NeutralizationGateOptions, type OptimizerEntryConfig, type PairedHoldout, type ParameterCandidate, type ParameterChange, type ParameterSweepProposerOptions, PlanCampaignRunOptions, type PlanEvalFixtureRunOptions, type PlaybackContext, type PlaybackDriver, type PlaybackStep, type PolicyEditProposerOptions, type ProfileDispatchFn, ProfileMatrixError, type ProfileSummary, ProposeContext, type ProposePatchesArgs, ProposedCandidate, type ProposerComparison, type ProposerEntry, type ProposerPairwise, type ProposerScore, type RejectedEdit, RunCampaignOptions, RunImprovementLoopOptions, type RunLineageLoopOptions, type RunLineageLoopResult, type RunLineageLoopSeed, type RunLineageOptions, type RunLineageResult, type RunLineageSeed, type RunLineageStepResult, type RunProfileMatrixOptions, type RunProfileMatrixResult, type RunSkillOptOptions, type RunSkillOptResult, Scenario, type ScenarioRollup, type ScenarioSignal, type ScoreboardRenderOptions, type ScoreboardRow, type ScoreboardSummary, type SequentialDecideFn, type SequentialDecideOptions, type SequentialDecision, type SequentialObservation, type SequentialPairedGate, type SequentialPairedGateOptions, type SkillOptEpochRecord, type SkillOptEvidence, type SkillOptProposer, type SkillOptProposerOptions, type SkillPatch, type SkillPatchOp, SkillPatchParseError, type SkillPatchRejection, SurfaceProposer, type SurfaceScore, type TraceAnalystProposerOptions, type UserStory, type UserStoryVerdict, type Worktree, type WorktreeAdapter, WorktreeAdapterError, aceProposer, applySkillPatch, buildAnalystSurfaceDispatch, callbackGovernor, campaignBreakdown, campaignMeanComposite, compareProposers, detectScale, dimensionRegressions, discoverEvalFixtures, extractFapoAttributionSignals, failureModeRecallJudge, fapoEscalationEntry, fapoProposer, fsLineageStore, gepaParetoEntry, gepaReflectionEntry, gitWorktreeAdapter, haloProposer, heldoutSignificance, heuristicGovernor, lineageNodeId, loadEvalFixture, loadEvalFixtureScenarios, makePlaybackDispatch, memLineageStore, memoryCurationProposer, neutralizationGate, neutralizeText, pairHoldout, parameterSweepProposer, parseSkillPatchResponse, patchEditCount, planEvalFixtureRun, policyEditProposer, renderScoreboardMarkdown, resolveRunDir, resolveWorktreePath, runLineage, runLineageLoop, runProfileMatrix, runSkillOpt, scoreDiscrimination, scoreUserStory, scoreboardSummary, selectDiscriminative, sequentialDecide, sequentialPairedGate, skillOptEntry, skillOptProposer, tangleTracesRoot, traceAnalystProposer, userStoryScoreboard };
@@ -7,7 +7,7 @@ import {
7
7
  paretoSignificanceGate,
8
8
  powerPreflight,
9
9
  runEval
10
- } from "../chunk-J22CKVXN.js";
10
+ } from "../chunk-V75NN2ZR.js";
11
11
  import {
12
12
  HARNESS_NATIVE_MODEL,
13
13
  agentProfileHash,
@@ -17,7 +17,7 @@ import {
17
17
  harnessAxisOf,
18
18
  llmJudge,
19
19
  verifyCompletion
20
- } from "../chunk-2HLM4SJJ.js";
20
+ } from "../chunk-6PF6LXRY.js";
21
21
  import {
22
22
  buildLoopProvenanceRecord,
23
23
  campaignBreakdown,
@@ -43,7 +43,7 @@ import {
43
43
  runOptimization,
44
44
  surfaceContentHash,
45
45
  surfaceHash
46
- } from "../chunk-3E5KUXYZ.js";
46
+ } from "../chunk-G4DLZAV5.js";
47
47
  import {
48
48
  assertRealBackend,
49
49
  contentHash,
@@ -737,6 +737,86 @@ function assertInside(root, target, label) {
737
737
  throw new Error(`loadEvalFixture: fixture path escapes evalsDir: ${label}`);
738
738
  }
739
739
 
740
+ // src/campaign/gates/neutralization-gate.ts
741
+ function mean(xs) {
742
+ return xs.length === 0 ? 0 : xs.reduce((a, b) => a + b, 0) / xs.length;
743
+ }
744
+ function pairedLift(arm, baseline, scenarioIds) {
745
+ const paired = pairHoldout(arm, baseline, scenarioIds, (s) => s.composite);
746
+ const deltas = paired.after.map((a, i) => a - (paired.before[i] ?? 0));
747
+ return { lift: mean(deltas), n: deltas.length };
748
+ }
749
+ function neutralizationGate(options) {
750
+ const maxDecorativeFraction = options.maxDecorativeFraction ?? 0.5;
751
+ return {
752
+ name: "neutralizationGate",
753
+ async decide(ctx) {
754
+ if (!ctx.baselineJudgeScores) {
755
+ throw new Error(
756
+ "neutralizationGate: ctx.baselineJudgeScores is required \u2014 the placebo control measures lift OVER baseline."
757
+ );
758
+ }
759
+ if (!ctx.neutralizedJudgeScores) {
760
+ throw new Error(
761
+ "neutralizationGate: ctx.neutralizedJudgeScores is required. It is populated by runImprovementLoop only when a `neutralize` function is supplied \u2014 composing this gate without that wiring would pass an unproven candidate."
762
+ );
763
+ }
764
+ const scenarioIds = new Set(options.scenarios.map((s) => s.id));
765
+ const cand = pairedLift(
766
+ ctx.judgeScores,
767
+ ctx.baselineJudgeScores,
768
+ scenarioIds
769
+ );
770
+ const neut = pairedLift(
771
+ ctx.neutralizedJudgeScores,
772
+ ctx.baselineJudgeScores,
773
+ scenarioIds
774
+ );
775
+ if (cand.lift <= 0) {
776
+ return {
777
+ decision: "hold",
778
+ reasons: [
779
+ `neutralization: candidate held-out lift ${cand.lift.toFixed(3)} \u2264 0 \u2014 no positive lift to attribute to content`
780
+ ],
781
+ contributingGates: [
782
+ {
783
+ name: "neutralizationGate",
784
+ passed: false,
785
+ detail: { candidateLift: cand.lift, neutralizedLift: neut.lift, n: cand.n }
786
+ }
787
+ ],
788
+ delta: cand.lift
789
+ };
790
+ }
791
+ const decorativeFraction = neut.lift / cand.lift;
792
+ const passed = decorativeFraction < maxDecorativeFraction;
793
+ const pct = (decorativeFraction * 100).toFixed(0);
794
+ return {
795
+ decision: passed ? "ship" : "hold",
796
+ reasons: passed ? [
797
+ `neutralization: content is causal \u2014 blanked variant reproduces ${pct}% of the lift (< ${(maxDecorativeFraction * 100).toFixed(0)}%); candidate \u0394 ${cand.lift.toFixed(3)}, neutralized \u0394 ${neut.lift.toFixed(3)}`
798
+ ] : [
799
+ `neutralization: lift is DECORATIVE \u2014 blanking the content (footprint-matched) reproduces ${pct}% of the lift (\u2265 ${(maxDecorativeFraction * 100).toFixed(0)}%); candidate \u0394 ${cand.lift.toFixed(3)}, neutralized \u0394 ${neut.lift.toFixed(3)}`
800
+ ],
801
+ contributingGates: [
802
+ {
803
+ name: "neutralizationGate",
804
+ passed,
805
+ detail: {
806
+ candidateLift: cand.lift,
807
+ neutralizedLift: neut.lift,
808
+ decorativeFraction,
809
+ maxDecorativeFraction,
810
+ n: cand.n
811
+ }
812
+ }
813
+ ],
814
+ delta: cand.lift
815
+ };
816
+ }
817
+ };
818
+ }
819
+
740
820
  // src/campaign/gates/sequential.ts
741
821
  import { createHash as createHash3 } from "crypto";
742
822
  function verifyManifestSync(m) {
@@ -1214,6 +1294,12 @@ function appendLine(path, line) {
1214
1294
  }
1215
1295
  }
1216
1296
 
1297
+ // src/campaign/neutralize.ts
1298
+ var FILLER = "#";
1299
+ function neutralizeText(content) {
1300
+ return content.replace(/\S/g, FILLER);
1301
+ }
1302
+
1217
1303
  // src/campaign/proposers/fapo.ts
1218
1304
  var FAPO_LEVELS = ["prompt", "parameter", "structural"];
1219
1305
  var MAX_FINDING_DEPTH = 16;
@@ -2155,8 +2241,8 @@ async function compareProposerEntries(opts) {
2155
2241
  });
2156
2242
  const score = {
2157
2243
  name: w.name,
2158
- baselineComposite: mean(baselineArr),
2159
- winnerComposite: mean(w.arr),
2244
+ baselineComposite: mean2(baselineArr),
2245
+ winnerComposite: mean2(w.arr),
2160
2246
  lift: boot.mean,
2161
2247
  liftCi: { low: boot.low, high: boot.high },
2162
2248
  costUsd: w.costUsd,
@@ -2193,7 +2279,7 @@ async function compareProposerEntries(opts) {
2193
2279
  });
2194
2280
  return { scores, best, pairwise, holdoutScenarioIds: scenarioIds };
2195
2281
  }
2196
- function mean(xs) {
2282
+ function mean2(xs) {
2197
2283
  return xs.length === 0 ? 0 : xs.reduce((a, b) => a + b, 0) / xs.length;
2198
2284
  }
2199
2285
  function slug(name) {
@@ -2614,12 +2700,12 @@ function sanitize(id) {
2614
2700
  function sha(input) {
2615
2701
  return createHash5("sha256").update(JSON.stringify(input)).digest("hex");
2616
2702
  }
2617
- function mean2(xs) {
2703
+ function mean3(xs) {
2618
2704
  return xs.length === 0 ? 0 : xs.reduce((a, b) => a + b, 0) / xs.length;
2619
2705
  }
2620
2706
  function cellComposite(cell) {
2621
2707
  const composites = Object.values(cell.judgeScores).map((s) => s.composite);
2622
- return composites.length === 0 ? 0 : mean2(composites);
2708
+ return composites.length === 0 ? 0 : mean3(composites);
2623
2709
  }
2624
2710
  function requireResolvedModel(cell, profileId) {
2625
2711
  const resolved = cell.resolvedModel?.trim();
@@ -2655,7 +2741,7 @@ function buildRunRecord(args) {
2655
2741
  if (js.notes) notes.push(`${judgeName}: ${js.notes}`);
2656
2742
  }
2657
2743
  const perDimMean = {};
2658
- for (const [dim, values] of Object.entries(dimAccum)) perDimMean[dim] = mean2(values);
2744
+ for (const [dim, values] of Object.entries(dimAccum)) perDimMean[dim] = mean3(values);
2659
2745
  let costUsd = cell.costUsd;
2660
2746
  let costEstimated = false;
2661
2747
  if (costUsd === 0 && cell.tokenUsage.output > 0 && isModelPriced(model)) {
@@ -2819,7 +2905,7 @@ async function runProfileMatrix(opts) {
2819
2905
  // one harness, so the first record's model is representative).
2820
2906
  model: declaredModel === HARNESS_NATIVE_MODEL ? profileRecords[0]?.model ?? declaredModel : declaredModel,
2821
2907
  records: profileRecords.length,
2822
- meanComposite: mean2(profileRecords.map(compositeOf)),
2908
+ meanComposite: mean3(profileRecords.map(compositeOf)),
2823
2909
  totalCostUsd: pricedTotalCostUsd,
2824
2910
  integrity: summarizeBackendIntegrity(profileRecords)
2825
2911
  };
@@ -2849,7 +2935,7 @@ function rollup(records, keyOf) {
2849
2935
  groups.set(key, arr);
2850
2936
  }
2851
2937
  const out = {};
2852
- for (const [key, xs] of groups) out[key] = { meanComposite: mean2(xs), n: xs.length };
2938
+ for (const [key, xs] of groups) out[key] = { meanComposite: mean3(xs), n: xs.length };
2853
2939
  return out;
2854
2940
  }
2855
2941
  function rollupByPersona(records, scenarios, personaOf) {
@@ -3409,6 +3495,8 @@ export {
3409
3495
  makePlaybackDispatch,
3410
3496
  memLineageStore,
3411
3497
  memoryCurationProposer,
3498
+ neutralizationGate,
3499
+ neutralizeText,
3412
3500
  openAutoPr,
3413
3501
  pairHoldout,
3414
3502
  parameterSweepProposer,