@tangle-network/agent-eval 0.106.3 → 0.107.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/adapters/http.d.ts +1 -1
- package/dist/adapters/langchain.d.ts +1 -1
- package/dist/adapters/otel.d.ts +1 -1
- package/dist/campaign/index.d.ts +73 -8
- package/dist/campaign/index.js +99 -11
- package/dist/campaign/index.js.map +1 -1
- package/dist/{chunk-2HLM4SJJ.js → chunk-6PF6LXRY.js} +2 -2
- package/dist/{chunk-3E5KUXYZ.js → chunk-G4DLZAV5.js} +21 -1
- package/dist/chunk-G4DLZAV5.js.map +1 -0
- package/dist/{chunk-J22CKVXN.js → chunk-V75NN2ZR.js} +2 -2
- package/dist/contract/index.d.ts +6 -6
- package/dist/contract/index.js +2 -2
- package/dist/{gepa-DQ3ruj18.d.ts → gepa-BmqTrdtg.d.ts} +9 -1
- package/dist/hosted/index.d.ts +1 -1
- package/dist/index.d.ts +4 -4
- package/dist/index.js +2 -2
- package/dist/multishot/index.d.ts +1 -1
- package/dist/openapi.json +1 -1
- package/dist/{pre-registration-BkkTmtQG.d.ts → pre-registration-x3f0VpxT.d.ts} +1 -1
- package/dist/{provenance-Bsyjc67Z.d.ts → provenance-CIjRcI6m.d.ts} +2 -2
- package/dist/rl.d.ts +1 -1
- package/dist/{types-Ctf7XIAL.d.ts → types-CNZ0tHET.d.ts} +10 -0
- package/package.json +1 -1
- package/dist/chunk-3E5KUXYZ.js.map +0 -1
- /package/dist/{chunk-2HLM4SJJ.js.map → chunk-6PF6LXRY.js.map} +0 -0
- /package/dist/{chunk-J22CKVXN.js.map → chunk-V75NN2ZR.js.map} +0 -0
package/dist/adapters/http.d.ts
CHANGED
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import { S as Scenario, D as DispatchFn, b as DispatchContext } from '../types-
|
|
1
|
+
import { S as Scenario, D as DispatchFn, b as DispatchContext } from '../types-CNZ0tHET.js';
|
|
2
2
|
import '../run-record-I-Z3JNvO.js';
|
|
3
3
|
import '@tangle-network/agent-interface';
|
|
4
4
|
import '../errors-oeQrLqXC.js';
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import { S as Scenario, J as JudgeScore, D as DispatchFn, a as JudgeConfig } from '../types-
|
|
1
|
+
import { S as Scenario, J as JudgeScore, D as DispatchFn, a as JudgeConfig } from '../types-CNZ0tHET.js';
|
|
2
2
|
import '../run-record-I-Z3JNvO.js';
|
|
3
3
|
import '@tangle-network/agent-interface';
|
|
4
4
|
import '../errors-oeQrLqXC.js';
|
package/dist/adapters/otel.d.ts
CHANGED
package/dist/campaign/index.d.ts
CHANGED
|
@@ -1,11 +1,11 @@
|
|
|
1
|
-
import { S as SignedManifest, B as BackendIntegrityReport, C as CompletionRequirement, R as RuntimeEventLike, a as CompletionVerdict, P as ProducedState, b as CorrectnessChecker } from '../pre-registration-
|
|
2
|
-
export { L as LlmJudgeDimension, c as LlmJudgeOptions, l as llmJudge } from '../pre-registration-
|
|
1
|
+
import { S as SignedManifest, B as BackendIntegrityReport, C as CompletionRequirement, R as RuntimeEventLike, a as CompletionVerdict, P as ProducedState, b as CorrectnessChecker } from '../pre-registration-x3f0VpxT.js';
|
|
2
|
+
export { L as LlmJudgeDimension, c as LlmJudgeOptions, l as llmJudge } from '../pre-registration-x3f0VpxT.js';
|
|
3
3
|
import { A as AnalyzeTracesOptions, a as AnalyzeTracesInput, b as AnalyzeTracesResult } from '../analyst-C8HHvfJp.js';
|
|
4
|
-
import { S as Scenario, M as MutableSurface, b as DispatchContext, a as JudgeConfig,
|
|
5
|
-
export { i as CampaignAggregates, j as CampaignArtifactWriter, k as CampaignCellResult, l as CampaignCostMeter, z as CampaignTokenUsage, d as CampaignTraceWriter, D as DispatchFn, n as GateContext, h as GateDecision, G as GateResult, o as GenerationCandidate, A as JudgeAggregate, c as JudgeDimension, p as Mutator, O as OptimizationProposer, q as OptimizerConfig, P as ParetoParent, R as RedactionStatus, B as ScenarioAggregate, r as SessionScript, T as TraceSpan, E as isProposedCandidate, F as labelTrustRank } from '../types-
|
|
6
|
-
import { e as CampaignRunPlan, P as PlanCampaignRunOptions, C as CampaignStorage, b as RunCampaignOptions, c as RunImprovementLoopOptions } from '../gepa-
|
|
7
|
-
export { h as CampaignRunPlanCell, j as GepaProposerConstraints, G as GepaProposerOptions, O as OpenAutoPrOptions, k as OpenAutoPrResult, a as RunImprovementLoopResult, R as RunOptimizationOptions, l as RunOptimizationResult, m as countSentenceEdits, n as defaultRenderDiff, o as extractH2Sections, f as fsCampaignStorage, g as gepaProposer, i as inMemoryCampaignStorage, p as openAutoPr, q as planCampaignRun, r as runCampaign, d as runImprovementLoop, s as runOptimization, t as surfaceHash } from '../gepa-
|
|
8
|
-
export { A as AxisEvidence, a as AxisVerdict, B as BuildEvidenceVectorOptions, l as BuildLoopProvenanceArgs, D as DefaultProductionGateOptions, m as EmitLoopProvenanceArgs, n as EmitLoopProvenanceResult, E as EvidenceVector, b as EvolutionaryProposerOptions, H as HeldOutGateOptions, o as LoopProvenanceBackend, q as LoopProvenanceCandidate, L as LoopProvenanceRecord, O as ObjectiveSource, c as ParetoSignificanceGateOptions, P as PowerPreflight, s as PowerPreflightOptions, d as PromotionObjective, e as PromotionPolicy, R as RunEvalOptions, f as buildEvidenceVector, t as buildLoopProvenanceRecord, g as composeGate, h as defaultProductionGate, u as emitLoopProvenance, i as evolutionaryProposer, j as heldOutGate, v as loopProvenanceSpans, p as paretoPolicy, k as paretoSignificanceGate, w as powerPreflight, x as provenanceRecordPath, y as provenanceSpansPath, r as runEval, z as surfaceContentHash } from '../provenance-
|
|
4
|
+
import { S as Scenario, M as MutableSurface, b as DispatchContext, a as JudgeConfig, g as Gate, e as GenerationRecord, J as JudgeScore, L as LabeledScenarioStore, s as LabeledScenarioWrite, t as LabeledScenarioSampleArgs, u as LabeledScenarioRecord, v as LabelTrust, f as SurfaceProposer, w as ProposedCandidate, x as ProposeContext, y as LabeledScenarioSource, C as CampaignResult, m as CodeSurface } from '../types-CNZ0tHET.js';
|
|
5
|
+
export { i as CampaignAggregates, j as CampaignArtifactWriter, k as CampaignCellResult, l as CampaignCostMeter, z as CampaignTokenUsage, d as CampaignTraceWriter, D as DispatchFn, n as GateContext, h as GateDecision, G as GateResult, o as GenerationCandidate, A as JudgeAggregate, c as JudgeDimension, p as Mutator, O as OptimizationProposer, q as OptimizerConfig, P as ParetoParent, R as RedactionStatus, B as ScenarioAggregate, r as SessionScript, T as TraceSpan, E as isProposedCandidate, F as labelTrustRank } from '../types-CNZ0tHET.js';
|
|
6
|
+
import { e as CampaignRunPlan, P as PlanCampaignRunOptions, C as CampaignStorage, b as RunCampaignOptions, c as RunImprovementLoopOptions } from '../gepa-BmqTrdtg.js';
|
|
7
|
+
export { h as CampaignRunPlanCell, j as GepaProposerConstraints, G as GepaProposerOptions, O as OpenAutoPrOptions, k as OpenAutoPrResult, a as RunImprovementLoopResult, R as RunOptimizationOptions, l as RunOptimizationResult, m as countSentenceEdits, n as defaultRenderDiff, o as extractH2Sections, f as fsCampaignStorage, g as gepaProposer, i as inMemoryCampaignStorage, p as openAutoPr, q as planCampaignRun, r as runCampaign, d as runImprovementLoop, s as runOptimization, t as surfaceHash } from '../gepa-BmqTrdtg.js';
|
|
8
|
+
export { A as AxisEvidence, a as AxisVerdict, B as BuildEvidenceVectorOptions, l as BuildLoopProvenanceArgs, D as DefaultProductionGateOptions, m as EmitLoopProvenanceArgs, n as EmitLoopProvenanceResult, E as EvidenceVector, b as EvolutionaryProposerOptions, H as HeldOutGateOptions, o as LoopProvenanceBackend, q as LoopProvenanceCandidate, L as LoopProvenanceRecord, O as ObjectiveSource, c as ParetoSignificanceGateOptions, P as PowerPreflight, s as PowerPreflightOptions, d as PromotionObjective, e as PromotionPolicy, R as RunEvalOptions, f as buildEvidenceVector, t as buildLoopProvenanceRecord, g as composeGate, h as defaultProductionGate, u as emitLoopProvenance, i as evolutionaryProposer, j as heldOutGate, v as loopProvenanceSpans, p as paretoPolicy, k as paretoSignificanceGate, w as powerPreflight, x as provenanceRecordPath, y as provenanceSpansPath, r as runEval, z as surfaceContentHash } from '../provenance-CIjRcI6m.js';
|
|
9
9
|
import { E as EProcessState, a as PairedBootstrapResult } from '../statistics-D88peojY.js';
|
|
10
10
|
import { L as LlmClientOptions } from '../llm-client-DyqEH4jH.js';
|
|
11
11
|
import { AgentProfile } from '@tangle-network/agent-interface';
|
|
@@ -435,6 +435,43 @@ declare function loadEvalFixtureScenarios(evalsDir: string, options?: LoadEvalFi
|
|
|
435
435
|
*/
|
|
436
436
|
declare function planEvalFixtureRun<TArtifact = unknown>(options: PlanEvalFixtureRunOptions<TArtifact>): EvalFixtureRunPlan;
|
|
437
437
|
|
|
438
|
+
/**
|
|
439
|
+
* @module
|
|
440
|
+
* Composable placebo / neutralization promotion gate.
|
|
441
|
+
*
|
|
442
|
+
* A held-out gate proves a candidate beat baseline. It CANNOT prove the lift came
|
|
443
|
+
* from the candidate's CONTENT rather than from the prompt/mount FOOTPRINT the
|
|
444
|
+
* content happened to add (more bytes, a longer prompt). This gate closes that
|
|
445
|
+
* hole: it compares the candidate's held-out lift against the lift of a
|
|
446
|
+
* FOOTPRINT-MATCHED neutralized variant (same layout + length, zero content, via
|
|
447
|
+
* `neutralizeText`). If the neutralized variant reproduces more than
|
|
448
|
+
* `maxDecorativeFraction` of the candidate's lift, the lift is decorative — it
|
|
449
|
+
* survives blanking the content — and the candidate is HELD regardless of how
|
|
450
|
+
* large or significant its raw lift is.
|
|
451
|
+
*
|
|
452
|
+
* Compose it AFTER the significance gate — significance says the lift is real,
|
|
453
|
+
* this says the lift is CAUSED BY THE CONTENT:
|
|
454
|
+
* composeGate(heldOutGate({ ... }), neutralizationGate({ ... }))
|
|
455
|
+
*
|
|
456
|
+
* Requires `ctx.neutralizedJudgeScores`, populated by `runImprovementLoop` when it
|
|
457
|
+
* is given a `neutralize` function. A gate composed without that wiring fails
|
|
458
|
+
* loud rather than silently passing an unproven candidate.
|
|
459
|
+
*/
|
|
460
|
+
|
|
461
|
+
interface NeutralizationGateOptions<TScenario extends Scenario = Scenario> {
|
|
462
|
+
scenarios: TScenario[];
|
|
463
|
+
/** Reject when the neutralized (content-blanked, footprint-matched) variant
|
|
464
|
+
* reproduces at least this fraction of the candidate's held-out lift. Default
|
|
465
|
+
* 0.5 — if blanking the content keeps half the lift, the content is decorative.
|
|
466
|
+
* Equality rejects: a neutralized lift == threshold·candidateLift is decorative. */
|
|
467
|
+
maxDecorativeFraction?: number;
|
|
468
|
+
}
|
|
469
|
+
/**
|
|
470
|
+
* Composable placebo gate: ships only when the candidate's held-out lift is NOT
|
|
471
|
+
* mostly reproduced by a footprint-matched neutralized variant.
|
|
472
|
+
*/
|
|
473
|
+
declare function neutralizationGate<TArtifact, TScenario extends Scenario>(options: NeutralizationGateOptions<TScenario>): Gate<TArtifact, TScenario>;
|
|
474
|
+
|
|
438
475
|
/**
|
|
439
476
|
* Anytime-valid sequential promotion gate — an e-process (betting
|
|
440
477
|
* test-martingale, see `eProcess` in `statistics.ts`) over paired
|
|
@@ -751,6 +788,34 @@ declare class FsLabeledScenarioStore implements LabeledScenarioStore {
|
|
|
751
788
|
private pathForSource;
|
|
752
789
|
}
|
|
753
790
|
|
|
791
|
+
/**
|
|
792
|
+
* @module
|
|
793
|
+
* Footprint-matched neutralization — the placebo control for content-vs-footprint
|
|
794
|
+
* attribution in a promotion gate.
|
|
795
|
+
*
|
|
796
|
+
* A promoted surface can raise a held-out score two different ways:
|
|
797
|
+
* 1. its CONTENT is informative (the thing we want to promote), or
|
|
798
|
+
* 2. it merely added prompt/mount FOOTPRINT — more bytes, more lines, a longer
|
|
799
|
+
* more authoritative-looking prompt — that the model spends attention on
|
|
800
|
+
* regardless of what the bytes say.
|
|
801
|
+
*
|
|
802
|
+
* A held-out gate proves the candidate beat baseline; it cannot separate (1) from
|
|
803
|
+
* (2). `neutralizeText` produces a variant that keeps the input's layout and
|
|
804
|
+
* length while carrying ZERO information, so scoring it isolates the footprint
|
|
805
|
+
* contribution (2). Feed the neutralized variant's scores to `neutralizationGate`:
|
|
806
|
+
* any lift it still holds over baseline is decorative, and a candidate whose lift
|
|
807
|
+
* survives neutralization is rejected however large its raw lift.
|
|
808
|
+
*/
|
|
809
|
+
/**
|
|
810
|
+
* Blank every non-whitespace character to a 1-byte filler while preserving all
|
|
811
|
+
* whitespace. Line count, indentation, and word/line lengths are unchanged — so
|
|
812
|
+
* the neutralized variant has the same layout and (for ASCII) the same byte
|
|
813
|
+
* footprint as the input, but no readable content. Whitespace is preserved
|
|
814
|
+
* deliberately: collapsing it would change the token structure and stop the
|
|
815
|
+
* variant from being a true footprint match.
|
|
816
|
+
*/
|
|
817
|
+
declare function neutralizeText(content: string): string;
|
|
818
|
+
|
|
754
819
|
/**
|
|
755
820
|
* FAPO (Fully Autonomous Prompt Optimization) is an orchestration policy, not
|
|
756
821
|
* a new prompt mutation primitive. The paper's loop evaluates an inspectable
|
|
@@ -2007,4 +2072,4 @@ declare function gitWorktreeAdapter(opts: GitWorktreeAdapterOptions): WorktreeAd
|
|
|
2007
2072
|
* as a ref under the adapter's worktree dir. */
|
|
2008
2073
|
declare function resolveWorktreePath(surface: CodeSurface, worktreeDir?: string): string;
|
|
2009
2074
|
|
|
2010
|
-
export { type AcceptedEdit, type AceProposerOptions, type AnalystArtifact, type AnalystScenario, type ApplySkillPatchResult, type BuildAnalystSurfaceDispatchOptions, type CampaignBreakdown, CampaignResult, CampaignRunPlan, CampaignStorage, CodeSurface, type CompareProposersOptions, type DimensionRegression, type DiscriminationScore, DispatchContext, type EvalFixture, type EvalFixtureFile, type EvalFixtureLoadOptions, type EvalFixtureRunPlan, type EvalFixtureScenario, type EvalFixtureValidationMode, type FailureModeRecallJudgeOptions, type FapoAttributionSignals, type FapoEntryConfig, type FapoFailureCluster, type FapoOptimizationLevel, type FapoProposerOptions, type FapoReviewInput, type FapoReviewIssue, type FapoReviewResult, type FapoScopeContract, FsLabeledScenarioStore, type FsLabeledScenarioStoreOptions, Gate, GenerationRecord, type GitWorktreeAdapterOptions, type Governor, type GovernorContext, type GovernorOp, type HaloProposerOptions, type HeldoutSignificance, type HeldoutSignificanceOptions, type HeuristicGovernorOptions, type JsonPrimitive, type JsonValue, JudgeConfig, JudgeScore, LabelTrust, LabeledScenarioRecord, LabeledScenarioSampleArgs, LabeledScenarioSource, LabeledScenarioStore, LabeledScenarioStoreError, LabeledScenarioWrite, Lineage, type LineageEdge, type LineageGraph, type LineageNode, type LineageNodeInput, type LineageStore, type LoadEvalFixtureScenariosOptions, type MemoryCurationProposerOptions, MutableSurface, type OptimizerEntryConfig, type PairedHoldout, type ParameterCandidate, type ParameterChange, type ParameterSweepProposerOptions, PlanCampaignRunOptions, type PlanEvalFixtureRunOptions, type PlaybackContext, type PlaybackDriver, type PlaybackStep, type PolicyEditProposerOptions, type ProfileDispatchFn, ProfileMatrixError, type ProfileSummary, ProposeContext, type ProposePatchesArgs, ProposedCandidate, type ProposerComparison, type ProposerEntry, type ProposerPairwise, type ProposerScore, type RejectedEdit, RunCampaignOptions, RunImprovementLoopOptions, type RunLineageLoopOptions, type RunLineageLoopResult, type RunLineageLoopSeed, type RunLineageOptions, type RunLineageResult, type RunLineageSeed, type RunLineageStepResult, type RunProfileMatrixOptions, type RunProfileMatrixResult, type RunSkillOptOptions, type RunSkillOptResult, Scenario, type ScenarioRollup, type ScenarioSignal, type ScoreboardRenderOptions, type ScoreboardRow, type ScoreboardSummary, type SequentialDecideFn, type SequentialDecideOptions, type SequentialDecision, type SequentialObservation, type SequentialPairedGate, type SequentialPairedGateOptions, type SkillOptEpochRecord, type SkillOptEvidence, type SkillOptProposer, type SkillOptProposerOptions, type SkillPatch, type SkillPatchOp, SkillPatchParseError, type SkillPatchRejection, SurfaceProposer, type SurfaceScore, type TraceAnalystProposerOptions, type UserStory, type UserStoryVerdict, type Worktree, type WorktreeAdapter, WorktreeAdapterError, aceProposer, applySkillPatch, buildAnalystSurfaceDispatch, callbackGovernor, campaignBreakdown, campaignMeanComposite, compareProposers, detectScale, dimensionRegressions, discoverEvalFixtures, extractFapoAttributionSignals, failureModeRecallJudge, fapoEscalationEntry, fapoProposer, fsLineageStore, gepaParetoEntry, gepaReflectionEntry, gitWorktreeAdapter, haloProposer, heldoutSignificance, heuristicGovernor, lineageNodeId, loadEvalFixture, loadEvalFixtureScenarios, makePlaybackDispatch, memLineageStore, memoryCurationProposer, pairHoldout, parameterSweepProposer, parseSkillPatchResponse, patchEditCount, planEvalFixtureRun, policyEditProposer, renderScoreboardMarkdown, resolveRunDir, resolveWorktreePath, runLineage, runLineageLoop, runProfileMatrix, runSkillOpt, scoreDiscrimination, scoreUserStory, scoreboardSummary, selectDiscriminative, sequentialDecide, sequentialPairedGate, skillOptEntry, skillOptProposer, tangleTracesRoot, traceAnalystProposer, userStoryScoreboard };
|
|
2075
|
+
export { type AcceptedEdit, type AceProposerOptions, type AnalystArtifact, type AnalystScenario, type ApplySkillPatchResult, type BuildAnalystSurfaceDispatchOptions, type CampaignBreakdown, CampaignResult, CampaignRunPlan, CampaignStorage, CodeSurface, type CompareProposersOptions, type DimensionRegression, type DiscriminationScore, DispatchContext, type EvalFixture, type EvalFixtureFile, type EvalFixtureLoadOptions, type EvalFixtureRunPlan, type EvalFixtureScenario, type EvalFixtureValidationMode, type FailureModeRecallJudgeOptions, type FapoAttributionSignals, type FapoEntryConfig, type FapoFailureCluster, type FapoOptimizationLevel, type FapoProposerOptions, type FapoReviewInput, type FapoReviewIssue, type FapoReviewResult, type FapoScopeContract, FsLabeledScenarioStore, type FsLabeledScenarioStoreOptions, Gate, GenerationRecord, type GitWorktreeAdapterOptions, type Governor, type GovernorContext, type GovernorOp, type HaloProposerOptions, type HeldoutSignificance, type HeldoutSignificanceOptions, type HeuristicGovernorOptions, type JsonPrimitive, type JsonValue, JudgeConfig, JudgeScore, LabelTrust, LabeledScenarioRecord, LabeledScenarioSampleArgs, LabeledScenarioSource, LabeledScenarioStore, LabeledScenarioStoreError, LabeledScenarioWrite, Lineage, type LineageEdge, type LineageGraph, type LineageNode, type LineageNodeInput, type LineageStore, type LoadEvalFixtureScenariosOptions, type MemoryCurationProposerOptions, MutableSurface, type NeutralizationGateOptions, type OptimizerEntryConfig, type PairedHoldout, type ParameterCandidate, type ParameterChange, type ParameterSweepProposerOptions, PlanCampaignRunOptions, type PlanEvalFixtureRunOptions, type PlaybackContext, type PlaybackDriver, type PlaybackStep, type PolicyEditProposerOptions, type ProfileDispatchFn, ProfileMatrixError, type ProfileSummary, ProposeContext, type ProposePatchesArgs, ProposedCandidate, type ProposerComparison, type ProposerEntry, type ProposerPairwise, type ProposerScore, type RejectedEdit, RunCampaignOptions, RunImprovementLoopOptions, type RunLineageLoopOptions, type RunLineageLoopResult, type RunLineageLoopSeed, type RunLineageOptions, type RunLineageResult, type RunLineageSeed, type RunLineageStepResult, type RunProfileMatrixOptions, type RunProfileMatrixResult, type RunSkillOptOptions, type RunSkillOptResult, Scenario, type ScenarioRollup, type ScenarioSignal, type ScoreboardRenderOptions, type ScoreboardRow, type ScoreboardSummary, type SequentialDecideFn, type SequentialDecideOptions, type SequentialDecision, type SequentialObservation, type SequentialPairedGate, type SequentialPairedGateOptions, type SkillOptEpochRecord, type SkillOptEvidence, type SkillOptProposer, type SkillOptProposerOptions, type SkillPatch, type SkillPatchOp, SkillPatchParseError, type SkillPatchRejection, SurfaceProposer, type SurfaceScore, type TraceAnalystProposerOptions, type UserStory, type UserStoryVerdict, type Worktree, type WorktreeAdapter, WorktreeAdapterError, aceProposer, applySkillPatch, buildAnalystSurfaceDispatch, callbackGovernor, campaignBreakdown, campaignMeanComposite, compareProposers, detectScale, dimensionRegressions, discoverEvalFixtures, extractFapoAttributionSignals, failureModeRecallJudge, fapoEscalationEntry, fapoProposer, fsLineageStore, gepaParetoEntry, gepaReflectionEntry, gitWorktreeAdapter, haloProposer, heldoutSignificance, heuristicGovernor, lineageNodeId, loadEvalFixture, loadEvalFixtureScenarios, makePlaybackDispatch, memLineageStore, memoryCurationProposer, neutralizationGate, neutralizeText, pairHoldout, parameterSweepProposer, parseSkillPatchResponse, patchEditCount, planEvalFixtureRun, policyEditProposer, renderScoreboardMarkdown, resolveRunDir, resolveWorktreePath, runLineage, runLineageLoop, runProfileMatrix, runSkillOpt, scoreDiscrimination, scoreUserStory, scoreboardSummary, selectDiscriminative, sequentialDecide, sequentialPairedGate, skillOptEntry, skillOptProposer, tangleTracesRoot, traceAnalystProposer, userStoryScoreboard };
|
package/dist/campaign/index.js
CHANGED
|
@@ -7,7 +7,7 @@ import {
|
|
|
7
7
|
paretoSignificanceGate,
|
|
8
8
|
powerPreflight,
|
|
9
9
|
runEval
|
|
10
|
-
} from "../chunk-
|
|
10
|
+
} from "../chunk-V75NN2ZR.js";
|
|
11
11
|
import {
|
|
12
12
|
HARNESS_NATIVE_MODEL,
|
|
13
13
|
agentProfileHash,
|
|
@@ -17,7 +17,7 @@ import {
|
|
|
17
17
|
harnessAxisOf,
|
|
18
18
|
llmJudge,
|
|
19
19
|
verifyCompletion
|
|
20
|
-
} from "../chunk-
|
|
20
|
+
} from "../chunk-6PF6LXRY.js";
|
|
21
21
|
import {
|
|
22
22
|
buildLoopProvenanceRecord,
|
|
23
23
|
campaignBreakdown,
|
|
@@ -43,7 +43,7 @@ import {
|
|
|
43
43
|
runOptimization,
|
|
44
44
|
surfaceContentHash,
|
|
45
45
|
surfaceHash
|
|
46
|
-
} from "../chunk-
|
|
46
|
+
} from "../chunk-G4DLZAV5.js";
|
|
47
47
|
import {
|
|
48
48
|
assertRealBackend,
|
|
49
49
|
contentHash,
|
|
@@ -737,6 +737,86 @@ function assertInside(root, target, label) {
|
|
|
737
737
|
throw new Error(`loadEvalFixture: fixture path escapes evalsDir: ${label}`);
|
|
738
738
|
}
|
|
739
739
|
|
|
740
|
+
// src/campaign/gates/neutralization-gate.ts
|
|
741
|
+
function mean(xs) {
|
|
742
|
+
return xs.length === 0 ? 0 : xs.reduce((a, b) => a + b, 0) / xs.length;
|
|
743
|
+
}
|
|
744
|
+
function pairedLift(arm, baseline, scenarioIds) {
|
|
745
|
+
const paired = pairHoldout(arm, baseline, scenarioIds, (s) => s.composite);
|
|
746
|
+
const deltas = paired.after.map((a, i) => a - (paired.before[i] ?? 0));
|
|
747
|
+
return { lift: mean(deltas), n: deltas.length };
|
|
748
|
+
}
|
|
749
|
+
function neutralizationGate(options) {
|
|
750
|
+
const maxDecorativeFraction = options.maxDecorativeFraction ?? 0.5;
|
|
751
|
+
return {
|
|
752
|
+
name: "neutralizationGate",
|
|
753
|
+
async decide(ctx) {
|
|
754
|
+
if (!ctx.baselineJudgeScores) {
|
|
755
|
+
throw new Error(
|
|
756
|
+
"neutralizationGate: ctx.baselineJudgeScores is required \u2014 the placebo control measures lift OVER baseline."
|
|
757
|
+
);
|
|
758
|
+
}
|
|
759
|
+
if (!ctx.neutralizedJudgeScores) {
|
|
760
|
+
throw new Error(
|
|
761
|
+
"neutralizationGate: ctx.neutralizedJudgeScores is required. It is populated by runImprovementLoop only when a `neutralize` function is supplied \u2014 composing this gate without that wiring would pass an unproven candidate."
|
|
762
|
+
);
|
|
763
|
+
}
|
|
764
|
+
const scenarioIds = new Set(options.scenarios.map((s) => s.id));
|
|
765
|
+
const cand = pairedLift(
|
|
766
|
+
ctx.judgeScores,
|
|
767
|
+
ctx.baselineJudgeScores,
|
|
768
|
+
scenarioIds
|
|
769
|
+
);
|
|
770
|
+
const neut = pairedLift(
|
|
771
|
+
ctx.neutralizedJudgeScores,
|
|
772
|
+
ctx.baselineJudgeScores,
|
|
773
|
+
scenarioIds
|
|
774
|
+
);
|
|
775
|
+
if (cand.lift <= 0) {
|
|
776
|
+
return {
|
|
777
|
+
decision: "hold",
|
|
778
|
+
reasons: [
|
|
779
|
+
`neutralization: candidate held-out lift ${cand.lift.toFixed(3)} \u2264 0 \u2014 no positive lift to attribute to content`
|
|
780
|
+
],
|
|
781
|
+
contributingGates: [
|
|
782
|
+
{
|
|
783
|
+
name: "neutralizationGate",
|
|
784
|
+
passed: false,
|
|
785
|
+
detail: { candidateLift: cand.lift, neutralizedLift: neut.lift, n: cand.n }
|
|
786
|
+
}
|
|
787
|
+
],
|
|
788
|
+
delta: cand.lift
|
|
789
|
+
};
|
|
790
|
+
}
|
|
791
|
+
const decorativeFraction = neut.lift / cand.lift;
|
|
792
|
+
const passed = decorativeFraction < maxDecorativeFraction;
|
|
793
|
+
const pct = (decorativeFraction * 100).toFixed(0);
|
|
794
|
+
return {
|
|
795
|
+
decision: passed ? "ship" : "hold",
|
|
796
|
+
reasons: passed ? [
|
|
797
|
+
`neutralization: content is causal \u2014 blanked variant reproduces ${pct}% of the lift (< ${(maxDecorativeFraction * 100).toFixed(0)}%); candidate \u0394 ${cand.lift.toFixed(3)}, neutralized \u0394 ${neut.lift.toFixed(3)}`
|
|
798
|
+
] : [
|
|
799
|
+
`neutralization: lift is DECORATIVE \u2014 blanking the content (footprint-matched) reproduces ${pct}% of the lift (\u2265 ${(maxDecorativeFraction * 100).toFixed(0)}%); candidate \u0394 ${cand.lift.toFixed(3)}, neutralized \u0394 ${neut.lift.toFixed(3)}`
|
|
800
|
+
],
|
|
801
|
+
contributingGates: [
|
|
802
|
+
{
|
|
803
|
+
name: "neutralizationGate",
|
|
804
|
+
passed,
|
|
805
|
+
detail: {
|
|
806
|
+
candidateLift: cand.lift,
|
|
807
|
+
neutralizedLift: neut.lift,
|
|
808
|
+
decorativeFraction,
|
|
809
|
+
maxDecorativeFraction,
|
|
810
|
+
n: cand.n
|
|
811
|
+
}
|
|
812
|
+
}
|
|
813
|
+
],
|
|
814
|
+
delta: cand.lift
|
|
815
|
+
};
|
|
816
|
+
}
|
|
817
|
+
};
|
|
818
|
+
}
|
|
819
|
+
|
|
740
820
|
// src/campaign/gates/sequential.ts
|
|
741
821
|
import { createHash as createHash3 } from "crypto";
|
|
742
822
|
function verifyManifestSync(m) {
|
|
@@ -1214,6 +1294,12 @@ function appendLine(path, line) {
|
|
|
1214
1294
|
}
|
|
1215
1295
|
}
|
|
1216
1296
|
|
|
1297
|
+
// src/campaign/neutralize.ts
|
|
1298
|
+
var FILLER = "#";
|
|
1299
|
+
function neutralizeText(content) {
|
|
1300
|
+
return content.replace(/\S/g, FILLER);
|
|
1301
|
+
}
|
|
1302
|
+
|
|
1217
1303
|
// src/campaign/proposers/fapo.ts
|
|
1218
1304
|
var FAPO_LEVELS = ["prompt", "parameter", "structural"];
|
|
1219
1305
|
var MAX_FINDING_DEPTH = 16;
|
|
@@ -2155,8 +2241,8 @@ async function compareProposerEntries(opts) {
|
|
|
2155
2241
|
});
|
|
2156
2242
|
const score = {
|
|
2157
2243
|
name: w.name,
|
|
2158
|
-
baselineComposite:
|
|
2159
|
-
winnerComposite:
|
|
2244
|
+
baselineComposite: mean2(baselineArr),
|
|
2245
|
+
winnerComposite: mean2(w.arr),
|
|
2160
2246
|
lift: boot.mean,
|
|
2161
2247
|
liftCi: { low: boot.low, high: boot.high },
|
|
2162
2248
|
costUsd: w.costUsd,
|
|
@@ -2193,7 +2279,7 @@ async function compareProposerEntries(opts) {
|
|
|
2193
2279
|
});
|
|
2194
2280
|
return { scores, best, pairwise, holdoutScenarioIds: scenarioIds };
|
|
2195
2281
|
}
|
|
2196
|
-
function
|
|
2282
|
+
function mean2(xs) {
|
|
2197
2283
|
return xs.length === 0 ? 0 : xs.reduce((a, b) => a + b, 0) / xs.length;
|
|
2198
2284
|
}
|
|
2199
2285
|
function slug(name) {
|
|
@@ -2614,12 +2700,12 @@ function sanitize(id) {
|
|
|
2614
2700
|
function sha(input) {
|
|
2615
2701
|
return createHash5("sha256").update(JSON.stringify(input)).digest("hex");
|
|
2616
2702
|
}
|
|
2617
|
-
function
|
|
2703
|
+
function mean3(xs) {
|
|
2618
2704
|
return xs.length === 0 ? 0 : xs.reduce((a, b) => a + b, 0) / xs.length;
|
|
2619
2705
|
}
|
|
2620
2706
|
function cellComposite(cell) {
|
|
2621
2707
|
const composites = Object.values(cell.judgeScores).map((s) => s.composite);
|
|
2622
|
-
return composites.length === 0 ? 0 :
|
|
2708
|
+
return composites.length === 0 ? 0 : mean3(composites);
|
|
2623
2709
|
}
|
|
2624
2710
|
function requireResolvedModel(cell, profileId) {
|
|
2625
2711
|
const resolved = cell.resolvedModel?.trim();
|
|
@@ -2655,7 +2741,7 @@ function buildRunRecord(args) {
|
|
|
2655
2741
|
if (js.notes) notes.push(`${judgeName}: ${js.notes}`);
|
|
2656
2742
|
}
|
|
2657
2743
|
const perDimMean = {};
|
|
2658
|
-
for (const [dim, values] of Object.entries(dimAccum)) perDimMean[dim] =
|
|
2744
|
+
for (const [dim, values] of Object.entries(dimAccum)) perDimMean[dim] = mean3(values);
|
|
2659
2745
|
let costUsd = cell.costUsd;
|
|
2660
2746
|
let costEstimated = false;
|
|
2661
2747
|
if (costUsd === 0 && cell.tokenUsage.output > 0 && isModelPriced(model)) {
|
|
@@ -2819,7 +2905,7 @@ async function runProfileMatrix(opts) {
|
|
|
2819
2905
|
// one harness, so the first record's model is representative).
|
|
2820
2906
|
model: declaredModel === HARNESS_NATIVE_MODEL ? profileRecords[0]?.model ?? declaredModel : declaredModel,
|
|
2821
2907
|
records: profileRecords.length,
|
|
2822
|
-
meanComposite:
|
|
2908
|
+
meanComposite: mean3(profileRecords.map(compositeOf)),
|
|
2823
2909
|
totalCostUsd: pricedTotalCostUsd,
|
|
2824
2910
|
integrity: summarizeBackendIntegrity(profileRecords)
|
|
2825
2911
|
};
|
|
@@ -2849,7 +2935,7 @@ function rollup(records, keyOf) {
|
|
|
2849
2935
|
groups.set(key, arr);
|
|
2850
2936
|
}
|
|
2851
2937
|
const out = {};
|
|
2852
|
-
for (const [key, xs] of groups) out[key] = { meanComposite:
|
|
2938
|
+
for (const [key, xs] of groups) out[key] = { meanComposite: mean3(xs), n: xs.length };
|
|
2853
2939
|
return out;
|
|
2854
2940
|
}
|
|
2855
2941
|
function rollupByPersona(records, scenarios, personaOf) {
|
|
@@ -3409,6 +3495,8 @@ export {
|
|
|
3409
3495
|
makePlaybackDispatch,
|
|
3410
3496
|
memLineageStore,
|
|
3411
3497
|
memoryCurationProposer,
|
|
3498
|
+
neutralizationGate,
|
|
3499
|
+
neutralizeText,
|
|
3412
3500
|
openAutoPr,
|
|
3413
3501
|
pairHoldout,
|
|
3414
3502
|
parameterSweepProposer,
|