@tangle-network/agent-eval 0.108.1 → 0.110.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/analyst/index.d.ts +10 -12
- package/dist/analyst/index.js +8 -11
- package/dist/analyst/index.js.map +1 -1
- package/dist/{analyze-runs-DJYpep3L.d.ts → analyze-runs-Dmz6LA9e.d.ts} +4 -4
- package/dist/{baseline-Bbid3WoO.d.ts → baseline-DsNteOgR.d.ts} +32 -2
- package/dist/belief-state/index.d.ts +6 -6
- package/dist/benchmarks/index.d.ts +4 -4
- package/dist/benchmarks/index.js +7 -8
- package/dist/builder-eval/index.d.ts +4 -4
- package/dist/builder-eval/index.js +1 -2
- package/dist/builder-eval/index.js.map +1 -1
- package/dist/{calibration-BPmzuVPk.d.ts → calibration-Dz8TQV4y.d.ts} +2 -2
- package/dist/campaign/index.d.ts +161 -20
- package/dist/campaign/index.js +15 -8
- package/dist/{chunk-LOJ2QVCE.js → chunk-2IY4ILP4.js} +2 -2
- package/dist/{chunk-LIEJUH2I.js → chunk-6PL5MGDL.js} +9 -9
- package/dist/{chunk-2OGPXHOB.js → chunk-7NX6ZSBG.js} +36 -7
- package/dist/chunk-7NX6ZSBG.js.map +1 -0
- package/dist/{chunk-OVPVM4JC.js → chunk-GTERJI6Q.js} +4 -4
- package/dist/{chunk-YEHAEDUD.js → chunk-IMWDSFUM.js} +604 -2
- package/dist/chunk-IMWDSFUM.js.map +1 -0
- package/dist/{chunk-JZXGWLK5.js → chunk-MHNQWM4I.js} +62 -6
- package/dist/chunk-MHNQWM4I.js.map +1 -0
- package/dist/{chunk-QRVS7MX4.js → chunk-OW47B5WA.js} +3 -5
- package/dist/{chunk-QRVS7MX4.js.map → chunk-OW47B5WA.js.map} +1 -1
- package/dist/{chunk-DBDRR6GF.js → chunk-PLOMR3HP.js} +48 -2
- package/dist/chunk-PLOMR3HP.js.map +1 -0
- package/dist/{chunk-GDZAWO2I.js → chunk-QFGTU7MT.js} +2 -2
- package/dist/{chunk-6SKVFBTR.js → chunk-RNB2NICW.js} +115 -13
- package/dist/chunk-RNB2NICW.js.map +1 -0
- package/dist/{chunk-V7HNA47Z.js → chunk-RSVSSZKF.js} +5 -5
- package/dist/{chunk-5PK3626Q.js → chunk-XRGOKCMO.js} +88 -17
- package/dist/chunk-XRGOKCMO.js.map +1 -0
- package/dist/{code-agent-session-rnJKlqmT.d.ts → code-agent-session-yitf9I-F.d.ts} +1 -1
- package/dist/contract/index.d.ts +20 -23
- package/dist/contract/index.js +11 -13
- package/dist/contract/index.js.map +1 -1
- package/dist/{control-B8UthSBL.d.ts → control-U8LBKUES.d.ts} +5 -6
- package/dist/control.d.ts +8 -9
- package/dist/control.js +6 -8
- package/dist/{dataset-DS7ytHZU.d.ts → dataset-NENEzRgk.d.ts} +1 -1
- package/dist/{default-registry-BswHCXnU.d.ts → default-registry-Bcf1uKVI.d.ts} +1 -2
- package/dist/{emitter-C2rqGH_l.d.ts → emitter-BRchAAAx.d.ts} +2 -2
- package/dist/{failure-cluster-DH9Flgcf.d.ts → failure-cluster-C48PiReX.d.ts} +2 -2
- package/dist/feedback-trajectory-pDcz1lQ1.d.ts +348 -0
- package/dist/{gepa-B3x5Ulcv.d.ts → gepa-BUNP3606.d.ts} +143 -2
- package/dist/hosted/index.d.ts +7 -7
- package/dist/{index-pPtfoIJO.d.ts → index-Dc3VLGhp.d.ts} +2 -2
- package/dist/index.d.ts +645 -61
- package/dist/index.js +1282 -190
- package/dist/index.js.map +1 -1
- package/dist/{insight-report-B4xrdwEK.d.ts → insight-report-D4cXFsLt.d.ts} +1 -1
- package/dist/{integrity-DqGZg3st.d.ts → integrity-qemeBAyx.d.ts} +1 -1
- package/dist/{types-D1ytG0Yg.d.ts → kind-factory-20hcaYpf.d.ts} +169 -2
- package/dist/meta-eval/index.d.ts +5 -5
- package/dist/meta-eval/index.js +1 -2
- package/dist/meta-eval/index.js.map +1 -1
- package/dist/{multi-layer-verifier-CI4jdX-q.d.ts → multi-layer-verifier-BsqKuLyN.d.ts} +1 -1
- package/dist/multishot/index.d.ts +3 -3
- package/dist/openapi.json +1 -1
- package/dist/pipelines/index.d.ts +6 -7
- package/dist/pipelines/index.js +3 -6
- package/dist/pipelines/index.js.map +1 -1
- package/dist/{policy-edit-DQUXYMDm.d.ts → policy-edit-D2bBDZDf.d.ts} +2 -2
- package/dist/{pre-registration-BUhVPzE7.d.ts → pre-registration-BepVVa6P.d.ts} +3 -3
- package/dist/{provenance-DdDhf6cg.d.ts → provenance-DMvsfknv.d.ts} +3 -5
- package/dist/{query-0aTmbmQe.d.ts → query-Ck190MOd.d.ts} +2 -2
- package/dist/{release-report-DeJpsBiA.d.ts → release-report-oBfOz8ku.d.ts} +3 -3
- package/dist/reporting.d.ts +8 -8
- package/dist/{researcher-Wc7dx6GM.d.ts → researcher-CaH0CwFC.d.ts} +6 -6
- package/dist/rl.d.ts +568 -15
- package/dist/rl.js +4 -4
- package/dist/{rubric-predictive-validity-DPnyG-CE.d.ts → rubric-predictive-validity-C-fMteAW.d.ts} +1 -1
- package/dist/{run-record-I-Z3JNvO.d.ts → run-record-DksGsfgv.d.ts} +1 -1
- package/dist/{runtime-trajectory-iW9IhV3e.d.ts → runtime-trajectory-h5i0SZUj.d.ts} +1 -1
- package/dist/{schema-m0gsnbt3.d.ts → schema-SGWcK9wa.d.ts} +1 -1
- package/dist/{semantic-concept-judge-BmNZPB_j.d.ts → semantic-concept-judge-D7z6JCLZ.d.ts} +57 -4
- package/dist/{store-BcFXE6LG.d.ts → store-BsVi7ncX.d.ts} +1 -1
- package/dist/storyboard/index.d.ts +1 -1
- package/dist/{summary-report-QMZVe3P-.d.ts → summary-report-Bz-0-t8v.d.ts} +2 -2
- package/dist/{test-graded-scenario-DeODGLra.d.ts → test-graded-scenario-mzYBKspu.d.ts} +3 -3
- package/dist/traces.d.ts +54 -11
- package/dist/traces.js +25 -27
- package/dist/{types-BdIv5dvA.d.ts → types-v--ctu-b.d.ts} +2 -2
- package/dist/wire/index.d.ts +5 -6
- package/package.json +1 -71
- package/dist/adapters/http.d.ts +0 -142
- package/dist/adapters/http.js +0 -203
- package/dist/adapters/http.js.map +0 -1
- package/dist/adapters/langchain.d.ts +0 -95
- package/dist/adapters/langchain.js +0 -34
- package/dist/adapters/langchain.js.map +0 -1
- package/dist/adapters/otel.d.ts +0 -112
- package/dist/adapters/otel.js +0 -110
- package/dist/adapters/otel.js.map +0 -1
- package/dist/chunk-2OGPXHOB.js.map +0 -1
- package/dist/chunk-45EEMHTC.js +0 -35
- package/dist/chunk-45EEMHTC.js.map +0 -1
- package/dist/chunk-5BKGXME7.js +0 -65
- package/dist/chunk-5BKGXME7.js.map +0 -1
- package/dist/chunk-5PK3626Q.js.map +0 -1
- package/dist/chunk-6SK5VFYK.js +0 -100
- package/dist/chunk-6SK5VFYK.js.map +0 -1
- package/dist/chunk-6SKVFBTR.js.map +0 -1
- package/dist/chunk-DBDRR6GF.js.map +0 -1
- package/dist/chunk-DJWX3GVS.js +0 -81
- package/dist/chunk-DJWX3GVS.js.map +0 -1
- package/dist/chunk-FOUG2VVS.js +0 -855
- package/dist/chunk-FOUG2VVS.js.map +0 -1
- package/dist/chunk-JZXGWLK5.js.map +0 -1
- package/dist/chunk-K7QEIHHJ.js +0 -613
- package/dist/chunk-K7QEIHHJ.js.map +0 -1
- package/dist/chunk-KKHDIONI.js +0 -414
- package/dist/chunk-KKHDIONI.js.map +0 -1
- package/dist/chunk-KMPRBJK4.js +0 -74
- package/dist/chunk-KMPRBJK4.js.map +0 -1
- package/dist/chunk-Q2JRAWRI.js +0 -196
- package/dist/chunk-Q2JRAWRI.js.map +0 -1
- package/dist/chunk-RZTMDUO7.js +0 -49
- package/dist/chunk-RZTMDUO7.js.map +0 -1
- package/dist/chunk-STGVSCDH.js +0 -202
- package/dist/chunk-STGVSCDH.js.map +0 -1
- package/dist/chunk-YEHAEDUD.js.map +0 -1
- package/dist/control-runtime-Acf9CGhw.d.ts +0 -182
- package/dist/corpus-eBVwhCp1.d.ts +0 -560
- package/dist/counterfactual-DlOz8PBx.d.ts +0 -85
- package/dist/diagnose.d.ts +0 -252
- package/dist/diagnose.js +0 -382
- package/dist/diagnose.js.map +0 -1
- package/dist/feedback-trajectory-C9KCo8ag.d.ts +0 -169
- package/dist/governance/index.d.ts +0 -135
- package/dist/governance/index.js +0 -18
- package/dist/governance/index.js.map +0 -1
- package/dist/groundedness/index.d.ts +0 -112
- package/dist/groundedness/index.js +0 -77
- package/dist/groundedness/index.js.map +0 -1
- package/dist/harness-optimizer-mOl9XX_O.d.ts +0 -106
- package/dist/kind-factory-DvIGo_cP.d.ts +0 -171
- package/dist/knowledge/index.d.ts +0 -103
- package/dist/knowledge/index.js +0 -18
- package/dist/knowledge/index.js.map +0 -1
- package/dist/pareto-E-pembql.d.ts +0 -81
- package/dist/perf/index.d.ts +0 -123
- package/dist/perf/index.js +0 -18
- package/dist/perf/index.js.map +0 -1
- package/dist/prm/index.d.ts +0 -104
- package/dist/prm/index.js +0 -265
- package/dist/prm/index.js.map +0 -1
- package/dist/product-benchmark/index.d.ts +0 -247
- package/dist/product-benchmark/index.js +0 -37
- package/dist/product-benchmark/index.js.map +0 -1
- package/dist/red-team-KmmiqBlY.d.ts +0 -63
- package/dist/redact-B40YG2M_.d.ts +0 -45
- package/dist/rubric-Cc6UHvUb.d.ts +0 -73
- package/dist/run-critic-CmMf05uV.d.ts +0 -56
- package/dist/sink-fetch-B1Yg4Til.d.ts +0 -101
- package/dist/telemetry/file.d.ts +0 -19
- package/dist/telemetry/file.js +0 -45
- package/dist/telemetry/file.js.map +0 -1
- package/dist/telemetry/index.d.ts +0 -38
- package/dist/telemetry/index.js +0 -130
- package/dist/telemetry/index.js.map +0 -1
- package/dist/testing-C21CHsq2.d.ts +0 -20
- package/dist/testing.d.ts +0 -1
- package/dist/testing.js +0 -8
- package/dist/testing.js.map +0 -1
- package/dist/trajectory-2TkpSEVh.d.ts +0 -33
- package/dist/workflow/index.d.ts +0 -496
- package/dist/workflow/index.js +0 -2178
- package/dist/workflow/index.js.map +0 -1
- /package/dist/{chunk-LOJ2QVCE.js.map → chunk-2IY4ILP4.js.map} +0 -0
- /package/dist/{chunk-LIEJUH2I.js.map → chunk-6PL5MGDL.js.map} +0 -0
- /package/dist/{chunk-OVPVM4JC.js.map → chunk-GTERJI6Q.js.map} +0 -0
- /package/dist/{chunk-GDZAWO2I.js.map → chunk-QFGTU7MT.js.map} +0 -0
- /package/dist/{chunk-V7HNA47Z.js.map → chunk-RSVSSZKF.js.map} +0 -0
package/dist/campaign/index.d.ts
CHANGED
|
@@ -1,35 +1,32 @@
|
|
|
1
|
-
import { S as SignedManifest, B as BackendIntegrityReport, C as CompletionRequirement, R as RuntimeEventLike, a as CompletionVerdict, P as ProducedState, b as CorrectnessChecker } from '../pre-registration-
|
|
2
|
-
export { L as LlmJudgeDimension, c as LlmJudgeOptions, l as llmJudge } from '../pre-registration-
|
|
1
|
+
import { S as SignedManifest, B as BackendIntegrityReport, C as CompletionRequirement, R as RuntimeEventLike, a as CompletionVerdict, P as ProducedState, b as CorrectnessChecker } from '../pre-registration-BepVVa6P.js';
|
|
2
|
+
export { L as LlmJudgeDimension, c as LlmJudgeOptions, l as llmJudge } from '../pre-registration-BepVVa6P.js';
|
|
3
3
|
import { A as AnalyzeTracesOptions, a as AnalyzeTracesInput, b as AnalyzeTracesResult } from '../analyst-C8HHvfJp.js';
|
|
4
|
-
import { S as Scenario, M as MutableSurface,
|
|
5
|
-
export { i as CampaignAggregates, j as CampaignArtifactWriter, k as CampaignCellResult, l as CampaignCostMeter, z as CampaignTokenUsage,
|
|
6
|
-
import { C as CampaignRunPlan, P as PlanCampaignRunOptions, b as RunCampaignOptions, c as RunImprovementLoopOptions } from '../gepa-
|
|
7
|
-
export {
|
|
4
|
+
import { S as Scenario, M as MutableSurface, D as DispatchContext, b as JudgeConfig, g as Gate, e as GenerationRecord, J as JudgeScore, L as LabeledScenarioStore, s as LabeledScenarioWrite, t as LabeledScenarioSampleArgs, u as LabeledScenarioRecord, v as LabelTrust, f as SurfaceProposer, w as ProposedCandidate, x as ProposeContext, y as LabeledScenarioSource, C as CampaignResult, m as CodeSurface } from '../types-v--ctu-b.js';
|
|
5
|
+
export { i as CampaignAggregates, j as CampaignArtifactWriter, k as CampaignCellResult, l as CampaignCostMeter, z as CampaignTokenUsage, d as CampaignTraceWriter, c as DispatchFn, n as GateContext, h as GateDecision, G as GateResult, o as GenerationCandidate, A as JudgeAggregate, a as JudgeDimension, p as Mutator, O as OptimizationProposer, q as OptimizerConfig, P as ParetoParent, R as RedactionStatus, B as ScenarioAggregate, r as SessionScript, T as TraceSpan, E as isProposedCandidate, F as labelTrustRank } from '../types-v--ctu-b.js';
|
|
6
|
+
import { C as CampaignRunPlan, P as PlanCampaignRunOptions, b as RunCampaignOptions, c as RunImprovementLoopOptions } from '../gepa-BUNP3606.js';
|
|
7
|
+
export { f as CampaignRunPlanCell, h as GepaProposerConstraints, G as GepaProposerOptions, O as OpenAutoPrOptions, i as OpenAutoPrResult, a as RunImprovementLoopResult, R as RunOptimizationOptions, j as RunOptimizationResult, k as countSentenceEdits, l as defaultRenderDiff, m as extractH2Sections, g as gepaProposer, o as openAutoPr, p as planCampaignRun, r as runCampaign, d as runImprovementLoop, n as runOptimization, s as surfaceHash } from '../gepa-BUNP3606.js';
|
|
8
8
|
import { C as CampaignStorage } from '../storage-Dw_f7WMt.js';
|
|
9
9
|
export { f as fsCampaignStorage, i as inMemoryCampaignStorage } from '../storage-Dw_f7WMt.js';
|
|
10
|
-
export { A as AxisEvidence, a as AxisVerdict, B as BuildEvidenceVectorOptions, l as BuildLoopProvenanceArgs, D as DefaultProductionGateOptions, m as EmitLoopProvenanceArgs, n as EmitLoopProvenanceResult, E as EvidenceVector, b as EvolutionaryProposerOptions, H as HeldOutGateOptions, o as LoopProvenanceBackend, q as LoopProvenanceCandidate, L as LoopProvenanceRecord, O as ObjectiveSource, c as ParetoSignificanceGateOptions, P as PowerPreflight, s as PowerPreflightOptions, d as PromotionObjective, e as PromotionPolicy, R as RunEvalOptions, f as buildEvidenceVector, t as buildLoopProvenanceRecord, g as composeGate, h as defaultProductionGate, u as emitLoopProvenance, i as evolutionaryProposer, j as heldOutGate, v as loopProvenanceSpans, p as paretoPolicy, k as paretoSignificanceGate, w as powerPreflight, x as provenanceRecordPath, y as provenanceSpansPath, r as runEval, z as surfaceContentHash } from '../provenance-
|
|
10
|
+
export { A as AxisEvidence, a as AxisVerdict, B as BuildEvidenceVectorOptions, l as BuildLoopProvenanceArgs, D as DefaultProductionGateOptions, m as EmitLoopProvenanceArgs, n as EmitLoopProvenanceResult, E as EvidenceVector, b as EvolutionaryProposerOptions, H as HeldOutGateOptions, o as LoopProvenanceBackend, q as LoopProvenanceCandidate, L as LoopProvenanceRecord, O as ObjectiveSource, c as ParetoSignificanceGateOptions, P as PowerPreflight, s as PowerPreflightOptions, d as PromotionObjective, e as PromotionPolicy, R as RunEvalOptions, f as buildEvidenceVector, t as buildLoopProvenanceRecord, g as composeGate, h as defaultProductionGate, u as emitLoopProvenance, i as evolutionaryProposer, j as heldOutGate, v as loopProvenanceSpans, p as paretoPolicy, k as paretoSignificanceGate, w as powerPreflight, x as provenanceRecordPath, y as provenanceSpansPath, r as runEval, z as surfaceContentHash } from '../provenance-DMvsfknv.js';
|
|
11
11
|
import { E as EProcessState, a as PairedBootstrapResult } from '../statistics-D88peojY.js';
|
|
12
12
|
import { L as LlmClientOptions } from '../llm-client-DyqEH4jH.js';
|
|
13
13
|
import { AgentProfile } from '@tangle-network/agent-interface';
|
|
14
14
|
import { A as AgentEvalError } from '../errors-oeQrLqXC.js';
|
|
15
|
-
import { b as RunSplitTag, R as RunRecord } from '../run-record-
|
|
16
|
-
import { b as PolicyEdit, F as FindingToPolicyEditOptions, d as PolicyEditAdmissionOptions, c as PolicyEditAdmission } from '../policy-edit-
|
|
17
|
-
import { T as TraceAnalystKindSpec } from '../kind-factory-
|
|
18
|
-
import { A as AnalystFinding } from '../types-D1ytG0Yg.js';
|
|
15
|
+
import { b as RunSplitTag, R as RunRecord } from '../run-record-DksGsfgv.js';
|
|
16
|
+
import { b as PolicyEdit, F as FindingToPolicyEditOptions, d as PolicyEditAdmissionOptions, c as PolicyEditAdmission } from '../policy-edit-D2bBDZDf.js';
|
|
17
|
+
import { T as TraceAnalystKindSpec, A as AnalystFinding } from '../kind-factory-20hcaYpf.js';
|
|
19
18
|
import '@tangle-network/tcloud';
|
|
20
19
|
import '../raw-provider-sink-C46HDghv.js';
|
|
21
20
|
import '../verdict-C9MlYujm.js';
|
|
22
21
|
import '@ax-llm/ax';
|
|
23
22
|
import '../store-C1YxJDEK.js';
|
|
24
|
-
import '../
|
|
25
|
-
import '../
|
|
26
|
-
import '../
|
|
27
|
-
import '../schema-m0gsnbt3.js';
|
|
28
|
-
import '../pareto-E-pembql.js';
|
|
23
|
+
import '../dataset-NENEzRgk.js';
|
|
24
|
+
import '../store-BsVi7ncX.js';
|
|
25
|
+
import '../schema-SGWcK9wa.js';
|
|
29
26
|
import '../hosted/index.js';
|
|
30
|
-
import '../insight-report-
|
|
31
|
-
import '../summary-report-
|
|
32
|
-
import '../failure-cluster-
|
|
27
|
+
import '../insight-report-D4cXFsLt.js';
|
|
28
|
+
import '../summary-report-Bz-0-t8v.js';
|
|
29
|
+
import '../failure-cluster-C48PiReX.js';
|
|
33
30
|
import '../judge-calibration-7C-IDmKr.js';
|
|
34
31
|
import '../types-C7DGg5ex.js';
|
|
35
32
|
import 'zod';
|
|
@@ -727,6 +724,80 @@ declare function dimensionRegressions(candidate: Map<string, Record<string, Judg
|
|
|
727
724
|
seed?: number;
|
|
728
725
|
}): DimensionRegression[];
|
|
729
726
|
|
|
727
|
+
/**
|
|
728
|
+
* Evidence grounding for reflective optimizers (GEPA-style revise loops).
|
|
729
|
+
*
|
|
730
|
+
* Two failure modes recur when an LLM revises an artifact from raw rollout
|
|
731
|
+
* traces (first measured in agent-lab R358, where naive reflection REGRESSED
|
|
732
|
+
* the score 0.375 -> 0.125 before these helpers fixed it):
|
|
733
|
+
*
|
|
734
|
+
* 1. The environment often hides WHY a rollout failed - a tool call can
|
|
735
|
+
* succeed while an invisible downstream check fails - so the reviser
|
|
736
|
+
* cannot see the cause in the transcript. The only reliable signal is the
|
|
737
|
+
* field-level difference between what passing and failing rollouts did.
|
|
738
|
+
* `rolloutArgumentDiff` computes that difference deterministically so the
|
|
739
|
+
* reviser is handed the diff instead of being trusted to derive it.
|
|
740
|
+
*
|
|
741
|
+
* 2. Revisers invent plausible-but-wrong literal values ("use 'new'",
|
|
742
|
+
* "use 'sent'") that no passing rollout ever used, turning every rollout
|
|
743
|
+
* into a failure. `classifyUngroundedLiterals` mechanically detects them,
|
|
744
|
+
* separating HARMFUL literals (ones failing rollouts actually used -
|
|
745
|
+
* proven damage) from benign illustrations (e.g. a name example like
|
|
746
|
+
* 'Doe'), so callers can hard-reject the former and merely log the latter.
|
|
747
|
+
* Rejecting every ungrounded quoted word is too blunt: it killed a run
|
|
748
|
+
* over a surname illustration before the severity split existed.
|
|
749
|
+
*
|
|
750
|
+
* Pure data in, data out: no LLM calls, no filesystem, no domain knowledge.
|
|
751
|
+
*/
|
|
752
|
+
/** One tool/action call observed in a rollout: a name plus its arguments. */
|
|
753
|
+
interface RolloutCall {
|
|
754
|
+
readonly name: string;
|
|
755
|
+
readonly args: Readonly<Record<string, unknown>>;
|
|
756
|
+
}
|
|
757
|
+
/** A scored rollout: its calls plus the scalar outcome used to split pass/fail. */
|
|
758
|
+
interface ScoredRollout {
|
|
759
|
+
/** Caller-meaningful identifier (task id, cell id) used only for reporting. */
|
|
760
|
+
readonly id: string;
|
|
761
|
+
/** Scalar outcome in [0, 1]; `passThreshold` splits passing from failing. */
|
|
762
|
+
readonly score: number;
|
|
763
|
+
readonly calls: readonly RolloutCall[];
|
|
764
|
+
}
|
|
765
|
+
interface RolloutArgumentDiffOptions {
|
|
766
|
+
/** Rollouts with `score >= passThreshold` count as passing. Default 1. */
|
|
767
|
+
readonly passThreshold?: number;
|
|
768
|
+
/** Max distinct values listed per field per side in the rendered text. Default 4. */
|
|
769
|
+
readonly maxValuesPerField?: number;
|
|
770
|
+
}
|
|
771
|
+
interface RolloutArgumentDiff {
|
|
772
|
+
/** Human/LLM-readable per-field diff, one line per field. */
|
|
773
|
+
readonly text: string;
|
|
774
|
+
/** Lowercased stringified argument values seen in passing rollouts. */
|
|
775
|
+
readonly passingValues: ReadonlySet<string>;
|
|
776
|
+
/** Lowercased stringified argument values seen in failing rollouts. */
|
|
777
|
+
readonly failingValues: ReadonlySet<string>;
|
|
778
|
+
}
|
|
779
|
+
/**
|
|
780
|
+
* Deterministic per-field diff of call arguments between passing and failing
|
|
781
|
+
* rollouts. A field set by failing rollouts but left unset by passing ones is
|
|
782
|
+
* the classic poison-input signature; a field whose values differ across the
|
|
783
|
+
* split points at the correct value. Feed `text` to the reviser verbatim.
|
|
784
|
+
*/
|
|
785
|
+
declare function rolloutArgumentDiff(rollouts: readonly ScoredRollout[], opts?: RolloutArgumentDiffOptions): RolloutArgumentDiff;
|
|
786
|
+
interface UngroundedLiteralReport {
|
|
787
|
+
/** Quoted single-word literals in the text that no passing rollout used. */
|
|
788
|
+
readonly ungrounded: readonly string[];
|
|
789
|
+
/** The subset failing rollouts actually used - prescribing these is proven harmful. */
|
|
790
|
+
readonly harmful: readonly string[];
|
|
791
|
+
}
|
|
792
|
+
/**
|
|
793
|
+
* Scan revised artifact text for single-quoted single-word literals (the
|
|
794
|
+
* "use exactly 'new'" pattern) that appear in no passing rollout's argument
|
|
795
|
+
* values. Multi-word quotes pass (they are prose, not prescriptions).
|
|
796
|
+
* Callers should reject on `harmful` (with a bounded retry) and at most log
|
|
797
|
+
* `ungrounded` - see the module header for why the severities differ.
|
|
798
|
+
*/
|
|
799
|
+
declare function classifyUngroundedLiterals(text: string, diff: Pick<RolloutArgumentDiff, 'passingValues' | 'failingValues'>): UngroundedLiteralReport;
|
|
800
|
+
|
|
730
801
|
/**
|
|
731
802
|
* Filesystem `LabeledScenarioStore` adapter. The default capture sink for
|
|
732
803
|
* traces + eval artifacts. Production deployments typically swap for a
|
|
@@ -2018,6 +2089,76 @@ interface CampaignBreakdown {
|
|
|
2018
2089
|
* on: mean score per judge dimension + per-scenario composite. */
|
|
2019
2090
|
declare function campaignBreakdown<TArtifact, TScenario extends Scenario>(campaign: CampaignResult<TArtifact, TScenario>): CampaignBreakdown;
|
|
2020
2091
|
|
|
2092
|
+
/**
|
|
2093
|
+
* Single-run lock for evaluations that share one mutable environment.
|
|
2094
|
+
*
|
|
2095
|
+
* Two concurrent runs against a shared stateful gym silently corrupt each
|
|
2096
|
+
* other: each resets/mutates environment state mid-cell of the other, and
|
|
2097
|
+
* every score from both becomes garbage that LOOKS like worker variance
|
|
2098
|
+
* (agent-lab R357 burned hours on flip-flopping scores before tracing them
|
|
2099
|
+
* to exactly this). The fix is a pid lockfile: refuse to start while a live
|
|
2100
|
+
* holder exists, reclaim stale locks whose pid is gone, release only if the
|
|
2101
|
+
* lock is still ours.
|
|
2102
|
+
*
|
|
2103
|
+
* `alsoCheck` exists because independent runners can guard the same shared
|
|
2104
|
+
* resource with differently named lockfiles; a runner must respect all of
|
|
2105
|
+
* them even though it writes only its own.
|
|
2106
|
+
*/
|
|
2107
|
+
interface SingleRunLockOptions {
|
|
2108
|
+
/** Lockfile this runner writes (and checks). */
|
|
2109
|
+
readonly lockPath: string;
|
|
2110
|
+
/** Other runners' lockfiles guarding the same resource; checked, never written. */
|
|
2111
|
+
readonly alsoCheck?: readonly string[];
|
|
2112
|
+
/** Install a process 'exit' hook that releases the lock. Default true. */
|
|
2113
|
+
readonly releaseOnExit?: boolean;
|
|
2114
|
+
/** Owner pid recorded in the lockfile. Default process.pid. */
|
|
2115
|
+
readonly pid?: number;
|
|
2116
|
+
}
|
|
2117
|
+
interface SingleRunLock {
|
|
2118
|
+
/** Remove the lockfile if this process still owns it. Idempotent. */
|
|
2119
|
+
release(): void;
|
|
2120
|
+
}
|
|
2121
|
+
/**
|
|
2122
|
+
* Acquire the lock or throw naming the live holder. A stale lock (holder pid
|
|
2123
|
+
* no longer running) is reclaimed silently.
|
|
2124
|
+
*/
|
|
2125
|
+
declare function acquireSingleRunLock(opts: SingleRunLockOptions): SingleRunLock;
|
|
2126
|
+
|
|
2127
|
+
/**
|
|
2128
|
+
* Transient-transport-failure classification for dispatch retry policies.
|
|
2129
|
+
*
|
|
2130
|
+
* When an eval cell dies, the harness must decide: retry (the infrastructure
|
|
2131
|
+
* hiccuped - a 502 storm, an admission-queue rejection, a dropped stream) or
|
|
2132
|
+
* score it (the agent genuinely failed). Getting this wrong corrupts results
|
|
2133
|
+
* in both directions: scoring transport hiccups as failures buries real
|
|
2134
|
+
* effects under noise (agent-lab R353 found 5/30 identical repeats were 502s
|
|
2135
|
+
* scored as task failures), while retrying genuine failures silently drops
|
|
2136
|
+
* the hard cells and inflates every arm.
|
|
2137
|
+
*
|
|
2138
|
+
* Full-duration timeouts are the deliberate knob: on saturated shared
|
|
2139
|
+
* infrastructure a timeout usually means the request never got a slot
|
|
2140
|
+
* (retry it), but on unthrottled infrastructure it means the agent flailed
|
|
2141
|
+
* on the task until the clock ran out (a real score-0). Both readings were
|
|
2142
|
+
* needed in practice within one week, so the classifier takes it as an
|
|
2143
|
+
* option instead of hardcoding either.
|
|
2144
|
+
*/
|
|
2145
|
+
interface TransientFailureOptions {
|
|
2146
|
+
/**
|
|
2147
|
+
* Treat full-duration timeouts ("timeout after 180000ms") as transient.
|
|
2148
|
+
* Enable on saturated shared infrastructure where queue starvation eats
|
|
2149
|
+
* the clock; leave off when the agent had the resources and simply failed.
|
|
2150
|
+
* Default false.
|
|
2151
|
+
*/
|
|
2152
|
+
readonly retryFullDurationTimeouts?: boolean;
|
|
2153
|
+
/** Additional caller-specific transient patterns. */
|
|
2154
|
+
readonly extraPatterns?: readonly RegExp[];
|
|
2155
|
+
}
|
|
2156
|
+
/**
|
|
2157
|
+
* True when the error text describes an infrastructure hiccup that should be
|
|
2158
|
+
* retried rather than scored. Empty/undefined input is not transient.
|
|
2159
|
+
*/
|
|
2160
|
+
declare function isTransientTransportFailure(message: string | null | undefined, opts?: TransientFailureOptions): boolean;
|
|
2161
|
+
|
|
2021
2162
|
/**
|
|
2022
2163
|
* VCS-pluggable worktree adapter. One improvement = one worktree, PR-like
|
|
2023
2164
|
* (multiple commits allowed). A code-tier proposer's `propose()` creates a
|
|
@@ -2075,4 +2216,4 @@ declare function gitWorktreeAdapter(opts: GitWorktreeAdapterOptions): WorktreeAd
|
|
|
2075
2216
|
* as a ref under the adapter's worktree dir. */
|
|
2076
2217
|
declare function resolveWorktreePath(surface: CodeSurface, worktreeDir?: string): string;
|
|
2077
2218
|
|
|
2078
|
-
export { type AcceptedEdit, type AceProposerOptions, type AnalystArtifact, type AnalystScenario, type ApplySkillPatchResult, type BuildAnalystSurfaceDispatchOptions, type CampaignBreakdown, CampaignResult, CampaignRunPlan, CampaignStorage, CodeSurface, type CompareProposersOptions, type DimensionRegression, type DiscriminationScore, DispatchContext, type EvalFixture, type EvalFixtureFile, type EvalFixtureLoadOptions, type EvalFixtureRunPlan, type EvalFixtureScenario, type EvalFixtureValidationMode, type FailureModeRecallJudgeOptions, type FapoAttributionSignals, type FapoEntryConfig, type FapoFailureCluster, type FapoOptimizationLevel, type FapoProposerOptions, type FapoReviewInput, type FapoReviewIssue, type FapoReviewResult, type FapoScopeContract, FsLabeledScenarioStore, type FsLabeledScenarioStoreOptions, Gate, GenerationRecord, type GitWorktreeAdapterOptions, type Governor, type GovernorContext, type GovernorOp, type HaloProposerOptions, type HeldoutSignificance, type HeldoutSignificanceOptions, type HeuristicGovernorOptions, type JsonPrimitive, type JsonValue, JudgeConfig, JudgeScore, LabelTrust, LabeledScenarioRecord, LabeledScenarioSampleArgs, LabeledScenarioSource, LabeledScenarioStore, LabeledScenarioStoreError, LabeledScenarioWrite, Lineage, type LineageEdge, type LineageGraph, type LineageNode, type LineageNodeInput, type LineageStore, type LoadEvalFixtureScenariosOptions, type MemoryCurationProposerOptions, MutableSurface, type NeutralizationGateOptions, type OptimizerEntryConfig, type PairedHoldout, type ParameterCandidate, type ParameterChange, type ParameterSweepProposerOptions, PlanCampaignRunOptions, type PlanEvalFixtureRunOptions, type PlaybackContext, type PlaybackDriver, type PlaybackStep, type PolicyEditProposerOptions, type ProfileDispatchFn, ProfileMatrixError, type ProfileSummary, ProposeContext, type ProposePatchesArgs, ProposedCandidate, type ProposerComparison, type ProposerEntry, type ProposerPairwise, type ProposerScore, type RejectedEdit, RunCampaignOptions, RunImprovementLoopOptions, type RunLineageLoopOptions, type RunLineageLoopResult, type RunLineageLoopSeed, type RunLineageOptions, type RunLineageResult, type RunLineageSeed, type RunLineageStepResult, type RunProfileMatrixOptions, type RunProfileMatrixResult, type RunSkillOptOptions, type RunSkillOptResult, Scenario, type ScenarioRollup, type ScenarioSignal, type ScoreboardRenderOptions, type ScoreboardRow, type ScoreboardSummary, type SequentialDecideFn, type SequentialDecideOptions, type SequentialDecision, type SequentialObservation, type SequentialPairedGate, type SequentialPairedGateOptions, type SkillOptEpochRecord, type SkillOptEvidence, type SkillOptProposer, type SkillOptProposerOptions, type SkillPatch, type SkillPatchOp, SkillPatchParseError, type SkillPatchRejection, SurfaceProposer, type SurfaceScore, type TraceAnalystProposerOptions, type UserStory, type UserStoryVerdict, type Worktree, type WorktreeAdapter, WorktreeAdapterError, aceProposer, applySkillPatch, buildAnalystSurfaceDispatch, callbackGovernor, campaignBreakdown, campaignMeanComposite, compareProposers, detectScale, dimensionRegressions, discoverEvalFixtures, extractFapoAttributionSignals, failureModeRecallJudge, fapoEscalationEntry, fapoProposer, fsLineageStore, gepaParetoEntry, gepaReflectionEntry, gitWorktreeAdapter, haloProposer, heldoutSignificance, heuristicGovernor, lineageNodeId, loadEvalFixture, loadEvalFixtureScenarios, makePlaybackDispatch, memLineageStore, memoryCurationProposer, neutralizationGate, neutralizeText, pairHoldout, parameterSweepProposer, parseSkillPatchResponse, patchEditCount, planEvalFixtureRun, policyEditProposer, renderScoreboardMarkdown, resolveRunDir, resolveWorktreePath, runLineage, runLineageLoop, runProfileMatrix, runSkillOpt, scoreDiscrimination, scoreUserStory, scoreboardSummary, selectDiscriminative, sequentialDecide, sequentialPairedGate, skillOptEntry, skillOptProposer, tangleTracesRoot, traceAnalystProposer, userStoryScoreboard };
|
|
2219
|
+
export { type AcceptedEdit, type AceProposerOptions, type AnalystArtifact, type AnalystScenario, type ApplySkillPatchResult, type BuildAnalystSurfaceDispatchOptions, type CampaignBreakdown, CampaignResult, CampaignRunPlan, CampaignStorage, CodeSurface, type CompareProposersOptions, type DimensionRegression, type DiscriminationScore, DispatchContext, type EvalFixture, type EvalFixtureFile, type EvalFixtureLoadOptions, type EvalFixtureRunPlan, type EvalFixtureScenario, type EvalFixtureValidationMode, type FailureModeRecallJudgeOptions, type FapoAttributionSignals, type FapoEntryConfig, type FapoFailureCluster, type FapoOptimizationLevel, type FapoProposerOptions, type FapoReviewInput, type FapoReviewIssue, type FapoReviewResult, type FapoScopeContract, FsLabeledScenarioStore, type FsLabeledScenarioStoreOptions, Gate, GenerationRecord, type GitWorktreeAdapterOptions, type Governor, type GovernorContext, type GovernorOp, type HaloProposerOptions, type HeldoutSignificance, type HeldoutSignificanceOptions, type HeuristicGovernorOptions, type JsonPrimitive, type JsonValue, JudgeConfig, JudgeScore, LabelTrust, LabeledScenarioRecord, LabeledScenarioSampleArgs, LabeledScenarioSource, LabeledScenarioStore, LabeledScenarioStoreError, LabeledScenarioWrite, Lineage, type LineageEdge, type LineageGraph, type LineageNode, type LineageNodeInput, type LineageStore, type LoadEvalFixtureScenariosOptions, type MemoryCurationProposerOptions, MutableSurface, type NeutralizationGateOptions, type OptimizerEntryConfig, type PairedHoldout, type ParameterCandidate, type ParameterChange, type ParameterSweepProposerOptions, PlanCampaignRunOptions, type PlanEvalFixtureRunOptions, type PlaybackContext, type PlaybackDriver, type PlaybackStep, type PolicyEditProposerOptions, type ProfileDispatchFn, ProfileMatrixError, type ProfileSummary, ProposeContext, type ProposePatchesArgs, ProposedCandidate, type ProposerComparison, type ProposerEntry, type ProposerPairwise, type ProposerScore, type RejectedEdit, type RolloutArgumentDiff, type RolloutArgumentDiffOptions, type RolloutCall, RunCampaignOptions, RunImprovementLoopOptions, type RunLineageLoopOptions, type RunLineageLoopResult, type RunLineageLoopSeed, type RunLineageOptions, type RunLineageResult, type RunLineageSeed, type RunLineageStepResult, type RunProfileMatrixOptions, type RunProfileMatrixResult, type RunSkillOptOptions, type RunSkillOptResult, Scenario, type ScenarioRollup, type ScenarioSignal, type ScoreboardRenderOptions, type ScoreboardRow, type ScoreboardSummary, type ScoredRollout, type SequentialDecideFn, type SequentialDecideOptions, type SequentialDecision, type SequentialObservation, type SequentialPairedGate, type SequentialPairedGateOptions, type SingleRunLock, type SingleRunLockOptions, type SkillOptEpochRecord, type SkillOptEvidence, type SkillOptProposer, type SkillOptProposerOptions, type SkillPatch, type SkillPatchOp, SkillPatchParseError, type SkillPatchRejection, SurfaceProposer, type SurfaceScore, type TraceAnalystProposerOptions, type TransientFailureOptions, type UngroundedLiteralReport, type UserStory, type UserStoryVerdict, type Worktree, type WorktreeAdapter, WorktreeAdapterError, aceProposer, acquireSingleRunLock, applySkillPatch, buildAnalystSurfaceDispatch, callbackGovernor, campaignBreakdown, campaignMeanComposite, classifyUngroundedLiterals, compareProposers, detectScale, dimensionRegressions, discoverEvalFixtures, extractFapoAttributionSignals, failureModeRecallJudge, fapoEscalationEntry, fapoProposer, fsLineageStore, gepaParetoEntry, gepaReflectionEntry, gitWorktreeAdapter, haloProposer, heldoutSignificance, heuristicGovernor, isTransientTransportFailure, lineageNodeId, loadEvalFixture, loadEvalFixtureScenarios, makePlaybackDispatch, memLineageStore, memoryCurationProposer, neutralizationGate, neutralizeText, pairHoldout, parameterSweepProposer, parseSkillPatchResponse, patchEditCount, planEvalFixtureRun, policyEditProposer, renderScoreboardMarkdown, resolveRunDir, resolveWorktreePath, rolloutArgumentDiff, runLineage, runLineageLoop, runProfileMatrix, runSkillOpt, scoreDiscrimination, scoreUserStory, scoreboardSummary, selectDiscriminative, sequentialDecide, sequentialPairedGate, skillOptEntry, skillOptProposer, tangleTracesRoot, traceAnalystProposer, userStoryScoreboard };
|
package/dist/campaign/index.js
CHANGED
|
@@ -6,9 +6,11 @@ import {
|
|
|
6
6
|
SkillPatchParseError,
|
|
7
7
|
WorktreeAdapterError,
|
|
8
8
|
aceProposer,
|
|
9
|
+
acquireSingleRunLock,
|
|
9
10
|
applySkillPatch,
|
|
10
11
|
buildAnalystSurfaceDispatch,
|
|
11
12
|
callbackGovernor,
|
|
13
|
+
classifyUngroundedLiterals,
|
|
12
14
|
compareProposers,
|
|
13
15
|
discoverEvalFixtures,
|
|
14
16
|
extractFapoAttributionSignals,
|
|
@@ -21,6 +23,7 @@ import {
|
|
|
21
23
|
gitWorktreeAdapter,
|
|
22
24
|
haloProposer,
|
|
23
25
|
heuristicGovernor,
|
|
26
|
+
isTransientTransportFailure,
|
|
24
27
|
lineageNodeId,
|
|
25
28
|
llmJudge,
|
|
26
29
|
loadEvalFixture,
|
|
@@ -37,6 +40,7 @@ import {
|
|
|
37
40
|
policyEditProposer,
|
|
38
41
|
renderScoreboardMarkdown,
|
|
39
42
|
resolveWorktreePath,
|
|
43
|
+
rolloutArgumentDiff,
|
|
40
44
|
runLineage,
|
|
41
45
|
runLineageLoop,
|
|
42
46
|
runProfileMatrix,
|
|
@@ -51,7 +55,7 @@ import {
|
|
|
51
55
|
skillOptProposer,
|
|
52
56
|
traceAnalystProposer,
|
|
53
57
|
userStoryScoreboard
|
|
54
|
-
} from "../chunk-
|
|
58
|
+
} from "../chunk-RNB2NICW.js";
|
|
55
59
|
import {
|
|
56
60
|
buildEvidenceVector,
|
|
57
61
|
buildLoopProvenanceRecord,
|
|
@@ -84,7 +88,8 @@ import {
|
|
|
84
88
|
runOptimization,
|
|
85
89
|
surfaceContentHash,
|
|
86
90
|
surfaceHash
|
|
87
|
-
} from "../chunk-
|
|
91
|
+
} from "../chunk-GTERJI6Q.js";
|
|
92
|
+
import "../chunk-VI2UW6B6.js";
|
|
88
93
|
import {
|
|
89
94
|
fsCampaignStorage,
|
|
90
95
|
inMemoryCampaignStorage,
|
|
@@ -93,20 +98,18 @@ import {
|
|
|
93
98
|
runCampaign,
|
|
94
99
|
tangleTracesRoot
|
|
95
100
|
} from "../chunk-3PFZBGMR.js";
|
|
96
|
-
import "../chunk-
|
|
97
|
-
import "../chunk-
|
|
98
|
-
import "../chunk-2OGPXHOB.js";
|
|
101
|
+
import "../chunk-2IY4ILP4.js";
|
|
102
|
+
import "../chunk-7NX6ZSBG.js";
|
|
99
103
|
import "../chunk-N22ZO7FV.js";
|
|
100
|
-
import "../chunk-FUCQVFMU.js";
|
|
101
|
-
import "../chunk-45EEMHTC.js";
|
|
102
104
|
import "../chunk-XMSYF4A7.js";
|
|
103
105
|
import "../chunk-RPDDVKI7.js";
|
|
104
106
|
import "../chunk-GGE4NNQT.js";
|
|
105
107
|
import "../chunk-LNQEP766.js";
|
|
106
|
-
import "../chunk-PC4UYEBM.js";
|
|
107
108
|
import "../chunk-VK6HBGAE.js";
|
|
108
109
|
import "../chunk-XJYR7XFV.js";
|
|
109
110
|
import "../chunk-VSMTAMNK.js";
|
|
111
|
+
import "../chunk-FUCQVFMU.js";
|
|
112
|
+
import "../chunk-PC4UYEBM.js";
|
|
110
113
|
import "../chunk-ONWEPEDO.js";
|
|
111
114
|
import "../chunk-PZ5AY32C.js";
|
|
112
115
|
export {
|
|
@@ -117,6 +120,7 @@ export {
|
|
|
117
120
|
SkillPatchParseError,
|
|
118
121
|
WorktreeAdapterError,
|
|
119
122
|
aceProposer,
|
|
123
|
+
acquireSingleRunLock,
|
|
120
124
|
applySkillPatch,
|
|
121
125
|
buildAnalystSurfaceDispatch,
|
|
122
126
|
buildEvidenceVector,
|
|
@@ -124,6 +128,7 @@ export {
|
|
|
124
128
|
callbackGovernor,
|
|
125
129
|
campaignBreakdown,
|
|
126
130
|
campaignMeanComposite,
|
|
131
|
+
classifyUngroundedLiterals,
|
|
127
132
|
compareProposers,
|
|
128
133
|
composeGate,
|
|
129
134
|
countSentenceEdits,
|
|
@@ -151,6 +156,7 @@ export {
|
|
|
151
156
|
heuristicGovernor,
|
|
152
157
|
inMemoryCampaignStorage,
|
|
153
158
|
isProposedCandidate,
|
|
159
|
+
isTransientTransportFailure,
|
|
154
160
|
labelTrustRank,
|
|
155
161
|
lineageNodeId,
|
|
156
162
|
llmJudge,
|
|
@@ -178,6 +184,7 @@ export {
|
|
|
178
184
|
renderScoreboardMarkdown,
|
|
179
185
|
resolveRunDir,
|
|
180
186
|
resolveWorktreePath,
|
|
187
|
+
rolloutArgumentDiff,
|
|
181
188
|
runCampaign,
|
|
182
189
|
runEval,
|
|
183
190
|
runImprovementLoop,
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
import {
|
|
2
2
|
parseFindingSubject
|
|
3
|
-
} from "./chunk-
|
|
3
|
+
} from "./chunk-7NX6ZSBG.js";
|
|
4
4
|
import {
|
|
5
5
|
validateAgentProfileCell
|
|
6
6
|
} from "./chunk-XJYR7XFV.js";
|
|
@@ -661,4 +661,4 @@ export {
|
|
|
661
661
|
aggregateRunScore,
|
|
662
662
|
clamp012 as clamp01
|
|
663
663
|
};
|
|
664
|
-
//# sourceMappingURL=chunk-
|
|
664
|
+
//# sourceMappingURL=chunk-2IY4ILP4.js.map
|
|
@@ -1,6 +1,3 @@
|
|
|
1
|
-
import {
|
|
2
|
-
assertLlmRoute
|
|
3
|
-
} from "./chunk-FUCQVFMU.js";
|
|
4
1
|
import {
|
|
5
2
|
researchReport
|
|
6
3
|
} from "./chunk-6MFFKPZ4.js";
|
|
@@ -9,14 +6,11 @@ import {
|
|
|
9
6
|
assertRunCaptured
|
|
10
7
|
} from "./chunk-TT4KNT67.js";
|
|
11
8
|
import {
|
|
12
|
-
|
|
13
|
-
} from "./chunk-
|
|
9
|
+
TraceEmitter
|
|
10
|
+
} from "./chunk-TVVP3ZZQ.js";
|
|
14
11
|
import {
|
|
15
12
|
validateRunRecord
|
|
16
13
|
} from "./chunk-VK6HBGAE.js";
|
|
17
|
-
import {
|
|
18
|
-
TraceEmitter
|
|
19
|
-
} from "./chunk-TVVP3ZZQ.js";
|
|
20
14
|
import {
|
|
21
15
|
buildAgentProfileCell,
|
|
22
16
|
verifyAgentProfileCell
|
|
@@ -25,6 +19,12 @@ import {
|
|
|
25
19
|
canonicalize,
|
|
26
20
|
hashJson
|
|
27
21
|
} from "./chunk-VSMTAMNK.js";
|
|
22
|
+
import {
|
|
23
|
+
assertLlmRoute
|
|
24
|
+
} from "./chunk-FUCQVFMU.js";
|
|
25
|
+
import {
|
|
26
|
+
FileSystemRawProviderSink
|
|
27
|
+
} from "./chunk-PC4UYEBM.js";
|
|
28
28
|
|
|
29
29
|
// src/eval-campaign.ts
|
|
30
30
|
var DEFAULT_INTEGRITY = {
|
|
@@ -354,4 +354,4 @@ function defaultRunId(params) {
|
|
|
354
354
|
export {
|
|
355
355
|
runEvalCampaign
|
|
356
356
|
};
|
|
357
|
-
//# sourceMappingURL=chunk-
|
|
357
|
+
//# sourceMappingURL=chunk-6PL5MGDL.js.map
|
|
@@ -1,13 +1,40 @@
|
|
|
1
|
-
import {
|
|
2
|
-
callLlm
|
|
3
|
-
} from "./chunk-FUCQVFMU.js";
|
|
4
|
-
import {
|
|
5
|
-
makeFinding
|
|
6
|
-
} from "./chunk-45EEMHTC.js";
|
|
7
1
|
import {
|
|
8
2
|
TraceFileMissingError,
|
|
9
3
|
buildTraceAnalystTools
|
|
10
4
|
} from "./chunk-LNQEP766.js";
|
|
5
|
+
import {
|
|
6
|
+
callLlm
|
|
7
|
+
} from "./chunk-FUCQVFMU.js";
|
|
8
|
+
|
|
9
|
+
// src/analyst/types.ts
|
|
10
|
+
import { createHash } from "crypto";
|
|
11
|
+
function computeFindingId(input) {
|
|
12
|
+
const basis = JSON.stringify({
|
|
13
|
+
a: input.analyst_id,
|
|
14
|
+
r: input.area,
|
|
15
|
+
s: input.subject ?? "",
|
|
16
|
+
c: normalizeClaim(input.id_basis ?? input.claim)
|
|
17
|
+
});
|
|
18
|
+
return `f_${createHash("sha256").update(basis).digest("hex").slice(0, 20)}`;
|
|
19
|
+
}
|
|
20
|
+
function normalizeClaim(c) {
|
|
21
|
+
return c.toLowerCase().replace(/\s+/g, " ").replace(/[.!?;:,]+$/g, "").trim();
|
|
22
|
+
}
|
|
23
|
+
function makeFinding(init) {
|
|
24
|
+
const { id_basis, produced_at, ...rest } = init;
|
|
25
|
+
return {
|
|
26
|
+
schema_version: "1.0.0",
|
|
27
|
+
finding_id: computeFindingId({
|
|
28
|
+
analyst_id: rest.analyst_id,
|
|
29
|
+
area: rest.area,
|
|
30
|
+
subject: rest.subject,
|
|
31
|
+
claim: rest.claim,
|
|
32
|
+
id_basis
|
|
33
|
+
}),
|
|
34
|
+
produced_at: produced_at ?? (/* @__PURE__ */ new Date()).toISOString(),
|
|
35
|
+
...rest
|
|
36
|
+
};
|
|
37
|
+
}
|
|
11
38
|
|
|
12
39
|
// src/analyst/finding-subject.ts
|
|
13
40
|
import { z } from "zod";
|
|
@@ -992,6 +1019,8 @@ function selectPriorFindings(source, analystId) {
|
|
|
992
1019
|
}
|
|
993
1020
|
|
|
994
1021
|
export {
|
|
1022
|
+
computeFindingId,
|
|
1023
|
+
makeFinding,
|
|
995
1024
|
FINDING_SUBJECT_KINDS,
|
|
996
1025
|
parseFindingSubject,
|
|
997
1026
|
renderFindingSubject,
|
|
@@ -1016,4 +1045,4 @@ export {
|
|
|
1016
1045
|
DEFAULT_TRACE_ANALYST_KINDS,
|
|
1017
1046
|
AnalystRegistry
|
|
1018
1047
|
};
|
|
1019
|
-
//# sourceMappingURL=chunk-
|
|
1048
|
+
//# sourceMappingURL=chunk-7NX6ZSBG.js.map
|