@tangle-network/agent-eval 0.145.20 → 0.145.22
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +28 -0
- package/dist/analyst/index.d.ts +13 -13
- package/dist/analyst/index.js +4 -4
- package/dist/{backend-integrity-BffcHGdm.d.ts → backend-integrity-HVmWyOaX.d.ts} +2 -2
- package/dist/{backend-integrity-BffcHGdm.d.ts.map → backend-integrity-HVmWyOaX.d.ts.map} +1 -1
- package/dist/{benchmark-DbiZbPdH.d.ts → benchmark-DTQ1RApl.d.ts} +3 -3
- package/dist/{benchmark-DbiZbPdH.d.ts.map → benchmark-DTQ1RApl.d.ts.map} +1 -1
- package/dist/{benchmark-command-CkgLXHoN.js → benchmark-command-CDB_yKqS.js} +7 -7
- package/dist/{benchmark-command-CkgLXHoN.js.map → benchmark-command-CDB_yKqS.js.map} +1 -1
- package/dist/benchmarks/index.d.ts +4 -4
- package/dist/benchmarks/index.js +3 -3
- package/dist/campaign/index.d.ts +8 -8
- package/dist/campaign/index.js +5 -5
- package/dist/{campaign-Ct6r9oNe.js → campaign-CbQXcARJ.js} +6 -6
- package/dist/{campaign-Ct6r9oNe.js.map → campaign-CbQXcARJ.js.map} +1 -1
- package/dist/{capture-fetch-hykwHfDI.d.ts → capture-fetch-BZiO2cEH.d.ts} +2 -2
- package/dist/{capture-fetch-hykwHfDI.d.ts.map → capture-fetch-BZiO2cEH.d.ts.map} +1 -1
- package/dist/{chat-client-Bp4Ebuuc.js → chat-client-BJmwfnjN.js} +2 -2
- package/dist/{chat-client-Bp4Ebuuc.js.map → chat-client-BJmwfnjN.js.map} +1 -1
- package/dist/cli.js +2 -2
- package/dist/{client-D_TIV9pJ.d.ts → client-YDkU5QST.d.ts} +2 -2
- package/dist/{client-D_TIV9pJ.d.ts.map → client-YDkU5QST.d.ts.map} +1 -1
- package/dist/contract/index.d.ts +10 -10
- package/dist/contract/index.js +5 -5
- package/dist/{default-registry-cpHtP2Og.d.ts → default-registry-DONui3mA.d.ts} +6 -6
- package/dist/{default-registry-cpHtP2Og.d.ts.map → default-registry-DONui3mA.d.ts.map} +1 -1
- package/dist/{define-agent-eval-BXusiVTQ.js → define-agent-eval-BGxmy4W6.js} +3 -3
- package/dist/{define-agent-eval-BXusiVTQ.js.map → define-agent-eval-BGxmy4W6.js.map} +1 -1
- package/dist/{define-agent-eval-BgG5DIOS.d.ts → define-agent-eval-DrpgTVPp.d.ts} +5 -5
- package/dist/{define-agent-eval-BgG5DIOS.d.ts.map → define-agent-eval-DrpgTVPp.d.ts.map} +1 -1
- package/dist/{dspy-rlm-engine-CqwhQQBw.js → dspy-rlm-engine-B2nst4NR.js} +2 -2
- package/dist/{dspy-rlm-engine-CqwhQQBw.js.map → dspy-rlm-engine-B2nst4NR.js.map} +1 -1
- package/dist/{engine-DJqRKbhs.d.ts → engine-Bc3seRPe.d.ts} +4 -4
- package/dist/{engine-DJqRKbhs.d.ts.map → engine-Bc3seRPe.d.ts.map} +1 -1
- package/dist/{eval-campaign-aNqpefCS.js → eval-campaign-DeGLwACc.js} +2 -2
- package/dist/{eval-campaign-aNqpefCS.js.map → eval-campaign-DeGLwACc.js.map} +1 -1
- package/dist/{exact-types-CzbhVDr2.d.ts → exact-types-CYFbvA5U.d.ts} +2 -2
- package/dist/{exact-types-CzbhVDr2.d.ts.map → exact-types-CYFbvA5U.d.ts.map} +1 -1
- package/dist/experiment/index.d.ts +2 -2
- package/dist/{external-optimizer-contracts-lixrOZdX.d.ts → external-optimizer-contracts-CKEY98bK.d.ts} +2 -2
- package/dist/{external-optimizer-contracts-lixrOZdX.d.ts.map → external-optimizer-contracts-CKEY98bK.d.ts.map} +1 -1
- package/dist/{external-optimizer-process-D-9iX64j.js → external-optimizer-process-DYNizz8W.js} +2 -2
- package/dist/{external-optimizer-process-D-9iX64j.js.map → external-optimizer-process-DYNizz8W.js.map} +1 -1
- package/dist/{external-optimizer-subprocess-x5AbhAYX.js → external-optimizer-subprocess-r6dP9DAj.js} +2 -2
- package/dist/{external-optimizer-subprocess-x5AbhAYX.js.map → external-optimizer-subprocess-r6dP9DAj.js.map} +1 -1
- package/dist/{feedback-trajectory-D9vYSob_.d.ts → feedback-trajectory-Cg9EDxKB.d.ts} +3 -3
- package/dist/{feedback-trajectory-D9vYSob_.d.ts.map → feedback-trajectory-Cg9EDxKB.d.ts.map} +1 -1
- package/dist/hosted/index.d.ts +2 -2
- package/dist/index-BgHYx1QG.d.ts +1 -0
- package/dist/{index-YrUFx3FU.d.ts → index-CevPFhDT.d.ts} +7 -7
- package/dist/{index-YrUFx3FU.d.ts.map → index-CevPFhDT.d.ts.map} +1 -1
- package/dist/{index-CM-e2LiV.d.ts → index-DGrcBMbC.d.ts} +9 -9
- package/dist/{index-CM-e2LiV.d.ts.map → index-DGrcBMbC.d.ts.map} +1 -1
- package/dist/index.d.ts +24 -25
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +18 -19
- package/dist/index.js.map +1 -1
- package/dist/{integrity-OrcI9Nau.d.ts → integrity-DDoeWibF.d.ts} +2 -2
- package/dist/{integrity-OrcI9Nau.d.ts.map → integrity-DDoeWibF.d.ts.map} +1 -1
- package/dist/{llm-client-d0-2TT1g.js → llm-client-_UE4fU7Z.js} +45 -2
- package/dist/{llm-client-d0-2TT1g.js.map → llm-client-_UE4fU7Z.js.map} +1 -1
- package/dist/{llm-judge-DqAkIQA1.js → llm-judge-B7ffRF8w.js} +3 -3
- package/dist/{llm-judge-DqAkIQA1.js.map → llm-judge-B7ffRF8w.js.map} +1 -1
- package/dist/meta-eval/index.d.ts +1 -1
- package/dist/metrics-Cl0L1KUy.js.map +1 -1
- package/dist/multishot/index.d.ts +1 -1
- package/dist/openapi.json +1 -1
- package/dist/{produced-state-B9skdU8u.js → produced-state-D3k45S9a.js} +3 -3
- package/dist/{produced-state-B9skdU8u.js.map → produced-state-D3k45S9a.js.map} +1 -1
- package/dist/{promotion-policy-u3wj6w3U.d.ts → promotion-policy-BKi5OK5l.d.ts} +2 -2
- package/dist/{promotion-policy-u3wj6w3U.d.ts.map → promotion-policy-BKi5OK5l.d.ts.map} +1 -1
- package/dist/{provenance-AACfvbUR.d.ts → provenance-Bb7Zmyj2.d.ts} +6 -6
- package/dist/{provenance-AACfvbUR.d.ts.map → provenance-Bb7Zmyj2.d.ts.map} +1 -1
- package/dist/{registry-B_1Frl8a.d.ts → registry-CZgEmVXb.d.ts} +4 -4
- package/dist/{registry-B_1Frl8a.d.ts.map → registry-CZgEmVXb.d.ts.map} +1 -1
- package/dist/{researcher-Du-oniHp.d.ts → researcher-CC305Ed-.d.ts} +3 -3
- package/dist/{researcher-Du-oniHp.d.ts.map → researcher-CC305Ed-.d.ts.map} +1 -1
- package/dist/rl.d.ts +2 -2
- package/dist/rl.js +1 -1
- package/dist/{semantic-concept-judge-C6M-qOeb.js → semantic-concept-judge-BsY2Q0Oq.js} +2 -2
- package/dist/{semantic-concept-judge-C6M-qOeb.js.map → semantic-concept-judge-BsY2Q0Oq.js.map} +1 -1
- package/dist/{series-convergence-D1cL1f-4.d.ts → series-convergence-C9G-GNYK.d.ts} +2 -2
- package/dist/{series-convergence-D1cL1f-4.d.ts.map → series-convergence-C9G-GNYK.d.ts.map} +1 -1
- package/dist/{server-ulsOdrTI.js → server-cQXve8i4.js} +2 -2
- package/dist/{server-ulsOdrTI.js.map → server-cQXve8i4.js.map} +1 -1
- package/dist/{skillopt-optimization-method-DO6OFgcz.d.ts → skillopt-optimization-method-C9MrMgW9.d.ts} +5 -5
- package/dist/{skillopt-optimization-method-DO6OFgcz.d.ts.map → skillopt-optimization-method-C9MrMgW9.d.ts.map} +1 -1
- package/dist/{skillopt-optimization-method-CcWZBaSz.js → skillopt-optimization-method-Cv3K4DaS.js} +4 -4
- package/dist/{skillopt-optimization-method-CcWZBaSz.js.map → skillopt-optimization-method-Cv3K4DaS.js.map} +1 -1
- package/dist/{statistical-heldout-CBGPSZ5X.d.ts → statistical-heldout-fvxtHnKQ.d.ts} +2 -2
- package/dist/{statistical-heldout-CBGPSZ5X.d.ts.map → statistical-heldout-fvxtHnKQ.d.ts.map} +1 -1
- package/dist/{store-tool-spans-C9c4R7ca.d.ts → store-tool-spans-CV0hRsUp.d.ts} +4 -4
- package/dist/{store-tool-spans-C9c4R7ca.d.ts.map → store-tool-spans-CV0hRsUp.d.ts.map} +1 -1
- package/dist/{tool-groups-CZotz-e_.d.ts → tool-groups-CDDXNKhd.d.ts} +3 -3
- package/dist/tool-groups-CDDXNKhd.d.ts.map +1 -0
- package/dist/trace-repair/index.d.ts +1 -1
- package/dist/traces.d.ts +7 -7
- package/dist/{types-C4bSVIr7.d.ts → types-CWcPguwd.d.ts} +2 -2
- package/dist/{types-C4bSVIr7.d.ts.map → types-CWcPguwd.d.ts.map} +1 -1
- package/dist/{types-BjsNDR49.d.ts → types-DABZgDGV.d.ts} +3 -3
- package/dist/{types-BjsNDR49.d.ts.map → types-DABZgDGV.d.ts.map} +1 -1
- package/dist/{types-Cx3YUh2r.d.ts → types-XBRdmbj8.d.ts} +61 -2
- package/dist/types-XBRdmbj8.d.ts.map +1 -0
- package/dist/{types-vXyshMwx.d.ts → types-oAsa-fs4.d.ts} +2 -2
- package/dist/{types-vXyshMwx.d.ts.map → types-oAsa-fs4.d.ts.map} +1 -1
- package/dist/wire/index.d.ts +2 -2
- package/dist/wire/index.js +1 -1
- package/package.json +4 -3
- package/dist/index-BKShPTwZ.d.ts +0 -1
- package/dist/tool-groups-CZotz-e_.d.ts.map +0 -1
- package/dist/types-Cx3YUh2r.d.ts.map +0 -1
package/dist/index.js
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
import { t as __exportAll } from "./rolldown-runtime-8H4AJuhK.js";
|
|
2
2
|
import { i as JudgeError, o as NotFoundError, r as ConfigError, s as ValidationError, t as AgentEvalError } from "./errors-Dngq5h35.js";
|
|
3
3
|
import { r as hashJson, t as canonicalize } from "./pre-registration-DakwTRXk.js";
|
|
4
|
-
import { a as CODING_HARNESSES, c as agentProfileId, d as harnessAxisOf, i as verifyCompletion, l as agentProfileModelId, n as completionVerdict, o as HARNESS_NATIVE_MODEL, r as createLlmCorrectnessChecker, s as agentProfileHash, t as extractProducedState, u as expandProfileAxes } from "./produced-state-
|
|
4
|
+
import { a as CODING_HARNESSES, c as agentProfileId, d as harnessAxisOf, i as verifyCompletion, l as agentProfileModelId, n as completionVerdict, o as HARNESS_NATIVE_MODEL, r as createLlmCorrectnessChecker, s as agentProfileHash, t as extractProducedState, u as expandProfileAxes } from "./produced-state-D3k45S9a.js";
|
|
5
5
|
import { AGENT_PROFILE_KINDS, agentProfileCellHashMaterial, agentProfileCellKey, buildAgentProfileCell, groupRunsByAgentProfileCell, toAgentProfileJson, verifyAgentProfileCell } from "./profile-cell.js";
|
|
6
6
|
import { c as mulberry32 } from "./internal-BDHPCnjk.js";
|
|
7
7
|
import { a as spearmanR, i as ranks, n as partialCredit, o as weightedComposite, r as pearsonR, s as weightedMean, t as confidenceInterval } from "./descriptive-B5MwKfbf.js";
|
|
@@ -15,12 +15,12 @@ import { t as eProcess } from "./sequential-eprocess-CbUt2htw.js";
|
|
|
15
15
|
import { n as iqr, r as welchsTTest } from "./baseline-BhPRQBVn.js";
|
|
16
16
|
import { i as isLlmSpan, r as isJudgeSpan, s as isToolSpan, t as FAILURE_CLASSES } from "./schema-CRhEY1SO.js";
|
|
17
17
|
import { a as judgeSpans, c as runsForScenario, n as argHash } from "./query-Di7eEQ79.js";
|
|
18
|
-
import { i as analyzeRuns, o as checkCanaries, r as selfImprove, t as defineAgentEval } from "./define-agent-eval-
|
|
18
|
+
import { i as analyzeRuns, o as checkCanaries, r as selfImprove, t as defineAgentEval } from "./define-agent-eval-BGxmy4W6.js";
|
|
19
19
|
import { r as observedSplitScore, t as isRealnessGated } from "./reward-nw2xZGZG.js";
|
|
20
20
|
import { a as parseRunRecordSafe, c as validateRunRecord, i as modelHasSnapshot, n as UNKNOWN_MODEL, o as roundTripRunRecord, r as isRunRecord, s as runTaskScore, t as RunRecordValidationError } from "./run-record-D2lDdSAz.js";
|
|
21
21
|
import { a as summaryTable, n as gainHistogram, r as paretoChart } from "./summary-report-Blysd6Z2.js";
|
|
22
22
|
import { n as contentHash, r as fileVerdictCache, t as canonicalJson } from "./verdict-cache-mZf5FEiY.js";
|
|
23
|
-
import { B as DEFAULT_RED_TEAM_CORPUS, F as parseReflectionResponse, H as redTeamReport, K as runCampaign, P as buildReflectionPrompt, Q as summarizeBackendIntegrity, U as scoreRedTeamOutput, V as redTeamDataset, W as runCanaries, X as BackendIntegrityError, Z as assertRealBackend, _ as paretoFrontier, g as dominates, k as surfaceContentHash, t as llmJudge } from "./llm-judge-
|
|
23
|
+
import { B as DEFAULT_RED_TEAM_CORPUS, F as parseReflectionResponse, H as redTeamReport, K as runCampaign, P as buildReflectionPrompt, Q as summarizeBackendIntegrity, U as scoreRedTeamOutput, V as redTeamDataset, W as runCanaries, X as BackendIntegrityError, Z as assertRealBackend, _ as paretoFrontier, g as dominates, k as surfaceContentHash, t as llmJudge } from "./llm-judge-B7ffRF8w.js";
|
|
24
24
|
import { a as resolveModelPricing, i as isModelPriced, n as estimateCost, r as estimateTokens, t as MODEL_PRICING } from "./metrics-Cl0L1KUy.js";
|
|
25
25
|
import { a as CostLedgerPersistenceError, c as costForTokenPricing, i as CostLedger, l as costForUsage, n as CostCallConflictError, o as CostReceiptCaptureError, r as CostCeilingReachedError, s as CostReservationExceededError, t as CostAccountingIncompleteError, u as modelPriceKey } from "./cost-ledger-BSe92yAV.js";
|
|
26
26
|
import { n as REDACTION_VERSION, r as redactString, t as DEFAULT_REDACTION_RULES } from "./redact-7Aq1ukl-.js";
|
|
@@ -32,7 +32,7 @@ import { t as TraceEmitter } from "./emitter-CPBAhxum.js";
|
|
|
32
32
|
import { t as buildTrajectory } from "./trajectory-D_7rLrvE.js";
|
|
33
33
|
import { t as runCounterfactual } from "./counterfactual-lDfCx0Uz.js";
|
|
34
34
|
import { l as createBoundedTraceAnalysisStore, m as TRACE_ANALYST_TRUNCATION_MARKER_PREFIX, p as DEFAULT_TRACE_ANALYST_BUDGETS, t as createTraceAnalyst } from "./kind-factory-CPmSd58s.js";
|
|
35
|
-
import { S as validateAnalystReviewDecisions, _ as analystRunDigest, b as readAnalystReview, g as analystFindingDigest, n as buildDefaultAnalystRegistry, o as DEFAULT_TRACE_ANALYST_KINDS, r as AnalystRegistry, t as createChatClient, u as FAILURE_MODE_KIND_SPEC, v as assertUniqueFindingIds, x as snapshotAnalystRun, y as completedAnalystReviewQuality } from "./chat-client-
|
|
35
|
+
import { S as validateAnalystReviewDecisions, _ as analystRunDigest, b as readAnalystReview, g as analystFindingDigest, n as buildDefaultAnalystRegistry, o as DEFAULT_TRACE_ANALYST_KINDS, r as AnalystRegistry, t as createChatClient, u as FAILURE_MODE_KIND_SPEC, v as assertUniqueFindingIds, x as snapshotAnalystRun, y as completedAnalystReviewQuality } from "./chat-client-BJmwfnjN.js";
|
|
36
36
|
import { OUTPUT_VALUE } from "./trace-attributes.js";
|
|
37
37
|
import { r as extractUsageFromSse, t as extractUsage } from "./extract-usage-BrQ8mCLX.js";
|
|
38
38
|
import { n as InMemoryRawProviderSink, r as NoopRawProviderSink, t as FileSystemRawProviderSink } from "./raw-provider-sink-BQd7mzyT.js";
|
|
@@ -40,14 +40,14 @@ import { S as captureFetchToRawSink, d as inferDomainKeywords, h as analyzeTrace
|
|
|
40
40
|
import { n as assertRunCaptured, t as RunIntegrityError } from "./integrity-Cy9WHAtb.js";
|
|
41
41
|
import { n as InMemoryTraceStore, t as FileSystemTraceStore } from "./store-DNe_Uv1Q.js";
|
|
42
42
|
import { t as packageVersion$1 } from "./package-version-D7lQHt_-.js";
|
|
43
|
-
import { _ as
|
|
44
|
-
import { t as runEvalCampaign } from "./eval-campaign-
|
|
43
|
+
import { C as assertCrossFamily, S as CrossFamilyError, _ as assertCrossFamilyServed, a as assertLlmRoute, b as checkServedModel, c as callLlmJson, d as isTransientLlmError, f as maximumChargeForLlmRequest, g as ServedCrossFamilyError, h as ModelSubstitutionError, l as costReceiptFromLlm, m as stripFencedJson, n as LlmClient, o as backoffMs, p as probeLlm, r as LlmResponseError, s as callLlm, t as LlmCallError, u as costReceiptFromLlmError, v as assertServedModel, w as judgeFamily, x as servedModelAcceptable, y as assertServedModels } from "./llm-client-_UE4fU7Z.js";
|
|
44
|
+
import { t as runEvalCampaign } from "./eval-campaign-DeGLwACc.js";
|
|
45
45
|
import { i as improvementVerdict, n as computeExperimentStats } from "./experiment-tracker-C29gXM4B.js";
|
|
46
46
|
import "./rollout-ytVQ7WT8.js";
|
|
47
47
|
import { t as mintRolloutRows } from "./mint-BV6tLVWl.js";
|
|
48
48
|
import { n as pairedEvalueSequence, t as evaluateInterimReleaseConfidence } from "./sequential-Br0mAPHA.js";
|
|
49
|
-
import { t as createDspyRlmTraceEngine } from "./dspy-rlm-engine-
|
|
50
|
-
import { a as diffFindings, n as runSemanticConceptJudge, r as FindingsStore, t as SEMANTIC_CONCEPT_JUDGE_VERSION } from "./semantic-concept-judge-
|
|
49
|
+
import { t as createDspyRlmTraceEngine } from "./dspy-rlm-engine-B2nst4NR.js";
|
|
50
|
+
import { a as diffFindings, n as runSemanticConceptJudge, r as FindingsStore, t as SEMANTIC_CONCEPT_JUDGE_VERSION } from "./semantic-concept-judge-BsY2Q0Oq.js";
|
|
51
51
|
import { t as analyzeSeries } from "./series-convergence-CjO2QdRW.js";
|
|
52
52
|
import { i as otlpTextToTraceAnalysisStore, n as OtlpFileTraceStore } from "./store-otlp-CsptLYpN.js";
|
|
53
53
|
import { n as evaluateReleaseConfidence, r as bootstrapCi } from "./release-confidence-BknrpBnO.js";
|
|
@@ -2844,11 +2844,11 @@ function describeCompletion(persona, state) {
|
|
|
2844
2844
|
* — exported so harness authors can inspect and regression-test it.
|
|
2845
2845
|
*
|
|
2846
2846
|
* @deprecated A role expressed as a code function can never be optimized.
|
|
2847
|
-
*
|
|
2848
|
-
*
|
|
2847
|
+
* Treat this output as SEED data for a registry-backed directive prompt.
|
|
2848
|
+
* Removal tracked by tangle-network/agent-eval#618.
|
|
2849
2849
|
*/
|
|
2850
2850
|
function buildDriverSystemPrompt(persona, state, productContext = "") {
|
|
2851
|
-
warnDeprecatedOnce("buildDriverSystemPrompt", "buildDriverSystemPrompt is deprecated: roles belong in registry-backed prompt data, not code (tangle-network/agent-
|
|
2851
|
+
warnDeprecatedOnce("buildDriverSystemPrompt", "buildDriverSystemPrompt is deprecated: roles belong in registry-backed prompt data, not code (tangle-network/agent-eval#618).");
|
|
2852
2852
|
const rigor = persona.rigor ?? "demanding";
|
|
2853
2853
|
const expertise = persona.expertise ? ` You are ${persona.expertise}.` : "";
|
|
2854
2854
|
const pressure = persona.pressurePoints && persona.pressurePoints.length > 0 ? `\nA competent ${persona.role} here MUST get the agent to address each of:\n${persona.pressurePoints.map((p) => ` - ${p}`).join("\n")}\nDo NOT hand these to the agent. Probe whether it surfaces them itself. If it misses one, press on exactly that gap until it delivers or demonstrably fails.\n` : "";
|
|
@@ -2878,17 +2878,16 @@ Output ONLY your next message to the agent — in character, first person, no me
|
|
|
2878
2878
|
}
|
|
2879
2879
|
/**
|
|
2880
2880
|
* Decide the simulated user's next turn — the reactive, adversarial
|
|
2881
|
-
* turn-
|
|
2882
|
-
*
|
|
2883
|
-
*
|
|
2884
|
-
* when the simulated professional would sign off.
|
|
2881
|
+
* turn-generator an in-process eval harness uses to drive a multi-shot
|
|
2882
|
+
* conversation. Returns the next user message, or the literal "DONE" when the
|
|
2883
|
+
* simulated professional would sign off.
|
|
2885
2884
|
*
|
|
2886
2885
|
* @deprecated The persona-driver loop becomes a 2-node agent graph (driver
|
|
2887
|
-
* profile + delegates edge)
|
|
2888
|
-
*
|
|
2886
|
+
* profile + delegates edge). Removal tracked by
|
|
2887
|
+
* tangle-network/agent-eval#618.
|
|
2889
2888
|
*/
|
|
2890
2889
|
async function decideNextUserTurn(chat, opts) {
|
|
2891
|
-
warnDeprecatedOnce("decideNextUserTurn", "decideNextUserTurn is deprecated: the persona-driver loop becomes a 2-node agent graph (tangle-network/agent-
|
|
2890
|
+
warnDeprecatedOnce("decideNextUserTurn", "decideNextUserTurn is deprecated: the persona-driver loop becomes a 2-node agent graph (tangle-network/agent-eval#618).");
|
|
2892
2891
|
const { persona, state, history, productContext = "", model = "claude-sonnet-4-6" } = opts;
|
|
2893
2892
|
const lastResponse = history.length > 0 ? history[history.length - 1].content.slice(0, 2e3) : "(no conversation yet — this is the first message)";
|
|
2894
2893
|
const recentHistory = history.slice(-6).map((h) => `${h.role}: ${h.content.slice(0, 500)}`).join("\n\n");
|
|
@@ -7402,6 +7401,6 @@ function rankRows(rows, weights) {
|
|
|
7402
7401
|
})).sort((a, b) => b.mean - a.mean);
|
|
7403
7402
|
}
|
|
7404
7403
|
//#endregion
|
|
7405
|
-
export { AGENT_PROFILE_KINDS, AgentEvalError, AnalystRegistry, BOOTSTRAP_GATE_MIN_N, BackendIntegrityError, BudgetBreachError, BudgetGuard, CODING_HARNESSES, ConfigError, CostAccountingIncompleteError, CostCallConflictError, CostCeilingReachedError, CostLedger, CostLedgerPersistenceError, CostReceiptCaptureError, CostReservationExceededError, CostTracker, CrossFamilyError, DECISION_PAIRED_DELTA_STATISTIC, DEFAULT_PERMUTATIONS, DEFAULT_REDACTION_RULES, DEFAULT_RED_TEAM_CORPUS, DEFAULT_TRACE_ANALYST_BUDGETS, DEFAULT_TRACE_ANALYST_KINDS, ERROR_COUNT_PATTERNS, EquivalenceProtocolError, FAILURE_CLASSES, FAILURE_MODE_KIND_SPEC, FileSystemFeedbackTrajectoryStore, FileSystemRawProviderSink, FileSystemTraceStore, FindingsStore, HARNESS_NATIVE_MODEL, HeldOutGate, InMemoryFeedbackTrajectoryStore, InMemoryRawProviderSink, InMemoryTraceStore, JudgeError, LlmCallError, LlmClient, LlmResponseError, MANN_WHITNEY_EXACT_MAX_STATES, MANN_WHITNEY_EXACT_MAX_WORK, MODEL_PRICING, MultiLayerVerifier, NoopRawProviderSink, NotFoundError, OUTPUT_VALUE, OtlpFileTraceStore, PairwiseSteeringOptimizer, ProductClient, PromptRegistry, REDACTION_VERSION, RunIntegrityError, RunRecordValidationError, SEMANTIC_CONCEPT_JUDGE_VERSION, TRACE_ANALYST_TRUNCATION_MARKER_PREFIX, TraceEmitter, UNKNOWN_MODEL, VERIFICATION_STRATEGIES, VERIFICATION_STRATEGY_SOURCES, ValidationError, WILCOXON_EXACT_MAX_N, acquisitionPlansForKnowledgeGaps, agentProfileCellHashMaterial, agentProfileCellKey, agentProfileHash, agentProfileId, aggregateJudgeVerdicts, aggregateRunScore, analystFindingDigest, analystRunDigest, analystRunToFeedbackTrajectory, analystRunToReviewRequests, analyzeAntiSlop, analyzeRuns, analyzeSeries, analyzeTraces, argHash, assertCapabilityHeadroom, assertCrossFamily, assertLlmRoute, assertNoHiddenLeak, assertProductBenchmarkRun, assertRealBackend, assertRunCaptured, assertSingleBackend, assignFeedbackSplit, benjaminiHochberg, blendHeldout, blockingKnowledgeEval, bonferroni, bootstrapCi, budgetBreachView, buildAgentProfileCell, buildDefaultAnalystRegistry, buildEquivalenceRecord, buildReflectionPrompt, buildTraceInsightContext, buildTraceInsightPrompt, buildTrajectory, calibrateJudge, calibrateJudgeContinuous, callLlm, callLlmJson, canonicalJson, canonicalize, capabilityHeadroom, captureFetchToRawSink, certificationEvidenceDigest, checkCanaries, checkTraceContracts, clamp01, classifyFailure, cliffsDelta, cohensD, comparePairedArms, completionVerdict, computeExperimentStats, computeFindingId, computeToolUseMetrics, confidenceInterval, contentHash, continuousAgreement, controlRunToFeedbackTrajectory, corpusInterRaterAgreement, corpusInterRaterAgreementFromJudgeScores, costForTokenPricing, costForUsage, costReceiptFromLlm, costReceiptFromLlmError, createAntiSlopJudge, createBoundedTraceAnalysisStore, createChatClient, createDspyRlmTraceEngine, createFeedbackTrajectory, createLlmCorrectnessChecker, createLlmReviewer, createTraceAnalyst, decideNextUserTurn, decidePairedPromotion, defaultBlendWeights, defineAgentEval, defineEquivalenceCheck, deployGateLayer, describeTraceInsightScope, diffFindings, diffScorecard, discoverPersonas, domainEvidencePattern, dominates, eProcess, ensembleJudge, equivalenceVerdict, errorStreakDetector, estimateCost, estimateTokens, evaluateActionPolicy, evaluateInterimReleaseConfidence, evaluateOracles, evaluateReleaseConfidence, expandProfileAxes, exportProductBenchmark, exportProductBenchmarkRuns, exportRunAsOtlp, extractErrorCount, extractProducedState, extractUsage, extractUsageFromSse, failureClusterView, feedbackTrajectoriesToDatasetScenarios, feedbackTrajectoriesToOptimizerRows, feedbackTrajectoryToOptimizerRow, fileVerdictCache, formatScorecardDiff, gainHistogram, gateTreatmentApplied, gradeOnHidden, gradeSemanticStatus, groupRunsByAgentProfileCell, harnessAxisOf, hashContent, hashJson, hiddenGrade, holm, improvementVerdict, inMemoryReviewStore, inferDomainKeywords, interRaterReliability, interpretCliffs, iqr, isBinaryOutcomeVector, isJudgeSpan, isLlmSpan, isModelPriced, isRunRecord, isToolSpan, isTransientLlmError, jsonShape, jsonlReviewStore, jsonlRunRecordBackend, judgeAgreementView, judgeFamily, judgeSpans, knowledgeReadinessTracePayload, leaderboard, llmJudge, loadScorecard, localCommandRunner, makeFinding, makeProposalFinding, mannWhitneyU, maximumChargeForLlmRequest, mcnemar, mcnemarPower, mcnemarRequiredN, minimumPairsForPairedDeltaTest, mintRolloutRows, modelHasSnapshot, modelPriceKey, mulberry32, notBlocked, objectiveEval, observeAll, otlpTextToTraceAnalysisStore, pairArms, pairRunRecords, pairedBinaryScale, pairedBootstrap, pairedCohensDz, pairedDeltaTest, pairedDeltaTieFraction, pairedEvalueSequence, pairedMde, pairedRiskDifference, pairedRiskDifferenceExact, pairedRiskDifferenceScore, pairedSignTest, pairedTTest, paretoChart, paretoFrontier, parseReflectionResponse, parseRunRecordSafe, parseRuntimeTrajectoryHookEvent, partialCredit, partitionHeldOut, passAtK, pearsonR, preflightModels, probeLlm, productBenchmarkRepoIdentity, profile_exports as profile, projectRuntimeTrajectoryEvidence, proposeSynthesisTargets, ranks, readProductBenchmarkManifest, recordRuns, recordRunsToScorecard, redTeamDataset, redTeamReport, redactString, regexMatches, renderPreferenceMemoryMarkdown, repeatedActionDetector, requiredPairedSampleSize, requiredSampleSize, resolveModelPricing, resolveSeat, roundTripRunRecord, routeFields, runAgentControlLoop, runCampaign, runCanaries, runCounterfactual, runEquivalenceCheck, runEvalCampaign, runIntentMatchJudge, runKeywordCoverageJudge, runKeywordCoverageJudgeUrl, runProposeReview, runProposeReviewAsControlLoop, runSemanticConceptJudge, runTaskScore, runsForScenario, scoreKnowledgeReadiness, scoreRedTeamOutput, scoreTraceInsightReadiness, seatPresets, selfImprove, spearmanR, stripFencedJson, subjectiveEval, summarizeBackendIntegrity, summarizePreferenceMemory, summaryTable, textInSnapshot, toAgentProfileJson, tokenizeDomainWords, toolSpansToTraceAnalysisStore, toolWasteView, traceContract, urlContains, userQuestionsForKnowledgeGaps, validateRunRecord, verbosityBias, verifyAgentProfileCell, verifyCompletion, viteDeployRunner, weightedComposite, weightedMean, wilcoxonSignedRank, wilson, withAssignedFeedbackSplit, withHeldoutBlend, withJudgeRetry, wranglerDeployRunner };
|
|
7404
|
+
export { AGENT_PROFILE_KINDS, AgentEvalError, AnalystRegistry, BOOTSTRAP_GATE_MIN_N, BackendIntegrityError, BudgetBreachError, BudgetGuard, CODING_HARNESSES, ConfigError, CostAccountingIncompleteError, CostCallConflictError, CostCeilingReachedError, CostLedger, CostLedgerPersistenceError, CostReceiptCaptureError, CostReservationExceededError, CostTracker, CrossFamilyError, DECISION_PAIRED_DELTA_STATISTIC, DEFAULT_PERMUTATIONS, DEFAULT_REDACTION_RULES, DEFAULT_RED_TEAM_CORPUS, DEFAULT_TRACE_ANALYST_BUDGETS, DEFAULT_TRACE_ANALYST_KINDS, ERROR_COUNT_PATTERNS, EquivalenceProtocolError, FAILURE_CLASSES, FAILURE_MODE_KIND_SPEC, FileSystemFeedbackTrajectoryStore, FileSystemRawProviderSink, FileSystemTraceStore, FindingsStore, HARNESS_NATIVE_MODEL, HeldOutGate, InMemoryFeedbackTrajectoryStore, InMemoryRawProviderSink, InMemoryTraceStore, JudgeError, LlmCallError, LlmClient, LlmResponseError, MANN_WHITNEY_EXACT_MAX_STATES, MANN_WHITNEY_EXACT_MAX_WORK, MODEL_PRICING, ModelSubstitutionError, MultiLayerVerifier, NoopRawProviderSink, NotFoundError, OUTPUT_VALUE, OtlpFileTraceStore, PairwiseSteeringOptimizer, ProductClient, PromptRegistry, REDACTION_VERSION, RunIntegrityError, RunRecordValidationError, SEMANTIC_CONCEPT_JUDGE_VERSION, ServedCrossFamilyError, TRACE_ANALYST_TRUNCATION_MARKER_PREFIX, TraceEmitter, UNKNOWN_MODEL, VERIFICATION_STRATEGIES, VERIFICATION_STRATEGY_SOURCES, ValidationError, WILCOXON_EXACT_MAX_N, acquisitionPlansForKnowledgeGaps, agentProfileCellHashMaterial, agentProfileCellKey, agentProfileHash, agentProfileId, aggregateJudgeVerdicts, aggregateRunScore, analystFindingDigest, analystRunDigest, analystRunToFeedbackTrajectory, analystRunToReviewRequests, analyzeAntiSlop, analyzeRuns, analyzeSeries, analyzeTraces, argHash, assertCapabilityHeadroom, assertCrossFamily, assertCrossFamilyServed, assertLlmRoute, assertNoHiddenLeak, assertProductBenchmarkRun, assertRealBackend, assertRunCaptured, assertServedModel, assertServedModels, assertSingleBackend, assignFeedbackSplit, benjaminiHochberg, blendHeldout, blockingKnowledgeEval, bonferroni, bootstrapCi, budgetBreachView, buildAgentProfileCell, buildDefaultAnalystRegistry, buildEquivalenceRecord, buildReflectionPrompt, buildTraceInsightContext, buildTraceInsightPrompt, buildTrajectory, calibrateJudge, calibrateJudgeContinuous, callLlm, callLlmJson, canonicalJson, canonicalize, capabilityHeadroom, captureFetchToRawSink, certificationEvidenceDigest, checkCanaries, checkServedModel, checkTraceContracts, clamp01, classifyFailure, cliffsDelta, cohensD, comparePairedArms, completionVerdict, computeExperimentStats, computeFindingId, computeToolUseMetrics, confidenceInterval, contentHash, continuousAgreement, controlRunToFeedbackTrajectory, corpusInterRaterAgreement, corpusInterRaterAgreementFromJudgeScores, costForTokenPricing, costForUsage, costReceiptFromLlm, costReceiptFromLlmError, createAntiSlopJudge, createBoundedTraceAnalysisStore, createChatClient, createDspyRlmTraceEngine, createFeedbackTrajectory, createLlmCorrectnessChecker, createLlmReviewer, createTraceAnalyst, decideNextUserTurn, decidePairedPromotion, defaultBlendWeights, defineAgentEval, defineEquivalenceCheck, deployGateLayer, describeTraceInsightScope, diffFindings, diffScorecard, discoverPersonas, domainEvidencePattern, dominates, eProcess, ensembleJudge, equivalenceVerdict, errorStreakDetector, estimateCost, estimateTokens, evaluateActionPolicy, evaluateInterimReleaseConfidence, evaluateOracles, evaluateReleaseConfidence, expandProfileAxes, exportProductBenchmark, exportProductBenchmarkRuns, exportRunAsOtlp, extractErrorCount, extractProducedState, extractUsage, extractUsageFromSse, failureClusterView, feedbackTrajectoriesToDatasetScenarios, feedbackTrajectoriesToOptimizerRows, feedbackTrajectoryToOptimizerRow, fileVerdictCache, formatScorecardDiff, gainHistogram, gateTreatmentApplied, gradeOnHidden, gradeSemanticStatus, groupRunsByAgentProfileCell, harnessAxisOf, hashContent, hashJson, hiddenGrade, holm, improvementVerdict, inMemoryReviewStore, inferDomainKeywords, interRaterReliability, interpretCliffs, iqr, isBinaryOutcomeVector, isJudgeSpan, isLlmSpan, isModelPriced, isRunRecord, isToolSpan, isTransientLlmError, jsonShape, jsonlReviewStore, jsonlRunRecordBackend, judgeAgreementView, judgeFamily, judgeSpans, knowledgeReadinessTracePayload, leaderboard, llmJudge, loadScorecard, localCommandRunner, makeFinding, makeProposalFinding, mannWhitneyU, maximumChargeForLlmRequest, mcnemar, mcnemarPower, mcnemarRequiredN, minimumPairsForPairedDeltaTest, mintRolloutRows, modelHasSnapshot, modelPriceKey, mulberry32, notBlocked, objectiveEval, observeAll, otlpTextToTraceAnalysisStore, pairArms, pairRunRecords, pairedBinaryScale, pairedBootstrap, pairedCohensDz, pairedDeltaTest, pairedDeltaTieFraction, pairedEvalueSequence, pairedMde, pairedRiskDifference, pairedRiskDifferenceExact, pairedRiskDifferenceScore, pairedSignTest, pairedTTest, paretoChart, paretoFrontier, parseReflectionResponse, parseRunRecordSafe, parseRuntimeTrajectoryHookEvent, partialCredit, partitionHeldOut, passAtK, pearsonR, preflightModels, probeLlm, productBenchmarkRepoIdentity, profile_exports as profile, projectRuntimeTrajectoryEvidence, proposeSynthesisTargets, ranks, readProductBenchmarkManifest, recordRuns, recordRunsToScorecard, redTeamDataset, redTeamReport, redactString, regexMatches, renderPreferenceMemoryMarkdown, repeatedActionDetector, requiredPairedSampleSize, requiredSampleSize, resolveModelPricing, resolveSeat, roundTripRunRecord, routeFields, runAgentControlLoop, runCampaign, runCanaries, runCounterfactual, runEquivalenceCheck, runEvalCampaign, runIntentMatchJudge, runKeywordCoverageJudge, runKeywordCoverageJudgeUrl, runProposeReview, runProposeReviewAsControlLoop, runSemanticConceptJudge, runTaskScore, runsForScenario, scoreKnowledgeReadiness, scoreRedTeamOutput, scoreTraceInsightReadiness, seatPresets, selfImprove, servedModelAcceptable, spearmanR, stripFencedJson, subjectiveEval, summarizeBackendIntegrity, summarizePreferenceMemory, summaryTable, textInSnapshot, toAgentProfileJson, tokenizeDomainWords, toolSpansToTraceAnalysisStore, toolWasteView, traceContract, urlContains, userQuestionsForKnowledgeGaps, validateRunRecord, verbosityBias, verifyAgentProfileCell, verifyCompletion, viteDeployRunner, weightedComposite, weightedMean, wilcoxonSignedRank, wilson, withAssignedFeedbackSplit, withHeldoutBlend, withJudgeRetry, wranglerDeployRunner };
|
|
7406
7405
|
|
|
7407
7406
|
//# sourceMappingURL=index.js.map
|