@tangle-network/agent-eval 0.139.3 → 0.140.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +14 -0
- package/dist/analyst/index.d.ts +15 -13
- package/dist/analyst/index.d.ts.map +1 -1
- package/dist/analyst/index.js +6 -6
- package/dist/{benchmark-DDVdWcwA.d.ts → benchmark-DxaZfy0w.d.ts} +3 -3
- package/dist/{benchmark-DDVdWcwA.d.ts.map → benchmark-DxaZfy0w.d.ts.map} +1 -1
- package/dist/{benchmark-command-xvi2liH7.js → benchmark-command-CK0UnXAD.js} +54 -33
- package/dist/benchmark-command-CK0UnXAD.js.map +1 -0
- package/dist/benchmarks/index.d.ts +1 -1
- package/dist/benchmarks/index.js +1 -1
- package/dist/{benchmarks-zxhy1QV3.js → benchmarks-HwoBE32G.js} +4 -4
- package/dist/{benchmarks-zxhy1QV3.js.map → benchmarks-HwoBE32G.js.map} +1 -1
- package/dist/campaign/index.d.ts +5 -5
- package/dist/campaign/index.js +3 -3
- package/dist/{campaign-DrS6_hLd.js → campaign-BzjYNYVZ.js} +5 -5
- package/dist/{campaign-DrS6_hLd.js.map → campaign-BzjYNYVZ.js.map} +1 -1
- package/dist/cli.js +2 -2
- package/dist/{client-BohnDFBq.d.ts → client-BoqGxEqx.d.ts} +4 -4
- package/dist/{client-BohnDFBq.d.ts.map → client-BoqGxEqx.d.ts.map} +1 -1
- package/dist/{completion-verifier-IPoP4fQO.d.ts → completion-verifier-D15NHYSk.d.ts} +5 -5
- package/dist/{completion-verifier-IPoP4fQO.d.ts.map → completion-verifier-D15NHYSk.d.ts.map} +1 -1
- package/dist/contract/index.d.ts +10 -10
- package/dist/contract/index.js +6 -6
- package/dist/control.d.ts +2 -2
- package/dist/{cost-ledger-CZ9diLxY.js → cost-ledger-DMFxsLKr.js} +22 -8
- package/dist/cost-ledger-DMFxsLKr.js.map +1 -0
- package/dist/{cost-ledger-DKgyIWRj.d.ts → cost-ledger-FuQvHxPm.d.ts} +8 -2
- package/dist/{cost-ledger-DKgyIWRj.d.ts.map → cost-ledger-FuQvHxPm.d.ts.map} +1 -1
- package/dist/{default-registry-B8vf7Rmf.d.ts → default-registry-Ci7wAAR8.d.ts} +5 -5
- package/dist/{default-registry-B8vf7Rmf.d.ts.map → default-registry-Ci7wAAR8.d.ts.map} +1 -1
- package/dist/{default-registry-BgJJItGr.js → default-registry-DCp-6hc-.js} +3 -3
- package/dist/{default-registry-BgJJItGr.js.map → default-registry-DCp-6hc-.js.map} +1 -1
- package/dist/{dspy-rlm-engine-DTkVyDX-.js → dspy-rlm-engine-Bkak4nzo.js} +11 -4
- package/dist/dspy-rlm-engine-Bkak4nzo.js.map +1 -0
- package/dist/{eval-campaign-BmptJj50.js → eval-campaign-YdkpWWoT.js} +2 -2
- package/dist/{eval-campaign-BmptJj50.js.map → eval-campaign-YdkpWWoT.js.map} +1 -1
- package/dist/{exact-types-MaaFcllV.d.ts → exact-types-B0lJV3tu.d.ts} +2 -2
- package/dist/{exact-types-MaaFcllV.d.ts.map → exact-types-B0lJV3tu.d.ts.map} +1 -1
- package/dist/{external-optimizer-contracts-BrxY2Sli.d.ts → external-optimizer-contracts-nb7c_WAR.d.ts} +12 -2
- package/dist/{external-optimizer-contracts-BrxY2Sli.d.ts.map → external-optimizer-contracts-nb7c_WAR.d.ts.map} +1 -1
- package/dist/{extract-usage-DZs601Va.js → extract-usage-C5vMw-0R.js} +2 -2
- package/dist/{extract-usage-DZs601Va.js.map → extract-usage-C5vMw-0R.js.map} +1 -1
- package/dist/{feedback-trajectory-BJUWOkJM.d.ts → feedback-trajectory-BCHqzLh3.d.ts} +3 -3
- package/dist/{feedback-trajectory-BJUWOkJM.d.ts.map → feedback-trajectory-BCHqzLh3.d.ts.map} +1 -1
- package/dist/fuzz.d.ts +1 -1
- package/dist/fuzz.js +1 -1
- package/dist/hosted/index.d.ts +3 -3
- package/dist/{index-_66rVpwN.d.ts → index-CKI1CXTL.d.ts} +5 -5
- package/dist/{index-_66rVpwN.d.ts.map → index-CKI1CXTL.d.ts.map} +1 -1
- package/dist/{index-CWOPCJiw.d.ts → index-D2enbqA0.d.ts} +2 -2
- package/dist/{index-CWOPCJiw.d.ts.map → index-D2enbqA0.d.ts.map} +1 -1
- package/dist/{index-BTm_P9aC.d.ts → index-DCP4I2Qx.d.ts} +10 -10
- package/dist/{index-BTm_P9aC.d.ts.map → index-DCP4I2Qx.d.ts.map} +1 -1
- package/dist/{index-CtR1xh4V.d.ts → index-DFLVtPZ9.d.ts} +3 -3
- package/dist/{index-CtR1xh4V.d.ts.map → index-DFLVtPZ9.d.ts.map} +1 -1
- package/dist/index.d.ts +23 -23
- package/dist/index.js +13 -13
- package/dist/{insight-report-Bu5Wi9tG.d.ts → insight-report-Bh_8ksel.d.ts} +4 -4
- package/dist/{insight-report-Bu5Wi9tG.d.ts.map → insight-report-Bh_8ksel.d.ts.map} +1 -1
- package/dist/{integrity-COTh3DTH.d.ts → integrity-DRXobPEs.d.ts} +2 -2
- package/dist/{integrity-COTh3DTH.d.ts.map → integrity-DRXobPEs.d.ts.map} +1 -1
- package/dist/{kind-factory-CFxA0JQX.js → kind-factory-DB7nIs35.js} +2 -2
- package/dist/{kind-factory-CFxA0JQX.js.map → kind-factory-DB7nIs35.js.map} +1 -1
- package/dist/{llm-client-bkztEfIx.js → llm-client-B3WXSH5Y.js} +2 -2
- package/dist/{llm-client-bkztEfIx.js.map → llm-client-B3WXSH5Y.js.map} +1 -1
- package/dist/meta-eval/index.d.ts +2 -2
- package/dist/multishot/index.d.ts +2 -2
- package/dist/openapi.json +1 -1
- package/dist/{release-report-fZarvIm-.d.ts → release-report-B_bQOHM-.d.ts} +3 -3
- package/dist/{release-report-fZarvIm-.d.ts.map → release-report-B_bQOHM-.d.ts.map} +1 -1
- package/dist/{replay-DjG4IG60.d.ts → replay-BqTgoioO.d.ts} +6 -6
- package/dist/{replay-DjG4IG60.d.ts.map → replay-BqTgoioO.d.ts.map} +1 -1
- package/dist/{replay-SA4OB7O7.js → replay-k2MsOmv5.js} +4 -4
- package/dist/{replay-SA4OB7O7.js.map → replay-k2MsOmv5.js.map} +1 -1
- package/dist/reporting.d.ts +4 -4
- package/dist/{researcher-BxhtGfKa.d.ts → researcher-C6lzl-rP.d.ts} +5 -5
- package/dist/{researcher-BxhtGfKa.d.ts.map → researcher-C6lzl-rP.d.ts.map} +1 -1
- package/dist/{reward-hacking-CqSLiV51.d.ts → reward-hacking-CEVVmy3h.d.ts} +2 -2
- package/dist/{reward-hacking-CqSLiV51.d.ts.map → reward-hacking-CEVVmy3h.d.ts.map} +1 -1
- package/dist/rl.d.ts +5 -5
- package/dist/rl.js +1 -1
- package/dist/rollout/index.d.ts +1 -1
- package/dist/{rubric-predictive-validity-DQBQj6uV.d.ts → rubric-predictive-validity-C2CthIfY.d.ts} +2 -2
- package/dist/{rubric-predictive-validity-DQBQj6uV.d.ts.map → rubric-predictive-validity-C2CthIfY.d.ts.map} +1 -1
- package/dist/{run-evidence-C4RcRQT5.d.ts → run-evidence-8Ou28QSa.d.ts} +3 -3
- package/dist/{run-evidence-C4RcRQT5.d.ts.map → run-evidence-8Ou28QSa.d.ts.map} +1 -1
- package/dist/{run-record-CztDMXVF.d.ts → run-record-Tb3TTtUn.d.ts} +2 -2
- package/dist/{run-record-CztDMXVF.d.ts.map → run-record-Tb3TTtUn.d.ts.map} +1 -1
- package/dist/{semantic-concept-judge-BuIJ9IfB.js → semantic-concept-judge-DJQtFr95.js} +3 -3
- package/dist/{semantic-concept-judge-BuIJ9IfB.js.map → semantic-concept-judge-DJQtFr95.js.map} +1 -1
- package/dist/{server-DaCpLfi0.js → server-Cu4M3NSO.js} +3 -3
- package/dist/{server-DaCpLfi0.js.map → server-Cu4M3NSO.js.map} +1 -1
- package/dist/{single-run-lock-BTTtPZ9N.js → single-run-lock-CiQThJxB.js} +22 -12
- package/dist/single-run-lock-CiQThJxB.js.map +1 -0
- package/dist/{skill-usage-B-BFS8M2.d.ts → skill-usage-CVVnoIx-.d.ts} +26 -10
- package/dist/skill-usage-CVVnoIx-.d.ts.map +1 -0
- package/dist/{skillopt-optimization-method-BbGnCC53.js → skillopt-optimization-method-CSBQ8Qma.js} +4 -4
- package/dist/{skillopt-optimization-method-BbGnCC53.js.map → skillopt-optimization-method-CSBQ8Qma.js.map} +1 -1
- package/dist/{skillopt-optimization-method-_s0Tub7Y.d.ts → skillopt-optimization-method-D1dqGzzH.d.ts} +11 -11
- package/dist/{skillopt-optimization-method-_s0Tub7Y.d.ts.map → skillopt-optimization-method-D1dqGzzH.d.ts.map} +1 -1
- package/dist/{statistics-B5d0Zd-z.d.ts → statistics-B4u_CiFd.d.ts} +2 -2
- package/dist/{statistics-B5d0Zd-z.d.ts.map → statistics-B4u_CiFd.d.ts.map} +1 -1
- package/dist/{store-otlp-DX4fGIcf.js → store-otlp-vRByAR6h.js} +2 -2
- package/dist/{store-otlp-DX4fGIcf.js.map → store-otlp-vRByAR6h.js.map} +1 -1
- package/dist/{summary-report-Cg7BifAM.d.ts → summary-report-o3eJ3gxG.d.ts} +3 -3
- package/dist/{summary-report-Cg7BifAM.d.ts.map → summary-report-o3eJ3gxG.d.ts.map} +1 -1
- package/dist/{tool-groups-CdYq22lX.d.ts → tool-groups-DVQTy9lq.d.ts} +8 -8
- package/dist/{tool-groups-CdYq22lX.d.ts.map → tool-groups-DVQTy9lq.d.ts.map} +1 -1
- package/dist/traces.d.ts +6 -6
- package/dist/traces.js +4 -4
- package/dist/{types-uPrS6mD-.d.ts → types-BjMFz88h.d.ts} +2 -2
- package/dist/{types-uPrS6mD-.d.ts.map → types-BjMFz88h.d.ts.map} +1 -1
- package/dist/{types-DoEYskCd.d.ts → types-D3jh6F98.d.ts} +4 -4
- package/dist/{types-DoEYskCd.d.ts.map → types-D3jh6F98.d.ts.map} +1 -1
- package/dist/{types-BBFNHxSK.d.ts → types-Dk7PB7vh.d.ts} +5 -5
- package/dist/{types-BBFNHxSK.d.ts.map → types-Dk7PB7vh.d.ts.map} +1 -1
- package/dist/wire/index.d.ts +3 -3
- package/dist/wire/index.js +1 -1
- package/package.json +1 -1
- package/dist/benchmark-command-xvi2liH7.js.map +0 -1
- package/dist/cost-ledger-CZ9diLxY.js.map +0 -1
- package/dist/dspy-rlm-engine-DTkVyDX-.js.map +0 -1
- package/dist/single-run-lock-BTTtPZ9N.js.map +0 -1
- package/dist/skill-usage-B-BFS8M2.d.ts.map +0 -1
package/CHANGELOG.md
CHANGED
|
@@ -4,6 +4,20 @@ All notable changes to `@tangle-network/agent-eval` and its sibling `agent-eval-
|
|
|
4
4
|
|
|
5
5
|
---
|
|
6
6
|
|
|
7
|
+
## [0.140.0] - 2026-07-31 - the recursive engine runs on real providers
|
|
8
|
+
|
|
9
|
+
### Changed
|
|
10
|
+
|
|
11
|
+
- The DSPy bridge takes a selectable control adapter, so a model that answers in prose and fenced code rather than DSPy's `[[ ## field ## ]]` markers no longer voids a completed investigation.
|
|
12
|
+
**The bridge's `analyze` input gains a required `controlAdapter` key**: a Python `agent-eval-rpc` and its npm peer must now be the same version, and a mismatch fails with `analyze input must contain exactly [...]`.
|
|
13
|
+
- A single malformed finding never voids a completed RLM case; rejections are recorded with a reason instead of raised.
|
|
14
|
+
- Raise the DSPy output-token default from 4096 to 16384. 4096 is below what current coding models emit for a full findings array — glm-5.2 through an OpenAI-compatible gateway returns 8192 and the request is rejected outright, so the old default failed before any analysis ran. `maxCostUsd` remains the real spend bound.
|
|
15
|
+
|
|
16
|
+
### Added
|
|
17
|
+
|
|
18
|
+
- First scored run of the recursive DSPy RLM analyst on the pinned 32-case CodeTraceBench corpus: F1 0.3644 over 60 of 64 completed cases at $6.73, against the retired one-shot runner's 0.3673 at $1.21 and CodeTracer's 0.3128.
|
|
19
|
+
Recursion did not improve step localization on this task, and the artifact says so; the engine's claimed value is verified findings on an arbitrary session, which this benchmark does not measure.
|
|
20
|
+
|
|
7
21
|
## [0.139.3] - 2026-07-31 - supervisor runs under `.agent`
|
|
8
22
|
|
|
9
23
|
### Changed
|
package/dist/analyst/index.d.ts
CHANGED
|
@@ -1,14 +1,14 @@
|
|
|
1
1
|
import { a as MultiLayerVerifier, l as VerifyOptions, o as Severity } from "../multi-layer-verifier-BHY1gWAc.js";
|
|
2
|
-
import { A as parseFindingSubject, C as FINDING_SUBJECT_KINDS, D as FindingSubjectStringSchema, E as FindingSubjectKind, F as DefineExactCustomAnalystOptions, G as SemanticConceptJudgeOptions, I as defineCustomAnalyst, L as defineTraceAnalyst, M as DspyRlmTraceEngineOptions, N as createDspyRlmTraceEngine, O as KIND_EXPECTED_SUBJECTS, P as DefineCustomAnalystOptions, S as FINDING_SUBJECT_GRAMMAR_PROMPT, T as FindingSubject, W as SemanticConceptJudgeInput, Y as RunCritic, Z as RunTrace, _ as FindingsDiff, a as SkillUsageScanConfig, b as defaultIsMaterial, c as DEFAULT_TRACE_ANALYST_KINDS, d as IMPROVEMENT_KIND_SPEC, f as FAILURE_MODE_KIND_SPEC, g as DiffPolicy, h as emitControlIntegrityFindings, i as SkillUsageReport, j as renderFindingSubject, k as findingSubjectGrammarPromptFor, l as KNOWLEDGE_POISONING_KIND_SPEC, m as ControlIntegrityAnalyst, n as SkillUsageAnalyst, o as buildSkillUsageReport, p as CONTROL_INTEGRITY_ANALYST, r as SkillUsageRecord, s as emitSkillUsageFindings, t as SKILL_USAGE_ANALYST, u as KNOWLEDGE_GAP_KIND_SPEC, v as FindingsStore, w as FINDING_SUBJECT_SYNTAX, x as diffFindings, y as PersistedFinding } from "../skill-usage-
|
|
3
|
-
import { b as CustomTokenPricing, c as CostLedgerHandle } from "../cost-ledger-
|
|
4
|
-
import { A as ChatClient, B as SandboxSdkTransportOpts, F as CreateChatClientOpts, I as CustomTransportOpts, L as DirectProviderTransportOpts, M as ChatResponse, N as ChatTransport, P as CliBridgeTransportOpts, R as MockTransportOpts, V as createChatClient, j as ChatRequest, k as ChatCallOpts, m as JudgeInput, p as JudgeFn, z as RouterTransportOpts } from "../types-
|
|
5
|
-
import { _ as makeFinding, a as AnalystInputKind, c as AnalystRunInputs, d as AnalystSeverity, f as AnalystUsageReceipt, g as computeFindingId, h as ProposalFindingOrigin, i as AnalystFinding, l as AnalystRunResult, m as ProposalFinding, n as AnalystContext, o as AnalystRequirements, p as EvidenceRef, r as AnalystCost, s as AnalystRunEvent, t as Analyst, u as AnalystRunSummary, v as makeProposalFinding, x as TraceAnalysisStore } from "../types-
|
|
6
|
-
import { a as createTraceAnalyst, c as runTraceAnalyst, d as deriveEfficiencyFindings, i as TraceAnalystDefinition, l as BehavioralAnalystOptions, n as buildDefaultAnalystRegistry, o as renderPriorFindings, r as CreateTraceAnalystOptions, s as renderUpstreamFindings, t as DefaultAnalystRegistryOptions, u as behavioralAnalyst } from "../default-registry-
|
|
7
|
-
import { a as ExactAnalystRunPolicySnapshot, c as ExactAnalystSnapshot, d as ExactExecutionComponentSnapshot, i as ExactAnalystRunEvent, l as ExactCapableAnalyst, n as ExactAnalystExecutionPlanSnapshot, o as ExactAnalystRunResult, r as ExactAnalystRunCompletion, s as ExactAnalystRunSummary, t as ExactAnalystBudgetSnapshot, u as ExactExecutionComponentIdentity } from "../exact-types-
|
|
8
|
-
import { A as ExactAnalystRunExecutionError, D as AnalystRegistryOptions, E as AnalystRegistry, M as RegistryRunOpts, N as assertExactRegistryRunOpts, O as BudgetPolicy, T as AnalystHooks, j as ExactRegistryRunOpts, k as ExactAnalystBudgetPolicy } from "../completion-verifier-
|
|
9
|
-
import { C as scoreAnalystFindings, S as traceStoreEvidenceResolver, _ as AnalystIssueExpectation, a as AnalystBenchmarkLabelState, b as registryBenchmarkRunner, c as AnalystBenchmarkProvenance, d as AnalystBenchmarkSummary, f as AnalystEvidenceExpectation, g as AnalystFindingScore, h as AnalystEvidenceResolver, i as AnalystBenchmarkError, l as AnalystBenchmarkResult, m as AnalystEvidenceResolutionError, n as AnalystBenchmarkDatasetRef, o as AnalystBenchmarkObservation, p as AnalystEvidenceResolution, r as AnalystBenchmarkDescriptor, s as AnalystBenchmarkOutput, t as AnalystBenchmarkCase, u as AnalystBenchmarkRunner, v as AnalystLatencyDistribution, x as runAnalystBenchmark, y as RunAnalystBenchmarkOptions } from "../benchmark-
|
|
10
|
-
import { r as ExternalOptimizerRunnerCommand } from "../external-optimizer-contracts-
|
|
11
|
-
import { a as TraceAnalysisEngineRequest, c as resolveTraceAnalystLimits, d as RawAnalystEvidence, f as RawAnalystEvidenceSchema, g as parseRawFinding, h as evidenceRefsFromRawFinding, i as TraceAnalysisEngine, l as ANALYST_SEVERITIES, m as RawAnalystFindingSchema, n as buildTraceToolsForGroup, o as TraceAnalysisEngineResult, p as RawAnalystFinding, r as DEFAULT_TRACE_ANALYST_LIMITS, s as TraceAnalystLimits, t as TraceToolGroupName, u as RAW_FINDING_SCHEMA_PROMPT } from "../tool-groups-
|
|
2
|
+
import { A as parseFindingSubject, C as FINDING_SUBJECT_KINDS, D as FindingSubjectStringSchema, E as FindingSubjectKind, F as DefineExactCustomAnalystOptions, G as SemanticConceptJudgeOptions, I as defineCustomAnalyst, L as defineTraceAnalyst, M as DspyRlmTraceEngineOptions, N as createDspyRlmTraceEngine, O as KIND_EXPECTED_SUBJECTS, P as DefineCustomAnalystOptions, S as FINDING_SUBJECT_GRAMMAR_PROMPT, T as FindingSubject, W as SemanticConceptJudgeInput, Y as RunCritic, Z as RunTrace, _ as FindingsDiff, a as SkillUsageScanConfig, b as defaultIsMaterial, c as DEFAULT_TRACE_ANALYST_KINDS, d as IMPROVEMENT_KIND_SPEC, f as FAILURE_MODE_KIND_SPEC, g as DiffPolicy, h as emitControlIntegrityFindings, i as SkillUsageReport, j as renderFindingSubject, k as findingSubjectGrammarPromptFor, l as KNOWLEDGE_POISONING_KIND_SPEC, m as ControlIntegrityAnalyst, n as SkillUsageAnalyst, o as buildSkillUsageReport, p as CONTROL_INTEGRITY_ANALYST, r as SkillUsageRecord, s as emitSkillUsageFindings, t as SKILL_USAGE_ANALYST, u as KNOWLEDGE_GAP_KIND_SPEC, v as FindingsStore, w as FINDING_SUBJECT_SYNTAX, x as diffFindings, y as PersistedFinding } from "../skill-usage-CVVnoIx-.js";
|
|
3
|
+
import { b as CustomTokenPricing, c as CostLedgerHandle } from "../cost-ledger-FuQvHxPm.js";
|
|
4
|
+
import { A as ChatClient, B as SandboxSdkTransportOpts, F as CreateChatClientOpts, I as CustomTransportOpts, L as DirectProviderTransportOpts, M as ChatResponse, N as ChatTransport, P as CliBridgeTransportOpts, R as MockTransportOpts, V as createChatClient, j as ChatRequest, k as ChatCallOpts, m as JudgeInput, p as JudgeFn, z as RouterTransportOpts } from "../types-BjMFz88h.js";
|
|
5
|
+
import { _ as makeFinding, a as AnalystInputKind, c as AnalystRunInputs, d as AnalystSeverity, f as AnalystUsageReceipt, g as computeFindingId, h as ProposalFindingOrigin, i as AnalystFinding, l as AnalystRunResult, m as ProposalFinding, n as AnalystContext, o as AnalystRequirements, p as EvidenceRef, r as AnalystCost, s as AnalystRunEvent, t as Analyst, u as AnalystRunSummary, v as makeProposalFinding, x as TraceAnalysisStore } from "../types-D3jh6F98.js";
|
|
6
|
+
import { a as createTraceAnalyst, c as runTraceAnalyst, d as deriveEfficiencyFindings, i as TraceAnalystDefinition, l as BehavioralAnalystOptions, n as buildDefaultAnalystRegistry, o as renderPriorFindings, r as CreateTraceAnalystOptions, s as renderUpstreamFindings, t as DefaultAnalystRegistryOptions, u as behavioralAnalyst } from "../default-registry-Ci7wAAR8.js";
|
|
7
|
+
import { a as ExactAnalystRunPolicySnapshot, c as ExactAnalystSnapshot, d as ExactExecutionComponentSnapshot, i as ExactAnalystRunEvent, l as ExactCapableAnalyst, n as ExactAnalystExecutionPlanSnapshot, o as ExactAnalystRunResult, r as ExactAnalystRunCompletion, s as ExactAnalystRunSummary, t as ExactAnalystBudgetSnapshot, u as ExactExecutionComponentIdentity } from "../exact-types-B0lJV3tu.js";
|
|
8
|
+
import { A as ExactAnalystRunExecutionError, D as AnalystRegistryOptions, E as AnalystRegistry, M as RegistryRunOpts, N as assertExactRegistryRunOpts, O as BudgetPolicy, T as AnalystHooks, j as ExactRegistryRunOpts, k as ExactAnalystBudgetPolicy } from "../completion-verifier-D15NHYSk.js";
|
|
9
|
+
import { C as scoreAnalystFindings, S as traceStoreEvidenceResolver, _ as AnalystIssueExpectation, a as AnalystBenchmarkLabelState, b as registryBenchmarkRunner, c as AnalystBenchmarkProvenance, d as AnalystBenchmarkSummary, f as AnalystEvidenceExpectation, g as AnalystFindingScore, h as AnalystEvidenceResolver, i as AnalystBenchmarkError, l as AnalystBenchmarkResult, m as AnalystEvidenceResolutionError, n as AnalystBenchmarkDatasetRef, o as AnalystBenchmarkObservation, p as AnalystEvidenceResolution, r as AnalystBenchmarkDescriptor, s as AnalystBenchmarkOutput, t as AnalystBenchmarkCase, u as AnalystBenchmarkRunner, v as AnalystLatencyDistribution, x as runAnalystBenchmark, y as RunAnalystBenchmarkOptions } from "../benchmark-DxaZfy0w.js";
|
|
10
|
+
import { r as ExternalOptimizerRunnerCommand } from "../external-optimizer-contracts-nb7c_WAR.js";
|
|
11
|
+
import { a as TraceAnalysisEngineRequest, c as resolveTraceAnalystLimits, d as RawAnalystEvidence, f as RawAnalystEvidenceSchema, g as parseRawFinding, h as evidenceRefsFromRawFinding, i as TraceAnalysisEngine, l as ANALYST_SEVERITIES, m as RawAnalystFindingSchema, n as buildTraceToolsForGroup, o as TraceAnalysisEngineResult, p as RawAnalystFinding, r as DEFAULT_TRACE_ANALYST_LIMITS, s as TraceAnalystLimits, t as TraceToolGroupName, u as RAW_FINDING_SCHEMA_PROMPT } from "../tool-groups-DVQTy9lq.js";
|
|
12
12
|
//#region src/analyst/adapters.d.ts
|
|
13
13
|
declare function liftSeverity(s: Severity): AnalystSeverity;
|
|
14
14
|
interface VerifierAdapterOpts<Env> {
|
|
@@ -366,6 +366,8 @@ interface CodeTraceBlockDiagnostics {
|
|
|
366
366
|
unresolvedBlockInteriorSteps: number[];
|
|
367
367
|
/** Steps claimed by more than one block; the first block keeps the step. */
|
|
368
368
|
overlappingBlockSteps: number[];
|
|
369
|
+
/** Findings dropped before expansion because their shape or evidence is invalid. */
|
|
370
|
+
rejectedFindings?: string[];
|
|
369
371
|
}
|
|
370
372
|
declare function emptyPublicBenchmarkRunner(): AnalystBenchmarkRunner<AnalystRunInputs>;
|
|
371
373
|
declare function adaptPublicBenchmarkFindings(options: {
|
|
@@ -685,13 +687,13 @@ interface AnalystBenchmarkCommandConfig {
|
|
|
685
687
|
resume: boolean;
|
|
686
688
|
}
|
|
687
689
|
declare function runAnalystBenchmarkCommand(argv: readonly string[], env?: NodeJS.ProcessEnv, dependencies?: AnalystBenchmarkCommandDependencies): Promise<number>;
|
|
688
|
-
declare const ANALYST_BENCHMARK_HELP = "agent-eval analyst-benchmark\n\nRun the recursive DSPy trace analyst against public AgentRx or CodeTraceBench labels.\n\nRequired:\n --dataset agentrx|codetracebench\n --analyst dspy-rlm|direct Scored analyst. Default: dspy-rlm.\n 'direct' is the retired one-shot runner that\n produced the published evidence.\n --labels <dataset.json|dataset.jsonl>\n --trace-dir <one-trace-per-file OTLP JSONL directory>\n --artifact-dir <extracted artifact root> Required for CodeTraceBench\n --out <new output directory>\n --revision <full 40- or 64-character hex digest>\n --split <dataset split>\n --base-url <OpenAI-compatible /v1 URL>\n --api-key-env <environment variable containing the bearer>\n --model <provider model id>\n --limit <positive case count>\n\nControls:\n --resume Continue an interrupted run in --out\n --seed <integer> Case-selection and comparison seed. Default: 0\n --concurrency <positive integer> Parallel benchmark jobs. Default: 1\n --repetitions <positive integer> Runs per case and runner. Default: 1\n --max-output-tokens <positive> Model output limit per call. Default:
|
|
690
|
+
declare const ANALYST_BENCHMARK_HELP = "agent-eval analyst-benchmark\n\nRun the recursive DSPy trace analyst against public AgentRx or CodeTraceBench labels.\n\nRequired:\n --dataset agentrx|codetracebench\n --analyst dspy-rlm|direct Scored analyst. Default: dspy-rlm.\n 'direct' is the retired one-shot runner that\n produced the published evidence.\n --labels <dataset.json|dataset.jsonl>\n --trace-dir <one-trace-per-file OTLP JSONL directory>\n --artifact-dir <extracted artifact root> Required for CodeTraceBench\n --out <new output directory>\n --revision <full 40- or 64-character hex digest>\n --split <dataset split>\n --base-url <OpenAI-compatible /v1 URL>\n --api-key-env <environment variable containing the bearer>\n --model <provider model id>\n --limit <positive case count>\n\nControls:\n --resume Continue an interrupted run in --out\n --seed <integer> Case-selection and comparison seed. Default: 0\n --concurrency <positive integer> Parallel benchmark jobs. Default: 1\n --repetitions <positive integer> Runs per case and runner. Default: 1\n --max-output-tokens <positive> Model output limit per call. Default: 16384\n --python <executable> Python with agent-eval-rpc[dspy]. Default: python\n --timeout-ms <positive> Model analyst deadline per case. Default: 300000\n --max-cost-usd <positive> Run-wide spend limit. Default: 5\n --max-artifact-bytes <positive> Final evidence bytes per case. Default: 8388608\n\nWrites result.json with every observation, metric, usage field, error, comparison,\ninput digest, artifact digest, case distribution, selected case id, and explicit\nunknown cost. Limited deterministic-hash subsets are marked non-representative.\nCompleted observations are fsynced to observations.jsonl. Shareable output is in\nresult.json and report.md. Machine-local paths, endpoint, and command are isolated\nin run.local.json.\nThe key is read from the named environment variable and is never written.";
|
|
689
691
|
//#endregion
|
|
690
692
|
//#region src/analyst/benchmark-implementation.d.ts
|
|
691
693
|
declare const ANALYST_BENCHMARK_IMPLEMENTATION_DIGEST_ALGORITHM = "sha256-canonical-source-manifest";
|
|
692
694
|
declare const ANALYST_BENCHMARK_DEPENDENCY_LOCK_DIGEST_ALGORITHM = "sha256-canonical-file-manifest";
|
|
693
695
|
declare const ANALYST_BENCHMARK_DEPENDENCY_LOCK_FILES: readonly string[];
|
|
694
|
-
declare const ANALYST_BENCHMARK_DEPENDENCY_LOCK_SHA256 = "
|
|
696
|
+
declare const ANALYST_BENCHMARK_DEPENDENCY_LOCK_SHA256 = "1e4c763f2b791c57b37c0e326c56546b433a7a99c0bc7d064abdb58b9cbf0d81";
|
|
695
697
|
/** The published benchmark evidence was produced at this package version, by
|
|
696
698
|
* the retired one-shot direct runner, before trace analysts moved to the
|
|
697
699
|
* recursive DSPy RLM engine. Both evidence digests below are historical facts
|
|
@@ -703,7 +705,7 @@ declare const ANALYST_BENCHMARK_EVIDENCE_PACKAGE_VERSION = "0.137.0";
|
|
|
703
705
|
declare const ANALYST_BENCHMARK_EVIDENCE_DEPENDENCY_LOCK_SHA256 = "1e03f2daed356d60316aabefb407ec1e437ac94d408d61eea4ae096e9c6fbb5b";
|
|
704
706
|
declare const ANALYST_BENCHMARK_EVIDENCE_IMPLEMENTATION_SHA256 = "4dba263b6256a30d56c7fdb2d992d3a953c0035d731f359b704db806f68f75ac";
|
|
705
707
|
declare const ANALYST_BENCHMARK_IMPLEMENTATION_FILES: readonly string[];
|
|
706
|
-
declare const ANALYST_BENCHMARK_IMPLEMENTATION_SHA256 = "
|
|
708
|
+
declare const ANALYST_BENCHMARK_IMPLEMENTATION_SHA256 = "56dd4c7ed19fc5855f99ad238464ad95cb1bea155ad58c49dab3eb0f6cbe7d6a";
|
|
707
709
|
declare function analystBenchmarkImplementationDigest(): string;
|
|
708
710
|
declare function analystBenchmarkDependencyLockDigest(): string;
|
|
709
711
|
//#endregion
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"index.d.ts","names":[],"sources":["../../src/analyst/adapters.ts","../../src/analyst/benchmark-agentrx-calibration.ts","../../src/analyst/benchmark-dataset-types.ts","../../src/analyst/benchmark-dataset-agentrx.ts","../../src/analyst/benchmark-dataset-codetrace.ts","../../src/analyst/benchmark-dataset-utils.ts","../../src/analyst/benchmark-verification-outcome.ts","../../src/analyst/benchmark-verification-artifacts.ts","../../src/analyst/benchmark-public-types.ts","../../src/analyst/benchmark-public-adapters.ts","../../src/analyst/benchmark-public-data.ts","../../src/analyst/benchmark-public-prompt.ts","../../src/analyst/benchmark-public-model.ts","../../src/analyst/benchmark-public-rlm.ts","../../src/analyst/benchmark-comparison.ts","../../src/analyst/benchmark-public-calibration.ts","../../src/analyst/benchmark-command-artifact.ts","../../src/analyst/benchmark-command-result.ts","../../src/analyst/benchmark-command.ts","../../src/analyst/benchmark-implementation.ts","../../src/analyst/benchmark-report.ts","../../src/analyst/parse-tolerant.ts","../../src/analyst/proposal-findings.ts"],"mappings":";;;;;;;;;;;;iBA4CgB,aAAa,GAAG,WAAgB;UAe/B,oBAAoB;EACnC;EACA;EACA,UAAU,mBAAmB;;;;;EAK7B,UAAU,KAAK,cAAc;;iBAGf,sBAAsB,KAAK,MAAM,oBAAoB,OAAO,QAAQ;UAwEnE;EACf;EACA;EACA,SAAS;;EAET;;iBAGc,uBAAuB,OAAM,uBAA4B,QAAQ;UAiEhE;EACf;EACA;EACA,OAAO;;EAEP,MAAM;;EAEN,OAAO;;EAEP;;iBAGc,mBAAmB,MAAM,mBAAmB,QAAQ;UAiDnD;EACf;EACA;;EAEA,UAAU,KAAK;;EAEf;;iBAGc,kCACd,OAAM,kCACL,QAAQ;;;cC5RE;UAEI;EACf;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;;UAGe;EACf;EACA;EACA;EACA,SAAS;;iBAGK,4BACd,QAAQ,wBACR,2BACC;iBAkBa,iCAAiC,SAAS;;;KCtD9C;UAEK;EACf,YAAY;EACZ;EACA;EACA;EACA;EACA;;UAGe;EACf,eAAe;EACf,mBAAmB;EACnB;IAAe,YAAY;IAAY;;EACvC,wBAAwB;EACxB;EACA;EACA;;UAGe;EACf,UAAU;EACV;EACA;EACA;EACA;;UAGe;EACf,UAAU;EACV,mBAAmB;EACnB;EACA;EACA;EACA;EACA;;UAGe;EACf;EACA;EACA;EACA;;UAGe;EACf;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;;UAGe;EACf;EACA;EACA;EACA,iBAAiB;;KAGP,0CAEC,sCACA,iCACA;UAEI;EACf;EACA;EACA;EACA;;EAEA;EACA;EACA;EACA;EACA;EACA;EACA,oCAAoC;;UAGrB;EACf,eAAe;EACf,WAAW,sBAAsB;;KAGvB;UAEK;;;;;EAKf,WAAW;;UAGI,kCACP,yBACN;UAEa,oCAAoC;EACnD;;EAEA;;UAGe,yCAAyC;EACxD;EACA;EACA;EACA;;UAGe,2CACP,kCACN;;;iBCtGY,qBAAqB,QACnC,KAAK,YACL,OAAO,QACP,UAAS,8BACR,qBAAqB;;iBAoGR,6BACd,mBAAmB,YACnB,iBACA,UAAS,mCACR;iBA6Ea,yBAAyB;;iBAoNzB,iBAAiB;;;iBC7YjB,mBAAmB,QACjC,KAAK,mBACL,OAAO,QACP,UAAS,4BACR,qBAAqB;;iBAwFR,gCACd,2BACA,aAAa,uBACb,UAAS,qCACR;;;iBClHa,wBAAwB;;;KCA5B;UAEK;EACf;EACA;EACA,QAAQ;;UAGO;EACf,QAAQ;EACR;EAKA;IAAe;IAAe;;EAC9B,SAAS;EACT;EACA;EACA;EACA;;UAGe;EACf;EACA;;iBA8Ec,yBACd,gBAAgB,2BACf;;;cChGU;KAED;UAEK;EACf,MAAM;EACN;EACA;EACA;EACA;EACA;;UAGe;EACf;EACA;EACA,SAAS;EACT;EACA;EACA;EACA;EACA;EACA,OAAO;EACP,cAAc;EACd,UAAU,OAAO;;UAGF;EACf,UAAU;EACV,SAAS;EACT,OAAO,MAAM;IAA6B;;;iBAatB,mCAAmC;EACvD;EACA,KAAK;EACL;IACE,QAAQ;iBAwHI,kCACd,kBACA,iBACA,WAAW,6BACX;;;KChLU;UAEK;EACf;EACA;EACA;EACA;EACA;;EAEA,UAAU;;EAEV;EACA;IACE,SAAS;IACT;IACA;IACA;IACA;;EAEF,aAAa;EACb;IACE;IACA;;;EAGF,mBAAmB;;UAGJ;EACf,OAAO,qBAAqB;EAC5B;EACA;EACA;EACA,YAAY;IACV;IACA;IACA;;EAEF,uBAAuB;EACvB,WAAW;;UAGI;EACf;EACA;EACA,QAAQ;;UAGO;EACf,OAAO;EACP,OAAO;EACP,OAAO;EACP,YAAY;EACZ,QAAQ;;UAGO;EACf;EACA;EACA;EACA;EACA;EACA;EACA,QAAQ;EACR,UAAU;;;;;;;;;;UCnDK;EACf;EACA;;EAEA;EACA;EACA,UAAU;EACV;EACA;EACA;EACA;EACA,WAAW;;;;;;;;;UAUI;EACf;EACA;;EAEA,kCAAkC;;EAElC;;EAEA;;iBAGc,8BAA8B,uBAAuB;iBAiB/C,6BAA6B;EACjD,SAAS;EACT;EACA,mBAAmB;EACnB;EACA,OAAO;EACP,SAAS;IACP;EAAU,UAAU;EAAkB,aAAa;;;;;;;;;
|
|
1
|
+
{"version":3,"file":"index.d.ts","names":[],"sources":["../../src/analyst/adapters.ts","../../src/analyst/benchmark-agentrx-calibration.ts","../../src/analyst/benchmark-dataset-types.ts","../../src/analyst/benchmark-dataset-agentrx.ts","../../src/analyst/benchmark-dataset-codetrace.ts","../../src/analyst/benchmark-dataset-utils.ts","../../src/analyst/benchmark-verification-outcome.ts","../../src/analyst/benchmark-verification-artifacts.ts","../../src/analyst/benchmark-public-types.ts","../../src/analyst/benchmark-public-adapters.ts","../../src/analyst/benchmark-public-data.ts","../../src/analyst/benchmark-public-prompt.ts","../../src/analyst/benchmark-public-model.ts","../../src/analyst/benchmark-public-rlm.ts","../../src/analyst/benchmark-comparison.ts","../../src/analyst/benchmark-public-calibration.ts","../../src/analyst/benchmark-command-artifact.ts","../../src/analyst/benchmark-command-result.ts","../../src/analyst/benchmark-command.ts","../../src/analyst/benchmark-implementation.ts","../../src/analyst/benchmark-report.ts","../../src/analyst/parse-tolerant.ts","../../src/analyst/proposal-findings.ts"],"mappings":";;;;;;;;;;;;iBA4CgB,aAAa,GAAG,WAAgB;UAe/B,oBAAoB;EACnC;EACA;EACA,UAAU,mBAAmB;;;;;EAK7B,UAAU,KAAK,cAAc;;iBAGf,sBAAsB,KAAK,MAAM,oBAAoB,OAAO,QAAQ;UAwEnE;EACf;EACA;EACA,SAAS;;EAET;;iBAGc,uBAAuB,OAAM,uBAA4B,QAAQ;UAiEhE;EACf;EACA;EACA,OAAO;;EAEP,MAAM;;EAEN,OAAO;;EAEP;;iBAGc,mBAAmB,MAAM,mBAAmB,QAAQ;UAiDnD;EACf;EACA;;EAEA,UAAU,KAAK;;EAEf;;iBAGc,kCACd,OAAM,kCACL,QAAQ;;;cC5RE;UAEI;EACf;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;;UAGe;EACf;EACA;EACA;EACA,SAAS;;iBAGK,4BACd,QAAQ,wBACR,2BACC;iBAkBa,iCAAiC,SAAS;;;KCtD9C;UAEK;EACf,YAAY;EACZ;EACA;EACA;EACA;EACA;;UAGe;EACf,eAAe;EACf,mBAAmB;EACnB;IAAe,YAAY;IAAY;;EACvC,wBAAwB;EACxB;EACA;EACA;;UAGe;EACf,UAAU;EACV;EACA;EACA;EACA;;UAGe;EACf,UAAU;EACV,mBAAmB;EACnB;EACA;EACA;EACA;EACA;;UAGe;EACf;EACA;EACA;EACA;;UAGe;EACf;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;;UAGe;EACf;EACA;EACA;EACA,iBAAiB;;KAGP,0CAEC,sCACA,iCACA;UAEI;EACf;EACA;EACA;EACA;;EAEA;EACA;EACA;EACA;EACA;EACA;EACA,oCAAoC;;UAGrB;EACf,eAAe;EACf,WAAW,sBAAsB;;KAGvB;UAEK;;;;;EAKf,WAAW;;UAGI,kCACP,yBACN;UAEa,oCAAoC;EACnD;;EAEA;;UAGe,yCAAyC;EACxD;EACA;EACA;EACA;;UAGe,2CACP,kCACN;;;iBCtGY,qBAAqB,QACnC,KAAK,YACL,OAAO,QACP,UAAS,8BACR,qBAAqB;;iBAoGR,6BACd,mBAAmB,YACnB,iBACA,UAAS,mCACR;iBA6Ea,yBAAyB;;iBAoNzB,iBAAiB;;;iBC7YjB,mBAAmB,QACjC,KAAK,mBACL,OAAO,QACP,UAAS,4BACR,qBAAqB;;iBAwFR,gCACd,2BACA,aAAa,uBACb,UAAS,qCACR;;;iBClHa,wBAAwB;;;KCA5B;UAEK;EACf;EACA;EACA,QAAQ;;UAGO;EACf,QAAQ;EACR;EAKA;IAAe;IAAe;;EAC9B,SAAS;EACT;EACA;EACA;EACA;;UAGe;EACf;EACA;;iBA8Ec,yBACd,gBAAgB,2BACf;;;cChGU;KAED;UAEK;EACf,MAAM;EACN;EACA;EACA;EACA;EACA;;UAGe;EACf;EACA;EACA,SAAS;EACT;EACA;EACA;EACA;EACA;EACA,OAAO;EACP,cAAc;EACd,UAAU,OAAO;;UAGF;EACf,UAAU;EACV,SAAS;EACT,OAAO,MAAM;IAA6B;;;iBAatB,mCAAmC;EACvD;EACA,KAAK;EACL;IACE,QAAQ;iBAwHI,kCACd,kBACA,iBACA,WAAW,6BACX;;;KChLU;UAEK;EACf;EACA;EACA;EACA;EACA;;EAEA,UAAU;;EAEV;EACA;IACE,SAAS;IACT;IACA;IACA;IACA;;EAEF,aAAa;EACb;IACE;IACA;;;EAGF,mBAAmB;;UAGJ;EACf,OAAO,qBAAqB;EAC5B;EACA;EACA;EACA,YAAY;IACV;IACA;IACA;;EAEF,uBAAuB;EACvB,WAAW;;UAGI;EACf;EACA;EACA,QAAQ;;UAGO;EACf,OAAO;EACP,OAAO;EACP,OAAO;EACP,YAAY;EACZ,QAAQ;;UAGO;EACf;EACA;EACA;EACA;EACA;EACA;EACA,QAAQ;EACR,UAAU;;;;;;;;;;UCnDK;EACf;EACA;;EAEA;EACA;EACA,UAAU;EACV;EACA;EACA;EACA;EACA,WAAW;;;;;;;;;UAUI;EACf;EACA;;EAEA,kCAAkC;;EAElC;;EAEA;;EAEA;;iBAGc,8BAA8B,uBAAuB;iBAiB/C,6BAA6B;EACjD,SAAS;EACT;EACA,mBAAmB;EACnB;EACA,OAAO;EACP,SAAS;IACP;EAAU,UAAU;EAAkB,aAAa;;;;;;;;;iBAqKjC,6BAA6B;EACjD;EACA,iBAAiB;EACjB,OAAO;EACP;EACA;EACA,SAAS;IACP;EAAU,UAAU;EAAkB,aAAa;;;;iBCzMjC,wBACpB,eACC,QAAQ,MAAM;iBA2BD,0BACd,SAAS,+BACT,eAAe,2BACf;EAAW;EAAe;IACzB,MAAM;iBAwBO,6BACd,SAAS,+BACT,eAAe,4BACd;iBAyCa,+BACd,SAAS,+BACT,iBAAiB,2BACjB,mBAAmB,2BACnB,eACC;iBAcmB,8BAA8B;EAClD,SAAS;EACT;EACA;EACA;EACA;EACA;EACA;IACE,QAAQ;;;;;;;cCxKC;;;;cAKA;cAMA;;iBA0FG,4BAA4B,SAAS;;iBAerC,+BAA+B,SAAS;;;;iBAaxC,8BAA8B,SAAS;;;;iBCzEvC,kCACd,SAAS,+BACT,QAAQ,oCACP,uBAAuB;;;;iBCtCV,+BACd,SAAS,+BACT,QAAQ,oCACP,uBAAuB;;;KC7Bd;UAoBK;EACf,QAAQ;EACR;;EAEA;;EAEA;;EAEA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;;UAGe;EACf;EACA;EACA,SAAS;;iBASK,sBACd,QAAQ,wBACR;EACE;EACA;EACA;EACA;EACA;IAED;;;UCnEc;EACf;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;;EAEA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;;UAGe;EACf;EACA;EACA,SAAS;;iBAGK,8BACd,QAAQ,yBACP;iBAca,mCAAmC,SAAS;;;UCrC3C;EACf;EACA;EACA;IACE,SAAS;IACT;IACA;IACA;IACA;IACA,YAAY;MAAQ;MAAiB;MAAsB;;IAC3D,uBAAuB;IACvB,0BAA0B;IAC1B;MACE;MACA;MACA;MACA,QAAQ;;IAEV;MACE;MACA;MACA;MACA;MACA;MACA;MACA;MACA;MACA;MACA;;;EAGJ,QAAQ;EACR,aAAa;EACb,uBAAuB;EACvB,qBAAqB;;UAGN;EACf;EACA;EACA;EACA;IACE;IACA;IACA;;;UAIa;EACf;IACE,SAAS;IACT;IACA;IACA;MACE;MACA;MACA;;IAEF;IACA;IACA;IACA;IACA;IACA;IACA;IACA;IACA;IACA;;EAEF;IACE;IACA;IACA;IACA,YAAY;MAAQ;MAAiB;MAAsB;;IAC3D;IACA;;;UAIa;EACf;EACA;EACA;EACA;EACA,UAAU;;UAGK;EACf;EACA;EACA;EACA;IACE;IACA;IACA;IACA;IACA;IACA;;EAEF;EACA;IACE;IACA;IACA;;EAEF;IACE;IACA;IACA;IACA;IACA;IACA;;;UAIa;EACf;EACA;EACA;EACA,aAAa;EACb;;cAGW;cACA;cACA;cACA;;;iBCxHS,6BACpB,eACC,QAAQ;;;UC8DM;EACf,uBACE,SAAS,+BACT,QAAQ,sCACL,uBAAuB;;;;;;;;;;KAWlB;UAEK;EACf,SAAS;EACT,SAAS;EACT;EACA;EACA;EACA;EACA;EACA;EACA,OAAO;EACP;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;;iBAKoB,2BACpB,yBACA,MAAK,OAAO,YACZ,eAAc,sCACb;cAsSU;;;cCraA;cAEA;cAEA;cAOA;;;;;;;;cAUA;cAEA;cAGA;cAGA;cAoFA;iBAGG;iBAIA;;;iBCrHA,+BACd,QAAQ,wBACR,uBAAsB;;;;;;;;;;;;;;;iBCQR,gBAAgB;;;;;iBAgBhB,WAAW;;;;;;;iBAeX,oBAAoB;;;;iBCXpB,kBAAkB,mBAAmB,WAAW;;;;;;iBAShD,uBACd,mBACA,mBACC,cAAc"}
|
package/dist/analyst/index.js
CHANGED
|
@@ -1,12 +1,12 @@
|
|
|
1
|
-
import { i as CostLedger } from "../cost-ledger-
|
|
2
|
-
import { C as createChatClient, _ as CONTROL_INTEGRITY_ANALYST, b as behavioralAnalyst, f as DEFAULT_TRACE_ANALYST_KINDS, g as FAILURE_MODE_KIND_SPEC, h as IMPROVEMENT_KIND_SPEC, i as assertExactRegistryRunOpts, m as KNOWLEDGE_GAP_KIND_SPEC, n as AnalystRegistry, p as KNOWLEDGE_POISONING_KIND_SPEC, r as ExactAnalystRunExecutionError, t as buildDefaultAnalystRegistry, v as ControlIntegrityAnalyst, x as deriveEfficiencyFindings, y as emitControlIntegrityFindings } from "../default-registry-
|
|
3
|
-
import { A as coerceJson, B as renderFindingSubject, D as RawAnalystFindingSchema, E as RawAnalystEvidenceSchema, F as FINDING_SUBJECT_SYNTAX, G as resolveTraceAnalystLimits, I as FindingSubjectStringSchema, L as KIND_EXPECTED_SUBJECTS, M as stripCodeFences, N as FINDING_SUBJECT_GRAMMAR_PROMPT, O as evidenceRefsFromRawFinding, P as FINDING_SUBJECT_KINDS, R as findingSubjectGrammarPromptFor, T as RAW_FINDING_SCHEMA_PROMPT, W as DEFAULT_TRACE_ANALYST_LIMITS, a as buildTraceToolsForGroup, i as runTraceAnalyst, j as coerceToFindingRows, k as parseRawFinding, n as renderPriorFindings, r as renderUpstreamFindings, t as createTraceAnalyst, w as ANALYST_SEVERITIES, z as parseFindingSubject } from "../kind-factory-
|
|
1
|
+
import { i as CostLedger } from "../cost-ledger-DMFxsLKr.js";
|
|
2
|
+
import { C as createChatClient, _ as CONTROL_INTEGRITY_ANALYST, b as behavioralAnalyst, f as DEFAULT_TRACE_ANALYST_KINDS, g as FAILURE_MODE_KIND_SPEC, h as IMPROVEMENT_KIND_SPEC, i as assertExactRegistryRunOpts, m as KNOWLEDGE_GAP_KIND_SPEC, n as AnalystRegistry, p as KNOWLEDGE_POISONING_KIND_SPEC, r as ExactAnalystRunExecutionError, t as buildDefaultAnalystRegistry, v as ControlIntegrityAnalyst, x as deriveEfficiencyFindings, y as emitControlIntegrityFindings } from "../default-registry-DCp-6hc-.js";
|
|
3
|
+
import { A as coerceJson, B as renderFindingSubject, D as RawAnalystFindingSchema, E as RawAnalystEvidenceSchema, F as FINDING_SUBJECT_SYNTAX, G as resolveTraceAnalystLimits, I as FindingSubjectStringSchema, L as KIND_EXPECTED_SUBJECTS, M as stripCodeFences, N as FINDING_SUBJECT_GRAMMAR_PROMPT, O as evidenceRefsFromRawFinding, P as FINDING_SUBJECT_KINDS, R as findingSubjectGrammarPromptFor, T as RAW_FINDING_SCHEMA_PROMPT, W as DEFAULT_TRACE_ANALYST_LIMITS, a as buildTraceToolsForGroup, i as runTraceAnalyst, j as coerceToFindingRows, k as parseRawFinding, n as renderPriorFindings, r as renderUpstreamFindings, t as createTraceAnalyst, w as ANALYST_SEVERITIES, z as parseFindingSubject } from "../kind-factory-DB7nIs35.js";
|
|
4
4
|
import { a as computeFindingId, i as validateUsageSettlementTimeout, n as settleUsageReceiptFromCostLedger, o as makeFinding, s as makeProposalFinding } from "../usage-receipt-CgxMEBZq.js";
|
|
5
5
|
import { n as isProposalFinding, t as assertProposalFindings } from "../proposal-findings-2GIUo1et.js";
|
|
6
|
-
import { a as RunCritic, c as buildSkillUsageReport, d as defaultIsMaterial, f as diffFindings, h as defineTraceAnalyst, i as runSemanticConceptJudge, l as emitSkillUsageFindings, m as defineCustomAnalyst, n as SEMANTIC_CONCEPT_JUDGE_VERSION, o as SKILL_USAGE_ANALYST, s as SkillUsageAnalyst, u as FindingsStore } from "../semantic-concept-judge-
|
|
7
|
-
import { t as createDspyRlmTraceEngine } from "../dspy-rlm-engine-
|
|
6
|
+
import { a as RunCritic, c as buildSkillUsageReport, d as defaultIsMaterial, f as diffFindings, h as defineTraceAnalyst, i as runSemanticConceptJudge, l as emitSkillUsageFindings, m as defineCustomAnalyst, n as SEMANTIC_CONCEPT_JUDGE_VERSION, o as SKILL_USAGE_ANALYST, s as SkillUsageAnalyst, u as FindingsStore } from "../semantic-concept-judge-DJQtFr95.js";
|
|
7
|
+
import { t as createDspyRlmTraceEngine } from "../dspy-rlm-engine-Bkak4nzo.js";
|
|
8
8
|
import { a as scoreAnalystFindings, n as runAnalystBenchmark, r as traceStoreEvidenceResolver, t as registryBenchmarkRunner } from "../benchmark-CYtcIF2V.js";
|
|
9
|
-
import { A as ANALYST_BENCHMARK_DEPENDENCY_LOCK_SHA256, B as ANALYST_BENCHMARK_LOCAL_RECEIPT_FILE, C as publicBenchmarkSystemPrompt, D as parseVerificationOutcome, E as loadCodeTraceVerificationArtifacts, F as ANALYST_BENCHMARK_IMPLEMENTATION_FILES, G as summarizeAgentRxCalibration, H as ANALYST_BENCHMARK_OBSERVATIONS_FILE, I as ANALYST_BENCHMARK_IMPLEMENTATION_SHA256, J as agentRxBenchmarkCase, K as codeTraceBenchCase, L as analystBenchmarkDependencyLockDigest, M as ANALYST_BENCHMARK_EVIDENCE_IMPLEMENTATION_SHA256, N as ANALYST_BENCHMARK_EVIDENCE_PACKAGE_VERSION, O as ANALYST_BENCHMARK_DEPENDENCY_LOCK_DIGEST_ALGORITHM, P as ANALYST_BENCHMARK_IMPLEMENTATION_DIGEST_ALGORITHM, Q as normalizeBenchmarkLabel, R as analystBenchmarkImplementationDigest, S as publicBenchmarkRlmInstructions, T as appendVerificationArtifactsToOtlp, U as AGENT_RX_UPSTREAM_REVISION, V as ANALYST_BENCHMARK_MANIFEST_FILE, W as renderAgentRxCalibrationMarkdown, X as normalizeAgentRxCategory, Y as agentRxPredictionsToFindings, Z as roundAgentRxStep, _ as expandCodeTraceFailureBlocks, a as renderCodeTraceCalibrationMarkdown, b as MAX_INCORRECT_BLOCK_STEPS, c as createPublicBenchmarkRlmRunner, d as preparePublicAnalystBenchmark, f as publicBenchmarkDistributions, g as emptyPublicBenchmarkRunner, h as adaptPublicBenchmarkFindings, i as readAnalystBenchmarkArtifact, j as ANALYST_BENCHMARK_EVIDENCE_DEPENDENCY_LOCK_SHA256, k as ANALYST_BENCHMARK_DEPENDENCY_LOCK_FILES, l as createPublicBenchmarkDirectRunner, m as selectPublicBenchmarkRows, n as runAnalystBenchmarkCommand, o as summarizeCodeTraceCalibration, p as publicBenchmarkSelectionReport, q as codeTracerPredictionsToFindings, r as renderAnalystBenchmarkMarkdown, s as compareAnalystRunners, t as ANALYST_BENCHMARK_HELP, u as loadPublicBenchmarkRows, v as CODE_TRACE_BENCH_ANALYST_PROMPT, w as DEFAULT_MAX_VERIFICATION_ARTIFACT_BYTES, x as publicBenchmarkProtocolSha256, y as MAX_INCORRECT_BLOCKS, z as ANALYST_BENCHMARK_COST_LEDGER_FILE } from "../benchmark-command-
|
|
9
|
+
import { A as ANALYST_BENCHMARK_DEPENDENCY_LOCK_SHA256, B as ANALYST_BENCHMARK_LOCAL_RECEIPT_FILE, C as publicBenchmarkSystemPrompt, D as parseVerificationOutcome, E as loadCodeTraceVerificationArtifacts, F as ANALYST_BENCHMARK_IMPLEMENTATION_FILES, G as summarizeAgentRxCalibration, H as ANALYST_BENCHMARK_OBSERVATIONS_FILE, I as ANALYST_BENCHMARK_IMPLEMENTATION_SHA256, J as agentRxBenchmarkCase, K as codeTraceBenchCase, L as analystBenchmarkDependencyLockDigest, M as ANALYST_BENCHMARK_EVIDENCE_IMPLEMENTATION_SHA256, N as ANALYST_BENCHMARK_EVIDENCE_PACKAGE_VERSION, O as ANALYST_BENCHMARK_DEPENDENCY_LOCK_DIGEST_ALGORITHM, P as ANALYST_BENCHMARK_IMPLEMENTATION_DIGEST_ALGORITHM, Q as normalizeBenchmarkLabel, R as analystBenchmarkImplementationDigest, S as publicBenchmarkRlmInstructions, T as appendVerificationArtifactsToOtlp, U as AGENT_RX_UPSTREAM_REVISION, V as ANALYST_BENCHMARK_MANIFEST_FILE, W as renderAgentRxCalibrationMarkdown, X as normalizeAgentRxCategory, Y as agentRxPredictionsToFindings, Z as roundAgentRxStep, _ as expandCodeTraceFailureBlocks, a as renderCodeTraceCalibrationMarkdown, b as MAX_INCORRECT_BLOCK_STEPS, c as createPublicBenchmarkRlmRunner, d as preparePublicAnalystBenchmark, f as publicBenchmarkDistributions, g as emptyPublicBenchmarkRunner, h as adaptPublicBenchmarkFindings, i as readAnalystBenchmarkArtifact, j as ANALYST_BENCHMARK_EVIDENCE_DEPENDENCY_LOCK_SHA256, k as ANALYST_BENCHMARK_DEPENDENCY_LOCK_FILES, l as createPublicBenchmarkDirectRunner, m as selectPublicBenchmarkRows, n as runAnalystBenchmarkCommand, o as summarizeCodeTraceCalibration, p as publicBenchmarkSelectionReport, q as codeTracerPredictionsToFindings, r as renderAnalystBenchmarkMarkdown, s as compareAnalystRunners, t as ANALYST_BENCHMARK_HELP, u as loadPublicBenchmarkRows, v as CODE_TRACE_BENCH_ANALYST_PROMPT, w as DEFAULT_MAX_VERIFICATION_ARTIFACT_BYTES, x as publicBenchmarkProtocolSha256, y as MAX_INCORRECT_BLOCKS, z as ANALYST_BENCHMARK_COST_LEDGER_FILE } from "../benchmark-command-CK0UnXAD.js";
|
|
10
10
|
//#region src/analyst/adapters.ts
|
|
11
11
|
/**
|
|
12
12
|
* Adapter factories — lift each existing agent-eval primitive into the
|
|
@@ -1,5 +1,5 @@
|
|
|
1
|
-
import { c as AnalystRunInputs, f as AnalystUsageReceipt, i as AnalystFinding, p as EvidenceRef, x as TraceAnalysisStore } from "./types-
|
|
2
|
-
import { E as AnalystRegistry, M as RegistryRunOpts } from "./completion-verifier-
|
|
1
|
+
import { c as AnalystRunInputs, f as AnalystUsageReceipt, i as AnalystFinding, p as EvidenceRef, x as TraceAnalysisStore } from "./types-D3jh6F98.js";
|
|
2
|
+
import { E as AnalystRegistry, M as RegistryRunOpts } from "./completion-verifier-D15NHYSk.js";
|
|
3
3
|
//#region src/analyst/benchmark-scoring.d.ts
|
|
4
4
|
declare function scoreAnalystFindings(testCase: Pick<AnalystBenchmarkCase, 'id' | 'expectedIssues' | 'labeledEvidence'>, findings: readonly AnalystFinding[]): AnalystFindingScore;
|
|
5
5
|
//#endregion
|
|
@@ -233,4 +233,4 @@ declare function registryBenchmarkRunner(options: {
|
|
|
233
233
|
}): AnalystBenchmarkRunner<AnalystRunInputs>;
|
|
234
234
|
//#endregion
|
|
235
235
|
export { scoreAnalystFindings as C, traceStoreEvidenceResolver as S, AnalystIssueExpectation as _, AnalystBenchmarkLabelState as a, registryBenchmarkRunner as b, AnalystBenchmarkProvenance as c, AnalystBenchmarkSummary as d, AnalystEvidenceExpectation as f, AnalystFindingScore as g, AnalystEvidenceResolver as h, AnalystBenchmarkError as i, AnalystBenchmarkResult as l, AnalystEvidenceResolutionError as m, AnalystBenchmarkDatasetRef as n, AnalystBenchmarkObservation as o, AnalystEvidenceResolution as p, AnalystBenchmarkDescriptor as r, AnalystBenchmarkOutput as s, AnalystBenchmarkCase as t, AnalystBenchmarkRunner as u, AnalystLatencyDistribution as v, runAnalystBenchmark as x, RunAnalystBenchmarkOptions as y };
|
|
236
|
-
//# sourceMappingURL=benchmark-
|
|
236
|
+
//# sourceMappingURL=benchmark-DxaZfy0w.d.ts.map
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"benchmark-
|
|
1
|
+
{"version":3,"file":"benchmark-DxaZfy0w.d.ts","names":[],"sources":["../src/analyst/benchmark-scoring.ts","../src/analyst/benchmark.ts"],"mappings":";;;iBASgB,qBACd,UAAU,KAAK,oEACf,mBAAmB,mBAClB;;;UCIc;EACf;EACA,OAAO;;UAGQ;EACf;EACA;EACA;EACA;EACA,oBAAoB;EACpB;;EAEA,4BAA4B;;KAGlB;UAEK,qBAAqB;EACpC;;EAEA;;EAEA,YAAY;EACZ,OAAO;EACP,yBAAyB;;EAEzB,2BAA2B;EAC3B;EACA,WAAW;;UAGI;EACf;EACA;EACA;EACA;EACA;EACA,mBAAmB;EACnB;EACA;EACA;EACA;;EAEA;;EAEA;;EAEA;EACA;;UAGe;EACf,UAAU;EACV;EACA;;UAGe;EACf;EACA;EACA,oBAAoB;EACpB,QAAQ;;EAER;;KAGU,wBAAwB,qBAAqB;EACvD;EACA,WAAW;EACX,UAAU;EACV,SAAS;gBACK;;;;;iBAMA,2BAA2B,QACzC,WAAW,OAAO,WAAW,qBAC5B,wBAAwB;UAoBV;EACf,mBAAmB;EACnB,QAAQ;EACR,WAAW;;;;;EAKX;;EAEA,QAAQ;;UAGO;EACf;EACA;EACA;EACA;;UAGe,uBAAuB;EACtC;EACA,QACE,OAAO,QACP;IAAW;IAAgB;IAAoB,SAAS;MACvD,yBAAyB,QAAQ;;UAGrB;EACf;EACA;EACA;EACA,YAAY;EACZ;EACA;EACA;EACA;EACA,mBAAmB;EACnB,OAAO;EACP,qBAAqB;EACrB;EACA,eAAe;EACf,QAAQ;EACR,iBAAiB;EACjB,QAAQ;;UAGO;EACf;EACA;EACA;EACA;EACA;;UAGe;EACf;EACA;EACA;EACA;EACA;EACA;EACA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;;EAEA;;EAEA;;EAEA;;EAEA;EACA,WAAW;EACX;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;;UAGe;EACf;EACA;EACA;;UAGe;EACf;EACA,UAAU;EACV;EACA,cAAc;EACd,WAAW;;UAGI,mCAAmC;EAClD;EACA;EACA;EACA;EACA;EACA;EACA;;UAGe;EACf,YAAY;EACZ,cAAc;EACd,WAAW;;UAGI,2BAA2B;EAC1C,gBAAgB,qBAAqB;EACrC,kBAAkB,uBAAuB;EACzC;EACA;EACA;EACA,kBAAkB,wBAAwB;EAC1C,YAAY;;EAEZ,+BAA+B;EAC/B,iBAAiB,aAAa,uCAAuC;EACrE,SAAS;;iBAGW,oBAAoB,QACxC,SAAS,2BAA2B,UACnC,QAAQ;iBAyDK,wBAAwB;EACtC;EACA,UAAU;EACV,aAAa,KAAK;;EAElB;IACE,uBAAuB"}
|
|
@@ -1,15 +1,15 @@
|
|
|
1
1
|
import { c as ValidationError, t as AgentEvalError } from "./errors-D-LKuDhb.js";
|
|
2
2
|
import { s as resolveModelPricing } from "./metrics-C9YY1OcL.js";
|
|
3
|
-
import { a as CostLedgerPersistenceError, i as CostLedger, n as CostCallConflictError, o as CostReceiptCaptureError, r as CostCeilingReachedError, s as CostReservationExceededError, t as CostAccountingIncompleteError } from "./cost-ledger-
|
|
4
|
-
import { c as callLlmJson, f as maximumChargeForLlmRequest, l as costReceiptFromLlm, r as LlmResponseError, t as LlmCallError, u as costReceiptFromLlmError } from "./llm-client-
|
|
5
|
-
import { O as evidenceRefsFromRawFinding, h as TRACE_ANALYSIS_LIMITS, i as runTraceAnalyst } from "./kind-factory-
|
|
3
|
+
import { a as CostLedgerPersistenceError, i as CostLedger, n as CostCallConflictError, o as CostReceiptCaptureError, r as CostCeilingReachedError, s as CostReservationExceededError, t as CostAccountingIncompleteError } from "./cost-ledger-DMFxsLKr.js";
|
|
4
|
+
import { c as callLlmJson, f as maximumChargeForLlmRequest, l as costReceiptFromLlm, r as LlmResponseError, t as LlmCallError, u as costReceiptFromLlmError } from "./llm-client-B3WXSH5Y.js";
|
|
5
|
+
import { O as evidenceRefsFromRawFinding, h as TRACE_ANALYSIS_LIMITS, i as runTraceAnalyst } from "./kind-factory-DB7nIs35.js";
|
|
6
6
|
import { o as makeFinding, r as usageReceiptFromCostLedger } from "./usage-receipt-CgxMEBZq.js";
|
|
7
7
|
import { i as hashCanonical, r as canonicalString } from "./canonical-D011XM8r.js";
|
|
8
|
-
import { n as createRunCostLedger, r as fsCampaignStorage, t as acquireSingleRunLock } from "./single-run-lock-
|
|
9
|
-
import { t as createDspyRlmTraceEngine } from "./dspy-rlm-engine-
|
|
8
|
+
import { n as createRunCostLedger, r as fsCampaignStorage, t as acquireSingleRunLock } from "./single-run-lock-CiQThJxB.js";
|
|
9
|
+
import { t as createDspyRlmTraceEngine } from "./dspy-rlm-engine-Bkak4nzo.js";
|
|
10
10
|
import { c as writeLedgerFileAtomically, s as withLedgerFileLock } from "./ledger-core-Dxz0Rkwa.js";
|
|
11
11
|
import { E as pairedBootstrap } from "./statistics-ByxzSiOM.js";
|
|
12
|
-
import { i as otlpTextToTraceAnalysisStore, r as createOtlpBufferTraceStore, t as DEFAULT_MAX_TRACE_FILE_BYTES } from "./store-otlp-
|
|
12
|
+
import { i as otlpTextToTraceAnalysisStore, r as createOtlpBufferTraceStore, t as DEFAULT_MAX_TRACE_FILE_BYTES } from "./store-otlp-vRByAR6h.js";
|
|
13
13
|
import { i as summarizeAnalystBenchmarkRunner, n as runAnalystBenchmark, r as traceStoreEvidenceResolver } from "./benchmark-CYtcIF2V.js";
|
|
14
14
|
import { z } from "zod";
|
|
15
15
|
import { constants, existsSync, lstatSync, readFileSync } from "node:fs";
|
|
@@ -1172,7 +1172,7 @@ const ANALYST_BENCHMARK_DEPENDENCY_LOCK_FILES = Object.freeze([
|
|
|
1172
1172
|
"package.json",
|
|
1173
1173
|
"pnpm-lock.yaml"
|
|
1174
1174
|
]);
|
|
1175
|
-
const ANALYST_BENCHMARK_DEPENDENCY_LOCK_SHA256 = "
|
|
1175
|
+
const ANALYST_BENCHMARK_DEPENDENCY_LOCK_SHA256 = "1e4c763f2b791c57b37c0e326c56546b433a7a99c0bc7d064abdb58b9cbf0d81";
|
|
1176
1176
|
/** The published benchmark evidence was produced at this package version, by
|
|
1177
1177
|
* the retired one-shot direct runner, before trace analysts moved to the
|
|
1178
1178
|
* recursive DSPy RLM engine. Both evidence digests below are historical facts
|
|
@@ -1266,7 +1266,7 @@ const ANALYST_BENCHMARK_IMPLEMENTATION_FILES = Object.freeze([
|
|
|
1266
1266
|
"src/trace/otlp-attributes.ts",
|
|
1267
1267
|
"src/trace/raw-provider-sink.ts"
|
|
1268
1268
|
]);
|
|
1269
|
-
const ANALYST_BENCHMARK_IMPLEMENTATION_SHA256 = "
|
|
1269
|
+
const ANALYST_BENCHMARK_IMPLEMENTATION_SHA256 = "56dd4c7ed19fc5855f99ad238464ad95cb1bea155ad58c49dab3eb0f6cbe7d6a";
|
|
1270
1270
|
function analystBenchmarkImplementationDigest() {
|
|
1271
1271
|
return ANALYST_BENCHMARK_IMPLEMENTATION_SHA256;
|
|
1272
1272
|
}
|
|
@@ -2178,6 +2178,7 @@ Never print an entire trace, full source file, or more than 12000 characters in
|
|
|
2178
2178
|
Build a compact table of assistant step ids, actions, following observations, and final verification.
|
|
2179
2179
|
Inspect suspicious steps with viewSpans or searchSpan instead of repeatedly printing the table.
|
|
2180
2180
|
This runner emits no JSON fields, so the block is encoded in the finding's subject.
|
|
2181
|
+
Only findings_json is scored; your prose answer is ignored, so every incorrect block you identify must appear as a finding, never only in the answer.
|
|
2181
2182
|
Emit exactly one finding per contiguous failure block.
|
|
2182
2183
|
Set the finding's subject to incorrect-steps-<first_step>-<last_step>-<escape_status>-consequence-<consequence_step>, using the same four values the task defines; for a block covering only step 7 that the agent never escaped and whose damage shows at step 9, the subject is incorrect-steps-7-7-unescaped-consequence-9.
|
|
2183
2184
|
The runner expands the block to one scored step per member and builds every scored citation itself.
|
|
@@ -2300,20 +2301,34 @@ async function adaptCodeTraceFindings(trajectoryId, findings, analystId, store,
|
|
|
2300
2301
|
diagnostics: emptyCodeTraceBlockDiagnostics()
|
|
2301
2302
|
};
|
|
2302
2303
|
}
|
|
2303
|
-
|
|
2304
|
-
|
|
2305
|
-
|
|
2306
|
-
|
|
2307
|
-
|
|
2308
|
-
|
|
2309
|
-
|
|
2304
|
+
const blocks = [];
|
|
2305
|
+
const rejectedFindings = [];
|
|
2306
|
+
for (const source of findings) try {
|
|
2307
|
+
await validateCodeTraceFindingEvidence({
|
|
2308
|
+
trajectoryId,
|
|
2309
|
+
findings: [source],
|
|
2310
|
+
store,
|
|
2311
|
+
...signal ? { signal } : {}
|
|
2312
|
+
});
|
|
2313
|
+
blocks.push(codeTraceBlockFromFinding(trajectoryId, source));
|
|
2314
|
+
} catch (error) {
|
|
2315
|
+
rejectedFindings.push(`${source.finding_id}: ${error instanceof Error ? error.message : String(error)}`);
|
|
2316
|
+
}
|
|
2317
|
+
const expanded = await expandCodeTraceFailureBlocks({
|
|
2310
2318
|
trajectoryId,
|
|
2311
|
-
blocks
|
|
2319
|
+
blocks,
|
|
2312
2320
|
store,
|
|
2313
2321
|
analystId,
|
|
2314
2322
|
...findings[0] ? { producedAt: findings[0].produced_at } : {},
|
|
2315
2323
|
...signal ? { signal } : {}
|
|
2316
2324
|
});
|
|
2325
|
+
return {
|
|
2326
|
+
findings: expanded.findings,
|
|
2327
|
+
diagnostics: {
|
|
2328
|
+
...expanded.diagnostics,
|
|
2329
|
+
rejectedFindings
|
|
2330
|
+
}
|
|
2331
|
+
};
|
|
2317
2332
|
}
|
|
2318
2333
|
function codeTraceBlockFromFinding(trajectoryId, source) {
|
|
2319
2334
|
const parsed = CODE_TRACE_BLOCK_SUBJECT.exec(source.subject ?? "");
|
|
@@ -3406,9 +3421,9 @@ function trajectoryIdFromCaseId$1(dataset, caseId) {
|
|
|
3406
3421
|
function createPublicBenchmarkRlmRunner(dataset, config) {
|
|
3407
3422
|
const costLedger = config.costLedger ?? new CostLedger();
|
|
3408
3423
|
const limits = {
|
|
3409
|
-
maxIterations: config.dspyRlm?.maxIterations ??
|
|
3410
|
-
maxLlmCalls: config.dspyRlm?.maxLlmCalls ??
|
|
3411
|
-
maxToolCalls: config.dspyRlm?.maxToolCalls ??
|
|
3424
|
+
maxIterations: config.dspyRlm?.maxIterations ?? 14,
|
|
3425
|
+
maxLlmCalls: config.dspyRlm?.maxLlmCalls ?? 8,
|
|
3426
|
+
maxToolCalls: config.dspyRlm?.maxToolCalls ?? 80,
|
|
3412
3427
|
maxOutputChars: config.dspyRlm?.maxOutputChars ?? 8e3
|
|
3413
3428
|
};
|
|
3414
3429
|
const pricing = config.pricing ?? pricingForModel(config.model);
|
|
@@ -4539,24 +4554,30 @@ const NON_SCORABLE_COST_ERRORS = /* @__PURE__ */ new Set([
|
|
|
4539
4554
|
function assertObservationAccountingComplete(observation, costLedger, analystRunnerId) {
|
|
4540
4555
|
if (observation.error && NON_SCORABLE_COST_ERRORS.has(observation.error.class)) throw new CostAccountingIncompleteError(`Analyst benchmark stopped before scoring: ${observation.error.message}`);
|
|
4541
4556
|
if (observation.runnerId !== analystRunnerId) return;
|
|
4542
|
-
const
|
|
4557
|
+
const filter = {
|
|
4543
4558
|
channel: "analyst",
|
|
4544
4559
|
tags: {
|
|
4545
4560
|
benchmarkCaseId: observation.caseId,
|
|
4546
4561
|
benchmarkRepetition: String(observation.repetition)
|
|
4547
4562
|
}
|
|
4548
|
-
}
|
|
4549
|
-
if (!
|
|
4550
|
-
|
|
4551
|
-
|
|
4552
|
-
|
|
4553
|
-
|
|
4554
|
-
|
|
4555
|
-
|
|
4563
|
+
};
|
|
4564
|
+
if (!costAccountingIsTrustworthy(costLedger.summary(filter))) throw accountingError(costLedger, "the recursive analyst has incomplete cost accounting", filter);
|
|
4565
|
+
}
|
|
4566
|
+
const BUDGET_BREACH_REASON = /exceeding its enforced maximum/;
|
|
4567
|
+
/**
|
|
4568
|
+
* Cost accounting is trustworthy when every call resolved and none breached its
|
|
4569
|
+
* budget. A recursive analyst on a real provider will occasionally receive a
|
|
4570
|
+
* settled response whose usage the provider omitted; that call is honestly
|
|
4571
|
+
* recorded as unknown and excluded from the reported cost, so it does not
|
|
4572
|
+
* invalidate a completed run. A call left pending, one lost, or one charged
|
|
4573
|
+
* beyond its maximum is a genuine integrity failure and still halts.
|
|
4574
|
+
*/
|
|
4575
|
+
function costAccountingIsTrustworthy(summary) {
|
|
4576
|
+
if (summary.pendingCalls > 0 || summary.unresolvedCalls > 0) return false;
|
|
4577
|
+
return !summary.incompleteReasons.some((reason) => BUDGET_BREACH_REASON.test(reason));
|
|
4556
4578
|
}
|
|
4557
4579
|
function assertCostLedgerFinalizable(costLedger) {
|
|
4558
|
-
|
|
4559
|
-
if (!summary.accountingComplete || summary.pendingCalls > 0 || summary.unresolvedCalls > 0) throw accountingError(costLedger, "the run has pending or incomplete cost entries");
|
|
4580
|
+
if (!costAccountingIsTrustworthy(costLedger.summary())) throw accountingError(costLedger, "the run has pending or budget-breaching cost entries");
|
|
4560
4581
|
}
|
|
4561
4582
|
function accountingError(costLedger, reason, filter) {
|
|
4562
4583
|
const details = costLedger.summary(filter).incompleteReasons.slice(0, 3).join("; ");
|
|
@@ -4587,7 +4608,7 @@ Controls:
|
|
|
4587
4608
|
--seed <integer> Case-selection and comparison seed. Default: 0
|
|
4588
4609
|
--concurrency <positive integer> Parallel benchmark jobs. Default: 1
|
|
4589
4610
|
--repetitions <positive integer> Runs per case and runner. Default: 1
|
|
4590
|
-
--max-output-tokens <positive> Model output limit per call. Default:
|
|
4611
|
+
--max-output-tokens <positive> Model output limit per call. Default: 16384
|
|
4591
4612
|
--python <executable> Python with agent-eval-rpc[dspy]. Default: python
|
|
4592
4613
|
--timeout-ms <positive> Model analyst deadline per case. Default: 300000
|
|
4593
4614
|
--max-cost-usd <positive> Run-wide spend limit. Default: 5
|
|
@@ -4628,7 +4649,7 @@ function parseCommandConfig(argv, env) {
|
|
|
4628
4649
|
baseUrl: openAiCompatibleBaseUrl(requiredFlag(flags, "base-url")),
|
|
4629
4650
|
apiKey,
|
|
4630
4651
|
model: requiredFlag(flags, "model"),
|
|
4631
|
-
maxOutputTokens: positiveFlag(flags, "max-output-tokens",
|
|
4652
|
+
maxOutputTokens: positiveFlag(flags, "max-output-tokens", 16384),
|
|
4632
4653
|
timeoutMs: positiveFlag(flags, "timeout-ms", 3e5),
|
|
4633
4654
|
maxCostUsdPerAnalysis: maxCostUsd,
|
|
4634
4655
|
...python ? { dspyRlm: { runner: { command: python } } } : {}
|
|
@@ -4807,4 +4828,4 @@ function shellQuote(value) {
|
|
|
4807
4828
|
//#endregion
|
|
4808
4829
|
export { ANALYST_BENCHMARK_DEPENDENCY_LOCK_SHA256 as A, ANALYST_BENCHMARK_LOCAL_RECEIPT_FILE as B, publicBenchmarkSystemPrompt as C, parseVerificationOutcome as D, loadCodeTraceVerificationArtifacts as E, ANALYST_BENCHMARK_IMPLEMENTATION_FILES as F, summarizeAgentRxCalibration as G, ANALYST_BENCHMARK_OBSERVATIONS_FILE as H, ANALYST_BENCHMARK_IMPLEMENTATION_SHA256 as I, agentRxBenchmarkCase as J, codeTraceBenchCase as K, analystBenchmarkDependencyLockDigest as L, ANALYST_BENCHMARK_EVIDENCE_IMPLEMENTATION_SHA256 as M, ANALYST_BENCHMARK_EVIDENCE_PACKAGE_VERSION as N, ANALYST_BENCHMARK_DEPENDENCY_LOCK_DIGEST_ALGORITHM as O, ANALYST_BENCHMARK_IMPLEMENTATION_DIGEST_ALGORITHM as P, normalizeBenchmarkLabel as Q, analystBenchmarkImplementationDigest as R, publicBenchmarkRlmInstructions as S, appendVerificationArtifactsToOtlp as T, AGENT_RX_UPSTREAM_REVISION as U, ANALYST_BENCHMARK_MANIFEST_FILE as V, renderAgentRxCalibrationMarkdown as W, normalizeAgentRxCategory as X, agentRxPredictionsToFindings as Y, roundAgentRxStep as Z, expandCodeTraceFailureBlocks as _, renderCodeTraceCalibrationMarkdown as a, MAX_INCORRECT_BLOCK_STEPS as b, createPublicBenchmarkRlmRunner as c, preparePublicAnalystBenchmark as d, publicBenchmarkDistributions as f, emptyPublicBenchmarkRunner as g, adaptPublicBenchmarkFindings as h, readAnalystBenchmarkArtifact as i, ANALYST_BENCHMARK_EVIDENCE_DEPENDENCY_LOCK_SHA256 as j, ANALYST_BENCHMARK_DEPENDENCY_LOCK_FILES as k, createPublicBenchmarkDirectRunner as l, selectPublicBenchmarkRows as m, runAnalystBenchmarkCommand as n, summarizeCodeTraceCalibration as o, publicBenchmarkSelectionReport as p, codeTracerPredictionsToFindings as q, renderAnalystBenchmarkMarkdown as r, compareAnalystRunners as s, ANALYST_BENCHMARK_HELP as t, loadPublicBenchmarkRows as u, CODE_TRACE_BENCH_ANALYST_PROMPT as v, DEFAULT_MAX_VERIFICATION_ARTIFACT_BYTES as w, publicBenchmarkProtocolSha256 as x, MAX_INCORRECT_BLOCKS as y, ANALYST_BENCHMARK_COST_LEDGER_FILE as z };
|
|
4809
4830
|
|
|
4810
|
-
//# sourceMappingURL=benchmark-command-
|
|
4831
|
+
//# sourceMappingURL=benchmark-command-CK0UnXAD.js.map
|