@tangle-network/agent-eval 0.150.1 → 0.150.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +29 -0
- package/dist/analyst/index.d.ts +3 -3
- package/dist/analyst/index.js +3 -3
- package/dist/{benchmark-command-CAFwbH0L.js → benchmark-command-BU1Las59.js} +6 -6
- package/dist/{benchmark-command-CAFwbH0L.js.map → benchmark-command-BU1Las59.js.map} +1 -1
- package/dist/benchmarks/index.js +3 -3
- package/dist/builder-eval/index.js +1 -1
- package/dist/campaign/index.js +5 -5
- package/dist/{campaign-CN_7xJdV.js → campaign-la-gEYNz.js} +7 -7
- package/dist/{campaign-CN_7xJdV.js.map → campaign-la-gEYNz.js.map} +1 -1
- package/dist/{chat-client-2bVfrzhN.js → chat-client-Bvmxedyv.js} +2 -2
- package/dist/{chat-client-2bVfrzhN.js.map → chat-client-Bvmxedyv.js.map} +1 -1
- package/dist/cli.js +1 -1
- package/dist/contract/index.js +5 -5
- package/dist/{define-agent-eval-rqNyVhVV.js → define-agent-eval-C8V8sMqP.js} +7 -7
- package/dist/{define-agent-eval-rqNyVhVV.js.map → define-agent-eval-C8V8sMqP.js.map} +1 -1
- package/dist/{descriptive-B5MwKfbf.js → descriptive-jDOuI6mz.js} +22 -2
- package/dist/descriptive-jDOuI6mz.js.map +1 -0
- package/dist/{dspy-rlm-engine-DHI0WrUU.js → dspy-rlm-engine-CBYlPvNy.js} +2 -2
- package/dist/{dspy-rlm-engine-DHI0WrUU.js.map → dspy-rlm-engine-CBYlPvNy.js.map} +1 -1
- package/dist/{eval-campaign-C4jmuM-b.js → eval-campaign-CQuZrLR_.js} +2 -2
- package/dist/{eval-campaign-C4jmuM-b.js.map → eval-campaign-CQuZrLR_.js.map} +1 -1
- package/dist/{external-optimizer-process-Bhmzngf-.js → external-optimizer-process-x9oEXKsU.js} +2 -2
- package/dist/{external-optimizer-process-Bhmzngf-.js.map → external-optimizer-process-x9oEXKsU.js.map} +1 -1
- package/dist/{external-optimizer-subprocess-Dn90UqN2.js → external-optimizer-subprocess-CKNb42oM.js} +7 -3
- package/dist/external-optimizer-subprocess-CKNb42oM.js.map +1 -0
- package/dist/{index-CTKpu9ry.d.ts → index-Aj3WO3_a.d.ts} +2 -2
- package/dist/{index-CTKpu9ry.d.ts.map → index-Aj3WO3_a.d.ts.map} +1 -1
- package/dist/index.d.ts +3 -72
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +11 -11
- package/dist/{integrity-DysDBWDu.js → integrity-DL91tucI.js} +16 -3
- package/dist/integrity-DL91tucI.js.map +1 -0
- package/dist/{judge-calibration-DZkWrm5H.js → judge-calibration-zZjLz8hr.js} +2 -2
- package/dist/{judge-calibration-DZkWrm5H.js.map → judge-calibration-zZjLz8hr.js.map} +1 -1
- package/dist/{llm-judge-CVq33oz1.js → llm-judge-BqqMS8t7.js} +4 -4
- package/dist/{llm-judge-CVq33oz1.js.map → llm-judge-BqqMS8t7.js.map} +1 -1
- package/dist/meta-eval/index.js +2 -2
- package/dist/openapi.json +1 -1
- package/dist/pipelines/index.js +1 -1
- package/dist/{produced-state-jfk8Du3b.js → produced-state-Bnq4FaDO.js} +2 -2
- package/dist/{produced-state-jfk8Du3b.js.map → produced-state-Bnq4FaDO.js.map} +1 -1
- package/dist/reporting.js +2 -2
- package/dist/{reward-hacking-DKI9T52l.js → reward-hacking-C0x0xihA.js} +2 -2
- package/dist/{reward-hacking-DKI9T52l.js.map → reward-hacking-C0x0xihA.js.map} +1 -1
- package/dist/rl.js +3 -3
- package/dist/{rubric-predictive-validity-Cwwyd7ah.js → rubric-predictive-validity-CzxLoZge.js} +2 -2
- package/dist/{rubric-predictive-validity-Cwwyd7ah.js.map → rubric-predictive-validity-CzxLoZge.js.map} +1 -1
- package/dist/{skillopt-optimization-method-DLeUcK-K.js → skillopt-optimization-method-CPBlTcj5.js} +4 -4
- package/dist/{skillopt-optimization-method-DLeUcK-K.js.map → skillopt-optimization-method-CPBlTcj5.js.map} +1 -1
- package/dist/{summary-report-Blysd6Z2.js → summary-report-DW2bEpdB.js} +2 -2
- package/dist/{summary-report-Blysd6Z2.js.map → summary-report-DW2bEpdB.js.map} +1 -1
- package/dist/supervisor-run/index.d.ts +26 -4
- package/dist/supervisor-run/index.d.ts.map +1 -1
- package/dist/supervisor-run/index.js +102 -24
- package/dist/supervisor-run/index.js.map +1 -1
- package/dist/{tool-waste-BDdBZG1F.js → tool-waste-C-VHSRwF.js} +3 -3
- package/dist/{tool-waste-BDdBZG1F.js.map → tool-waste-C-VHSRwF.js.map} +1 -1
- package/dist/{types-yLK8gXE9.d.ts → types-I5WwVzQ7.d.ts} +159 -9
- package/dist/types-I5WwVzQ7.d.ts.map +1 -0
- package/docs/adapters-observability.md +9 -23
- package/docs/campaign-proposers.md +13 -5
- package/docs/concepts.md +3 -4
- package/docs/wire-protocol.md +1 -1
- package/package.json +1 -1
- package/dist/descriptive-B5MwKfbf.js.map +0 -1
- package/dist/external-optimizer-subprocess-Dn90UqN2.js.map +0 -1
- package/dist/integrity-DysDBWDu.js.map +0 -1
- package/dist/types-yLK8gXE9.d.ts.map +0 -1
package/CHANGELOG.md
CHANGED
|
@@ -8,6 +8,35 @@ All notable changes to `@tangle-network/agent-eval` and its sibling `agent-eval-
|
|
|
8
8
|
|
|
9
9
|
---
|
|
10
10
|
|
|
11
|
+
## [0.150.2] — 2026-08-20
|
|
12
|
+
|
|
13
|
+
### Added
|
|
14
|
+
|
|
15
|
+
- `economics.spend` on the supervisor-run report and `spendUsd` on the rollup (#660): both total-spend measurements as named fields with their denominators, never one bare number. `spend.journalDerived` (journal `metered` + `settled` rows) answers execution accounting — what execution observably consumed; `spend.closeRecord` (loops `state.json` `result.spentUsd`, or Runtime `result.json` `spentTotal.usd`, which the analyzer now reads) answers billing — what the store recorded as settled at close. Each run-level measurement carries its record count; each rollup measurement carries `runs`, its own denominator, because the two sums cover different run sets (measured in discovery-lab: 4.762B input tokens journal-derived over 918 runs vs 4.780B close-record over 868 runs, ~0.4% apart). Divergence is a signal to read, not an error. `totalUsd` stays as the collapsed compatibility field; its docstring names the pick order and points to `spend`.
|
|
16
|
+
- `summarizeNumberSeries(values)` on the statistics surface (#659): the distribution fold (`n` / `min` / `p50` / `p90` / `max` / `sum`, nearest-rank quantiles) that previously lived only inside the supervisor-run analyzer as the worker-wall summary. Returns the exported `SeriesDistribution`, or `null` for an empty series. `WallDistribution` is now a type alias of `SeriesDistribution` and the analyzer reuses the helper in place, so the two folds cannot drift.
|
|
17
|
+
|
|
18
|
+
### Fixed
|
|
19
|
+
|
|
20
|
+
- The `/v1/messages` shim translation accepts `system`-role messages inside the `messages` array. Measured blocker (live proof against published 0.150.1): claude CLI 2.1.232 injects system-role turns mid-conversation — the `--max-budget-usd` status line ("USD budget: $0/$1; $1 remaining", sent on every engine run because the bridge always passes the budget flag), session-front listings, and task-tool reminders — and the translation refused each with a 400 `message role must be 'user' or 'assistant'`. The canonical contract already allows the system role at any position, so an injected system turn now translates to a canonical system message in place, with the same text-block joining and `cache_control` stripping as the top-level `system` field, which keeps its slot at the front. Two verbatim captured wire bodies ship as test fixtures.
|
|
21
|
+
- `orchestration.supervisorWallMs` now populates on Runtime's file-backed layout (#658). Runtime writes no completion stamp, so the field read `unavailable` on 918 of 918 analyzable discovery-lab runs. When the completion stamp is missing, the analyzer derives the wall from the journal's first-to-last event stamps (spawned / settled / cancelled / metered) and reports the derivation in the new `supervisorWallSource: 'stamps' | 'journal-span'` field — never a silent substitution. `journal-span` is a lower bound; `idleMs`, `idlePct`, and `workerUtilization` integrate to the same bound. A run with explicit stamps reports `stamps` and is unchanged; a run with no journal keeps both fields `unavailable` with the same reason. An inverted stamp pair stays unavailable (corruption, not absence).
|
|
22
|
+
- `readRuntimeSupervisorRun` no longer throws on a run dir without `spawn-journal.jsonl` (#657). It returns the same absent-shaped sources `readLoopsSupervisorRun` returns for a missing store — `journal` and `workers` null, each with a reason — so every dependent metric reads `unavailable`, never 0. Measured in discovery-lab: 294 of 1,212 real run dirs have no spawn journal, and every fleet consumer wrapped the reader in its own guard. `RuntimeReaderOptions.strict: true` opts back into the throw. A journal that exists but cannot be parsed still throws. `SupervisorRunSources` gains optional `journalMissingReason`, which the analyzer uses verbatim so a non-loops layout names its own journal file.
|
|
23
|
+
- Docs freshness sweep against the 0.150.1 surface. `docs/campaign-proposers.md`: the metered agent CLI path records its measured status (tool calls translate, dual-wire receipts must match admitted attempts), names the `-inf` trap (an agent engine's one registering evaluation costs the full train set against `maxEvaluations`), and the Runtime Knobs table gains `maxEvaluations` (agent engines), `budget.maxRequests` (agent engines), and `expectUsage`; the unproxied-engine example is labeled as the unproxied path. `docs/concepts.md` no longer claims three exported bias probes: `verbosityBias` is the one shipped probe, and `JudgeInsight.positionalBias`/`selfPreference` are caller-filled fields. `docs/adapters-observability.md` drops the deleted `createOtelBridge` design note (the module left in the stranded-module deletions) and points at `createHostedClient` + `/v1/traces/ingest` instead. `clients/python/README.md` documents the fully metered agent-engine path behind `optimizer.anthropicEndpoint`. `docs/wire-protocol.md` version example matches the current release.
|
|
24
|
+
- `examples/agent-engine-optimizer/`: the first example of the metered agent-engine path. `selfImprove` + `gepaOptimizationMethod` with the `autoresearch` engine drives a real `claude` CLI through the loopback Anthropic route (`optimizer.anthropicEndpoint: true`). The example encodes the measured constraints as code: `maxEvaluations` at least the train-set size (one registering aggregate eval costs the whole training pool, enforced at startup), `expectUsage: 'off'` for a deterministic no-LLM evaluator, and output-token headroom for reasoning models. The README documents the receipt fields the run prints and the 0.150.2 system-role translation prerequisite for unmodified CLI runs.
|
|
25
|
+
- `runDispatchServer` no longer aborts every dispatch on current Node. The server keyed client-disconnect detection on the request stream's `close` event, which Node fires when the request BODY completes, so every worker call aborted ~immediately and the two-process `distributed-driver` example failed end to end. Disconnect detection now keys on the response stream's `close` with `writableEnded` still false. `src/adapters/http.test.ts` covers both directions: a slow dispatch completes untouched, and a mid-flight client abort reaches the worker's `ctx.signal`.
|
|
26
|
+
- Examples freshness sweep — every directory under `examples/` re-verified compile + run as documented:
|
|
27
|
+
- `distributed-driver` names its dispatch (`dispatchRef`) — `httpDispatch` returns an anonymous function, which the campaign manifest rejects — and sets `expectUsage: 'off'` for its stub worker.
|
|
28
|
+
- `multi-shot-optimization` and `selfimprove-quickstart` printed `hold` while their READMEs promised `ship`: the gate's exact paired test cannot reach 95% significance below six paired holdout observations. Both now hold six cases and ship; the READMEs state the floor.
|
|
29
|
+
- `fine-tune-with-prime-rl`: the shipped fixture predated the 0.126 `RunRecord` shape (`costProvenance`, `terminalOutcome`) and its rows sat on the holdout split, so the documented command crashed and, once past validation, exported ZERO rows silently. The fixture is regenerated on the `search` split with top-level `prompt`/`completion` (`outcome.raw` is numeric-only by contract), text lookups throw on missing fields instead of exporting placeholder rows, an empty export now fails loudly, and the README command paths and pinned output are verified.
|
|
30
|
+
- `customer-otel-traces` read the retired `failureMode` field, so its "Failures" section silently vanished; it now counts `outcome.raw.error_span_count` and the README pins the real report, including the failure-class recommendation.
|
|
31
|
+
- `hosted-ingest-server` gains the README the index always pointed at, and the file-header `curl` now carries the required `X-Tangle-Wire-Version` header (verified against the live server: without it, 400).
|
|
32
|
+
- `same-sandbox-harness`'s documented `tsx -e` snippet used top-level await, which tsx rejects in string-eval; the README now uses a dynamic import (verified to run).
|
|
33
|
+
- `foreign-agent-quickstart` moved to the standard `LLM_BASE_URL`/`LLM_API_KEY`/`LLM_MODEL` variables and declares `expectUsage: 'off'` for its unmetered demo transport.
|
|
34
|
+
- Optimizer examples (`self-improve-optimizer`, `compare-optimization-methods`, GSM8K, AppWorld) encode the measured GEPA lessons: `LLM_BASE_URL` is required (no silent `api.openai.com` default), every GEPA engine run pins `reflection_lm_kwargs: { num_retries: 0 }` via the shared `examples/_shared/gepa-reflection.ts`, worker `maxTokens` is env-tunable with a reasoning-model headroom note, and `GEPA_MAX_EVALUATIONS` documents the >= train-partition-size floor.
|
|
35
|
+
- Model ids refreshed across examples: retired `gpt-4o`/`gpt-4.1-mini`/`claude-sonnet-4-6` literals replaced with served ids (`deepseek-v4-flash`, `glm-5.3`; RunRecord literals carry the required snapshot suffix).
|
|
36
|
+
- Removed `examples/benchmarks/appworld/halo_chat.py` and `halo-chat.sh`: launchers for an external analysis engine that no documented example references. `run-bench.ts` requires `APPWORLD_DIR` explicitly instead of defaulting to a machine-local path.
|
|
37
|
+
|
|
38
|
+
---
|
|
39
|
+
|
|
11
40
|
## [0.150.1] — 2026-08-20
|
|
12
41
|
|
|
13
42
|
### Added
|
package/dist/analyst/index.d.ts
CHANGED
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
import { a as MultiLayerVerifier, c as VerifyOptions, o as Severity } from "../multi-layer-verifier-BUaQ4C17.js";
|
|
2
|
-
import { A as SemanticConceptJudgeInput, C as parseFindingSubject, E as createDspyRlmTraceEngine, F as RunScoreWeights, P as RunScore, S as findingSubjectGrammarPromptFor, T as DspyRlmTraceEngineOptions, _ as FINDING_SUBJECT_SYNTAX, a as FAILURE_MODE_KIND_SPEC, b as FindingSubjectStringSchema, c as emitControlIntegrityFindings, d as FindingsStore, f as PersistedFinding, g as FINDING_SUBJECT_KINDS, h as FINDING_SUBJECT_GRAMMAR_PROMPT, i as IMPROVEMENT_KIND_SPEC, j as SemanticConceptJudgeOptions, l as DiffPolicy, m as diffFindings, n as KNOWLEDGE_POISONING_KIND_SPEC, o as CONTROL_INTEGRITY_ANALYST, p as defaultIsMaterial, r as KNOWLEDGE_GAP_KIND_SPEC, s as ControlIntegrityAnalyst, t as DEFAULT_TRACE_ANALYST_KINDS, u as FindingsDiff, v as FindingSubject, w as renderFindingSubject, x as KIND_EXPECTED_SUBJECTS, y as FindingSubjectKind } from "../index-
|
|
2
|
+
import { A as SemanticConceptJudgeInput, C as parseFindingSubject, E as createDspyRlmTraceEngine, F as RunScoreWeights, P as RunScore, S as findingSubjectGrammarPromptFor, T as DspyRlmTraceEngineOptions, _ as FINDING_SUBJECT_SYNTAX, a as FAILURE_MODE_KIND_SPEC, b as FindingSubjectStringSchema, c as emitControlIntegrityFindings, d as FindingsStore, f as PersistedFinding, g as FINDING_SUBJECT_KINDS, h as FINDING_SUBJECT_GRAMMAR_PROMPT, i as IMPROVEMENT_KIND_SPEC, j as SemanticConceptJudgeOptions, l as DiffPolicy, m as diffFindings, n as KNOWLEDGE_POISONING_KIND_SPEC, o as CONTROL_INTEGRITY_ANALYST, p as defaultIsMaterial, r as KNOWLEDGE_GAP_KIND_SPEC, s as ControlIntegrityAnalyst, t as DEFAULT_TRACE_ANALYST_KINDS, u as FindingsDiff, v as FindingSubject, w as renderFindingSubject, x as KIND_EXPECTED_SUBJECTS, y as FindingSubjectKind } from "../index-Aj3WO3_a.js";
|
|
3
3
|
import { b as CustomTokenPricing, c as CostLedgerHandle } from "../cost-ledger-DbQdN3nO.js";
|
|
4
4
|
import { C as TraceEvent, _ as Span, f as Run, n as BudgetLedgerEntry, t as Artifact } from "../schema-BtVldJ3T.js";
|
|
5
5
|
import { C as SandboxSdkTransportOpts, S as RouterTransportOpts, _ as CliBridgeTransportOpts, a as JudgeInput, b as DirectProviderTransportOpts, f as ChatCallOpts, g as ChatTransport, h as ChatResponse, i as JudgeFn, m as ChatRequest, p as ChatClient, v as CreateChatClientOpts, w as createChatClient, x as MockTransportOpts, y as CustomTransportOpts } from "../types-jUBXJ7Iz.js";
|
|
@@ -1299,7 +1299,7 @@ declare const ANALYST_BENCHMARK_HELP = "agent-eval analyst-benchmark\n\nRun the
|
|
|
1299
1299
|
declare const ANALYST_BENCHMARK_IMPLEMENTATION_DIGEST_ALGORITHM = "sha256-canonical-source-manifest";
|
|
1300
1300
|
declare const ANALYST_BENCHMARK_DEPENDENCY_LOCK_DIGEST_ALGORITHM = "sha256-canonical-file-manifest";
|
|
1301
1301
|
declare const ANALYST_BENCHMARK_DEPENDENCY_LOCK_FILES: readonly string[];
|
|
1302
|
-
declare const ANALYST_BENCHMARK_DEPENDENCY_LOCK_SHA256 = "
|
|
1302
|
+
declare const ANALYST_BENCHMARK_DEPENDENCY_LOCK_SHA256 = "61ad75e662e240404f523d7dad66e18145df9c35b1ce8630468970fef2a00162";
|
|
1303
1303
|
/** The published benchmark evidence was produced at this package version, by
|
|
1304
1304
|
* the retired one-shot direct runner, before trace analysts moved to the
|
|
1305
1305
|
* recursive DSPy RLM engine. Both evidence digests below are historical facts
|
|
@@ -1311,7 +1311,7 @@ declare const ANALYST_BENCHMARK_EVIDENCE_PACKAGE_VERSION = "0.137.0";
|
|
|
1311
1311
|
declare const ANALYST_BENCHMARK_EVIDENCE_DEPENDENCY_LOCK_SHA256 = "1e03f2daed356d60316aabefb407ec1e437ac94d408d61eea4ae096e9c6fbb5b";
|
|
1312
1312
|
declare const ANALYST_BENCHMARK_EVIDENCE_IMPLEMENTATION_SHA256 = "4dba263b6256a30d56c7fdb2d992d3a953c0035d731f359b704db806f68f75ac";
|
|
1313
1313
|
declare const ANALYST_BENCHMARK_IMPLEMENTATION_FILES: readonly string[];
|
|
1314
|
-
declare const ANALYST_BENCHMARK_IMPLEMENTATION_SHA256 = "
|
|
1314
|
+
declare const ANALYST_BENCHMARK_IMPLEMENTATION_SHA256 = "40b66a9ab0dce2cfc8da9e3bfac96665f79e3bd666ba8e5d9a2b31d2d8a2c161";
|
|
1315
1315
|
declare function analystBenchmarkImplementationDigest(): string;
|
|
1316
1316
|
declare function analystBenchmarkDependencyLockDigest(): string;
|
|
1317
1317
|
//#endregion
|
package/dist/analyst/index.js
CHANGED
|
@@ -3,11 +3,11 @@ import { i as CostLedger } from "../cost-ledger-B1qx30B4.js";
|
|
|
3
3
|
import { i as isProposalFinding, n as clamp01, r as assertProposalFindings, t as aggregateRunScore } from "../run-score-lDzV0X8j.js";
|
|
4
4
|
import { a as settleUsageReceiptFromCostLedger, n as makeFinding, r as makeProposalFinding, s as validateUsageSettlementTimeout, t as computeFindingId } from "../types-BI4fT3HN.js";
|
|
5
5
|
import { A as coerceJson, B as renderFindingSubject, D as RawAnalystFindingSchema, E as RawAnalystEvidenceSchema, F as FINDING_SUBJECT_SYNTAX, H as resolveTraceAnalystLimits, I as FindingSubjectStringSchema, L as KIND_EXPECTED_SUBJECTS, M as stripCodeFences, N as FINDING_SUBJECT_GRAMMAR_PROMPT, O as evidenceRefsFromRawFinding, P as FINDING_SUBJECT_KINDS, R as findingSubjectGrammarPromptFor, T as RAW_FINDING_SCHEMA_PROMPT, V as DEFAULT_TRACE_ANALYST_LIMITS, a as buildTraceToolsForGroup, i as runTraceAnalyst, j as coerceToFindingRows, k as parseRawFinding, n as renderPriorFindings, r as renderUpstreamFindings, t as createTraceAnalyst, w as ANALYST_SEVERITIES, z as parseFindingSubject } from "../kind-factory-DmAa0h3K.js";
|
|
6
|
-
import { a as assertExactRegistryRunOpts, c as KNOWLEDGE_GAP_KIND_SPEC, d as CONTROL_INTEGRITY_ANALYST, f as ControlIntegrityAnalyst, h as deriveEfficiencyFindings, i as ExactAnalystRunExecutionError, l as IMPROVEMENT_KIND_SPEC, m as behavioralAnalyst, n as buildDefaultAnalystRegistry, o as DEFAULT_TRACE_ANALYST_KINDS, p as emitControlIntegrityFindings, r as AnalystRegistry, s as KNOWLEDGE_POISONING_KIND_SPEC, t as createChatClient, u as FAILURE_MODE_KIND_SPEC } from "../chat-client-
|
|
7
|
-
import { t as createDspyRlmTraceEngine } from "../dspy-rlm-engine-
|
|
6
|
+
import { a as assertExactRegistryRunOpts, c as KNOWLEDGE_GAP_KIND_SPEC, d as CONTROL_INTEGRITY_ANALYST, f as ControlIntegrityAnalyst, h as deriveEfficiencyFindings, i as ExactAnalystRunExecutionError, l as IMPROVEMENT_KIND_SPEC, m as behavioralAnalyst, n as buildDefaultAnalystRegistry, o as DEFAULT_TRACE_ANALYST_KINDS, p as emitControlIntegrityFindings, r as AnalystRegistry, s as KNOWLEDGE_POISONING_KIND_SPEC, t as createChatClient, u as FAILURE_MODE_KIND_SPEC } from "../chat-client-Bvmxedyv.js";
|
|
7
|
+
import { t as createDspyRlmTraceEngine } from "../dspy-rlm-engine-CBYlPvNy.js";
|
|
8
8
|
import { a as diffFindings, i as defaultIsMaterial, n as runSemanticConceptJudge, r as FindingsStore, t as SEMANTIC_CONCEPT_JUDGE_VERSION } from "../semantic-concept-judge-laMCnTLn.js";
|
|
9
9
|
import { a as scoreAnalystFindings, i as summarizeAnalystBenchmarkRunner, n as runAnalystBenchmark, r as traceStoreEvidenceResolver, t as registryBenchmarkRunner } from "../benchmark-BhT16ep9.js";
|
|
10
|
-
import { $ as analystBenchmarkDependencyLockDigest, A as compareAnalystRunners, B as DEFAULT_MAX_VERIFICATION_ARTIFACT_BYTES, C as analystDefinitionProtocolSha256, D as readAnalystBenchmarkArtifact, E as expandCodeTraceFailureBlocks, F as MAX_INCORRECT_BLOCKS, G as ANALYST_BENCHMARK_DEPENDENCY_LOCK_FILES, H as loadCodeTraceVerificationArtifacts, I as MAX_INCORRECT_BLOCK_STEPS, J as ANALYST_BENCHMARK_EVIDENCE_IMPLEMENTATION_SHA256, K as ANALYST_BENCHMARK_DEPENDENCY_LOCK_SHA256, L as publicBenchmarkProtocolSha256, M as effectiveAnalystProtocolSha256, N as readAnalystInstructionsOverride, O as renderCodeTraceCalibrationMarkdown, P as CODE_TRACE_BENCH_ANALYST_PROMPT, Q as ANALYST_BENCHMARK_IMPLEMENTATION_SHA256, R as publicBenchmarkRlmInstructions, S as analystDefinitionAsymmetries, T as emptyPublicBenchmarkRunner, U as parseVerificationOutcome, V as appendVerificationArtifactsToOtlp, W as ANALYST_BENCHMARK_DEPENDENCY_LOCK_DIGEST_ALGORITHM, X as ANALYST_BENCHMARK_IMPLEMENTATION_DIGEST_ALGORITHM, Y as ANALYST_BENCHMARK_EVIDENCE_PACKAGE_VERSION, Z as ANALYST_BENCHMARK_IMPLEMENTATION_FILES, _ as runReplVariableAnalystDefinition, a as primeAnalystProtocolSha256, at as AGENT_RX_UPSTREAM_REVISION, b as runChunkedAnalystDefinition, c as nodeHttpPrimeBridgeTransport, ct as codeTraceBenchCase, d as publicBenchmarkDistributions, dt as agentRxPredictionsToFindings, et as analystBenchmarkImplementationDigest, f as publicBenchmarkSelectionReport, ft as normalizeAgentRxCategory, g as rlmEngineLimits, h as publicRlmAnalystDefinition, i as createPrimeBenchmarkRunner, it as ANALYST_BENCHMARK_OBSERVATIONS_FILE, j as analystInstructionsOverrideFromText, k as summarizeCodeTraceCalibration, l as loadPublicBenchmarkRows, lt as codeTracerPredictionsToFindings, m as createPublicBenchmarkRlmRunner, mt as normalizeBenchmarkLabel, n as runAnalystBenchmarkCommand, nt as ANALYST_BENCHMARK_LOCAL_RECEIPT_FILE, o as primeCodeTraceAnalystDefinition, ot as renderAgentRxCalibrationMarkdown, p as selectPublicBenchmarkRows, pt as roundAgentRxStep, q as ANALYST_BENCHMARK_EVIDENCE_DEPENDENCY_LOCK_SHA256, r as renderAnalystBenchmarkMarkdown, rt as ANALYST_BENCHMARK_MANIFEST_FILE, s as runInlineAnalystDefinition, st as summarizeAgentRxCalibration, t as ANALYST_BENCHMARK_HELP, tt as ANALYST_BENCHMARK_COST_LEDGER_FILE, u as preparePublicAnalystBenchmark, ut as agentRxBenchmarkCase, v as createPublicBenchmarkDirectRunner, w as adaptPublicBenchmarkFindings, x as AnalystExpressivenessError, y as publicDirectAnalystDefinition, z as publicBenchmarkSystemPrompt } from "../benchmark-command-
|
|
10
|
+
import { $ as analystBenchmarkDependencyLockDigest, A as compareAnalystRunners, B as DEFAULT_MAX_VERIFICATION_ARTIFACT_BYTES, C as analystDefinitionProtocolSha256, D as readAnalystBenchmarkArtifact, E as expandCodeTraceFailureBlocks, F as MAX_INCORRECT_BLOCKS, G as ANALYST_BENCHMARK_DEPENDENCY_LOCK_FILES, H as loadCodeTraceVerificationArtifacts, I as MAX_INCORRECT_BLOCK_STEPS, J as ANALYST_BENCHMARK_EVIDENCE_IMPLEMENTATION_SHA256, K as ANALYST_BENCHMARK_DEPENDENCY_LOCK_SHA256, L as publicBenchmarkProtocolSha256, M as effectiveAnalystProtocolSha256, N as readAnalystInstructionsOverride, O as renderCodeTraceCalibrationMarkdown, P as CODE_TRACE_BENCH_ANALYST_PROMPT, Q as ANALYST_BENCHMARK_IMPLEMENTATION_SHA256, R as publicBenchmarkRlmInstructions, S as analystDefinitionAsymmetries, T as emptyPublicBenchmarkRunner, U as parseVerificationOutcome, V as appendVerificationArtifactsToOtlp, W as ANALYST_BENCHMARK_DEPENDENCY_LOCK_DIGEST_ALGORITHM, X as ANALYST_BENCHMARK_IMPLEMENTATION_DIGEST_ALGORITHM, Y as ANALYST_BENCHMARK_EVIDENCE_PACKAGE_VERSION, Z as ANALYST_BENCHMARK_IMPLEMENTATION_FILES, _ as runReplVariableAnalystDefinition, a as primeAnalystProtocolSha256, at as AGENT_RX_UPSTREAM_REVISION, b as runChunkedAnalystDefinition, c as nodeHttpPrimeBridgeTransport, ct as codeTraceBenchCase, d as publicBenchmarkDistributions, dt as agentRxPredictionsToFindings, et as analystBenchmarkImplementationDigest, f as publicBenchmarkSelectionReport, ft as normalizeAgentRxCategory, g as rlmEngineLimits, h as publicRlmAnalystDefinition, i as createPrimeBenchmarkRunner, it as ANALYST_BENCHMARK_OBSERVATIONS_FILE, j as analystInstructionsOverrideFromText, k as summarizeCodeTraceCalibration, l as loadPublicBenchmarkRows, lt as codeTracerPredictionsToFindings, m as createPublicBenchmarkRlmRunner, mt as normalizeBenchmarkLabel, n as runAnalystBenchmarkCommand, nt as ANALYST_BENCHMARK_LOCAL_RECEIPT_FILE, o as primeCodeTraceAnalystDefinition, ot as renderAgentRxCalibrationMarkdown, p as selectPublicBenchmarkRows, pt as roundAgentRxStep, q as ANALYST_BENCHMARK_EVIDENCE_DEPENDENCY_LOCK_SHA256, r as renderAnalystBenchmarkMarkdown, rt as ANALYST_BENCHMARK_MANIFEST_FILE, s as runInlineAnalystDefinition, st as summarizeAgentRxCalibration, t as ANALYST_BENCHMARK_HELP, tt as ANALYST_BENCHMARK_COST_LEDGER_FILE, u as preparePublicAnalystBenchmark, ut as agentRxBenchmarkCase, v as createPublicBenchmarkDirectRunner, w as adaptPublicBenchmarkFindings, x as AnalystExpressivenessError, y as publicDirectAnalystDefinition, z as publicBenchmarkSystemPrompt } from "../benchmark-command-BU1Las59.js";
|
|
11
11
|
import { a as extractPrimeJsonObject, c as primeProtocolSha256, d as runPrimeExchange, f as decodeReplyRows, i as emptyPrimeRawUsage, l as primeReplyDefect, n as buildPrimePrompt, o as mergePrimeRawUsage, r as buildPrimeRepairPrompt, s as normalizePrimeUsage, t as analystUsageReceiptFromPrimeUsage, u as projectPrimeTrajectory } from "../prime-protocol-6tZTVsWm.js";
|
|
12
12
|
import { existsSync, readFileSync, readdirSync, statSync } from "node:fs";
|
|
13
13
|
import { join } from "node:path";
|
|
@@ -4,14 +4,14 @@ import { a as resolveModelPricing } from "./metrics-Qv-cpptD.js";
|
|
|
4
4
|
import { a as CostLedgerPersistenceError, i as CostLedger, n as CostCallConflictError, o as CostReceiptCaptureError, r as CostCeilingReachedError, s as CostReservationExceededError, t as CostAccountingIncompleteError } from "./cost-ledger-B1qx30B4.js";
|
|
5
5
|
import { c as writeLedgerFileAtomically, s as withLedgerFileLock } from "./ledger-core-DTae9rv_.js";
|
|
6
6
|
import { i as hashCanonical, r as canonicalString } from "./canonical-D-XsTQ6_.js";
|
|
7
|
-
import { b as createRunCostLedger, i as startExternalOptimizerModelProxy, r as runWithCleanup, v as resolveExternalOptimizerProcessLimits, x as fsCampaignStorage } from "./external-optimizer-subprocess-
|
|
7
|
+
import { b as createRunCostLedger, i as startExternalOptimizerModelProxy, r as runWithCleanup, v as resolveExternalOptimizerProcessLimits, x as fsCampaignStorage } from "./external-optimizer-subprocess-CKNb42oM.js";
|
|
8
8
|
import { n as makeFinding, o as usageReceiptFromCostLedger } from "./types-BI4fT3HN.js";
|
|
9
9
|
import { D as RawAnalystFindingSchema, O as evidenceRefsFromRawFinding, T as RAW_FINDING_SCHEMA_PROMPT, h as TRACE_ANALYSIS_LIMITS, i as runTraceAnalyst } from "./kind-factory-DmAa0h3K.js";
|
|
10
10
|
import { c as callLlmJson, r as LlmResponseError, t as LlmCallError } from "./llm-client-Bg32RW0j.js";
|
|
11
|
-
import { t as createDspyRlmTraceEngine } from "./dspy-rlm-engine-
|
|
11
|
+
import { t as createDspyRlmTraceEngine } from "./dspy-rlm-engine-CBYlPvNy.js";
|
|
12
12
|
import { i as otlpTextToTraceAnalysisStore, r as createOtlpBufferTraceStore, t as DEFAULT_MAX_TRACE_FILE_BYTES } from "./store-otlp-CDYWW_8N.js";
|
|
13
13
|
import { i as summarizeAnalystBenchmarkRunner, n as runAnalystBenchmark, r as traceStoreEvidenceResolver } from "./benchmark-BhT16ep9.js";
|
|
14
|
-
import { n as acquireSingleRunLock } from "./external-optimizer-process-
|
|
14
|
+
import { n as acquireSingleRunLock } from "./external-optimizer-process-x9oEXKsU.js";
|
|
15
15
|
import { c as primeProtocolSha256, d as runPrimeExchange, f as decodeReplyRows, n as buildPrimePrompt, p as assertEqualDeclarativeTerms, t as analystUsageReceiptFromPrimeUsage, u as projectPrimeTrajectory } from "./prime-protocol-6tZTVsWm.js";
|
|
16
16
|
import { createHash, randomUUID } from "node:crypto";
|
|
17
17
|
import { constants, existsSync, lstatSync, readFileSync } from "node:fs";
|
|
@@ -1209,7 +1209,7 @@ const ANALYST_BENCHMARK_DEPENDENCY_LOCK_FILES = Object.freeze([
|
|
|
1209
1209
|
"package.json",
|
|
1210
1210
|
"pnpm-lock.yaml"
|
|
1211
1211
|
]);
|
|
1212
|
-
const ANALYST_BENCHMARK_DEPENDENCY_LOCK_SHA256 = "
|
|
1212
|
+
const ANALYST_BENCHMARK_DEPENDENCY_LOCK_SHA256 = "61ad75e662e240404f523d7dad66e18145df9c35b1ce8630468970fef2a00162";
|
|
1213
1213
|
/** The published benchmark evidence was produced at this package version, by
|
|
1214
1214
|
* the retired one-shot direct runner, before trace analysts moved to the
|
|
1215
1215
|
* recursive DSPy RLM engine. Both evidence digests below are historical facts
|
|
@@ -1328,7 +1328,7 @@ const ANALYST_BENCHMARK_IMPLEMENTATION_FILES = Object.freeze([
|
|
|
1328
1328
|
"src/trace/raw-provider-sink.ts",
|
|
1329
1329
|
"src/verdict-cache.ts"
|
|
1330
1330
|
]);
|
|
1331
|
-
const ANALYST_BENCHMARK_IMPLEMENTATION_SHA256 = "
|
|
1331
|
+
const ANALYST_BENCHMARK_IMPLEMENTATION_SHA256 = "40b66a9ab0dce2cfc8da9e3bfac96665f79e3bd666ba8e5d9a2b31d2d8a2c161";
|
|
1332
1332
|
function analystBenchmarkImplementationDigest() {
|
|
1333
1333
|
return ANALYST_BENCHMARK_IMPLEMENTATION_SHA256;
|
|
1334
1334
|
}
|
|
@@ -6370,4 +6370,4 @@ function shellQuote(value) {
|
|
|
6370
6370
|
//#endregion
|
|
6371
6371
|
export { analystBenchmarkDependencyLockDigest as $, compareAnalystRunners as A, DEFAULT_MAX_VERIFICATION_ARTIFACT_BYTES as B, analystDefinitionProtocolSha256 as C, readAnalystBenchmarkArtifact as D, expandCodeTraceFailureBlocks as E, MAX_INCORRECT_BLOCKS as F, ANALYST_BENCHMARK_DEPENDENCY_LOCK_FILES as G, loadCodeTraceVerificationArtifacts as H, MAX_INCORRECT_BLOCK_STEPS as I, ANALYST_BENCHMARK_EVIDENCE_IMPLEMENTATION_SHA256 as J, ANALYST_BENCHMARK_DEPENDENCY_LOCK_SHA256 as K, publicBenchmarkProtocolSha256 as L, effectiveAnalystProtocolSha256 as M, readAnalystInstructionsOverride as N, renderCodeTraceCalibrationMarkdown as O, CODE_TRACE_BENCH_ANALYST_PROMPT as P, ANALYST_BENCHMARK_IMPLEMENTATION_SHA256 as Q, publicBenchmarkRlmInstructions as R, analystDefinitionAsymmetries as S, emptyPublicBenchmarkRunner as T, parseVerificationOutcome as U, appendVerificationArtifactsToOtlp as V, ANALYST_BENCHMARK_DEPENDENCY_LOCK_DIGEST_ALGORITHM as W, ANALYST_BENCHMARK_IMPLEMENTATION_DIGEST_ALGORITHM as X, ANALYST_BENCHMARK_EVIDENCE_PACKAGE_VERSION as Y, ANALYST_BENCHMARK_IMPLEMENTATION_FILES as Z, runReplVariableAnalystDefinition as _, primeAnalystProtocolSha256 as a, AGENT_RX_UPSTREAM_REVISION as at, runChunkedAnalystDefinition as b, nodeHttpPrimeBridgeTransport as c, codeTraceBenchCase as ct, publicBenchmarkDistributions as d, agentRxPredictionsToFindings as dt, analystBenchmarkImplementationDigest as et, publicBenchmarkSelectionReport as f, normalizeAgentRxCategory as ft, rlmEngineLimits as g, publicRlmAnalystDefinition as h, createPrimeBenchmarkRunner as i, ANALYST_BENCHMARK_OBSERVATIONS_FILE as it, analystInstructionsOverrideFromText as j, summarizeCodeTraceCalibration as k, loadPublicBenchmarkRows as l, codeTracerPredictionsToFindings as lt, createPublicBenchmarkRlmRunner as m, normalizeBenchmarkLabel as mt, runAnalystBenchmarkCommand as n, ANALYST_BENCHMARK_LOCAL_RECEIPT_FILE as nt, primeCodeTraceAnalystDefinition as o, renderAgentRxCalibrationMarkdown as ot, selectPublicBenchmarkRows as p, roundAgentRxStep as pt, ANALYST_BENCHMARK_EVIDENCE_DEPENDENCY_LOCK_SHA256 as q, renderAnalystBenchmarkMarkdown as r, ANALYST_BENCHMARK_MANIFEST_FILE as rt, runInlineAnalystDefinition as s, summarizeAgentRxCalibration as st, ANALYST_BENCHMARK_HELP as t, ANALYST_BENCHMARK_COST_LEDGER_FILE as tt, preparePublicAnalystBenchmark as u, agentRxBenchmarkCase as ut, createPublicBenchmarkDirectRunner as v, adaptPublicBenchmarkFindings as w, AnalystExpressivenessError as x, publicDirectAnalystDefinition as y, publicBenchmarkSystemPrompt as z };
|
|
6372
6372
|
|
|
6373
|
-
//# sourceMappingURL=benchmark-command-
|
|
6373
|
+
//# sourceMappingURL=benchmark-command-BU1Las59.js.map
|