@tangle-network/agent-eval 0.144.3 → 0.144.5

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (82) hide show
  1. package/CHANGELOG.md +14 -0
  2. package/dist/analyst/index.d.ts +363 -16
  3. package/dist/analyst/index.d.ts.map +1 -1
  4. package/dist/analyst/index.js +8 -8
  5. package/dist/{benchmark-CYtcIF2V.js → benchmark-B181aMF9.js} +2 -2
  6. package/dist/{benchmark-CYtcIF2V.js.map → benchmark-B181aMF9.js.map} +1 -1
  7. package/dist/{benchmark-CP6kWfj8.d.ts → benchmark-Fmo42QVE.d.ts} +3 -3
  8. package/dist/{benchmark-CP6kWfj8.d.ts.map → benchmark-Fmo42QVE.d.ts.map} +1 -1
  9. package/dist/{benchmark-command-CQPKRUr-.js → benchmark-command-CA_NFOmy.js} +919 -35
  10. package/dist/benchmark-command-CA_NFOmy.js.map +1 -0
  11. package/dist/benchmarks/index.d.ts +1 -1
  12. package/dist/benchmarks/index.js +1 -1
  13. package/dist/{benchmarks-CkG1bWFa.js → benchmarks-CWbsj0t4.js} +2 -2
  14. package/dist/{benchmarks-CkG1bWFa.js.map → benchmarks-CWbsj0t4.js.map} +1 -1
  15. package/dist/campaign/index.d.ts +4 -4
  16. package/dist/campaign/index.js +2 -2
  17. package/dist/{campaign-DjGFyPxH.js → campaign-ClpnD7Ug.js} +2 -2
  18. package/dist/{campaign-DjGFyPxH.js.map → campaign-ClpnD7Ug.js.map} +1 -1
  19. package/dist/cli.js +1 -1
  20. package/dist/{client-Bbht4xxl.d.ts → client-DAb7MWtL.d.ts} +2 -2
  21. package/dist/{client-Bbht4xxl.d.ts.map → client-DAb7MWtL.d.ts.map} +1 -1
  22. package/dist/{completion-verifier-EJERfFwF.d.ts → completion-verifier-VvpHRu78.d.ts} +3 -3
  23. package/dist/{completion-verifier-EJERfFwF.d.ts.map → completion-verifier-VvpHRu78.d.ts.map} +1 -1
  24. package/dist/contract/index.d.ts +6 -6
  25. package/dist/contract/index.js +4 -4
  26. package/dist/control.d.ts +2 -2
  27. package/dist/{default-registry-iXfu2trt.d.ts → default-registry-BwbZ9N9v.d.ts} +5 -5
  28. package/dist/{default-registry-iXfu2trt.d.ts.map → default-registry-BwbZ9N9v.d.ts.map} +1 -1
  29. package/dist/{default-registry-SOyHB6qG.js → default-registry-RLNNoeEP.js} +3 -3
  30. package/dist/{default-registry-SOyHB6qG.js.map → default-registry-RLNNoeEP.js.map} +1 -1
  31. package/dist/{dspy-rlm-engine-IRCG8kdi.js → dspy-rlm-engine-19FQEMBK.js} +2 -2
  32. package/dist/{dspy-rlm-engine-IRCG8kdi.js.map → dspy-rlm-engine-19FQEMBK.js.map} +1 -1
  33. package/dist/{exact-types-BygCBR4L.d.ts → exact-types-BQ7W90C4.d.ts} +2 -2
  34. package/dist/{exact-types-BygCBR4L.d.ts.map → exact-types-BQ7W90C4.d.ts.map} +1 -1
  35. package/dist/{extract-usage-7l1Xq5ti.js → extract-usage-BW27f3XW.js} +2 -2
  36. package/dist/{extract-usage-7l1Xq5ti.js.map → extract-usage-BW27f3XW.js.map} +1 -1
  37. package/dist/{feedback-trajectory-CSIkRLQX.d.ts → feedback-trajectory-WK7x4mhy.d.ts} +3 -3
  38. package/dist/{feedback-trajectory-CSIkRLQX.d.ts.map → feedback-trajectory-WK7x4mhy.d.ts.map} +1 -1
  39. package/dist/hosted/index.d.ts +2 -2
  40. package/dist/{index-BuHs_OnD.d.ts → index-CpxZSlB7.d.ts} +6 -6
  41. package/dist/{index-BuHs_OnD.d.ts.map → index-CpxZSlB7.d.ts.map} +1 -1
  42. package/dist/{index-D_qTihaQ.d.ts → index-CsuAo2-J.d.ts} +4 -4
  43. package/dist/{index-D_qTihaQ.d.ts.map → index-CsuAo2-J.d.ts.map} +1 -1
  44. package/dist/index.d.ts +13 -13
  45. package/dist/index.js +10 -10
  46. package/dist/{kind-factory-Bvwe3pup.js → kind-factory-BHIgPmzS.js} +2 -2
  47. package/dist/{kind-factory-Bvwe3pup.js.map → kind-factory-BHIgPmzS.js.map} +1 -1
  48. package/dist/multishot/index.d.ts +1 -1
  49. package/dist/multishot/index.js +1 -1
  50. package/dist/multishot/index.js.map +1 -1
  51. package/dist/openapi.json +1 -1
  52. package/dist/{replay-DQ-55DC_.d.ts → replay-B7S7Pdbw.d.ts} +3 -3
  53. package/dist/{replay-DQ-55DC_.d.ts.map → replay-B7S7Pdbw.d.ts.map} +1 -1
  54. package/dist/{replay-CqOsGjzU.js → replay-GW61ezMW.js} +4 -4
  55. package/dist/{replay-CqOsGjzU.js.map → replay-GW61ezMW.js.map} +1 -1
  56. package/dist/rl.d.ts +1 -1
  57. package/dist/{run-evidence-H1vRpIdT.d.ts → run-evidence-j5Ynww6L.d.ts} +2 -2
  58. package/dist/{run-evidence-H1vRpIdT.d.ts.map → run-evidence-j5Ynww6L.d.ts.map} +1 -1
  59. package/dist/{semantic-concept-judge-Do5aM9wP.js → semantic-concept-judge-DKRtp2sY.js} +2 -2
  60. package/dist/{semantic-concept-judge-Do5aM9wP.js.map → semantic-concept-judge-DKRtp2sY.js.map} +1 -1
  61. package/dist/{skill-usage-DtpLou9L.d.ts → skill-usage-DUvvudWR.d.ts} +5 -5
  62. package/dist/{skill-usage-DtpLou9L.d.ts.map → skill-usage-DUvvudWR.d.ts.map} +1 -1
  63. package/dist/{skillopt-optimization-method-CvSJGdm3.d.ts → skillopt-optimization-method-CGz9ywhM.d.ts} +4 -4
  64. package/dist/{skillopt-optimization-method-CvSJGdm3.d.ts.map → skillopt-optimization-method-CGz9ywhM.d.ts.map} +1 -1
  65. package/dist/{store-otlp-D4I90_vR.js → store-otlp-CKtTpRhv.js} +2 -2
  66. package/dist/{store-otlp-D4I90_vR.js.map → store-otlp-CKtTpRhv.js.map} +1 -1
  67. package/dist/{tool-groups-CMmsgTzj.d.ts → tool-groups-ByZiqpVk.d.ts} +4 -4
  68. package/dist/{tool-groups-CMmsgTzj.d.ts.map → tool-groups-ByZiqpVk.d.ts.map} +1 -1
  69. package/dist/traces.d.ts +3 -3
  70. package/dist/traces.js +4 -4
  71. package/dist/{types-y8jrxXWd.d.ts → types-CZt1PBIk.d.ts} +19 -1
  72. package/dist/{types-y8jrxXWd.d.ts.map → types-CZt1PBIk.d.ts.map} +1 -1
  73. package/dist/{types-DcJxgsLy.d.ts → types-Dcoaqcsc.d.ts} +2 -2
  74. package/dist/{types-DcJxgsLy.d.ts.map → types-Dcoaqcsc.d.ts.map} +1 -1
  75. package/dist/{usage-receipt-CgxMEBZq.js → usage-receipt-EVI8B8Xu.js} +8 -1
  76. package/dist/usage-receipt-EVI8B8Xu.js.map +1 -0
  77. package/dist/wire/index.d.ts +1 -1
  78. package/docs/prime-analyst.md +120 -0
  79. package/docs/trace-analysis.md +1 -1
  80. package/package.json +3 -3
  81. package/dist/benchmark-command-CQPKRUr-.js.map +0 -1
  82. package/dist/usage-receipt-CgxMEBZq.js.map +0 -1
@@ -0,0 +1,120 @@
1
+ # Prime Analyst Runner
2
+
3
+ The `prime` analyst runs an RLM coding agent as a trace analyst through an OpenAI-compatible cli-bridge.
4
+ It is the third scored arm of `agent-eval analyst-benchmark`, beside the recursive `dspy-rlm` engine and the one-shot `direct` baseline, and it exists so the prime-vs-dspy comparison is reproducible from this repository alone.
5
+ It speaks the CodeTraceBench failure-block contract only; `--analyst prime` with `--dataset agentrx` is rejected.
6
+
7
+ Implementation: `src/analyst/benchmark-runner-prime.ts` (`createPrimeBenchmarkRunner`) binds the CodeTraceBench block grammar to the shared protocol in `src/analyst/prime-protocol.ts` and `src/analyst/prime-bridge-transport.ts`.
8
+ Wiring: `--analyst prime` in `src/analyst/benchmark-command.ts`.
9
+
10
+ ## What the runner does
11
+
12
+ The benchmark command prepares every case identically for every runner: it loads the label-free trajectory, appends the row's final-verification artifacts as benchmark-verification spans, and hands each runner the same trace store.
13
+ The prime runner adds nothing to that input.
14
+
15
+ Per case it:
16
+
17
+ 1. Projects the full span set with `viewTrace` and serializes it as inline JSON in the prompt.
18
+ Prime has no REPL and no trace tools, so the same projection the dspy typed path binds as a REPL variable is delivered as text.
19
+ 2. Sends one user message to `<bridge>/v1/chat/completions`: the CodeTraceBench task definition (`CODE_TRACE_BENCH_ANALYST_PROMPT`), a strict block-JSON output contract, the trajectory JSON, and the final-verification spans.
20
+ 3. Parses the reply's fenced JSON object (`{ "answer", "blocks": [...] }`), validates each block row, and expands accepted blocks into one scored finding per member step with the published `expandCodeTraceFailureBlocks` — the exact conversion every benchmark runner uses.
21
+
22
+ Findings carry `analyst_id: 'prime'` and score through the same evidence resolution, comparison, and calibration paths as every other arm.
23
+
24
+ ## Reproduce: prime vs dspy-rlm on the same rows
25
+
26
+ Case selection is deterministic in `--labels`, `--limit`, and `--seed`, so two runs that share those flags (and the same `--trace-dir`, `--artifact-dir`, `--revision`, `--split`) score exactly the same rows.
27
+ Run the two arms into two output directories and compare:
28
+
29
+ ```sh
30
+ # Arm 1: prime through the cli-bridge (no model-owner module; the bridge owns execution)
31
+ agent-eval analyst-benchmark \
32
+ --dataset codetracebench \
33
+ --analyst prime \
34
+ --bridge-url http://localhost:4181 \
35
+ --labels .artifacts/manifest.jsonl \
36
+ --trace-dir .artifacts/traces \
37
+ --artifact-dir .artifacts/results \
38
+ --out .artifacts/prime-run \
39
+ --revision aa213b84ffb6690fc37ca15766d6ca174ec36d4d \
40
+ --split verified \
41
+ --model prime/zai/glm-5.2 \
42
+ --timeout-ms 1200000 \
43
+ --limit 20 \
44
+ --seed 7 \
45
+ --concurrency 1
46
+
47
+ # Arm 2: the recursive DSPy engine on the SAME rows (same labels/limit/seed)
48
+ agent-eval analyst-benchmark \
49
+ --dataset codetracebench \
50
+ --analyst dspy-rlm \
51
+ --labels .artifacts/manifest.jsonl \
52
+ --trace-dir .artifacts/traces \
53
+ --artifact-dir .artifacts/results \
54
+ --out .artifacts/dspy-run \
55
+ --revision aa213b84ffb6690fc37ca15766d6ca174ec36d4d \
56
+ --split verified \
57
+ --model-owner-module ./dist/runtime-model-owner.mjs \
58
+ --model opencode/zai-coding-plan/glm-5.2 \
59
+ --python .venv/bin/python \
60
+ --timeout-ms 1200000 \
61
+ --limit 20 \
62
+ --seed 7 \
63
+ --concurrency 1
64
+
65
+ node benchmarks/trace-analysis/tools/compare-analyst-runs.mjs \
66
+ .artifacts/prime-run/result.json .artifacts/dspy-run/result.json
67
+ ```
68
+
69
+ `--model` keeps its normal semantics; for prime it is the bridge model id in `<backend>/<provider>/<model>` form (`prime/zai/glm-5.2`), which the bridge maps to its configured backend model.
70
+ Prime analyses routinely exceed the 300-second default deadline, so set `--timeout-ms` explicitly (the proven external rig used 1200000).
71
+ `--no-repair` disables the bounded repair turn described below.
72
+
73
+ ## Bridge prerequisites
74
+
75
+ The runner needs a running cli-bridge whose `prime` backend is enabled:
76
+
77
+ - `BRIDGE_BACKENDS=prime` — enable the prime backend in the bridge.
78
+ - `PRIME_BIN` — absolute path to the prime binary (on nix installs, the nix store path of the `prime` executable).
79
+ - `PRIME_MODELS_JSON` — the bridge's model table mapping bridge model ids such as `prime/zai/glm-5.2` to the backend provider and model the prime agent runs.
80
+
81
+ The bridge listens on `http://localhost:4181` by default; pass `--bridge-url` when it listens elsewhere.
82
+ Provider credentials live in the bridge process, never in this command: for `--analyst prime` there is no `--model-owner-module`, and passing one is an error.
83
+
84
+ ## Protocol notes
85
+
86
+ - **Short-strings contract, no rationale.**
87
+ The output contract caps every string the model must emit and forbids a `rationale` field.
88
+ Why: stream-splice corruption on long strings was measured on the live bridge path — the bridge splices its backend's streamed output into one reply, and long strings arrive corrupted often enough to void otherwise-correct work.
89
+ Short claims survive the splice; block coordinates carry the signal.
90
+ - **One bounded repair turn.**
91
+ A structurally malformed reply (no parseable JSON object, or no `blocks` array) gets exactly one stateless follow-up call carrying the malformed reply plus the output contract — never the trajectory — mirroring the dspy arm's typed-adapter repair so both arms face the same structured-output affordance.
92
+ Still malformed after repair = failed observation with a typed error (`PrimeMalformedReplyError`), recorded exactly as a dspy-rlm failure is.
93
+ Zero valid blocks from a well-formed reply is an honest null, not a failure.
94
+ - **Oversized traces fall back to chunked projection.**
95
+ When the full `viewTrace` response is oversized, or the rendered JSON exceeds the 360k-char inline budget, the runner re-projects every span through chunked `viewSpans` at a 1200-byte per-attribute cap, in store order, and fails loud if any span drops or the result is still oversized.
96
+ - **Usage receipts.**
97
+ Token counts are the bridge's exact reported counts; USD is a rate-based estimate from the model's catalog rates (for `prime/zai/glm-5.2`, the z.ai coding-plan list rates: 0.6/2.2 USD per million input/output tokens).
98
+ A reply without usage stays uncaptured — never a silent zero — and a repair turn's usage merges into the case's receipt, poisoning each side independently so a measured count survives a missing partner.
99
+ A reply that reports only one side lands in `AnalystUsageReceipt.partialTokens` with `tokens: null`, `cost` uncaptured, and the reported side priced into `knownCostUsd` as a lower bound: `RunTokenUsage` has no nullable side, so carrying a one-sided count in `tokens` would mean writing a zero nobody measured.
100
+ When the bridge reports `estimated: true` — it derived the counts from character lengths because the backend CLI reported none — the receipt carries `tokensEstimated: true`, which is what separates a rate estimate over exact tokens from one over derived tokens.
101
+ - **Per-observation protocol digest.**
102
+ Every prime observation records `primeAnalystProtocolSha256()` in its runner metadata, hashing the question, task prompt, output contract, repair contract, and projection limits that actually ran.
103
+
104
+ ## Reusing the protocol outside this benchmark
105
+
106
+ `src/analyst/prime-protocol.ts` is the consumer-agnostic core, exported from `@tangle-network/agent-eval/analyst`.
107
+ It speaks raw rows and names no finding type, so an analyzer with a different row grammar — span-grounded findings against its own artifact, say — binds it without importing CodeTraceBench types:
108
+
109
+ - `PrimeReplyContract<TRow>` supplies the rows field name, the contract lines spliced into both prompts, a single-pass `decodeRow`, and an optional `maxRows` cap applied to ACCEPTED rows so malformed rows never consume a slot.
110
+ - `buildPrimePrompt` / `buildPrimeRepairPrompt` compose the prompts; the repair prompt never carries the trajectory.
111
+ - `runPrimeExchange` runs the call, the bounded repair turn, and row decoding, returning one typed outcome whose `PrimeFailure.kind` separates `transport`, `http-status`, `unparseable-json`, `no-content`, `deadline`, `malformed-reply`, and `aborted`.
112
+ A cancelled run is never recorded as an analyzer verdict.
113
+ - `projectPrimeTrajectory` runs the render → measure → fall back → re-measure → fail-loud ladder over a caller-supplied `PrimeProjectionSource`, so the source of the projection stays the consumer's choice.
114
+ - `normalizePrimeUsage` / `mergePrimeRawUsage` keep the bridge's report lossless; `analystUsageReceiptFromPrimeUsage` is the agent-eval-only binding to the typed receipt, so a consumer with no pricing table simply does not call it.
115
+ - `primeProtocolSha256` hashes the ACTUALLY composed contract, so two consumers that both stamp `analyst_id: 'prime'` while asking different questions get different digests by construction.
116
+
117
+ ## Status
118
+
119
+ The first prime-vs-dspy comparison batch (20+ live CodeTraceBench cases through cli-bridge on `prime/zai/glm-5.2`) is in flight on the proven external rig this runner was ported from.
120
+ Numbers land in `benchmarks/trace-analysis/` when the batch completes; until then this document makes no accuracy claim for the prime arm.
@@ -227,7 +227,7 @@ Cross-run and pooled comparisons use `benchmarks/trace-analysis/tools/compare-an
227
227
  `agent-eval analyst-benchmark` runs the public AgentRx or CodeTraceBench adapters with:
228
228
 
229
229
  1. an empty-finding baseline,
230
- 2. the actual DSPy RLM trace analyst.
230
+ 2. the scored analyst `--analyst` selects: the recursive DSPy RLM engine (`dspy-rlm`, default), the one-shot `direct` baseline, or the `prime` RLM coding agent behind an OpenAI-compatible cli-bridge (CodeTraceBench only; see [prime-analyst.md](./prime-analyst.md) for bridge prerequisites and the prime-vs-dspy reproduce commands).
231
231
 
232
232
  ```sh
233
233
  agent-eval analyst-benchmark \
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@tangle-network/agent-eval",
3
- "version": "0.144.3",
3
+ "version": "0.144.5",
4
4
  "description": "Evaluate and improve AI agents from runs, traces, judges, and feedback. Compare candidates, cluster failures, measure lift, and gate releases.",
5
5
  "homepage": "https://github.com/tangle-network/agent-eval#readme",
6
6
  "repository": {
@@ -168,8 +168,8 @@
168
168
  "dependencies": {
169
169
  "@asteasolutions/zod-to-openapi": "^9.1.0",
170
170
  "@hono/node-server": "^2.0.12",
171
- "@tangle-network/agent-core": "0.4.33",
172
- "@tangle-network/agent-interface": "0.43.0",
171
+ "@tangle-network/agent-core": "0.5.3",
172
+ "@tangle-network/agent-interface": "0.46.0",
173
173
  "@tangle-network/agent-trace-contract": "^1.0.2",
174
174
  "hono": "^4.12.32",
175
175
  "linear-sum-assignment": "1.0.9",