@tangle-network/agent-eval 0.138.0 → 0.139.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +30 -0
- package/README.md +2 -1
- package/dist/analyst/index.d.ts +41 -94
- package/dist/analyst/index.d.ts.map +1 -1
- package/dist/analyst/index.js +9 -24
- package/dist/analyst/index.js.map +1 -1
- package/dist/{benchmark-D8dkki-J.js → benchmark-CYtcIF2V.js} +2 -2
- package/dist/{benchmark-D8dkki-J.js.map → benchmark-CYtcIF2V.js.map} +1 -1
- package/dist/{benchmark-DlQgU_XI.d.ts → benchmark-DDVdWcwA.d.ts} +3 -3
- package/dist/{benchmark-DlQgU_XI.d.ts.map → benchmark-DDVdWcwA.d.ts.map} +1 -1
- package/dist/{benchmark-command-CMqVqReF.js → benchmark-command-BKfjOBJ5.js} +243 -38
- package/dist/benchmark-command-BKfjOBJ5.js.map +1 -0
- package/dist/benchmarks/index.d.ts +1 -1
- package/dist/benchmarks/index.js +1 -1
- package/dist/{benchmarks-BJ_xK5rQ.js → benchmarks-zxhy1QV3.js} +4 -4
- package/dist/{benchmarks-BJ_xK5rQ.js.map → benchmarks-zxhy1QV3.js.map} +1 -1
- package/dist/campaign/index.d.ts +5 -5
- package/dist/campaign/index.js +4 -3
- package/dist/{campaign-BIBS-NHV.js → campaign-DrS6_hLd.js} +10 -9
- package/dist/campaign-DrS6_hLd.js.map +1 -0
- package/dist/canonical-D011XM8r.js +86 -0
- package/dist/canonical-D011XM8r.js.map +1 -0
- package/dist/cli.js +3 -3
- package/dist/{client-BwPKohkJ.d.ts → client-BohnDFBq.d.ts} +4 -4
- package/dist/{client-BwPKohkJ.d.ts.map → client-BohnDFBq.d.ts.map} +1 -1
- package/dist/{completion-verifier-B4-IMYcS.d.ts → completion-verifier-IPoP4fQO.d.ts} +178 -4
- package/dist/completion-verifier-IPoP4fQO.d.ts.map +1 -0
- package/dist/contract/index.d.ts +10 -10
- package/dist/contract/index.js +8 -7
- package/dist/contract/index.js.map +1 -1
- package/dist/control.d.ts +2 -2
- package/dist/{cost-ledger-CHDLA0Ss.js → cost-ledger-CZ9diLxY.js} +7 -7
- package/dist/cost-ledger-CZ9diLxY.js.map +1 -0
- package/dist/{cost-ledger-B1D3COAc.d.ts → cost-ledger-DKgyIWRj.d.ts} +5 -2
- package/dist/cost-ledger-DKgyIWRj.d.ts.map +1 -0
- package/dist/default-registry-B8vf7Rmf.d.ts +118 -0
- package/dist/default-registry-B8vf7Rmf.d.ts.map +1 -0
- package/dist/{default-registry-lp5R0lve.js → default-registry-BgJJItGr.js} +57 -1532
- package/dist/default-registry-BgJJItGr.js.map +1 -0
- package/dist/dspy-rlm-engine-DTkVyDX-.js +344 -0
- package/dist/dspy-rlm-engine-DTkVyDX-.js.map +1 -0
- package/dist/{eval-campaign-9MozgKL7.js → eval-campaign-BmptJj50.js} +2 -2
- package/dist/{eval-campaign-9MozgKL7.js.map → eval-campaign-BmptJj50.js.map} +1 -1
- package/dist/{exact-types-Dpw2LeHA.d.ts → exact-types-MaaFcllV.d.ts} +2 -2
- package/dist/{exact-types-Dpw2LeHA.d.ts.map → exact-types-MaaFcllV.d.ts.map} +1 -1
- package/dist/external-optimizer-contracts-BrxY2Sli.d.ts +32 -0
- package/dist/external-optimizer-contracts-BrxY2Sli.d.ts.map +1 -0
- package/dist/{extract-usage-CS391dOE.js → extract-usage-DZs601Va.js} +2 -2
- package/dist/{extract-usage-CS391dOE.js.map → extract-usage-DZs601Va.js.map} +1 -1
- package/dist/{feedback-trajectory-CoNep7rl.d.ts → feedback-trajectory-BJUWOkJM.d.ts} +3 -3
- package/dist/{feedback-trajectory-CoNep7rl.d.ts.map → feedback-trajectory-BJUWOkJM.d.ts.map} +1 -1
- package/dist/fuzz.d.ts +1 -1
- package/dist/fuzz.js +1 -1
- package/dist/{hf-dataset-DBJXXoY1.js → hf-dataset-XggBupCr.js} +2 -2
- package/dist/{hf-dataset-DBJXXoY1.js.map → hf-dataset-XggBupCr.js.map} +1 -1
- package/dist/hosted/index.d.ts +3 -3
- package/dist/{index-D0cxAdaV.d.ts → index-BTm_P9aC.d.ts} +11 -11
- package/dist/{index-D0cxAdaV.d.ts.map → index-BTm_P9aC.d.ts.map} +1 -1
- package/dist/{index-B2-IxCMB.d.ts → index-CWOPCJiw.d.ts} +2 -2
- package/dist/{index-B2-IxCMB.d.ts.map → index-CWOPCJiw.d.ts.map} +1 -1
- package/dist/{index-sMN_hI4E.d.ts → index-CtR1xh4V.d.ts} +3 -3
- package/dist/{index-sMN_hI4E.d.ts.map → index-CtR1xh4V.d.ts.map} +1 -1
- package/dist/{index-CjVYlVBK.d.ts → index-_66rVpwN.d.ts} +5 -5
- package/dist/{index-CjVYlVBK.d.ts.map → index-_66rVpwN.d.ts.map} +1 -1
- package/dist/index.d.ts +35 -56
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +51 -176
- package/dist/index.js.map +1 -1
- package/dist/{insight-report-CXd8VBDR.d.ts → insight-report-Bu5Wi9tG.d.ts} +4 -4
- package/dist/{insight-report-CXd8VBDR.d.ts.map → insight-report-Bu5Wi9tG.d.ts.map} +1 -1
- package/dist/{integrity-B-MLFz0I.d.ts → integrity-COTh3DTH.d.ts} +2 -2
- package/dist/{integrity-B-MLFz0I.d.ts.map → integrity-COTh3DTH.d.ts.map} +1 -1
- package/dist/kind-factory-CFxA0JQX.js +2133 -0
- package/dist/kind-factory-CFxA0JQX.js.map +1 -0
- package/dist/ledger-core/index.js +2 -1
- package/dist/{ledger-core-C0Yx1I14.js → ledger-core-Dxz0Rkwa.js} +3 -85
- package/dist/ledger-core-Dxz0Rkwa.js.map +1 -0
- package/dist/{llm-client-Cj3c7PEm.js → llm-client-bkztEfIx.js} +2 -2
- package/dist/{llm-client-Cj3c7PEm.js.map → llm-client-bkztEfIx.js.map} +1 -1
- package/dist/meta-eval/index.d.ts +2 -2
- package/dist/multishot/index.d.ts +2 -2
- package/dist/openapi.json +1 -1
- package/dist/{release-report-CoyvyLBs.d.ts → release-report-fZarvIm-.d.ts} +3 -3
- package/dist/{release-report-CoyvyLBs.d.ts.map → release-report-fZarvIm-.d.ts.map} +1 -1
- package/dist/{replay-DbIYwso6.d.ts → replay-DjG4IG60.d.ts} +34 -143
- package/dist/replay-DjG4IG60.d.ts.map +1 -0
- package/dist/{replay-Cb-4Vf0k.js → replay-SA4OB7O7.js} +48 -137
- package/dist/replay-SA4OB7O7.js.map +1 -0
- package/dist/reporting.d.ts +4 -4
- package/dist/{researcher-BCeOEjtR.d.ts → researcher-BxhtGfKa.d.ts} +5 -5
- package/dist/{researcher-BCeOEjtR.d.ts.map → researcher-BxhtGfKa.d.ts.map} +1 -1
- package/dist/{reward-hacking-sE2l_NV6.d.ts → reward-hacking-CqSLiV51.d.ts} +2 -2
- package/dist/{reward-hacking-sE2l_NV6.d.ts.map → reward-hacking-CqSLiV51.d.ts.map} +1 -1
- package/dist/rl.d.ts +5 -5
- package/dist/rl.js +1 -1
- package/dist/rollout/index.d.ts +1 -1
- package/dist/rollout/index.js +2 -2
- package/dist/{rollout-DQFl0UXA.js → rollout-8nj3mYvx.js} +2 -2
- package/dist/{rollout-DQFl0UXA.js.map → rollout-8nj3mYvx.js.map} +1 -1
- package/dist/{rubric-predictive-validity-w2klGv1u.d.ts → rubric-predictive-validity-DQBQj6uV.d.ts} +2 -2
- package/dist/{rubric-predictive-validity-w2klGv1u.d.ts.map → rubric-predictive-validity-DQBQj6uV.d.ts.map} +1 -1
- package/dist/{run-evidence-CbE0A8Xg.d.ts → run-evidence-C4RcRQT5.d.ts} +3 -3
- package/dist/{run-evidence-CbE0A8Xg.d.ts.map → run-evidence-C4RcRQT5.d.ts.map} +1 -1
- package/dist/{run-record-DwHMk1Ai.d.ts → run-record-CztDMXVF.d.ts} +2 -2
- package/dist/{run-record-DwHMk1Ai.d.ts.map → run-record-CztDMXVF.d.ts.map} +1 -1
- package/dist/{semantic-concept-judge-DYXDPZW0.js → semantic-concept-judge-BuIJ9IfB.js} +43 -6
- package/dist/semantic-concept-judge-BuIJ9IfB.js.map +1 -0
- package/dist/{server-DLEvyW2z.js → server-DaCpLfi0.js} +3 -3
- package/dist/{server-DLEvyW2z.js.map → server-DaCpLfi0.js.map} +1 -1
- package/dist/single-run-lock-BTTtPZ9N.js +989 -0
- package/dist/single-run-lock-BTTtPZ9N.js.map +1 -0
- package/dist/{skill-usage-Bv3G4VkA.d.ts → skill-usage-B-BFS8M2.d.ts} +54 -39
- package/dist/skill-usage-B-BFS8M2.d.ts.map +1 -0
- package/dist/{skillopt-optimization-method-CjKMZy0d.js → skillopt-optimization-method-BbGnCC53.js} +18 -802
- package/dist/skillopt-optimization-method-BbGnCC53.js.map +1 -0
- package/dist/{skillopt-optimization-method-CzfnA8O-.d.ts → skillopt-optimization-method-_s0Tub7Y.d.ts} +11 -39
- package/dist/skillopt-optimization-method-_s0Tub7Y.d.ts.map +1 -0
- package/dist/{statistics-mf70aXKp.d.ts → statistics-B5d0Zd-z.d.ts} +2 -2
- package/dist/{statistics-mf70aXKp.d.ts.map → statistics-B5d0Zd-z.d.ts.map} +1 -1
- package/dist/store-otlp-DX4fGIcf.js +757 -0
- package/dist/store-otlp-DX4fGIcf.js.map +1 -0
- package/dist/{summary-report-BKinV4yD.d.ts → summary-report-Cg7BifAM.d.ts} +3 -3
- package/dist/{summary-report-BKinV4yD.d.ts.map → summary-report-Cg7BifAM.d.ts.map} +1 -1
- package/dist/tool-groups-CdYq22lX.d.ts +258 -0
- package/dist/tool-groups-CdYq22lX.d.ts.map +1 -0
- package/dist/traces.d.ts +7 -6
- package/dist/traces.js +5 -5
- package/dist/{types-zFYez3PK.d.ts → types-BBFNHxSK.d.ts} +5 -5
- package/dist/{types-zFYez3PK.d.ts.map → types-BBFNHxSK.d.ts.map} +1 -1
- package/dist/{types-BtJhn8v6.d.ts → types-DoEYskCd.d.ts} +5 -5
- package/dist/{types-BtJhn8v6.d.ts.map → types-DoEYskCd.d.ts.map} +1 -1
- package/dist/{types-5q2T25iW.d.ts → types-uPrS6mD-.d.ts} +2 -2
- package/dist/{types-5q2T25iW.d.ts.map → types-uPrS6mD-.d.ts.map} +1 -1
- package/dist/usage-receipt-CgxMEBZq.js +134 -0
- package/dist/usage-receipt-CgxMEBZq.js.map +1 -0
- package/dist/wire/index.d.ts +3 -3
- package/dist/wire/index.js +1 -1
- package/docs/trace-analysis.md +170 -484
- package/package.json +1 -2
- package/dist/analyze-runs-CPYxfPWT.d.ts +0 -72
- package/dist/analyze-runs-CPYxfPWT.d.ts.map +0 -1
- package/dist/benchmark-command-CMqVqReF.js.map +0 -1
- package/dist/campaign-BIBS-NHV.js.map +0 -1
- package/dist/completion-verifier-B4-IMYcS.d.ts.map +0 -1
- package/dist/cost-ledger-B1D3COAc.d.ts.map +0 -1
- package/dist/cost-ledger-CHDLA0Ss.js.map +0 -1
- package/dist/default-registry-PUhIVRWz.d.ts +0 -215
- package/dist/default-registry-PUhIVRWz.d.ts.map +0 -1
- package/dist/default-registry-lp5R0lve.js.map +0 -1
- package/dist/ledger-core-C0Yx1I14.js.map +0 -1
- package/dist/registry-C4yJTza7.d.ts +0 -178
- package/dist/registry-C4yJTza7.d.ts.map +0 -1
- package/dist/replay-Cb-4Vf0k.js.map +0 -1
- package/dist/replay-DbIYwso6.d.ts.map +0 -1
- package/dist/semantic-concept-judge-DYXDPZW0.js.map +0 -1
- package/dist/single-run-lock-D_bS5xhj.js +0 -318
- package/dist/single-run-lock-D_bS5xhj.js.map +0 -1
- package/dist/skill-usage-Bv3G4VkA.d.ts.map +0 -1
- package/dist/skillopt-optimization-method-CjKMZy0d.js.map +0 -1
- package/dist/skillopt-optimization-method-CzfnA8O-.d.ts.map +0 -1
- package/dist/store-otlp-BenKynPE.js +0 -1688
- package/dist/store-otlp-BenKynPE.js.map +0 -1
- package/dist/tools-DZGdROtG.js +0 -255
- package/dist/tools-DZGdROtG.js.map +0 -1
package/docs/trace-analysis.md
CHANGED
|
@@ -1,558 +1,244 @@
|
|
|
1
1
|
# Trace Analysis
|
|
2
2
|
|
|
3
|
-
|
|
3
|
+
A trace analyst is a recursive research program that answers a question by choosing which trace data to inspect.
|
|
4
|
+
It is not a single prompt over a prebuilt trace dump.
|
|
4
5
|
|
|
5
|
-
|
|
6
|
-
2. Which exact steps support a suspected problem?
|
|
7
|
-
3. Does an analyst find labeled problems reliably enough to use?
|
|
6
|
+
Agent Eval separates five concerns:
|
|
8
7
|
|
|
9
|
-
|
|
10
|
-
|
|
8
|
+
| Part | Responsibility |
|
|
9
|
+
|---|---|
|
|
10
|
+
| Analyst definition | The question, investigation policy, allowed tools, and limits |
|
|
11
|
+
| Analysis engine | Runs the recursive investigation |
|
|
12
|
+
| Trace store | Provides bounded reads over OTLP traces |
|
|
13
|
+
| Finding | Records one structured claim with exact evidence |
|
|
14
|
+
| Analysis result | Returns the prose answer, findings, investigation steps, model calls, tool calls, and runtime |
|
|
11
15
|
|
|
12
|
-
|
|
16
|
+
The built-in model-backed analysts use the official DSPy `RLM`.
|
|
17
|
+
DSPy runs the research loop and a sandboxed Python interpreter.
|
|
18
|
+
Agent Eval owns trace access, credentials, cancellation, cost accounting, and output validation.
|
|
13
19
|
|
|
14
|
-
|
|
15
|
-
|
|
20
|
+
`callLlmJson()` and `createPublicBenchmarkDirectRunner()` are direct-call baselines.
|
|
21
|
+
They are not trace analysts.
|
|
16
22
|
|
|
17
|
-
|
|
18
|
-
import {
|
|
19
|
-
buildDefaultAnalystRegistry,
|
|
20
|
-
} from '@tangle-network/agent-eval/analyst'
|
|
21
|
-
import { OtlpFileTraceStore } from '@tangle-network/agent-eval/traces'
|
|
22
|
-
|
|
23
|
-
const traceStore = new OtlpFileTraceStore({ path: 'traces.otlp.jsonl' })
|
|
24
|
-
const analysts = buildDefaultAnalystRegistry()
|
|
23
|
+
## Install
|
|
25
24
|
|
|
26
|
-
|
|
25
|
+
Install the TypeScript package and the matching Python package with DSPy:
|
|
27
26
|
|
|
28
|
-
|
|
29
|
-
|
|
30
|
-
|
|
27
|
+
```sh
|
|
28
|
+
pnpm add @tangle-network/agent-eval
|
|
29
|
+
python -m venv .venv
|
|
30
|
+
.venv/bin/pip install "agent-eval-rpc[dspy]"
|
|
31
31
|
```
|
|
32
32
|
|
|
33
|
-
|
|
34
|
-
An analyst failure is recorded separately from an agent failure.
|
|
33
|
+
The Python extra pins the tested stable DSPy and Deno versions.
|
|
35
34
|
|
|
36
|
-
|
|
37
|
-
Custom stores provide `hasTrace` and batched `hasSpans` alongside the seven reads, and accept a `TraceAnalysisStoreContext` so cancellation reaches storage and scans.
|
|
38
|
-
The binding validates every custom-store result.
|
|
39
|
-
Missing fields, undeclared fields, inconsistent counts, and false continuation flags throw `TraceAnalysisStoreContractError` with code `backend_integrity`.
|
|
40
|
-
The analyst runs one Ax executor loop and accepts only an explicit structured `final(task, { report, findings })` result; max-turn fallback text fails loud.
|
|
41
|
-
|
|
42
|
-
### Bind the same reads into another agent environment
|
|
43
|
-
|
|
44
|
-
`buildTraceAnalysisToolDescriptors()` is the canonical definition of the analyst's seven bounded read operations and does not expose Ax types.
|
|
45
|
-
Each descriptor carries the stable `traces` namespace, function name, description, JSON input schema in `parameters`, and a handler already bound to the supplied `TraceAnalysisStore`.
|
|
46
|
-
`buildTraceAnalystTools()` adapts those descriptors into Ax functions; it does not define a second tool surface.
|
|
47
|
-
The bound handlers wrap custom stores with `createBoundedTraceAnalysisStore()`, so page limits, byte ceilings, not-found errors, and cancellation do not depend on the transport or adapter.
|
|
35
|
+
## Answer One Question
|
|
48
36
|
|
|
49
37
|
```ts
|
|
50
38
|
import {
|
|
51
|
-
|
|
52
|
-
|
|
53
|
-
} from '@tangle-network/agent-eval/traces'
|
|
54
|
-
|
|
55
|
-
declare const store: TraceAnalysisStore
|
|
56
|
-
declare function qualifyToolName(namespace: string, name: string): string
|
|
57
|
-
|
|
58
|
-
const tools = buildTraceAnalysisToolDescriptors({ store }).map(
|
|
59
|
-
({ namespace, name, description, parameters, handler }) => ({
|
|
60
|
-
name: qualifyToolName(namespace, name),
|
|
61
|
-
description,
|
|
62
|
-
inputSchema: parameters,
|
|
63
|
-
handler,
|
|
64
|
-
}),
|
|
65
|
-
)
|
|
66
|
-
```
|
|
67
|
-
|
|
68
|
-
Map these fields into the host's existing tool transport.
|
|
69
|
-
The host owns namespace encoding; use its existing convention instead of inventing one here.
|
|
70
|
-
Do not copy the schemas or reimplement the handlers in an MCP, Runtime, or provider adapter.
|
|
71
|
-
|
|
72
|
-
`queryTraces.limit`, `viewSpans.span_ids`, and search `max_matches` caps are present in the JSON Schemas and enforced before store calls.
|
|
73
|
-
Invalid arguments throw `TraceAnalysisValidationError` with code `validation`; responses that cannot fit their byte ceiling throw `TraceAnalysisLimitError` with code `limit_exceeded`.
|
|
74
|
-
Search patterns use RE2 syntax, which rejects backreferences and lookaround instead of allowing exponential-time expressions.
|
|
75
|
-
Search results return `hits` and an exact `has_more` flag; they do not invent a total after a capped scan.
|
|
76
|
-
`viewSpans` partitions every requested id across `spans`, `missing_span_ids`, and `omitted_span_ids`; `has_more` is true when omitted ids must be retried.
|
|
77
|
-
Attribute and match text shortening includes a deterministic marker.
|
|
78
|
-
Trace pages set `has_more`, and the overview returns every error cluster or fails explicitly when the configured response limit is too small.
|
|
79
|
-
|
|
80
|
-
### Analyze captured tool spans in memory
|
|
81
|
-
|
|
82
|
-
Use `toolSpansToTraceAnalysisStore()` when a live worker already returns canonical `ToolSpan[]` records.
|
|
83
|
-
The function snapshots the records immediately, groups them by `runId`, and exposes the same bounded reads and searches as the file-backed store.
|
|
84
|
-
|
|
85
|
-
```ts
|
|
39
|
+
createDspyRlmTraceEngine,
|
|
40
|
+
} from '@tangle-network/agent-eval/analyst'
|
|
86
41
|
import {
|
|
87
42
|
analyzeTraces,
|
|
88
|
-
toolSpansToTraceAnalysisStore,
|
|
89
|
-
type ToolSpan,
|
|
90
|
-
ToolTraceMissingError,
|
|
91
43
|
} from '@tangle-network/agent-eval/traces'
|
|
92
44
|
|
|
93
|
-
|
|
94
|
-
|
|
95
|
-
|
|
96
|
-
|
|
97
|
-
|
|
98
|
-
|
|
45
|
+
const engine = createDspyRlmTraceEngine({
|
|
46
|
+
baseUrl: process.env.LLM_BASE_URL!,
|
|
47
|
+
apiKey: process.env.LLM_API_KEY!,
|
|
48
|
+
model: process.env.LLM_MODEL!,
|
|
49
|
+
pricing: {
|
|
50
|
+
inputUsdPerMillion: 3,
|
|
51
|
+
outputUsdPerMillion: 15,
|
|
52
|
+
},
|
|
53
|
+
maxCostUsd: 0.50,
|
|
54
|
+
runner: { command: '.venv/bin/python' },
|
|
55
|
+
})
|
|
99
56
|
|
|
100
|
-
|
|
101
|
-
|
|
102
|
-
|
|
57
|
+
const result = await analyzeTraces(
|
|
58
|
+
{ question: 'What first caused this run to fail?' },
|
|
59
|
+
{
|
|
60
|
+
source: 'run.otlp.jsonl',
|
|
61
|
+
engine,
|
|
62
|
+
toolGroup: 'singleTrace',
|
|
63
|
+
limits: {
|
|
64
|
+
maxIterations: 8,
|
|
65
|
+
maxLlmCalls: 4,
|
|
66
|
+
maxToolCalls: 32,
|
|
67
|
+
},
|
|
68
|
+
},
|
|
69
|
+
)
|
|
103
70
|
|
|
104
|
-
|
|
71
|
+
console.log(result.answer)
|
|
72
|
+
console.log(result.findings)
|
|
73
|
+
console.log(result.trajectory)
|
|
74
|
+
```
|
|
105
75
|
|
|
106
|
-
|
|
107
|
-
|
|
108
|
-
spans are grouped by a normalized failure signature (uuids / hex ids / numbers /
|
|
109
|
-
absolute paths / durations collapsed), each cluster carrying its prevalence,
|
|
110
|
-
exemplar `trace_id`/`span_id`, and a verbatim sample. This is a zero-LLM,
|
|
111
|
-
reproducible failure checklist the analyst then explains and closes:
|
|
76
|
+
Omit `pricing` only when Agent Eval already recognizes the exact model or model family.
|
|
77
|
+
Unknown pricing fails before the first model call.
|
|
112
78
|
|
|
113
|
-
|
|
114
|
-
|
|
115
|
-
|
|
116
|
-
console.log(`${c.trace_count}× ${c.signature}: e.g. trace ${c.exemplar_trace_ids[0]}`)
|
|
117
|
-
}
|
|
118
|
-
```
|
|
79
|
+
The provider key remains in the Node process.
|
|
80
|
+
The Python process receives an authenticated loopback model endpoint with an ephemeral credential.
|
|
81
|
+
Each trace read also crosses an authenticated loopback callback and counts against `maxToolCalls`.
|
|
119
82
|
|
|
120
|
-
##
|
|
83
|
+
## Define A Reusable Analyst
|
|
121
84
|
|
|
122
|
-
|
|
123
|
-
|
|
124
|
-
Register it as a custom-input analyst and pass the existing value under its stable id:
|
|
85
|
+
An analyst definition contains no model, credentials, or execution state.
|
|
86
|
+
The same definition can run with DSPy or another engine that implements `TraceAnalysisEngine`.
|
|
125
87
|
|
|
126
88
|
```ts
|
|
127
89
|
import {
|
|
128
|
-
|
|
129
|
-
|
|
90
|
+
defineTraceAnalyst,
|
|
91
|
+
runTraceAnalyst,
|
|
130
92
|
} from '@tangle-network/agent-eval/analyst'
|
|
131
93
|
import {
|
|
132
|
-
|
|
133
|
-
} from '@tangle-network/agent-eval/
|
|
134
|
-
|
|
135
|
-
const sources = await readLoopsSupervisorRun(runDir)
|
|
136
|
-
const registry = new AnalystRegistry()
|
|
137
|
-
registry.register(CONTROL_INTEGRITY_ANALYST)
|
|
94
|
+
OtlpFileTraceStore,
|
|
95
|
+
} from '@tangle-network/agent-eval/traces'
|
|
138
96
|
|
|
139
|
-
const
|
|
140
|
-
|
|
97
|
+
const repeatedFailure = defineTraceAnalyst({
|
|
98
|
+
id: 'repeated-tool-failure',
|
|
99
|
+
description: 'Finds the repeated tool failure with the largest impact.',
|
|
100
|
+
area: 'tool-use',
|
|
101
|
+
version: '1.0.0',
|
|
102
|
+
question: 'Which repeated tool failure should we fix first?',
|
|
103
|
+
instructions: [
|
|
104
|
+
'Compare frequency and downstream impact.',
|
|
105
|
+
'Cite the failing span and at least one affected downstream span.',
|
|
106
|
+
'Return no finding when the traces do not support a repeated failure.',
|
|
107
|
+
].join('\n'),
|
|
108
|
+
toolGroup: 'all',
|
|
109
|
+
minimumEvidenceCitations: 2,
|
|
110
|
+
limits: {
|
|
111
|
+
maxIterations: 10,
|
|
112
|
+
maxLlmCalls: 6,
|
|
113
|
+
maxToolCalls: 40,
|
|
114
|
+
},
|
|
141
115
|
})
|
|
142
|
-
```
|
|
143
116
|
|
|
144
|
-
|
|
145
|
-
|
|
117
|
+
const store = new OtlpFileTraceStore({ path: 'runs.otlp.jsonl' })
|
|
118
|
+
await store.ensureIndexed()
|
|
146
119
|
|
|
147
|
-
|
|
148
|
-
|
|
149
|
-
|
|
150
|
-
|
|
151
|
-
|
|
152
|
-
|
|
153
|
-
|
|
154
|
-
| Did a queued steer reach the worker? | `SupervisorRunSources.workers[].inbox` and `.events` | Requests and acknowledgements are joined by request id, not compared as totals. Missing, malformed, duplicate, or uncorrelated rows make the affected count unavailable. |
|
|
155
|
-
| Can behavior be attributed to an exact profile? | `policy.agent_profile_cell_id` | An absent id is reported as unavailable. |
|
|
156
|
-
| Can action authorship or reasoning be inspected? | `messages[]` | Empty gap rows are reported as unavailable. |
|
|
120
|
+
const result = await runTraceAnalyst({
|
|
121
|
+
definition: repeatedFailure,
|
|
122
|
+
engine,
|
|
123
|
+
store,
|
|
124
|
+
context: { runId: 'release-42' },
|
|
125
|
+
})
|
|
126
|
+
```
|
|
157
127
|
|
|
158
|
-
|
|
159
|
-
|
|
160
|
-
|
|
161
|
-
Resume integrity additionally requires an explicit prior-session id and resumed-session id rather than a prose summary.
|
|
128
|
+
Keep product policy in the definition.
|
|
129
|
+
Keep model transport, recursion mechanics, secrets, and accounting in the engine.
|
|
130
|
+
Keep exact trace facts in deterministic tools or checks.
|
|
162
131
|
|
|
163
|
-
|
|
164
|
-
Their count is retained in `SupervisorRunTree.gaps`, so analyzing a projected tree later cannot turn an unreadable parent row into a missing-parent violation.
|
|
132
|
+
## Run The Built-In Set
|
|
165
133
|
|
|
166
|
-
|
|
134
|
+
The default registry always includes deterministic behavior checks.
|
|
135
|
+
It adds the four recursive analysts only when an engine is supplied:
|
|
167
136
|
|
|
168
|
-
`
|
|
169
|
-
|
|
137
|
+
- `failure-mode`
|
|
138
|
+
- `knowledge-gap`
|
|
139
|
+
- `knowledge-poisoning`
|
|
140
|
+
- `improvement`
|
|
170
141
|
|
|
171
142
|
```ts
|
|
172
143
|
import {
|
|
173
|
-
|
|
174
|
-
defineTraceAnalyst,
|
|
175
|
-
makeFinding,
|
|
144
|
+
buildDefaultAnalystRegistry,
|
|
176
145
|
} from '@tangle-network/agent-eval/analyst'
|
|
177
146
|
|
|
178
|
-
const
|
|
179
|
-
|
|
180
|
-
|
|
181
|
-
|
|
182
|
-
|
|
183
|
-
|
|
184
|
-
async analyze(store) {
|
|
185
|
-
const overview = await store.getOverview({ has_errors: true })
|
|
186
|
-
const cluster = overview.error_clusters[0]
|
|
187
|
-
if (!cluster) return []
|
|
188
|
-
|
|
189
|
-
return [makeFinding({
|
|
190
|
-
analyst_id: 'repeated-tool-errors',
|
|
191
|
-
area: 'tool-use',
|
|
192
|
-
subject: cluster.signature,
|
|
193
|
-
claim: `${cluster.span_count} failed spans share one error`,
|
|
194
|
-
severity: 'high',
|
|
195
|
-
confidence: 1,
|
|
196
|
-
evidence_refs: [{
|
|
197
|
-
kind: 'span',
|
|
198
|
-
uri: `trace://${encodeURIComponent(cluster.exemplar_trace_ids[0])}/span/${encodeURIComponent(cluster.exemplar_span_ids[0])}`,
|
|
199
|
-
excerpt: cluster.status_message_sample,
|
|
200
|
-
}],
|
|
201
|
-
recommended_action: 'Fix the highest-frequency tool error before changing prompts.',
|
|
202
|
-
validation_plan: 'Run fresh cases and require this error signature to disappear.',
|
|
203
|
-
})]
|
|
204
|
-
},
|
|
205
|
-
}))
|
|
147
|
+
const registry = buildDefaultAnalystRegistry({ engine })
|
|
148
|
+
const run = await registry.run('release-42', { traceStore: store })
|
|
149
|
+
|
|
150
|
+
for (const finding of run.findings) {
|
|
151
|
+
console.log(finding.analyst_id, finding.claim, finding.evidence_refs)
|
|
152
|
+
}
|
|
206
153
|
```
|
|
207
154
|
|
|
208
|
-
|
|
209
|
-
|
|
155
|
+
`run.per_analyst` records each analyst's status, latency, calls, tokens, and cost.
|
|
156
|
+
One analyst failure does not become an agent failure.
|
|
210
157
|
|
|
211
|
-
|
|
158
|
+
Pass `definitions` to replace the four built-ins.
|
|
159
|
+
Omit `engine` for deterministic-only analysis.
|
|
212
160
|
|
|
213
|
-
|
|
214
|
-
Label the issue identity and exact evidence locations, then run the same cases through every implementation.
|
|
161
|
+
## Trace Tools
|
|
215
162
|
|
|
216
|
-
|
|
217
|
-
import {
|
|
218
|
-
compareAnalystRunners,
|
|
219
|
-
registryBenchmarkRunner,
|
|
220
|
-
renderAnalystBenchmarkMarkdown,
|
|
221
|
-
runAnalystBenchmark,
|
|
222
|
-
traceStoreEvidenceResolver,
|
|
223
|
-
} from '@tangle-network/agent-eval/analyst'
|
|
163
|
+
The engine receives only the tool group declared by the analyst:
|
|
224
164
|
|
|
225
|
-
|
|
226
|
-
|
|
227
|
-
|
|
228
|
-
|
|
229
|
-
|
|
230
|
-
input: { traceStore },
|
|
231
|
-
expectedIssues: [{
|
|
232
|
-
id: 'repeated-command',
|
|
233
|
-
subjects: ['failure-mode:repeated-command'],
|
|
234
|
-
evidence: [{ kind: 'span', uri: 'trace://run-1/span/tool-3' }],
|
|
235
|
-
criticalEvidence: [{ kind: 'span', uri: 'trace://run-1/span/tool-1' }],
|
|
236
|
-
}],
|
|
237
|
-
labeledEvidence: [
|
|
238
|
-
{ kind: 'span', uri: 'trace://run-1/span/tool-1' },
|
|
239
|
-
{ kind: 'span', uri: 'trace://run-1/span/tool-3' },
|
|
240
|
-
],
|
|
241
|
-
}],
|
|
242
|
-
runners: [registryBenchmarkRunner({ id: 'built-in', registry: analysts })],
|
|
243
|
-
repetitions: 3,
|
|
244
|
-
resolveEvidence: traceStoreEvidenceResolver((input) => input.traceStore),
|
|
245
|
-
benchmark: {
|
|
246
|
-
id: 'failure-localization',
|
|
247
|
-
dataset: {
|
|
248
|
-
id: 'my-team/trace-failures',
|
|
249
|
-
revision: 'git-sha-or-content-digest',
|
|
250
|
-
split: 'test',
|
|
251
|
-
},
|
|
252
|
-
},
|
|
253
|
-
})
|
|
254
|
-
|
|
255
|
-
console.log(renderAnalystBenchmarkMarkdown(benchmark))
|
|
256
|
-
```
|
|
165
|
+
| Group | Use |
|
|
166
|
+
|---|---|
|
|
167
|
+
| `singleTrace` | Inspect one known trajectory |
|
|
168
|
+
| `discoveryAndSearch` | Find relevant traces and spans across a dataset |
|
|
169
|
+
| `all` | Use every bounded trace operation |
|
|
257
170
|
|
|
258
|
-
The
|
|
171
|
+
The canonical operations are:
|
|
259
172
|
|
|
260
|
-
-
|
|
261
|
-
-
|
|
262
|
-
-
|
|
263
|
-
-
|
|
264
|
-
-
|
|
265
|
-
-
|
|
266
|
-
-
|
|
267
|
-
- latency, calls, every reported token counter, and known or missing cost,
|
|
268
|
-
- dataset revision, case tags, case metadata, and runner metadata.
|
|
173
|
+
- `getDatasetOverview`
|
|
174
|
+
- `queryTraces`
|
|
175
|
+
- `countTraces`
|
|
176
|
+
- `viewTrace`
|
|
177
|
+
- `viewSpans`
|
|
178
|
+
- `searchTrace`
|
|
179
|
+
- `searchSpan`
|
|
269
180
|
|
|
270
|
-
Use `
|
|
271
|
-
|
|
272
|
-
|
|
181
|
+
Use `buildTraceAnalysisToolDescriptors({ store })` to bind the same operations into another runtime.
|
|
182
|
+
Each descriptor includes its name, namespace, description, JSON input schema, and bounded handler.
|
|
183
|
+
Do not copy the schemas or reimplement the handlers in another adapter.
|
|
273
184
|
|
|
274
|
-
|
|
185
|
+
Custom stores implement `TraceAnalysisStore`.
|
|
186
|
+
`OtlpFileTraceStore`, `otlpTextToTraceAnalysisStore()`, and `toolSpansToTraceAnalysisStore()` provide common adapters.
|
|
187
|
+
Store results are checked for missing fields, undeclared fields, inconsistent counts, invalid continuation flags, oversized responses, and unsafe search patterns.
|
|
275
188
|
|
|
276
|
-
|
|
277
|
-
Agent Eval does not download datasets or own trace capture.
|
|
278
|
-
Load public data at an immutable commit and record that commit in `benchmark.dataset.revision`.
|
|
189
|
+
## Result Contract
|
|
279
190
|
|
|
280
|
-
|
|
281
|
-
import {
|
|
282
|
-
agentRxBenchmarkCase,
|
|
283
|
-
agentRxPredictionsToFindings,
|
|
284
|
-
codeTraceBenchCase,
|
|
285
|
-
codeTracerPredictionsToFindings,
|
|
286
|
-
} from '@tangle-network/agent-eval/analyst'
|
|
287
|
-
import { otlpTextToTraceAnalysisStore } from '@tangle-network/agent-eval/traces'
|
|
288
|
-
import { chatTrajectoryToSpans, serializeSpans } from '@tangle-network/traces'
|
|
191
|
+
Every recursive run returns:
|
|
289
192
|
|
|
290
|
-
|
|
291
|
-
|
|
292
|
-
|
|
293
|
-
|
|
294
|
-
|
|
295
|
-
|
|
193
|
+
| Field | Meaning |
|
|
194
|
+
|---|---|
|
|
195
|
+
| `answer` | Direct answer to the analyst question |
|
|
196
|
+
| `findings` | Valid cited claims accepted by the analyst policy |
|
|
197
|
+
| `trajectory` | DSPy RLM investigation steps |
|
|
198
|
+
| `modelCalls` | Successful provider completions used by the engine |
|
|
199
|
+
| `toolCalls` | Trace reads admitted for execution, including a read that later fails |
|
|
200
|
+
| `runtime` | Engine, package, sandbox identity, provider attempts, and successful completions |
|
|
296
201
|
|
|
297
|
-
|
|
298
|
-
|
|
299
|
-
|
|
300
|
-
|
|
301
|
-
|
|
302
|
-
|
|
303
|
-
}, {
|
|
304
|
-
stepCount: agentRxMessages.length,
|
|
305
|
-
})
|
|
202
|
+
Each finding requires exact `trace://` or `finding://` evidence.
|
|
203
|
+
Trace citations must resolve to an existing span.
|
|
204
|
+
When a citation includes an excerpt, the exact text must occur in that span or referenced finding.
|
|
205
|
+
Unknown, transformed, or fabricated identifiers are rejected.
|
|
206
|
+
An empty findings array means no submitted claim passed the evidence rules.
|
|
207
|
+
It does not prove the run was correct.
|
|
306
208
|
|
|
307
|
-
|
|
308
|
-
codeTraceRow.traj_id,
|
|
309
|
-
codetracerLabels,
|
|
310
|
-
{ stepCount: codeTraceRow.step_count },
|
|
311
|
-
)
|
|
312
|
-
const agentRxFindings = agentRxPredictionsToFindings(
|
|
313
|
-
agentRxRow.trajectory_id,
|
|
314
|
-
agentRxJudgeOutput,
|
|
315
|
-
{ stepCount: agentRxMessages.length },
|
|
316
|
-
)
|
|
317
|
-
```
|
|
209
|
+
## Measure Analyst Quality
|
|
318
210
|
|
|
319
|
-
|
|
320
|
-
|
|
321
|
-
Pass `labelSet: 'incorrect-and-unuseful'` to both the case and prediction adapters only for an explicitly combined experiment.
|
|
322
|
-
Every cited step is checked against `step_count`.
|
|
323
|
-
|
|
324
|
-
`agentRxBenchmarkCase()` accepts the public [AgentRx](https://huggingface.co/datasets/microsoft/AgentRx) label format.
|
|
325
|
-
AgentRx category quality and root-step accuracy are scored independently.
|
|
326
|
-
`traceAnalystQualityJudge` averages them when a root-step label exists.
|
|
327
|
-
Pass `target: 'all-failures'` only when the analyst is designed to identify every annotated failure.
|
|
328
|
-
|
|
329
|
-
Both adapters emit `trace://<id>/span/step-<n>` evidence by default.
|
|
330
|
-
`@tangle-network/traces` uses the same IDs when converting chat trajectories.
|
|
331
|
-
Pass `stepUri` when your trace store uses another URI scheme.
|
|
332
|
-
`codeTracerPredictionsToFindings()` and `agentRxPredictionsToFindings()` translate the maintained upstream engines' native outputs into the same evidence and category shape.
|
|
333
|
-
The CodeTracer adapter accepts the published `stage_id` format and the flat or grouped step-label formats emitted by CodeTracer 0.2.
|
|
334
|
-
AgentRx `Report.to_dict()` judge votes reduce to the upstream majority failure type and Python-rounded mean step, and direct `failures` arrays use the same reduction.
|
|
335
|
-
`failure_case: 0` produces no finding, which scores as a missed root cause on AgentRx's failed trajectories.
|
|
336
|
-
External runners can return `observedLatencyMs`, `usage`, `metadata`, and `error` together.
|
|
337
|
-
This records an upstream failure without discarding work already performed.
|
|
338
|
-
Set `observedLatencyMs: null` when an imported run did not record duration.
|
|
339
|
-
The report keeps it unknown instead of timing the import code.
|
|
340
|
-
|
|
341
|
-
## Run A Real-Model Public Benchmark
|
|
342
|
-
|
|
343
|
-
`agent-eval analyst-benchmark` runs the existing label adapters and `runAnalystBenchmark()` with two runners: an empty-finding baseline and a benchmark-specific model analyst.
|
|
344
|
-
The CodeTraceBench runner emits one prediction per incorrect step, including wrong actions that the agent later recovers from.
|
|
345
|
-
Final task success is evidence about the final state, not proof that every earlier action was correct.
|
|
346
|
-
Unuseful but correct exploration is a separate CodeTraceBench label and is not scored by the default run.
|
|
347
|
-
The AgentRx runner emits one taxonomy label and one root-cause step.
|
|
348
|
-
The generic `failure-mode` analyst is not used because its `failure-mode` area does not match either public task.
|
|
349
|
-
|
|
350
|
-
Convert CodeTraceBench trajectories with the maintained importer.
|
|
351
|
-
It writes one OTLP JSONL file per trajectory, preserves assistant step ids, and produces a receipt with source and output hashes.
|
|
352
|
-
Each label row's `traj_id` or `trajectory_id` must exactly equal the OTLP `trace_id`.
|
|
353
|
-
Use a domain-qualified ID when an upstream dataset reuses local trajectory numbers.
|
|
354
|
-
Keep the extracted CodeTraceBench artifact tree intact because each row's `source_relpath` locates its final test output.
|
|
355
|
-
|
|
356
|
-
```bash
|
|
357
|
-
traces import-codetracebench \
|
|
358
|
-
.artifacts/bench_manifest.verified.jsonl \
|
|
359
|
-
--trajectory-dir .artifacts/codetrace-normalized \
|
|
360
|
-
--out .artifacts/codetrace-otlp \
|
|
361
|
-
--revision aa213b84ffb6690fc37ca15766d6ca174ec36d4d \
|
|
362
|
-
--concurrency 8
|
|
363
|
-
```
|
|
211
|
+
Measure the analyst on labeled traces before using its findings for automated changes.
|
|
212
|
+
At minimum, report issue recall, finding precision, exact evidence accuracy, trusted-negative false positives, repeat agreement, failures, calls, tokens, cost, and latency.
|
|
364
213
|
|
|
365
|
-
|
|
366
|
-
|
|
214
|
+
`runAnalystBenchmark()` compares any `AnalystBenchmarkRunner` implementations.
|
|
215
|
+
`agent-eval analyst-benchmark` runs the public AgentRx or CodeTraceBench adapters with:
|
|
367
216
|
|
|
368
|
-
|
|
369
|
-
|
|
217
|
+
1. an empty-finding baseline,
|
|
218
|
+
2. the actual DSPy RLM trace analyst.
|
|
370
219
|
|
|
220
|
+
```sh
|
|
371
221
|
agent-eval analyst-benchmark \
|
|
372
222
|
--dataset codetracebench \
|
|
373
|
-
--labels .artifacts/
|
|
374
|
-
--trace-dir .artifacts/
|
|
375
|
-
--artifact-dir .artifacts/
|
|
376
|
-
--out .artifacts/
|
|
223
|
+
--labels .artifacts/manifest.jsonl \
|
|
224
|
+
--trace-dir .artifacts/traces \
|
|
225
|
+
--artifact-dir .artifacts/results \
|
|
226
|
+
--out .artifacts/analyst-run \
|
|
377
227
|
--revision aa213b84ffb6690fc37ca15766d6ca174ec36d4d \
|
|
378
228
|
--split verified \
|
|
379
229
|
--base-url http://127.0.0.1:3355/v1 \
|
|
380
230
|
--api-key-env CLI_BRIDGE_BEARER \
|
|
381
|
-
--model
|
|
231
|
+
--model claude-code/sonnet \
|
|
232
|
+
--python .venv/bin/python \
|
|
382
233
|
--limit 20 \
|
|
383
234
|
--seed 7 \
|
|
384
|
-
--concurrency
|
|
235
|
+
--concurrency 1 \
|
|
385
236
|
--max-cost-usd 5
|
|
386
237
|
```
|
|
387
238
|
|
|
388
|
-
The
|
|
389
|
-
|
|
390
|
-
|
|
391
|
-
Raw result JSON is hashed and parsed but is not sent to the model.
|
|
392
|
-
`--artifact-dir` may be a shared extraction root with one directory per `traj_id`, or the extraction root for one archive.
|
|
393
|
-
It prefers `panes/post-test.txt` over the duplicate `sessions/tests.log`.
|
|
394
|
-
It also recognizes `test_output.txt`, `results.json`, `result.json`, `report.json`, `*_result.json`, and `*_metrics.json`.
|
|
395
|
-
Each discovered file records its role, path, byte count, and SHA-256.
|
|
396
|
-
Files exposed as spans also record their span ids.
|
|
397
|
-
Missing evidence roles are explicit.
|
|
398
|
-
Known Terminal-Bench, SWE-Bench, and SWE-Multi result formats become one explicit passed or failed outcome span.
|
|
399
|
-
A missing or unparseable result becomes `unavailable`; it is never inferred from the trajectory or raw test text.
|
|
400
|
-
Raw test output is optional because some public cases retain only structured results.
|
|
401
|
-
|
|
402
|
-
Before the first model call, the command also checks that every selected label has a matching `step-<n>` span.
|
|
403
|
-
It refuses a missing dataset revision, implicit all-case run, duplicate trajectory id, multi-trace file, missing step, oversized evidence, or existing `result.json`.
|
|
404
|
-
The model selects positive integer assistant step ids.
|
|
405
|
-
The runner constructs each canonical trace URI and exact action excerpt from the selected span.
|
|
406
|
-
A missing, non-assistant, or empty step fails that model run while preserving its raw output and usage.
|
|
407
|
-
The command consumes already-downloaded labels, normalized traces, and extracted artifacts.
|
|
408
|
-
Dataset download, archive extraction, and trajectory conversion remain separate import steps.
|
|
409
|
-
|
|
410
|
-
The output directory contains:
|
|
411
|
-
|
|
412
|
-
- `manifest.json` with the immutable dataset, model, case, and input identity.
|
|
413
|
-
- `initialization-complete.json` written only after every initial file is durable.
|
|
414
|
-
- `observations.jsonl` with one fsynced, hash-chained row per completed case and runner.
|
|
415
|
-
- `cost-ledger.jsonl` with durable run-wide model reservations and receipts.
|
|
416
|
-
- `model-responses/` with one strict, content-hashed response and receipt per paid call for crash recovery.
|
|
417
|
-
- `result.json` with every observation, summary metric, comparison, error, latency, token counter, measured or estimated cost, explicit unknown cost, selected case id, source digest, dependency-lock digest, artifact digest, analyst protocol digest, implementation digest, and case distribution.
|
|
418
|
-
- `report.md` with the same run and selection distribution rendered for review.
|
|
419
|
-
- `run.local.json` with machine-local paths, endpoint, and command, kept out of the shareable result.
|
|
420
|
-
|
|
421
|
-
When active calls temporarily hold the remaining money limit, later calls wait for their final receipts.
|
|
422
|
-
The command rejects new work only when committed spend plus the next enforced maximum cannot fit.
|
|
423
|
-
|
|
424
|
-
Resume only the exact same run after an interruption:
|
|
425
|
-
|
|
426
|
-
```bash
|
|
427
|
-
agent-eval analyst-benchmark <the same flags> --resume
|
|
428
|
-
```
|
|
429
|
-
|
|
430
|
-
Resume rejects changed labels, traces, artifacts, model settings, case selection, local paths, or endpoint.
|
|
431
|
-
It runs only missing case and runner pairs.
|
|
432
|
-
If the provider response was saved before the process stopped, resume settles that exact response without another provider call.
|
|
433
|
-
If the call stopped before a response was saved, resume reuses the same provider request id.
|
|
434
|
-
An already complete run is read and checked without another model call.
|
|
435
|
-
|
|
436
|
-
The command exits `2` when any model analyst fails.
|
|
437
|
-
Its failed row still records latency, calls, available token counters, known spend, and the analyst error.
|
|
438
|
-
See the [32-case GLM-5.2 reference run](https://github.com/tangle-network/agent-eval/tree/main/benchmarks/trace-analysis/codetracebench-glm52-20260730) for a complete measured result and its stated limits.
|
|
439
|
-
|
|
440
|
-
`--limit` uses deterministic hash selection, not stratified sampling.
|
|
441
|
-
Limited runs are marked `representativeOfInput: false`.
|
|
442
|
-
The result compares source and selected distributions for label class, agent, model, difficulty, and solved state.
|
|
443
|
-
CodeTraceBench label classes distinguish `positive`, `trusted-negative`, `unlabeled-failure`, and `unlabeled-unknown`.
|
|
444
|
-
Micro precision, recall, and F1 pool all labeled steps and predictions.
|
|
445
|
-
Macro precision, recall, and F1 average per-case scores over issue-bearing cases, matching step-localization papers that report per-trajectory means.
|
|
446
|
-
The all-row result remains intact for comparison with published work.
|
|
447
|
-
The additional calibrated view measures labeled positives against solved, label-empty negatives and reports failed, label-empty rows separately instead of calling them clean.
|
|
448
|
-
The report separately counts final-result files and passed, failed, or unavailable outcomes.
|
|
449
|
-
Only a full census of the supplied input is marked representative.
|
|
450
|
-
|
|
451
|
-
AgentRx uses the same command with `--dataset agentrx` after obtaining its contact-gated label and trajectory files.
|
|
452
|
-
|
|
453
|
-
## Use Upstream Scorers
|
|
454
|
-
|
|
455
|
-
Agent Eval adapts upstream evaluators instead of copying them.
|
|
456
|
-
|
|
457
|
-
```ts
|
|
458
|
-
import { createEvaluator } from '@arizeai/phoenix-evals'
|
|
459
|
-
import { ExactMatch } from 'autoevals'
|
|
460
|
-
import {
|
|
461
|
-
autoevalsScorerJudge,
|
|
462
|
-
phoenixEvaluatorJudge,
|
|
463
|
-
} from '@tangle-network/agent-eval/campaign'
|
|
464
|
-
|
|
465
|
-
const phoenix = createEvaluator(
|
|
466
|
-
({ output, expected }) => output === expected ? 1 : 0,
|
|
467
|
-
{ name: 'exact-match', kind: 'CODE', telemetry: { isEnabled: false } },
|
|
468
|
-
)
|
|
469
|
-
|
|
470
|
-
const phoenixJudge = phoenixEvaluatorJudge(phoenix, {
|
|
471
|
-
mapInput: ({ artifact, scenario }) => ({ output: artifact, expected: scenario.expected }),
|
|
472
|
-
})
|
|
473
|
-
|
|
474
|
-
const autoevalsJudge = autoevalsScorerJudge(ExactMatch, {
|
|
475
|
-
name: 'exact-match',
|
|
476
|
-
kind: 'CODE',
|
|
477
|
-
mapInput: ({ artifact, scenario }) => ({ output: artifact, expected: scenario.expected }),
|
|
478
|
-
})
|
|
479
|
-
```
|
|
480
|
-
|
|
481
|
-
These adapters do not install either upstream package for consumers.
|
|
482
|
-
Install only the scorer package you use.
|
|
483
|
-
Missing or non-finite scores throw instead of becoming passes.
|
|
484
|
-
Phoenix evaluators marked `MINIMIZE` or `NEUTRAL` require `toComposite` so candidate selection never assumes the wrong direction.
|
|
485
|
-
Mark model-backed evaluators as `kind: 'LLM'` and provide `paidCall` with the model and a receipt mapper.
|
|
486
|
-
The campaign then passes its cancellation signal and cost ledger through the adapter.
|
|
487
|
-
An LLM evaluator is rejected before execution when either cost capture or the campaign ledger is missing.
|
|
488
|
-
|
|
489
|
-
## Turn Reviewed Findings Into Eval Data
|
|
490
|
-
|
|
491
|
-
Generated findings can populate a review queue.
|
|
492
|
-
They cannot promote themselves into learning data.
|
|
493
|
-
|
|
494
|
-
```ts
|
|
495
|
-
import {
|
|
496
|
-
analystFindingDigest,
|
|
497
|
-
analystRunDigest,
|
|
498
|
-
analystRunToFeedbackTrajectory,
|
|
499
|
-
analystRunToReviewRequests,
|
|
500
|
-
} from '@tangle-network/agent-eval'
|
|
501
|
-
|
|
502
|
-
const runDigest = analystRunDigest(result)
|
|
503
|
-
const requests = analystRunToReviewRequests(result)
|
|
504
|
-
await reviewQueue.add(requests)
|
|
505
|
-
|
|
506
|
-
declare const acceptedFindingIds: ReadonlySet<string>
|
|
507
|
-
|
|
508
|
-
const trajectory = analystRunToFeedbackTrajectory(result, {
|
|
509
|
-
task: { intent: 'Find why the command failed.' },
|
|
510
|
-
reviewRequests: requests,
|
|
511
|
-
reviewDecisions: [
|
|
512
|
-
...result.findings.map((finding) => ({
|
|
513
|
-
runDigest,
|
|
514
|
-
findingId: finding.finding_id,
|
|
515
|
-
findingDigest: analystFindingDigest(finding),
|
|
516
|
-
verdict: acceptedFindingIds.has(finding.finding_id) ? 'confirmed' as const : 'rejected' as const,
|
|
517
|
-
source: 'user' as const,
|
|
518
|
-
reviewerId: 'reviewer-42',
|
|
519
|
-
reviewId: 'trace-review-918',
|
|
520
|
-
reason: 'Reviewed against the cited span.',
|
|
521
|
-
decidedAt: new Date().toISOString(),
|
|
522
|
-
})),
|
|
523
|
-
{
|
|
524
|
-
runDigest,
|
|
525
|
-
verdict: 'completeness_assessed',
|
|
526
|
-
missedIssues: [],
|
|
527
|
-
source: 'user',
|
|
528
|
-
reviewerId: 'reviewer-42',
|
|
529
|
-
reviewId: 'trace-review-918',
|
|
530
|
-
reason: 'Reviewed the full run for omitted findings.',
|
|
531
|
-
decidedAt: new Date().toISOString(),
|
|
532
|
-
},
|
|
533
|
-
],
|
|
534
|
-
trace: { artifactUri: 'traces.otlp.jsonl', traceIds: ['run-1'] },
|
|
535
|
-
})
|
|
536
|
-
```
|
|
537
|
-
|
|
538
|
-
`analystRunToFeedbackTrajectory()` stores review requests separately from labels.
|
|
539
|
-
It can archive an unreviewed run.
|
|
540
|
-
`feedbackTrajectoryToOptimizerRow()` requires every decision to match the complete run digest, every finding decision to match the finding digest, and one independent completeness assessment.
|
|
541
|
-
Its score is F1 over confirmed findings and independently identified misses.
|
|
542
|
-
Generic labels and run-level outcomes do not satisfy these requirements.
|
|
543
|
-
|
|
544
|
-
## Required Trace Data
|
|
545
|
-
|
|
546
|
-
Useful analysis needs:
|
|
547
|
-
|
|
548
|
-
- stable run, trace, and span IDs,
|
|
549
|
-
- parent-child links and ordered timestamps,
|
|
550
|
-
- model, prompt, and configuration identity,
|
|
551
|
-
- complete tool names, arguments, results, and error codes,
|
|
552
|
-
- token, cost, and latency data when available,
|
|
553
|
-
- retrieval source IDs and scores when retrieval is involved,
|
|
554
|
-
- final environment outcomes such as tests, task completion, or policy blocks.
|
|
555
|
-
|
|
556
|
-
Do not include secrets, raw OAuth tokens, or unredacted personal data.
|
|
239
|
+
The command writes every observation before producing `result.json` and `report.md`.
|
|
240
|
+
It can resume without repeating completed cases.
|
|
241
|
+
Dataset revisions, selected case IDs, input hashes, trace hashes, result artifacts, usage, errors, and comparison settings remain in the output.
|
|
557
242
|
|
|
558
|
-
Use
|
|
243
|
+
Use fresh development cases while changing questions or instructions.
|
|
244
|
+
Report final quality only on an untouched holdout.
|