@tangle-network/agent-eval 0.136.0 → 0.137.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +38 -1
- package/README.md +4 -2
- package/dist/{agent-profile-cell-OhuTee9n.js → agent-profile-cell-CbfBm2g6.js} +2 -2
- package/dist/{agent-profile-cell-OhuTee9n.js.map → agent-profile-cell-CbfBm2g6.js.map} +1 -1
- package/dist/{agent-profile-cell-CCm3l2v2.d.ts → agent-profile-cell-Cw0PVwDr.d.ts} +2 -2
- package/dist/{agent-profile-cell-CCm3l2v2.d.ts.map → agent-profile-cell-Cw0PVwDr.d.ts.map} +1 -1
- package/dist/analyst/index.d.ts +139 -17
- package/dist/analyst/index.d.ts.map +1 -1
- package/dist/analyst/index.js +606 -4
- package/dist/analyst/index.js.map +1 -1
- package/dist/{analyze-runs-jjCmF8pU.js → analyze-runs-BScZvqMV.js} +6 -6
- package/dist/{analyze-runs-jjCmF8pU.js.map → analyze-runs-BScZvqMV.js.map} +1 -1
- package/dist/{analyze-runs-Cda5Xkj1.d.ts → analyze-runs-PVtnfjvA.d.ts} +6 -6
- package/dist/{analyze-runs-Cda5Xkj1.d.ts.map → analyze-runs-PVtnfjvA.d.ts.map} +1 -1
- package/dist/{baseline-BUeFcgrn.js → baseline-C-GocmIW.js} +2 -2
- package/dist/{baseline-BUeFcgrn.js.map → baseline-C-GocmIW.js.map} +1 -1
- package/dist/benchmark-CHX4orG7.d.ts +184 -0
- package/dist/benchmark-CHX4orG7.d.ts.map +1 -0
- package/dist/benchmark-YDrpumqB.js +414 -0
- package/dist/benchmark-YDrpumqB.js.map +1 -0
- package/dist/benchmarks/index.d.ts +1 -1
- package/dist/benchmarks/index.js +1 -1
- package/dist/{benchmarks-Dfgm9ts5.js → benchmarks-DCLkQOmc.js} +3 -3
- package/dist/{benchmarks-Dfgm9ts5.js.map → benchmarks-DCLkQOmc.js.map} +1 -1
- package/dist/builder-eval/index.js +2 -2
- package/dist/campaign/index.d.ts +6 -6
- package/dist/campaign/index.js +3 -3
- package/dist/{campaign-Dz8uQnhC.js → campaign-lgObcHFC.js} +212 -78
- package/dist/campaign-lgObcHFC.js.map +1 -0
- package/dist/cli.js +1 -1
- package/dist/client-C8L6h6Wf.d.ts +202 -0
- package/dist/client-C8L6h6Wf.d.ts.map +1 -0
- package/dist/completion-verifier-DSyRNVzU.d.ts +240 -0
- package/dist/completion-verifier-DSyRNVzU.d.ts.map +1 -0
- package/dist/contract/index.d.ts +10 -9
- package/dist/contract/index.d.ts.map +1 -1
- package/dist/contract/index.js +10 -10
- package/dist/control.d.ts +2 -2
- package/dist/control.js +1 -1
- package/dist/{cost-ledger-DHAjwNj7.js → cost-ledger-D-5_-dhi.js} +2 -2
- package/dist/{cost-ledger-DHAjwNj7.js.map → cost-ledger-D-5_-dhi.js.map} +1 -1
- package/dist/{cost-ledger-fGS_u_O1.d.ts → cost-ledger-D2o6JOrL.d.ts} +2 -2
- package/dist/{cost-ledger-fGS_u_O1.d.ts.map → cost-ledger-D2o6JOrL.d.ts.map} +1 -1
- package/dist/{dataset-BvtnC8Dc.d.ts → dataset-v_Y5902-.d.ts} +2 -2
- package/dist/{dataset-BvtnC8Dc.d.ts.map → dataset-v_Y5902-.d.ts.map} +1 -1
- package/dist/{default-registry-CHmdy2An.js → default-registry-CLXbRt0f.js} +119 -38
- package/dist/default-registry-CLXbRt0f.js.map +1 -0
- package/dist/{default-registry-Brxr728w.d.ts → default-registry-Dc5D_Loc.d.ts} +63 -137
- package/dist/default-registry-Dc5D_Loc.d.ts.map +1 -0
- package/dist/{errors-8YnH8WlF.js → errors-D-LKuDhb.js} +8 -2
- package/dist/errors-D-LKuDhb.js.map +1 -0
- package/dist/{errors-CEk209JS.d.ts → errors-DkfjIDvD.d.ts} +9 -3
- package/dist/errors-DkfjIDvD.d.ts.map +1 -0
- package/dist/{eval-campaign-Cc8WZJ6b.js → eval-campaign-CHqfLnff.js} +6 -6
- package/dist/{eval-campaign-Cc8WZJ6b.js.map → eval-campaign-CHqfLnff.js.map} +1 -1
- package/dist/{extract-usage-DIQpN-ww.js → extract-usage-p-56bh8q.js} +3 -3
- package/dist/{extract-usage-DIQpN-ww.js.map → extract-usage-p-56bh8q.js.map} +1 -1
- package/dist/{feedback-trajectory-CVaeREXV.d.ts → feedback-trajectory-N_F0PwHz.d.ts} +90 -3
- package/dist/feedback-trajectory-N_F0PwHz.d.ts.map +1 -0
- package/dist/fuzz.d.ts +1 -1
- package/dist/fuzz.js +2 -2
- package/dist/hosted/index.d.ts +3 -2
- package/dist/hosted/index.d.ts.map +1 -1
- package/dist/index-BipJlj-C.d.ts +316 -0
- package/dist/index-BipJlj-C.d.ts.map +1 -0
- package/dist/{index-CQsJcqch.d.ts → index-BnP1QJUv.d.ts} +5 -5
- package/dist/{index-CQsJcqch.d.ts.map → index-BnP1QJUv.d.ts.map} +1 -1
- package/dist/{index-B4Fjfo5U.d.ts → index-C-Pr4OWg.d.ts} +88 -317
- package/dist/index-C-Pr4OWg.d.ts.map +1 -0
- package/dist/{index-DuhJaaiH.d.ts → index-DEb46kc6.d.ts} +2 -2
- package/dist/{index-DuhJaaiH.d.ts.map → index-DEb46kc6.d.ts.map} +1 -1
- package/dist/{index-C2fkZhv_.d.ts → index-DRNl6g_N.d.ts} +3 -3
- package/dist/{index-C2fkZhv_.d.ts.map → index-DRNl6g_N.d.ts.map} +1 -1
- package/dist/{index-AbhwHp0V.d.ts → index-U3RHOShi.d.ts} +2 -2
- package/dist/{index-AbhwHp0V.d.ts.map → index-U3RHOShi.d.ts.map} +1 -1
- package/dist/index.d.ts +29 -70
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +507 -33
- package/dist/index.js.map +1 -1
- package/dist/{client-DcvgkaZi.d.ts → insight-report-B9ooYH_g.d.ts} +5 -203
- package/dist/insight-report-B9ooYH_g.d.ts.map +1 -0
- package/dist/integrity-CCXTftiL.js +1360 -0
- package/dist/integrity-CCXTftiL.js.map +1 -0
- package/dist/{integrity-rmVhXWA7.d.ts → integrity-CKxosZ5Z.d.ts} +3 -3
- package/dist/{integrity-rmVhXWA7.d.ts.map → integrity-CKxosZ5Z.d.ts.map} +1 -1
- package/dist/{integrity-BzRbCHzi.js → integrity-fdt8XPAv.js} +2 -2
- package/dist/{integrity-BzRbCHzi.js.map → integrity-fdt8XPAv.js.map} +1 -1
- package/dist/ledger-core/index.d.ts +1 -1
- package/dist/ledger-core/index.js +1 -1
- package/dist/{ledger-core-DAKFKRzi.js → ledger-core-t6sItivm.js} +85 -85
- package/dist/{ledger-core-DAKFKRzi.js.map → ledger-core-t6sItivm.js.map} +1 -1
- package/dist/{llm-client-DHx8pzyJ.js → llm-client-DKB25jV8.js} +3 -3
- package/dist/{llm-client-DHx8pzyJ.js.map → llm-client-DKB25jV8.js.map} +1 -1
- package/dist/meta-eval/index.d.ts +2 -2
- package/dist/meta-eval/index.js +3 -3
- package/dist/{mint-DyRUc9k6.js → mint-Ctwk079K.js} +4 -4
- package/dist/{mint-DyRUc9k6.js.map → mint-Ctwk079K.js.map} +1 -1
- package/dist/multishot/index.d.ts +2 -2
- package/dist/openapi.json +1 -1
- package/dist/{paired-arms-BbFKrAU-.js → paired-arms-iZ08VFMN.js} +3 -3
- package/dist/{paired-arms-BbFKrAU-.js.map → paired-arms-iZ08VFMN.js.map} +1 -1
- package/dist/pipelines/index.js +2 -2
- package/dist/profile-cell.d.ts +1 -1
- package/dist/profile-cell.js +1 -1
- package/dist/{propose-review-control-SQ-n9-We.js → propose-review-control-DLXz4FCX.js} +2 -2
- package/dist/{propose-review-control-SQ-n9-We.js.map → propose-review-control-DLXz4FCX.js.map} +1 -1
- package/dist/registry-BdM7SuTr.d.ts +124 -0
- package/dist/registry-BdM7SuTr.d.ts.map +1 -0
- package/dist/{release-report-DooPguBc.js → release-report-B5XPBvAU.js} +4 -4
- package/dist/{release-report-DooPguBc.js.map → release-report-B5XPBvAU.js.map} +1 -1
- package/dist/{release-report-DpBxGGI1.d.ts → release-report-CofgVNZt.d.ts} +4 -4
- package/dist/{release-report-DpBxGGI1.d.ts.map → release-report-CofgVNZt.d.ts.map} +1 -1
- package/dist/{replay-C6wRg47C.js → replay-Bju0T8Ls.js} +248 -8
- package/dist/replay-Bju0T8Ls.js.map +1 -0
- package/dist/{replay-BRfMIs81.d.ts → replay-K8FaC0CB.d.ts} +227 -52
- package/dist/replay-K8FaC0CB.d.ts.map +1 -0
- package/dist/reporting.d.ts +4 -4
- package/dist/reporting.js +4 -4
- package/dist/{researcher-Doo95b50.d.ts → researcher-Da0Wj-bt.d.ts} +6 -7
- package/dist/researcher-Da0Wj-bt.d.ts.map +1 -0
- package/dist/{reward-hacking-D-QqXvg-.d.ts → reward-hacking-CQ3hTCO3.d.ts} +2 -2
- package/dist/{reward-hacking-D-QqXvg-.d.ts.map → reward-hacking-CQ3hTCO3.d.ts.map} +1 -1
- package/dist/{reward-hacking-a-kYs0-i.js → reward-hacking-GyN0kMd8.js} +3 -3
- package/dist/{reward-hacking-a-kYs0-i.js.map → reward-hacking-GyN0kMd8.js.map} +1 -1
- package/dist/rl.d.ts +6 -6
- package/dist/rl.js +9 -9
- package/dist/rollout/index.d.ts +1 -1
- package/dist/rollout/index.js +3 -3
- package/dist/{rollout-DLSUIWLu.js → rollout-DQFl0UXA.js} +2 -2
- package/dist/{rollout-DLSUIWLu.js.map → rollout-DQFl0UXA.js.map} +1 -1
- package/dist/{rubric-predictive-validity-BJf-8ejY.js → rubric-predictive-validity-BRR632r1.js} +2 -2
- package/dist/{rubric-predictive-validity-BJf-8ejY.js.map → rubric-predictive-validity-BRR632r1.js.map} +1 -1
- package/dist/{rubric-predictive-validity-C1dCLcvb.d.ts → rubric-predictive-validity-C4sztLR3.d.ts} +2 -2
- package/dist/{rubric-predictive-validity-C1dCLcvb.d.ts.map → rubric-predictive-validity-C4sztLR3.d.ts.map} +1 -1
- package/dist/{run-evidence-DokQtX0-.d.ts → run-evidence-BDIircdA.d.ts} +3 -3
- package/dist/{run-evidence-DokQtX0-.d.ts.map → run-evidence-BDIircdA.d.ts.map} +1 -1
- package/dist/{run-record-DcObtIGh.d.ts → run-record-BPCa2rQ8.d.ts} +4 -4
- package/dist/{run-record-DcObtIGh.d.ts.map → run-record-BPCa2rQ8.d.ts.map} +1 -1
- package/dist/{run-record-BIwU2wdV.js → run-record-vRgqWmJw.js} +3 -3
- package/dist/{run-record-BIwU2wdV.js.map → run-record-vRgqWmJw.js.map} +1 -1
- package/dist/{semantic-concept-judge-Btozx3Vc.js → semantic-concept-judge-Bz64IckK.js} +4 -4
- package/dist/{semantic-concept-judge-Btozx3Vc.js.map → semantic-concept-judge-Bz64IckK.js.map} +1 -1
- package/dist/{server-Bz3WQJs6.js → server-KjXZZUDX.js} +3 -3
- package/dist/{server-Bz3WQJs6.js.map → server-KjXZZUDX.js.map} +1 -1
- package/dist/{skill-usage-BDQVPIG1.d.ts → skill-usage-CFDLLlhF.d.ts} +25 -47
- package/dist/skill-usage-CFDLLlhF.d.ts.map +1 -0
- package/dist/{skillopt-optimization-method-CwSYkv35.d.ts → skillopt-optimization-method-BpbnlvAZ.d.ts} +11 -12
- package/dist/skillopt-optimization-method-BpbnlvAZ.d.ts.map +1 -0
- package/dist/{skillopt-optimization-method-0UmPD6aP.js → skillopt-optimization-method-f4o9sUT4.js} +7 -7
- package/dist/{skillopt-optimization-method-0UmPD6aP.js.map → skillopt-optimization-method-f4o9sUT4.js.map} +1 -1
- package/dist/{statistics-CnGCLLqc.js → statistics-ByxzSiOM.js} +2 -2
- package/dist/{statistics-CnGCLLqc.js.map → statistics-ByxzSiOM.js.map} +1 -1
- package/dist/{statistics-CKOqre5S.d.ts → statistics-_7P642CN.d.ts} +2 -2
- package/dist/{statistics-CKOqre5S.d.ts.map → statistics-_7P642CN.d.ts.map} +1 -1
- package/dist/{summary-report-BEk8OFLs.js → summary-report-9A5y7EsK.js} +4 -4
- package/dist/{summary-report-BEk8OFLs.js.map → summary-report-9A5y7EsK.js.map} +1 -1
- package/dist/{summary-report-CPMINBqs.d.ts → summary-report-DHipz9Kx.d.ts} +3 -3
- package/dist/{summary-report-CPMINBqs.d.ts.map → summary-report-DHipz9Kx.d.ts.map} +1 -1
- package/dist/supervisor-run/index.d.ts +3 -2
- package/dist/supervisor-run/index.js +3 -2
- package/dist/{supervisor-run-Dr5HnTup.js → supervisor-run-B2EWUmQY.js} +28 -454
- package/dist/supervisor-run-B2EWUmQY.js.map +1 -0
- package/dist/{test-graded-scenario-BsqWLmPt.js → test-graded-scenario-JHcKQNpq.js} +2 -2
- package/dist/{test-graded-scenario-BsqWLmPt.js.map → test-graded-scenario-JHcKQNpq.js.map} +1 -1
- package/dist/tools-DZk2Jn64.js +1876 -0
- package/dist/tools-DZk2Jn64.js.map +1 -0
- package/dist/traces.d.ts +6 -7
- package/dist/traces.js +5 -6
- package/dist/{types-DVjczBM9.d.ts → types-CKswbJGO.d.ts} +260 -6
- package/dist/types-CKswbJGO.d.ts.map +1 -0
- package/dist/{types-DiWLru6Z.d.ts → types-CTGbIm57.d.ts} +5 -5
- package/dist/{types-DiWLru6Z.d.ts.map → types-CTGbIm57.d.ts.map} +1 -1
- package/dist/types-CTvKfr5F.d.ts +804 -0
- package/dist/types-CTvKfr5F.d.ts.map +1 -0
- package/dist/{index-CyC1BTmn.d.ts → types-Dea6tiVI.d.ts} +16 -238
- package/dist/types-Dea6tiVI.d.ts.map +1 -0
- package/dist/wire/index.d.ts +3 -3
- package/dist/wire/index.js +1 -1
- package/docs/feedback-trajectories.md +100 -1
- package/docs/trace-analysis.md +374 -58
- package/package.json +5 -1
- package/dist/analyst-BkTS3C58.d.ts +0 -89
- package/dist/analyst-BkTS3C58.d.ts.map +0 -1
- package/dist/analyst-j5je5J7c.js +0 -152
- package/dist/analyst-j5je5J7c.js.map +0 -1
- package/dist/campaign-Dz8uQnhC.js.map +0 -1
- package/dist/client-DcvgkaZi.d.ts.map +0 -1
- package/dist/default-registry-Brxr728w.d.ts.map +0 -1
- package/dist/default-registry-CHmdy2An.js.map +0 -1
- package/dist/errors-8YnH8WlF.js.map +0 -1
- package/dist/errors-CEk209JS.d.ts.map +0 -1
- package/dist/feedback-trajectory-CVaeREXV.d.ts.map +0 -1
- package/dist/index-B4Fjfo5U.d.ts.map +0 -1
- package/dist/index-CyC1BTmn.d.ts.map +0 -1
- package/dist/llm-client-BiK4HW0u.d.ts +0 -290
- package/dist/llm-client-BiK4HW0u.d.ts.map +0 -1
- package/dist/raw-provider-sink-BU29Sh8h.d.ts +0 -134
- package/dist/raw-provider-sink-BU29Sh8h.d.ts.map +0 -1
- package/dist/replay-BRfMIs81.d.ts.map +0 -1
- package/dist/replay-C6wRg47C.js.map +0 -1
- package/dist/researcher-Doo95b50.d.ts.map +0 -1
- package/dist/skill-usage-BDQVPIG1.d.ts.map +0 -1
- package/dist/skillopt-optimization-method-CwSYkv35.d.ts.map +0 -1
- package/dist/store-CxJry_cs.d.ts +0 -229
- package/dist/store-CxJry_cs.d.ts.map +0 -1
- package/dist/supervisor-run-Dr5HnTup.js.map +0 -1
- package/dist/tools-D8yTtNSN.js +0 -1190
- package/dist/tools-D8yTtNSN.js.map +0 -1
- package/dist/types-Cc3qbqzj.d.ts +0 -387
- package/dist/types-Cc3qbqzj.d.ts.map +0 -1
- package/dist/types-DVjczBM9.d.ts.map +0 -1
package/docs/trace-analysis.md
CHANGED
|
@@ -1,52 +1,81 @@
|
|
|
1
1
|
# Trace Analysis
|
|
2
2
|
|
|
3
|
-
Trace analysis
|
|
3
|
+
Trace analysis answers three different questions:
|
|
4
4
|
|
|
5
|
-
|
|
6
|
-
|
|
7
|
-
|
|
8
|
-
|
|
9
|
-
|
|
10
|
-
|
|
5
|
+
1. What happened in the run?
|
|
6
|
+
2. Which exact steps support a suspected problem?
|
|
7
|
+
3. Does an analyst find labeled problems reliably enough to use?
|
|
8
|
+
|
|
9
|
+
Keep those answers separate.
|
|
10
|
+
A generated finding is a review request, not training truth.
|
|
11
|
+
|
|
12
|
+
## Run The Built-In Analysts
|
|
13
|
+
|
|
14
|
+
The default registry always includes deterministic checks.
|
|
15
|
+
Model-assisted analysts are added only when you provide a model client.
|
|
16
|
+
|
|
17
|
+
```ts
|
|
18
|
+
import {
|
|
19
|
+
buildDefaultAnalystRegistry,
|
|
20
|
+
} from '@tangle-network/agent-eval/analyst'
|
|
21
|
+
import { OtlpFileTraceStore } from '@tangle-network/agent-eval/traces'
|
|
22
|
+
|
|
23
|
+
const traceStore = new OtlpFileTraceStore({ path: 'traces.otlp.jsonl' })
|
|
24
|
+
const analysts = buildDefaultAnalystRegistry()
|
|
25
|
+
|
|
26
|
+
const result = await analysts.run('release-42', { traceStore })
|
|
11
27
|
|
|
12
|
-
|
|
28
|
+
for (const finding of result.findings) {
|
|
29
|
+
console.log(finding.claim, finding.evidence_refs)
|
|
30
|
+
}
|
|
31
|
+
```
|
|
13
32
|
|
|
14
|
-
Use `
|
|
33
|
+
Use `result.per_analyst` to inspect failures, latency, calls, tokens, and cost.
|
|
34
|
+
An analyst failure is recorded separately from an agent failure.
|
|
15
35
|
|
|
16
|
-
|
|
17
|
-
|
|
18
|
-
|
|
19
|
-
|
|
20
|
-
|
|
36
|
+
Products can implement `TraceAnalysisStore`; they do not need to use the file store in production.
|
|
37
|
+
Custom stores provide `hasTrace` and batched `hasSpans` alongside the seven reads, and accept a `TraceAnalysisStoreContext` so cancellation reaches storage and scans.
|
|
38
|
+
The binding validates every custom-store result.
|
|
39
|
+
Missing fields, undeclared fields, inconsistent counts, and false continuation flags throw `TraceAnalysisStoreContractError` with code `backend_integrity`.
|
|
40
|
+
The analyst runs one Ax executor loop and accepts only an explicit structured `final(task, { report, findings })` result; max-turn fallback text fails loud.
|
|
21
41
|
|
|
22
|
-
|
|
23
|
-
TraceAnalyst to explain the evidence behind those decisions.
|
|
42
|
+
### Bind the same reads into another agent environment
|
|
24
43
|
|
|
25
|
-
|
|
44
|
+
`buildTraceAnalysisToolDescriptors()` is the canonical definition of the analyst's seven bounded read operations and does not expose Ax types.
|
|
45
|
+
Each descriptor carries the stable `traces` namespace, function name, description, JSON input schema in `parameters`, and a handler already bound to the supplied `TraceAnalysisStore`.
|
|
46
|
+
`buildTraceAnalystTools()` adapts those descriptors into Ax functions; it does not define a second tool surface.
|
|
47
|
+
The bound handlers wrap custom stores with `createBoundedTraceAnalysisStore()`, so page limits, byte ceilings, not-found errors, and cancellation do not depend on the transport or adapter.
|
|
26
48
|
|
|
27
49
|
```ts
|
|
28
50
|
import {
|
|
29
|
-
|
|
30
|
-
|
|
31
|
-
} from '@tangle-network/agent-eval'
|
|
51
|
+
buildTraceAnalysisToolDescriptors,
|
|
52
|
+
type TraceAnalysisStore,
|
|
53
|
+
} from '@tangle-network/agent-eval/traces'
|
|
32
54
|
|
|
33
|
-
const
|
|
34
|
-
|
|
35
|
-
question: 'Why did app-runtime holdout runs fail this week?',
|
|
36
|
-
}, {
|
|
37
|
-
source: new OtlpFileTraceStore({ path: 'traces/otlp.jsonl' }),
|
|
38
|
-
ai,
|
|
39
|
-
model: 'gpt-4o-2024-11-20',
|
|
40
|
-
maxSubqueries: 4,
|
|
41
|
-
maxParallelSubqueries: 2,
|
|
42
|
-
signal: abortController.signal,
|
|
43
|
-
})
|
|
55
|
+
declare const store: TraceAnalysisStore
|
|
56
|
+
declare function qualifyToolName(namespace: string, name: string): string
|
|
44
57
|
|
|
45
|
-
|
|
58
|
+
const tools = buildTraceAnalysisToolDescriptors({ store }).map(
|
|
59
|
+
({ namespace, name, description, parameters, handler }) => ({
|
|
60
|
+
name: qualifyToolName(namespace, name),
|
|
61
|
+
description,
|
|
62
|
+
inputSchema: parameters,
|
|
63
|
+
handler,
|
|
64
|
+
}),
|
|
65
|
+
)
|
|
46
66
|
```
|
|
47
67
|
|
|
48
|
-
|
|
49
|
-
The
|
|
68
|
+
Map these fields into the host's existing tool transport.
|
|
69
|
+
The host owns namespace encoding; use its existing convention instead of inventing one here.
|
|
70
|
+
Do not copy the schemas or reimplement the handlers in an MCP, Runtime, or provider adapter.
|
|
71
|
+
|
|
72
|
+
`queryTraces.limit`, `viewSpans.span_ids`, and search `max_matches` caps are present in the JSON Schemas and enforced before store calls.
|
|
73
|
+
Invalid arguments throw `TraceAnalysisValidationError` with code `validation`; responses that cannot fit their byte ceiling throw `TraceAnalysisLimitError` with code `limit_exceeded`.
|
|
74
|
+
Search patterns use RE2 syntax, which rejects backreferences and lookaround instead of allowing exponential-time expressions.
|
|
75
|
+
Search results return `hits` and an exact `has_more` flag; they do not invent a total after a capped scan.
|
|
76
|
+
`viewSpans` partitions every requested id across `spans`, `missing_span_ids`, and `omitted_span_ids`; `has_more` is true when omitted ids must be retried.
|
|
77
|
+
Attribute and match text shortening includes a deterministic marker.
|
|
78
|
+
Trace pages set `has_more`, and the overview returns every error cluster or fails explicitly when the configured response limit is too small.
|
|
50
79
|
|
|
51
80
|
### Analyze captured tool spans in memory
|
|
52
81
|
|
|
@@ -88,35 +117,322 @@ for (const c of overview.error_clusters) {
|
|
|
88
117
|
}
|
|
89
118
|
```
|
|
90
119
|
|
|
91
|
-
|
|
92
|
-
|
|
120
|
+
## Recursive control integrity (no LLM)
|
|
121
|
+
|
|
122
|
+
`CONTROL_INTEGRITY_ANALYST` checks the existing `SupervisorRunSources` or `SupervisorRunTree` directly.
|
|
123
|
+
It does not define another run format.
|
|
124
|
+
Register it as a custom-input analyst and pass the existing value under its stable id:
|
|
125
|
+
|
|
126
|
+
```ts
|
|
127
|
+
import {
|
|
128
|
+
AnalystRegistry,
|
|
129
|
+
CONTROL_INTEGRITY_ANALYST,
|
|
130
|
+
} from '@tangle-network/agent-eval/analyst'
|
|
131
|
+
import {
|
|
132
|
+
readLoopsSupervisorRun,
|
|
133
|
+
} from '@tangle-network/agent-eval/supervisor-run'
|
|
93
134
|
|
|
94
|
-
|
|
135
|
+
const sources = await readLoopsSupervisorRun(runDir)
|
|
136
|
+
const registry = new AnalystRegistry()
|
|
137
|
+
registry.register(CONTROL_INTEGRITY_ANALYST)
|
|
138
|
+
|
|
139
|
+
const result = await registry.run('run-123', {
|
|
140
|
+
custom: { 'control-integrity': sources },
|
|
141
|
+
})
|
|
142
|
+
```
|
|
143
|
+
|
|
144
|
+
Pass `SupervisorRunSources` when it is available.
|
|
145
|
+
A `SupervisorRunTree` does not retain raw journal multiplicity or worker request and acknowledgement rows, so tree input explicitly reports those checks as unavailable.
|
|
146
|
+
|
|
147
|
+
The deterministic pass can prove only facts represented by these two existing surfaces.
|
|
148
|
+
|
|
149
|
+
| Question | Current evidence | What the analyst can say |
|
|
150
|
+
|---|---|---|
|
|
151
|
+
| Is every invocation attached to one unambiguous tree? | `rootId`, `rollout_id`, `parent_rollout_id`, `run_id` | Duplicate ids, missing parents, extra parentless roots, cross-run edges, and ancestry cycles are violations with exact field references. |
|
|
152
|
+
| Did invocation roles survive capture? | Explicit journal and `RolloutLine.role` values | The root must remain `supervisor`; non-root roles are consumed as recorded, and workers may spawn workers. |
|
|
153
|
+
| Is the causal order possible? | `outcome.metrics.spawned_at`, `started_at`, `settled_at`, `completed_at`, `finished_at` when present | A child before its parent, a child after its parent closed, or a close before a start is a violation; absent timestamps produce no timing claim. |
|
|
154
|
+
| Did a queued steer reach the worker? | `SupervisorRunSources.workers[].inbox` and `.events` | Requests and acknowledgements are joined by request id, not compared as totals. Missing, malformed, duplicate, or uncorrelated rows make the affected count unavailable. |
|
|
155
|
+
| Can behavior be attributed to an exact profile? | `policy.agent_profile_cell_id` | An absent id is reported as unavailable. |
|
|
156
|
+
| Can action authorship or reasoning be inspected? | `messages[]` | Empty gap rows are reported as unavailable. |
|
|
157
|
+
|
|
158
|
+
An empty finding list means only that no implemented rule fired on the captured fields.
|
|
159
|
+
It does not certify that an agent chose the action, that the action was authorized, that a budget or depth limit was enforced, or that a finding caused a later decision.
|
|
160
|
+
Those claims require upstream action-decision events carrying `action_id`, `actor_rollout_id`, `target_rollout_id`, `action_kind`, `authority_snapshot_id`, requested and granted resource/depth values, the authorization result, and any `finding_id` or evidence references that caused the action.
|
|
161
|
+
Resume integrity additionally requires an explicit prior-session id and resumed-session id rather than a prose summary.
|
|
162
|
+
|
|
163
|
+
Malformed source rows are excluded from structural claims.
|
|
164
|
+
Their count is retained in `SupervisorRunTree.gaps`, so analyzing a projected tree later cannot turn an unreadable parent row into a missing-parent violation.
|
|
165
|
+
|
|
166
|
+
## Add A Custom Analyst
|
|
167
|
+
|
|
168
|
+
`defineTraceAnalyst()` fills the fixed registry fields.
|
|
169
|
+
The custom function receives the same bounded `TraceAnalysisStore` used by the built-ins.
|
|
170
|
+
|
|
171
|
+
```ts
|
|
172
|
+
import {
|
|
173
|
+
AnalystRegistry,
|
|
174
|
+
defineTraceAnalyst,
|
|
175
|
+
makeFinding,
|
|
176
|
+
} from '@tangle-network/agent-eval/analyst'
|
|
177
|
+
|
|
178
|
+
const analysts = new AnalystRegistry()
|
|
179
|
+
|
|
180
|
+
analysts.register(defineTraceAnalyst({
|
|
181
|
+
id: 'repeated-tool-errors',
|
|
182
|
+
description: 'Reports the largest repeated tool error cluster.',
|
|
183
|
+
cost: { kind: 'deterministic' },
|
|
184
|
+
async analyze(store) {
|
|
185
|
+
const overview = await store.getOverview({ has_errors: true })
|
|
186
|
+
const cluster = overview.error_clusters[0]
|
|
187
|
+
if (!cluster) return []
|
|
188
|
+
|
|
189
|
+
return [makeFinding({
|
|
190
|
+
analyst_id: 'repeated-tool-errors',
|
|
191
|
+
area: 'tool-use',
|
|
192
|
+
subject: cluster.signature,
|
|
193
|
+
claim: `${cluster.span_count} failed spans share one error`,
|
|
194
|
+
severity: 'high',
|
|
195
|
+
confidence: 1,
|
|
196
|
+
evidence_refs: [{
|
|
197
|
+
kind: 'span',
|
|
198
|
+
uri: `trace://${encodeURIComponent(cluster.exemplar_trace_ids[0])}/span/${encodeURIComponent(cluster.exemplar_span_ids[0])}`,
|
|
199
|
+
excerpt: cluster.status_message_sample,
|
|
200
|
+
}],
|
|
201
|
+
recommended_action: 'Fix the highest-frequency tool error before changing prompts.',
|
|
202
|
+
validation_plan: 'Run fresh cases and require this error signature to disappear.',
|
|
203
|
+
})]
|
|
204
|
+
},
|
|
205
|
+
}))
|
|
206
|
+
```
|
|
207
|
+
|
|
208
|
+
Use code for exact facts such as exit codes, missing fields, and repeated calls.
|
|
209
|
+
Use model-assisted analysts for semantic questions such as whether a response ignored user intent.
|
|
210
|
+
|
|
211
|
+
## Measure An Analyst
|
|
212
|
+
|
|
213
|
+
Do not judge an analyst by persuasive prose.
|
|
214
|
+
Label the issue identity and exact evidence locations, then run the same cases through every implementation.
|
|
215
|
+
|
|
216
|
+
```ts
|
|
217
|
+
import {
|
|
218
|
+
compareAnalystRunners,
|
|
219
|
+
registryBenchmarkRunner,
|
|
220
|
+
renderAnalystBenchmarkMarkdown,
|
|
221
|
+
runAnalystBenchmark,
|
|
222
|
+
traceStoreEvidenceResolver,
|
|
223
|
+
} from '@tangle-network/agent-eval/analyst'
|
|
95
224
|
|
|
96
|
-
|
|
225
|
+
const benchmark = await runAnalystBenchmark({
|
|
226
|
+
cases: [{
|
|
227
|
+
id: 'failed-command',
|
|
228
|
+
input: { traceStore },
|
|
229
|
+
expectedIssues: [{
|
|
230
|
+
id: 'repeated-command',
|
|
231
|
+
subjects: ['failure-mode:repeated-command'],
|
|
232
|
+
evidence: [{ kind: 'span', uri: 'trace://run-1/span/tool-3' }],
|
|
233
|
+
criticalEvidence: [{ kind: 'span', uri: 'trace://run-1/span/tool-1' }],
|
|
234
|
+
}],
|
|
235
|
+
labeledEvidence: [
|
|
236
|
+
{ kind: 'span', uri: 'trace://run-1/span/tool-1' },
|
|
237
|
+
{ kind: 'span', uri: 'trace://run-1/span/tool-3' },
|
|
238
|
+
],
|
|
239
|
+
}],
|
|
240
|
+
runners: [registryBenchmarkRunner({ id: 'built-in', registry: analysts })],
|
|
241
|
+
repetitions: 3,
|
|
242
|
+
resolveEvidence: traceStoreEvidenceResolver((input) => input.traceStore),
|
|
243
|
+
benchmark: {
|
|
244
|
+
id: 'failure-localization',
|
|
245
|
+
dataset: {
|
|
246
|
+
id: 'my-team/trace-failures',
|
|
247
|
+
revision: 'git-sha-or-content-digest',
|
|
248
|
+
split: 'test',
|
|
249
|
+
},
|
|
250
|
+
},
|
|
251
|
+
})
|
|
252
|
+
|
|
253
|
+
console.log(renderAnalystBenchmarkMarkdown(benchmark))
|
|
254
|
+
```
|
|
255
|
+
|
|
256
|
+
The result reports:
|
|
257
|
+
|
|
258
|
+
- issue recall and finding precision,
|
|
259
|
+
- first bad step accuracy,
|
|
260
|
+
- citation coverage, agreement with labeled locations, and actual location resolution,
|
|
261
|
+
- false positives on clean cases,
|
|
262
|
+
- repeat agreement,
|
|
263
|
+
- failed runs,
|
|
264
|
+
- latency, calls, every reported token counter, and known or missing cost,
|
|
265
|
+
- dataset revision, case tags, case metadata, and runner metadata.
|
|
266
|
+
|
|
267
|
+
Use `compareAnalystRunners()` for paired differences between two implementations.
|
|
268
|
+
Repetitions are averaged within each case before comparison.
|
|
269
|
+
Treat its interval as inferential only with at least 20 independent cases.
|
|
270
|
+
|
|
271
|
+
## Load Public Trace Labels
|
|
272
|
+
|
|
273
|
+
Use the published label adapters with `@tangle-network/traces` or your own trajectory loader.
|
|
274
|
+
Agent Eval does not download datasets or own trace capture.
|
|
275
|
+
Load public data at an immutable commit and record that commit in `benchmark.dataset.revision`.
|
|
276
|
+
|
|
277
|
+
```ts
|
|
278
|
+
import {
|
|
279
|
+
agentRxBenchmarkCase,
|
|
280
|
+
agentRxPredictionsToFindings,
|
|
281
|
+
codeTraceBenchCase,
|
|
282
|
+
codeTracerPredictionsToFindings,
|
|
283
|
+
} from '@tangle-network/agent-eval/analyst'
|
|
284
|
+
import { otlpTextToTraceAnalysisStore } from '@tangle-network/agent-eval/traces'
|
|
285
|
+
import { chatTrajectoryToSpans, serializeSpans } from '@tangle-network/traces'
|
|
286
|
+
|
|
287
|
+
const codeSpans = chatTrajectoryToSpans(codeTraceTrajectory, {
|
|
288
|
+
traceId: codeTraceRow.traj_id,
|
|
289
|
+
})
|
|
290
|
+
const codeCase = codeTraceBenchCase(codeTraceRow, {
|
|
291
|
+
traceStore: otlpTextToTraceAnalysisStore(serializeSpans(codeSpans)),
|
|
292
|
+
})
|
|
293
|
+
|
|
294
|
+
const agentRxSpans = chatTrajectoryToSpans(agentRxMessages, {
|
|
295
|
+
traceId: String(agentRxRow.trajectory_id),
|
|
296
|
+
stepMode: 'message',
|
|
297
|
+
})
|
|
298
|
+
const rootCauseCase = agentRxBenchmarkCase(agentRxRow, {
|
|
299
|
+
traceStore: otlpTextToTraceAnalysisStore(serializeSpans(agentRxSpans)),
|
|
300
|
+
}, {
|
|
301
|
+
stepCount: agentRxMessages.length,
|
|
302
|
+
})
|
|
303
|
+
|
|
304
|
+
const codeTracerFindings = codeTracerPredictionsToFindings(
|
|
305
|
+
codeTraceRow.traj_id,
|
|
306
|
+
codetracerLabels,
|
|
307
|
+
{ stepCount: codeTraceRow.step_count },
|
|
308
|
+
)
|
|
309
|
+
const agentRxFindings = agentRxPredictionsToFindings(
|
|
310
|
+
agentRxRow.trajectory_id,
|
|
311
|
+
agentRxJudgeOutput,
|
|
312
|
+
{ stepCount: agentRxMessages.length },
|
|
313
|
+
)
|
|
314
|
+
```
|
|
315
|
+
|
|
316
|
+
`codeTraceBenchCase()` accepts the public [CodeTraceBench](https://huggingface.co/datasets/NJU-LINK/CodeTraceBench) JSONL format.
|
|
317
|
+
It scores the published incorrect-step task by default, including clean trajectories.
|
|
318
|
+
Pass `labelSet: 'incorrect-and-unuseful'` to both the case and prediction adapters only for an explicitly combined experiment.
|
|
319
|
+
Every cited step is checked against `step_count`.
|
|
320
|
+
|
|
321
|
+
`agentRxBenchmarkCase()` accepts the public [AgentRx](https://huggingface.co/datasets/microsoft/AgentRx) label format.
|
|
322
|
+
AgentRx category quality and root-step accuracy are scored independently.
|
|
323
|
+
`traceAnalystQualityJudge` averages them when a root-step label exists.
|
|
324
|
+
Pass `target: 'all-failures'` only when the analyst is designed to identify every annotated failure.
|
|
325
|
+
|
|
326
|
+
Both adapters emit `trace://<id>/span/step-<n>` evidence by default.
|
|
327
|
+
`@tangle-network/traces` uses the same IDs when converting chat trajectories.
|
|
328
|
+
Pass `stepUri` when your trace store uses another URI scheme.
|
|
329
|
+
`codeTracerPredictionsToFindings()` and `agentRxPredictionsToFindings()` translate the maintained upstream engines' native outputs into the same evidence and category shape.
|
|
330
|
+
AgentRx `Report.to_dict()` judge votes reduce to the upstream majority failure type and Python-rounded mean step, and direct `failures` arrays use the same reduction.
|
|
331
|
+
`failure_case: 0` produces no finding, which scores as a missed root cause on AgentRx's failed trajectories.
|
|
332
|
+
|
|
333
|
+
## Use Upstream Scorers
|
|
334
|
+
|
|
335
|
+
Agent Eval adapts upstream evaluators instead of copying them.
|
|
336
|
+
|
|
337
|
+
```ts
|
|
338
|
+
import { createEvaluator } from '@arizeai/phoenix-evals'
|
|
339
|
+
import { ExactMatch } from 'autoevals'
|
|
340
|
+
import {
|
|
341
|
+
autoevalsScorerJudge,
|
|
342
|
+
phoenixEvaluatorJudge,
|
|
343
|
+
} from '@tangle-network/agent-eval/campaign'
|
|
344
|
+
|
|
345
|
+
const phoenix = createEvaluator(
|
|
346
|
+
({ output, expected }) => output === expected ? 1 : 0,
|
|
347
|
+
{ name: 'exact-match', kind: 'CODE', telemetry: { isEnabled: false } },
|
|
348
|
+
)
|
|
349
|
+
|
|
350
|
+
const phoenixJudge = phoenixEvaluatorJudge(phoenix, {
|
|
351
|
+
mapInput: ({ artifact, scenario }) => ({ output: artifact, expected: scenario.expected }),
|
|
352
|
+
})
|
|
353
|
+
|
|
354
|
+
const autoevalsJudge = autoevalsScorerJudge(ExactMatch, {
|
|
355
|
+
name: 'exact-match',
|
|
356
|
+
kind: 'CODE',
|
|
357
|
+
mapInput: ({ artifact, scenario }) => ({ output: artifact, expected: scenario.expected }),
|
|
358
|
+
})
|
|
359
|
+
```
|
|
360
|
+
|
|
361
|
+
These adapters do not install either upstream package for consumers.
|
|
362
|
+
Install only the scorer package you use.
|
|
363
|
+
Missing or non-finite scores throw instead of becoming passes.
|
|
364
|
+
Phoenix evaluators marked `MINIMIZE` or `NEUTRAL` require `toComposite` so candidate selection never assumes the wrong direction.
|
|
365
|
+
Mark model-backed evaluators as `kind: 'LLM'` and provide `paidCall` with the model and a receipt mapper.
|
|
366
|
+
The campaign then passes its cancellation signal and cost ledger through the adapter.
|
|
367
|
+
An LLM evaluator is rejected before execution when either cost capture or the campaign ledger is missing.
|
|
368
|
+
|
|
369
|
+
## Turn Reviewed Findings Into Eval Data
|
|
370
|
+
|
|
371
|
+
Generated findings can populate a review queue.
|
|
372
|
+
They cannot promote themselves into learning data.
|
|
373
|
+
|
|
374
|
+
```ts
|
|
375
|
+
import {
|
|
376
|
+
analystFindingDigest,
|
|
377
|
+
analystRunDigest,
|
|
378
|
+
analystRunToFeedbackTrajectory,
|
|
379
|
+
analystRunToReviewRequests,
|
|
380
|
+
} from '@tangle-network/agent-eval'
|
|
381
|
+
|
|
382
|
+
const runDigest = analystRunDigest(result)
|
|
383
|
+
const requests = analystRunToReviewRequests(result)
|
|
384
|
+
await reviewQueue.add(requests)
|
|
385
|
+
|
|
386
|
+
declare const acceptedFindingIds: ReadonlySet<string>
|
|
387
|
+
|
|
388
|
+
const trajectory = analystRunToFeedbackTrajectory(result, {
|
|
389
|
+
task: { intent: 'Find why the command failed.' },
|
|
390
|
+
reviewRequests: requests,
|
|
391
|
+
reviewDecisions: [
|
|
392
|
+
...result.findings.map((finding) => ({
|
|
393
|
+
runDigest,
|
|
394
|
+
findingId: finding.finding_id,
|
|
395
|
+
findingDigest: analystFindingDigest(finding),
|
|
396
|
+
verdict: acceptedFindingIds.has(finding.finding_id) ? 'confirmed' as const : 'rejected' as const,
|
|
397
|
+
source: 'user' as const,
|
|
398
|
+
reviewerId: 'reviewer-42',
|
|
399
|
+
reviewId: 'trace-review-918',
|
|
400
|
+
reason: 'Reviewed against the cited span.',
|
|
401
|
+
decidedAt: new Date().toISOString(),
|
|
402
|
+
})),
|
|
403
|
+
{
|
|
404
|
+
runDigest,
|
|
405
|
+
verdict: 'completeness_assessed',
|
|
406
|
+
missedIssues: [],
|
|
407
|
+
source: 'user',
|
|
408
|
+
reviewerId: 'reviewer-42',
|
|
409
|
+
reviewId: 'trace-review-918',
|
|
410
|
+
reason: 'Reviewed the full run for omitted findings.',
|
|
411
|
+
decidedAt: new Date().toISOString(),
|
|
412
|
+
},
|
|
413
|
+
],
|
|
414
|
+
trace: { artifactUri: 'traces.otlp.jsonl', traceIds: ['run-1'] },
|
|
415
|
+
})
|
|
416
|
+
```
|
|
97
417
|
|
|
98
|
-
|
|
99
|
-
|
|
100
|
-
|
|
101
|
-
|
|
102
|
-
|
|
103
|
-
- sandbox/build/test/deploy spans with exit codes and log artifacts
|
|
104
|
-
- custom events for knowledge readiness and integration gates
|
|
105
|
-
- final run outcome with pass/score/failure class
|
|
418
|
+
`analystRunToFeedbackTrajectory()` stores review requests separately from labels.
|
|
419
|
+
It can archive an unreviewed run.
|
|
420
|
+
`feedbackTrajectoryToOptimizerRow()` requires every decision to match the complete run digest, every finding decision to match the finding digest, and one independent completeness assessment.
|
|
421
|
+
Its score is F1 over confirmed findings and independently identified misses.
|
|
422
|
+
Generic labels and run-level outcomes do not satisfy these requirements.
|
|
106
423
|
|
|
107
|
-
|
|
424
|
+
## Required Trace Data
|
|
108
425
|
|
|
109
|
-
|
|
426
|
+
Useful analysis needs:
|
|
110
427
|
|
|
111
|
-
|
|
112
|
-
|
|
428
|
+
- stable run, trace, and span IDs,
|
|
429
|
+
- parent-child links and ordered timestamps,
|
|
430
|
+
- model, prompt, and configuration identity,
|
|
431
|
+
- complete tool names, arguments, results, and error codes,
|
|
432
|
+
- token, cost, and latency data when available,
|
|
433
|
+
- retrieval source IDs and scores when retrieval is involved,
|
|
434
|
+
- final environment outcomes such as tests, task completion, or policy blocks.
|
|
113
435
|
|
|
114
|
-
|
|
115
|
-
2. Emit canonical spans/events while the user task runs.
|
|
116
|
-
3. Convert the completed run to `FeedbackTrajectory` for replay.
|
|
117
|
-
4. Convert promotion-grade runs to `RunRecord` with `controlRunToRunRecord`.
|
|
118
|
-
5. Run TraceAnalyst over failure-heavy trace sets.
|
|
119
|
-
6. Feed findings into `ActionableSideInfo`, failure clusters, and release
|
|
120
|
-
reports.
|
|
436
|
+
Do not include secrets, raw OAuth tokens, or unredacted personal data.
|
|
121
437
|
|
|
122
|
-
|
|
438
|
+
Use [`@tangle-network/traces`](https://github.com/tangle-network/traces) to normalize coding-agent sessions, run HALO as an external report engine, or run Hodoscope as a behavior-discovery engine.
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@tangle-network/agent-eval",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.137.0",
|
|
4
4
|
"description": "Evaluate and improve AI agents from runs, traces, judges, and feedback. Compare candidates, cluster failures, measure lift, and gate releases.",
|
|
5
5
|
"homepage": "https://github.com/tangle-network/agent-eval#readme",
|
|
6
6
|
"repository": {
|
|
@@ -171,12 +171,16 @@
|
|
|
171
171
|
"@tangle-network/agent-core": "0.4.28",
|
|
172
172
|
"@tangle-network/agent-interface": "0.39.0",
|
|
173
173
|
"hono": "^4.12.32",
|
|
174
|
+
"linear-sum-assignment": "1.0.9",
|
|
175
|
+
"re2js": "2.8.6",
|
|
174
176
|
"zod": "^4.4.3"
|
|
175
177
|
},
|
|
176
178
|
"devDependencies": {
|
|
177
179
|
"@arethetypeswrong/cli": "^0.18.5",
|
|
180
|
+
"@arizeai/phoenix-evals": "^2.1.0",
|
|
178
181
|
"@biomejs/biome": "^2.5.5",
|
|
179
182
|
"@types/node": "^26.1.1",
|
|
183
|
+
"autoevals": "^0.3.0",
|
|
180
184
|
"esbuild": "^0.28.1",
|
|
181
185
|
"fast-check": "^4.9.0",
|
|
182
186
|
"husky": "^9.1.7",
|
|
@@ -1,89 +0,0 @@
|
|
|
1
|
-
import { t as TraceAnalysisStore } from "./store-CxJry_cs.js";
|
|
2
|
-
import { AxAIService, AxAgentActorTurnCallbackArgs } from "@ax-llm/ax";
|
|
3
|
-
//#region src/trace-analyst/analyst.d.ts
|
|
4
|
-
interface AnalyzeTracesInput {
|
|
5
|
-
/** The user-facing question. Domain framing belongs here, not in the
|
|
6
|
-
* actor description. */
|
|
7
|
-
question: string;
|
|
8
|
-
}
|
|
9
|
-
interface AnalyzeTracesResult {
|
|
10
|
-
/** The actor's submitted prose answer. */
|
|
11
|
-
answer: string;
|
|
12
|
-
/** Bulleted findings from the actor's structured completion. */
|
|
13
|
-
findings: string[];
|
|
14
|
-
/** Per-turn snapshots captured via `actorTurnCallback`. */
|
|
15
|
-
turns: AnalyzeTracesTurnSnapshot[];
|
|
16
|
-
/** Total turns the actor took. */
|
|
17
|
-
turnCount: number;
|
|
18
|
-
/** Token usage by role. */
|
|
19
|
-
usage: TraceAnalystUsage;
|
|
20
|
-
/** Full system + assistant + tool message log by role. */
|
|
21
|
-
chatLog: TraceAnalystChatLog;
|
|
22
|
-
/** Prompt version that produced this run. */
|
|
23
|
-
actorPromptVersion: string;
|
|
24
|
-
}
|
|
25
|
-
interface TraceAnalystUsage {
|
|
26
|
-
actor: TraceAnalystUsageEntry[];
|
|
27
|
-
responder: TraceAnalystUsageEntry[];
|
|
28
|
-
}
|
|
29
|
-
interface TraceAnalystUsageEntry {
|
|
30
|
-
[key: string]: unknown;
|
|
31
|
-
}
|
|
32
|
-
interface TraceAnalystChatLog {
|
|
33
|
-
actor: TraceAnalystChatMessage[];
|
|
34
|
-
responder: TraceAnalystChatMessage[];
|
|
35
|
-
}
|
|
36
|
-
interface TraceAnalystChatMessage {
|
|
37
|
-
[key: string]: unknown;
|
|
38
|
-
}
|
|
39
|
-
interface AnalyzeTracesTurnSnapshot {
|
|
40
|
-
stage: AxAgentActorTurnCallbackArgs['stage'];
|
|
41
|
-
turn: number;
|
|
42
|
-
isError: boolean;
|
|
43
|
-
/** The JS code the actor produced for this turn. */
|
|
44
|
-
code: string;
|
|
45
|
-
/** The formatted action-log entry the actor sees on the next turn. */
|
|
46
|
-
output: string;
|
|
47
|
-
/** Provider thought (when `executorOptions.showThoughts` is true and the
|
|
48
|
-
* provider returns it). */
|
|
49
|
-
thought?: string;
|
|
50
|
-
}
|
|
51
|
-
interface AnalyzeTracesOptions {
|
|
52
|
-
/** Trace data source. Pass either an OTLP-JSONL path or a custom store. */
|
|
53
|
-
source: string | TraceAnalysisStore;
|
|
54
|
-
/** Caller-provided AxAIService. */
|
|
55
|
-
ai: AxAIService;
|
|
56
|
-
/** Model id forwarded to the actor. */
|
|
57
|
-
model?: string;
|
|
58
|
-
/** Maximum model subqueries. 0 disables model fan-out. Default 4. */
|
|
59
|
-
maxSubqueries?: number;
|
|
60
|
-
/** Maximum actor turns. Default 12. */
|
|
61
|
-
maxTurns?: number;
|
|
62
|
-
/** Maximum parallel model subqueries. Default 2. */
|
|
63
|
-
maxParallelSubqueries?: number;
|
|
64
|
-
/** Cancels in-flight model and tool work. */
|
|
65
|
-
signal?: AbortSignal;
|
|
66
|
-
/** Override the actor description. */
|
|
67
|
-
actorDescription?: string;
|
|
68
|
-
/** Per-turn observability hook. */
|
|
69
|
-
onTurn?: (turn: AnalyzeTracesTurnSnapshot) => void | Promise<void>;
|
|
70
|
-
/** Override max runtime characters per turn. Default 6000. */
|
|
71
|
-
maxRuntimeChars?: number;
|
|
72
|
-
/** When set, every turn's snapshot is appended to this JSONL file
|
|
73
|
-
* immediately. If the analyst crashes mid-loop (provider 503,
|
|
74
|
-
* network error, validator reject) the partial reasoning is still
|
|
75
|
-
* on disk for diagnosis and recovery. */
|
|
76
|
-
progressLogPath?: string;
|
|
77
|
-
}
|
|
78
|
-
/**
|
|
79
|
-
* Run the trace analyst.
|
|
80
|
-
*
|
|
81
|
-
* Throws:
|
|
82
|
-
* - `TraceFileMissingError` if `source` is a path and doesn't exist.
|
|
83
|
-
* - `AxAgentClarificationError` if the analyst asks for clarification.
|
|
84
|
-
* - Provider errors (auth, rate limits) propagate from the AI service.
|
|
85
|
-
*/
|
|
86
|
-
declare function analyzeTraces(input: AnalyzeTracesInput, options: AnalyzeTracesOptions): Promise<AnalyzeTracesResult>;
|
|
87
|
-
//#endregion
|
|
88
|
-
export { analyzeTraces as a, AnalyzeTracesTurnSnapshot as i, AnalyzeTracesOptions as n, AnalyzeTracesResult as r, AnalyzeTracesInput as t };
|
|
89
|
-
//# sourceMappingURL=analyst-BkTS3C58.d.ts.map
|
|
@@ -1 +0,0 @@
|
|
|
1
|
-
{"version":3,"file":"analyst-BkTS3C58.d.ts","names":[],"sources":["../src/trace-analyst/analyst.ts"],"mappings":";;;UAOiB;;;EAGf;;UAGe;;EAEf;;EAEA;;EAEA,OAAO;;EAEP;;EAEA,OAAO;;EAEP,SAAS;;EAET;;UAGe;EACf,OAAO;EACP,WAAW;;UAGI;GACd;;UAGc;EACf,OAAO;EACP,WAAW;;UAGI;GACd;;UAGc;EACf,OAAO;EACP;EACA;;EAEA;;EAEA;;;EAGA;;UAGe;;EAEf,iBAAiB;;EAEjB,IAAI;;EAEJ;;EAEA;;EAEA;;EAEA;;EAEA,SAAS;;EAET;;EAEA,UAAU,MAAM,qCAAqC;;EAErD;;;;;EAKA;;;;;;;;;;iBAWoB,cACpB,OAAO,oBACP,SAAS,uBACR,QAAQ"}
|