@tangle-network/agent-eval 0.115.3 → 0.117.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +55 -0
- package/dist/analyst/index.d.ts +16 -11
- package/dist/analyst/index.js +33 -25
- package/dist/analyst/index.js.map +1 -1
- package/dist/{analyze-runs-BYHg6Irm.d.ts → analyze-runs--2x39HZ7.d.ts} +3 -3
- package/dist/{baseline-DsNteOgR.d.ts → baseline-DKq3gJpP.d.ts} +6 -3
- package/dist/belief-state/index.d.ts +6 -6
- package/dist/belief-state/index.js +1 -1
- package/dist/benchmarks/index.d.ts +12 -5
- package/dist/benchmarks/index.js +11 -10
- package/dist/builder-eval/index.d.ts +4 -4
- package/dist/builder-eval/index.js +1 -1
- package/dist/{calibration-Dz8TQV4y.d.ts → calibration-C8MTS7cw.d.ts} +2 -2
- package/dist/campaign/index.d.ts +247 -34
- package/dist/campaign/index.js +33 -13
- package/dist/chunk-3YYRZDON.js +45 -0
- package/dist/chunk-3YYRZDON.js.map +1 -0
- package/dist/{chunk-RPDDVKI7.js → chunk-4JLWXDYA.js} +2 -2
- package/dist/{chunk-WSBUZMBU.js → chunk-CCZIVI3F.js} +54 -115
- package/dist/chunk-CCZIVI3F.js.map +1 -0
- package/dist/{chunk-J6P6PK2R.js → chunk-FQNLDL4D.js} +3 -3
- package/dist/{chunk-ONM6PEAE.js → chunk-GQCZRZ7L.js} +2 -2
- package/dist/chunk-HHWE3POT.js +94 -0
- package/dist/chunk-HHWE3POT.js.map +1 -0
- package/dist/{chunk-ADYLPOSX.js → chunk-HQPHZGL6.js} +1112 -135
- package/dist/chunk-HQPHZGL6.js.map +1 -0
- package/dist/{chunk-FAOEFFRT.js → chunk-IDZTTFRR.js} +390 -78
- package/dist/chunk-IDZTTFRR.js.map +1 -0
- package/dist/{chunk-3LXTCTWL.js → chunk-JSDVRFAP.js} +2 -2
- package/dist/{chunk-MHNQWM4I.js → chunk-LQUTGLOZ.js} +5 -1
- package/dist/chunk-LQUTGLOZ.js.map +1 -0
- package/dist/{chunk-4D5RVB3W.js → chunk-LTVG32KX.js} +30 -5
- package/dist/chunk-LTVG32KX.js.map +1 -0
- package/dist/{chunk-5S5NJ63F.js → chunk-MGEHEHSN.js} +807 -15
- package/dist/chunk-MGEHEHSN.js.map +1 -0
- package/dist/{chunk-GY4SYVPJ.js → chunk-NJC7U437.js} +97 -25
- package/dist/chunk-NJC7U437.js.map +1 -0
- package/dist/{chunk-NYFUT3B3.js → chunk-ODVOOEWQ.js} +31 -10
- package/dist/chunk-ODVOOEWQ.js.map +1 -0
- package/dist/{chunk-LNQEP766.js → chunk-S2F4J57L.js} +44 -4
- package/dist/chunk-S2F4J57L.js.map +1 -0
- package/dist/chunk-VCTY3W6J.js +798 -0
- package/dist/chunk-VCTY3W6J.js.map +1 -0
- package/dist/chunk-VF3XSYTI.js +545 -0
- package/dist/chunk-VF3XSYTI.js.map +1 -0
- package/dist/{chunk-TLDB7WRY.js → chunk-YZPO4UHR.js} +28 -31
- package/dist/chunk-YZPO4UHR.js.map +1 -0
- package/dist/{chunk-KG4TD7EQ.js → chunk-ZUXV7UWZ.js} +1425 -697
- package/dist/chunk-ZUXV7UWZ.js.map +1 -0
- package/dist/cli.js +4 -2
- package/dist/cli.js.map +1 -1
- package/dist/{code-agent-session-D-g04tcy.d.ts → code-agent-session-CjZsVd19.d.ts} +1 -1
- package/dist/contract/index.d.ts +45 -31
- package/dist/contract/index.js +58 -19
- package/dist/contract/index.js.map +1 -1
- package/dist/{control-CcBiAEnn.d.ts → control-6vuGfmDH.d.ts} +5 -5
- package/dist/control.d.ts +6 -6
- package/dist/cost-ledger-DWy3XdJc.d.ts +183 -0
- package/dist/{default-registry-DltpYR5u.d.ts → default-registry-DaK8b3fv.d.ts} +2 -1
- package/dist/{emitter-BRchAAAx.d.ts → emitter-CjD7vUwv.d.ts} +2 -2
- package/dist/{failure-cluster-C48PiReX.d.ts → failure-cluster-DOAcSJ87.d.ts} +2 -2
- package/dist/{feedback-trajectory-pDcz1lQ1.d.ts → feedback-trajectory-BUnM58xL.d.ts} +3 -3
- package/dist/fuzz.d.ts +8 -16
- package/dist/fuzz.js +72 -42
- package/dist/fuzz.js.map +1 -1
- package/dist/{gepa-dne9JDPL.d.ts → gepa-eESocoDi.d.ts} +64 -12
- package/dist/hosted/index.d.ts +14 -7
- package/dist/{index-BTEpx9He.d.ts → index-PdX4VnPA.d.ts} +3 -3
- package/dist/index.d.ts +97 -55
- package/dist/index.js +343 -244
- package/dist/index.js.map +1 -1
- package/dist/{insight-report-IwwvqZZv.d.ts → insight-report-DY4nDW9Q.d.ts} +1 -1
- package/dist/{integrity-qemeBAyx.d.ts → integrity-DqlBiLyK.d.ts} +1 -1
- package/dist/kind-factory-ClZmO25A.d.ts +171 -0
- package/dist/{llm-client-DyqEH4jH.d.ts → llm-client-qoDd18Qz.d.ts} +27 -3
- package/dist/meta-eval/index.d.ts +8 -7
- package/dist/meta-eval/index.js +1 -1
- package/dist/multishot/index.d.ts +10 -3
- package/dist/openapi.json +1 -1
- package/dist/pipelines/index.d.ts +16 -6
- package/dist/pipelines/index.js +119 -23
- package/dist/pipelines/index.js.map +1 -1
- package/dist/{kind-factory-DcNg13sZ.d.ts → policy-edit-wG9uFEFm.d.ts} +114 -167
- package/dist/{pre-registration-D8h7ZxNL.d.ts → pre-registration-BWQhJ3vz.d.ts} +23 -4
- package/dist/{provenance-Bibyg1U9.d.ts → provenance-DpjwyseI.d.ts} +28 -16
- package/dist/{query-Ck190MOd.d.ts → query-CF7PG61p.d.ts} +5 -3
- package/dist/{release-report-CCtzajxP.d.ts → release-report-C8G2i5Xi.d.ts} +2 -2
- package/dist/reporting.d.ts +10 -9
- package/dist/{researcher-Dq-EtpbE.d.ts → researcher-C8XyxQsu.d.ts} +7 -7
- package/dist/rl.d.ts +17 -12
- package/dist/rl.js +2 -2
- package/dist/{rubric-predictive-validity-DYTLjGWu.d.ts → rubric-predictive-validity-p49lLVrE.d.ts} +1 -1
- package/dist/{run-campaign-UADIM77S.js → run-campaign-IM26A6PD.js} +4 -2
- package/dist/{run-record-B7RTi_ix.d.ts → run-record-BDH49H2E.d.ts} +2 -2
- package/dist/{runtime-trajectory-Dws7Kpgi.d.ts → runtime-trajectory-DGBIUt4B.d.ts} +1 -1
- package/dist/{schema-SGWcK9wa.d.ts → schema-B3Q3l9Z_.d.ts} +2 -0
- package/dist/{semantic-concept-judge-DxJmRkyJ.d.ts → semantic-concept-judge-CXnPEJbf.d.ts} +23 -5
- package/dist/{statistics-oUbOJe-S.d.ts → statistics-KUnG73jH.d.ts} +1 -1
- package/dist/{storage-Dw_f7WMt.d.ts → storage-DrX3v_5B.d.ts} +12 -1
- package/dist/{store-BsVi7ncX.d.ts → store-DGqD0Pyo.d.ts} +1 -1
- package/dist/storyboard/index.d.ts +1 -1
- package/dist/{summary-report-BJ5aNwZ1.d.ts → summary-report-C5bKFfm-.d.ts} +2 -2
- package/dist/{test-graded-scenario-mzYBKspu.d.ts → test-graded-scenario-B0ybnPY7.d.ts} +3 -3
- package/dist/traces.d.ts +19 -10
- package/dist/traces.js +16 -4
- package/dist/{types-C5gJrOVT.d.ts → types-BSw1rOUB.d.ts} +97 -38
- package/dist/{types-C7DGg5ex.d.ts → types-BkfcQnxV.d.ts} +15 -0
- package/dist/wire/index.d.ts +28 -19
- package/dist/wire/index.js +4 -2
- package/docs/design/loop-taxonomy.md +1 -2
- package/docs/distributed-driver.md +1 -1
- package/package.json +3 -3
- package/dist/chunk-4D5RVB3W.js.map +0 -1
- package/dist/chunk-5S5NJ63F.js.map +0 -1
- package/dist/chunk-ADYLPOSX.js.map +0 -1
- package/dist/chunk-FAOEFFRT.js.map +0 -1
- package/dist/chunk-GY4SYVPJ.js.map +0 -1
- package/dist/chunk-I6LVHOV3.js +0 -205
- package/dist/chunk-I6LVHOV3.js.map +0 -1
- package/dist/chunk-KG4TD7EQ.js.map +0 -1
- package/dist/chunk-LNQEP766.js.map +0 -1
- package/dist/chunk-MHNQWM4I.js.map +0 -1
- package/dist/chunk-NYFUT3B3.js.map +0 -1
- package/dist/chunk-QMXXSNC4.js +0 -761
- package/dist/chunk-QMXXSNC4.js.map +0 -1
- package/dist/chunk-TLDB7WRY.js.map +0 -1
- package/dist/chunk-WSBUZMBU.js.map +0 -1
- package/dist/cost-ledger-DuSqlw5B.d.ts +0 -113
- package/dist/policy-edit-RLn8GWof.d.ts +0 -103
- /package/dist/{chunk-RPDDVKI7.js.map → chunk-4JLWXDYA.js.map} +0 -0
- /package/dist/{chunk-J6P6PK2R.js.map → chunk-FQNLDL4D.js.map} +0 -0
- /package/dist/{chunk-ONM6PEAE.js.map → chunk-GQCZRZ7L.js.map} +0 -0
- /package/dist/{chunk-3LXTCTWL.js.map → chunk-JSDVRFAP.js.map} +0 -0
- /package/dist/{run-campaign-UADIM77S.js.map → run-campaign-IM26A6PD.js.map} +0 -0
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import { G as GainDistributionBin, P as ParetoFigureSpec } from './summary-report-
|
|
1
|
+
import { G as GainDistributionBin, P as ParetoFigureSpec } from './summary-report-C5bKFfm-.js';
|
|
2
2
|
import { C as ContinuousAgreement } from './judge-calibration-7C-IDmKr.js';
|
|
3
3
|
|
|
4
4
|
/**
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
import { C as CaptureIntegrityError } from './errors-oeQrLqXC.js';
|
|
2
2
|
import { R as RawProviderSink } from './raw-provider-sink-C46HDghv.js';
|
|
3
|
-
import { T as TraceStore } from './store-
|
|
3
|
+
import { T as TraceStore } from './store-DGqD0Pyo.js';
|
|
4
4
|
|
|
5
5
|
/**
|
|
6
6
|
* Run-completion integrity check — at end of run, verify the expected event
|
|
@@ -0,0 +1,171 @@
|
|
|
1
|
+
import { AxAIService, AxFunction } from '@ax-llm/ax';
|
|
2
|
+
import { T as TraceAnalysisStore } from './store-C1YxJDEK.js';
|
|
3
|
+
import { z } from 'zod';
|
|
4
|
+
import { g as AnalystCost, a as AnalystContext, A as Analyst } from './policy-edit-wG9uFEFm.js';
|
|
5
|
+
|
|
6
|
+
/**
|
|
7
|
+
* Typed Ax output for analyst findings.
|
|
8
|
+
*
|
|
9
|
+
* Replaces the legacy `findings:string[]` pattern (where every bullet
|
|
10
|
+
* became a flat-severity `AnalystFinding`) with a structured object
|
|
11
|
+
* array. Ax binds the field as `findings:json[]` so the provider emits
|
|
12
|
+
* native structured output; at the kind-factory boundary we Zod-validate
|
|
13
|
+
* each emitted finding so malformed rows fail loud instead of being
|
|
14
|
+
* silently lifted with default severity.
|
|
15
|
+
*
|
|
16
|
+
* Why not `f.object().array()` directly in the signature? The Ax
|
|
17
|
+
* signature string `question:string -> findings:json[]` already lets
|
|
18
|
+
* the provider emit JSON arrays. A Zod boundary is required either
|
|
19
|
+
* way (the provider can return any JSON), and Zod gives us a single
|
|
20
|
+
* validation surface independent of which Ax version is installed.
|
|
21
|
+
*/
|
|
22
|
+
|
|
23
|
+
declare const ANALYST_SEVERITIES: readonly ["critical", "high", "medium", "low", "info"];
|
|
24
|
+
declare const RawAnalystFindingSchema: z.ZodObject<{
|
|
25
|
+
severity: z.ZodEnum<{
|
|
26
|
+
info: "info";
|
|
27
|
+
critical: "critical";
|
|
28
|
+
medium: "medium";
|
|
29
|
+
low: "low";
|
|
30
|
+
high: "high";
|
|
31
|
+
}>;
|
|
32
|
+
claim: z.ZodString;
|
|
33
|
+
subject: z.ZodOptional<z.ZodString>;
|
|
34
|
+
evidence_uri: z.ZodString;
|
|
35
|
+
evidence_excerpt: z.ZodOptional<z.ZodString>;
|
|
36
|
+
confidence: z.ZodNumber;
|
|
37
|
+
rationale: z.ZodOptional<z.ZodString>;
|
|
38
|
+
recommended_action: z.ZodOptional<z.ZodString>;
|
|
39
|
+
}, z.core.$strict>;
|
|
40
|
+
type RawAnalystFinding = z.infer<typeof RawAnalystFindingSchema>;
|
|
41
|
+
/**
|
|
42
|
+
* Description embedded into the actor prompt so the LLM knows what
|
|
43
|
+
* shape to emit. Kept here so kinds share one source of truth rather
|
|
44
|
+
* than restating the schema in every prompt.
|
|
45
|
+
*/
|
|
46
|
+
declare const RAW_FINDING_SCHEMA_PROMPT = "Each finding MUST be a JSON object with these fields:\n - severity: one of \"critical\" | \"high\" | \"medium\" | \"low\" | \"info\"\n - claim: one-sentence statement (max 2000 chars)\n - subject?: the routing locus this finding is about. It MUST be one of the exact subject forms listed in this kind's instructions above (e.g. `system-prompt:<section>`, `agent-knowledge:wiki:<slug>`, `tool-doc:<tool>`). A free phrase, a bare noun, or any form not in that list is REJECTED at parse time and the finding is discarded \u2014 omit subject entirely rather than guess a form.\n - evidence_uri: REQUIRED, never blank. Exactly one of \"span://<trace_id>/<span_id>\" (trace evidence), \"artifact://<relative-path>\" (files), \"metric://<name>\" (named scalars) \u2014 ALWAYS cite a real id surfaced by the tools. If you have no citable id, do not emit the finding.\n - evidence_excerpt?: short quote (<=2000 chars) from the cited span/artifact\n - confidence: number 0..1 \u2014 0.9+ when backed by exact quotes, 0.6-0.8 for inferred patterns, <0.5 for speculative\n - rationale?: one or two sentences explaining the reasoning\n - recommended_action?: concrete change phrased as an imperative (\"Add ...\", \"Replace ...\", \"Stop ...\") \u2014 omit when the finding is purely descriptive\n\nEmit an empty array when the question has no findings to report. Do not fabricate evidence.";
|
|
47
|
+
/**
|
|
48
|
+
* Validate one row emitted by the LLM. Returns the typed finding on
|
|
49
|
+
* success; returns `null` and logs the reason on failure so the kind
|
|
50
|
+
* factory can skip-and-count rather than abort the whole analyst run.
|
|
51
|
+
*/
|
|
52
|
+
declare function parseRawFinding(row: unknown, log?: (msg: string, fields?: Record<string, unknown>) => void): RawAnalystFinding | null;
|
|
53
|
+
|
|
54
|
+
/**
|
|
55
|
+
* Analyst-kind factory — the typed way to define trace analysts.
|
|
56
|
+
*
|
|
57
|
+
* A "kind" is a specialized analyst whose actor prompt, tool subset,
|
|
58
|
+
* and Ax recursion config target one failure-mode lens (failure-mode
|
|
59
|
+
* classification, knowledge gap discovery, knowledge poisoning, recursive
|
|
60
|
+
* self-improvement, ...). Kinds emit findings in the typed `RawAnalystFinding`
|
|
61
|
+
* shape via a JSON-array Ax output; the factory validates each row with
|
|
62
|
+
* Zod and lifts it into `AnalystFinding[]` with no shape guessing.
|
|
63
|
+
*
|
|
64
|
+
* Composition rules:
|
|
65
|
+
* - Each kind owns its actor description. No generic "answer this
|
|
66
|
+
* question" prompt — the prompt names the failure lens.
|
|
67
|
+
* - Each kind picks a narrow tool subset from `ANALYST_TOOL_GROUPS`.
|
|
68
|
+
* A kind that never needs full-trace dumps can drop `viewTrace` /
|
|
69
|
+
* `viewSpans` and stay cheap.
|
|
70
|
+
* - Each kind declares its recursion + parallelism budget. Discovery-
|
|
71
|
+
* heavy kinds (failure-mode) get higher `maxDepth`; lens kinds
|
|
72
|
+
* (poisoning) usually stay at 0 since they have a tighter brief.
|
|
73
|
+
*
|
|
74
|
+
* Optimizer hook: kinds may declare `goldens` — labeled examples used
|
|
75
|
+
* by `AxMiPRO` / `AxBootstrapFewShot` / `AxGEPA` to fit the actor
|
|
76
|
+
* description programmatically. Stored on the kind, not the registry,
|
|
77
|
+
* because the right metric is kind-specific.
|
|
78
|
+
*/
|
|
79
|
+
|
|
80
|
+
/**
|
|
81
|
+
* Per-kind specification. The factory turns this into a regular
|
|
82
|
+
* `Analyst<TraceAnalysisStore>` ready for `AnalystRegistry.register()`.
|
|
83
|
+
*/
|
|
84
|
+
interface TraceAnalystKindSpec {
|
|
85
|
+
/** Stable id. Appears in finding_id, telemetry, and registry exclusions. */
|
|
86
|
+
id: string;
|
|
87
|
+
/** One-sentence description shown in `registry.list()`. */
|
|
88
|
+
description: string;
|
|
89
|
+
/** Coarse classification stamped on every emitted finding (`failure-mode`, `knowledge-gap`, ...). */
|
|
90
|
+
area: string;
|
|
91
|
+
/** Bump on any breaking change to the actor prompt or output schema. */
|
|
92
|
+
version: string;
|
|
93
|
+
/** Actor system prompt. Must instruct the LLM to emit `findings` per the schema. */
|
|
94
|
+
actorDescription: string;
|
|
95
|
+
/** Responder system prompt; falls back to a minimal "format the findings" instruction. */
|
|
96
|
+
responderDescription?: string;
|
|
97
|
+
/** Tool functions the actor may call. Pick narrow subsets via `ANALYST_TOOL_GROUPS`. */
|
|
98
|
+
buildTools: (store: TraceAnalysisStore) => AxFunction[];
|
|
99
|
+
/** Recursion budget. `maxDepth: 0` disables subagents. */
|
|
100
|
+
recursion?: {
|
|
101
|
+
maxDepth: number;
|
|
102
|
+
maxParallelSubagents?: number;
|
|
103
|
+
};
|
|
104
|
+
/** Actor turn cap. Default 12. */
|
|
105
|
+
maxTurns?: number;
|
|
106
|
+
/** Runtime char cap. Default 6000. */
|
|
107
|
+
maxRuntimeChars?: number;
|
|
108
|
+
/** Cost classification surfaced in `registry.list()` and budget enforcement. */
|
|
109
|
+
cost: AnalystCost;
|
|
110
|
+
/** Per-finding-row hook — kinds may reject / rewrite before lifting. */
|
|
111
|
+
postProcess?: (row: RawAnalystFinding, ctx: AnalystContext) => RawAnalystFinding | null;
|
|
112
|
+
/** Optional optimizer hook — populated when a kind wants to fit its prompt against labeled examples. */
|
|
113
|
+
goldens?: TraceAnalystGolden[];
|
|
114
|
+
}
|
|
115
|
+
/**
|
|
116
|
+
* One labeled example consumed by Ax optimizers (MIPRO / GEPA / Bootstrap).
|
|
117
|
+
* Each input is the same `{question}` an analyst would receive; `expected`
|
|
118
|
+
* is the ground-truth finding set a fitted prompt should produce on this
|
|
119
|
+
* input. Metric: kind-specific (default: F1 on `finding_id` overlap).
|
|
120
|
+
*/
|
|
121
|
+
interface TraceAnalystGolden {
|
|
122
|
+
question: string;
|
|
123
|
+
expected: ReadonlyArray<Omit<RawAnalystFinding, 'confidence'>>;
|
|
124
|
+
}
|
|
125
|
+
interface CreateTraceAnalystKindOpts {
|
|
126
|
+
/** AxAIService bound at registration time. */
|
|
127
|
+
ai: AxAIService;
|
|
128
|
+
/** Optional model override; falls back to the AI service's default. */
|
|
129
|
+
model?: string;
|
|
130
|
+
/** Override the spec's `version` (e.g. when an optimizer has fitted a new prompt). */
|
|
131
|
+
versionSuffix?: string;
|
|
132
|
+
/**
|
|
133
|
+
* Optional two-phase recovery: when the agentic harvest is empty but the
|
|
134
|
+
* actor produced a substantive free-form `report`, extract findings from that
|
|
135
|
+
* prose via a tolerant chat-completions pass (`structureFindings`) — no
|
|
136
|
+
* strict-emission contract, so it works on weak models. Omit to leave the
|
|
137
|
+
* actor's harvest as-is (the report is still surfaced fail-loud either way).
|
|
138
|
+
*/
|
|
139
|
+
recovery?: {
|
|
140
|
+
baseUrl: string;
|
|
141
|
+
apiKey?: string;
|
|
142
|
+
model?: string;
|
|
143
|
+
fetchImpl?: typeof fetch;
|
|
144
|
+
};
|
|
145
|
+
}
|
|
146
|
+
/**
|
|
147
|
+
* Build an `Analyst<TraceAnalysisStore>` from a kind spec.
|
|
148
|
+
*
|
|
149
|
+
* Lifts the Ax pipeline once at registration time so the registry
|
|
150
|
+
* gets a stateless analyst. The Ax agent is freshly constructed per
|
|
151
|
+
* `analyze()` call (the agent carries chat-log + usage state we don't
|
|
152
|
+
* want shared across analyst runs).
|
|
153
|
+
*/
|
|
154
|
+
declare function createTraceAnalystKind(spec: TraceAnalystKindSpec, opts: CreateTraceAnalystKindOpts): Analyst<TraceAnalysisStore>;
|
|
155
|
+
/**
|
|
156
|
+
* Render a compact prior-findings block the actor reads alongside its
|
|
157
|
+
* brief. Each row is one line so the actor can scan dozens cheaply.
|
|
158
|
+
* The kind's prompt instructs the actor to (a) check whether a new
|
|
159
|
+
* cluster matches a prior `finding_id` (carry the id forward via
|
|
160
|
+
* `id_basis` to keep diffs stable) and (b) raise severity / confidence
|
|
161
|
+
* when a prior finding has reappeared without remediation.
|
|
162
|
+
*
|
|
163
|
+
* Returns the empty string when there are no prior findings — most
|
|
164
|
+
* runs are "first-of-its-kind" and the prompt stays unchanged.
|
|
165
|
+
*
|
|
166
|
+
* Exported for tests + for consumers that build their own actor
|
|
167
|
+
* prompts (e.g. specialized analysts living outside the default kinds).
|
|
168
|
+
*/
|
|
169
|
+
declare function renderPriorFindings(prior: AnalystContext['priorFindings']): string;
|
|
170
|
+
|
|
171
|
+
export { ANALYST_SEVERITIES as A, type CreateTraceAnalystKindOpts as C, RAW_FINDING_SCHEMA_PROMPT as R, type TraceAnalystKindSpec as T, type RawAnalystFinding as a, RawAnalystFindingSchema as b, type TraceAnalystGolden as c, createTraceAnalystKind as d, parseRawFinding as p, renderPriorFindings as r };
|
|
@@ -1,3 +1,4 @@
|
|
|
1
|
+
import { c as CostReceiptInput, M as MaximumCharge } from './cost-ledger-DWy3XdJc.js';
|
|
1
2
|
import { A as AgentEvalError, C as CaptureIntegrityError } from './errors-oeQrLqXC.js';
|
|
2
3
|
import { R as RawProviderSink, P as ProviderRedactor } from './raw-provider-sink-C46HDghv.js';
|
|
3
4
|
|
|
@@ -55,10 +56,16 @@ interface LlmCallRequest {
|
|
|
55
56
|
/** Per-call timeout, default 300s. */
|
|
56
57
|
timeoutMs?: number;
|
|
57
58
|
}
|
|
59
|
+
/** Conservative priced bound for the exact text request sent to a provider.
|
|
60
|
+
* Returns undefined when output or multimodal input is not bounded, causing a
|
|
61
|
+
* capped CostLedger to reject the call before execution. */
|
|
62
|
+
declare function maximumChargeForLlmRequest(request: Pick<LlmCallRequest, 'model' | 'messages' | 'jsonSchema' | 'maxTokens'>, options?: LlmClientOptions): MaximumCharge | undefined;
|
|
58
63
|
interface LlmUsage {
|
|
59
64
|
promptTokens: number;
|
|
60
65
|
completionTokens: number;
|
|
61
66
|
totalTokens: number;
|
|
67
|
+
/** False when the provider omitted or malformed prompt/completion usage. */
|
|
68
|
+
captured?: boolean;
|
|
62
69
|
/** Proxies populate this when prompt caching is on. */
|
|
63
70
|
cachedPromptTokens?: number;
|
|
64
71
|
}
|
|
@@ -94,12 +101,26 @@ interface LlmCallResult {
|
|
|
94
101
|
/** Raw response body. */
|
|
95
102
|
raw: Record<string, unknown>;
|
|
96
103
|
}
|
|
104
|
+
type LlmCallMetadata = Pick<LlmCallResult, 'usage' | 'costUsd' | 'model' | 'durationMs'>;
|
|
105
|
+
/** Convert a provider result into the canonical paid-call receipt input. */
|
|
106
|
+
declare function costReceiptFromLlm(result: LlmCallResult): CostReceiptInput;
|
|
107
|
+
/** Structured-response failures retain their completed provider receipt. */
|
|
108
|
+
declare function costReceiptFromLlmError(error: Error): CostReceiptInput | undefined;
|
|
97
109
|
declare class LlmCallError extends AgentEvalError {
|
|
98
110
|
readonly status: number;
|
|
99
111
|
readonly body: string;
|
|
100
112
|
readonly model: string;
|
|
101
113
|
constructor(message: string, status: number, body: string, model: string);
|
|
102
114
|
}
|
|
115
|
+
/** A provider response completed and incurred measurable usage, but its content
|
|
116
|
+
* could not satisfy the caller's response contract. The response envelope is
|
|
117
|
+
* retained so accounting can commit the receipt before the error propagates. */
|
|
118
|
+
declare class LlmResponseError extends AgentEvalError {
|
|
119
|
+
readonly result: LlmCallResult;
|
|
120
|
+
constructor(message: string, result: LlmCallResult, options?: {
|
|
121
|
+
cause?: unknown;
|
|
122
|
+
});
|
|
123
|
+
}
|
|
103
124
|
interface LlmClientOptions {
|
|
104
125
|
/** Base URL (without trailing slash). Must end at the `/v1` prefix. */
|
|
105
126
|
baseUrl?: string;
|
|
@@ -111,6 +132,8 @@ interface LlmClientOptions {
|
|
|
111
132
|
name: string;
|
|
112
133
|
value: string;
|
|
113
134
|
};
|
|
135
|
+
/** Stable provider idempotency key, reused across retries of this logical call. */
|
|
136
|
+
idempotencyKey?: string;
|
|
114
137
|
/** Default timeout in ms. Per-call can override. */
|
|
115
138
|
defaultTimeoutMs?: number;
|
|
116
139
|
/**
|
|
@@ -125,10 +148,10 @@ interface LlmClientOptions {
|
|
|
125
148
|
* Before launching each attempt the loop checks the remaining budget and
|
|
126
149
|
* stops retrying once it is exhausted, rather than waiting the full
|
|
127
150
|
* per-attempt timeout on every retry. Bounds total time independent of
|
|
128
|
-
*
|
|
151
|
+
* total attempts × `timeoutMs`.
|
|
129
152
|
*/
|
|
130
153
|
deadlineMs?: number;
|
|
131
|
-
/**
|
|
154
|
+
/** Total provider attempts. Legacy option name; default 3 (1 initial + 2 retries). */
|
|
132
155
|
maxRetries?: number;
|
|
133
156
|
/** Fetch implementation — defaults to global `fetch`. Override for custom transport (e.g. tests). */
|
|
134
157
|
fetch?: typeof fetch;
|
|
@@ -253,6 +276,7 @@ declare function probeLlm(model: string, opts?: LlmClientOptions & {
|
|
|
253
276
|
* to inject a single configured instance into multiple primitives.
|
|
254
277
|
*/
|
|
255
278
|
declare class LlmClient {
|
|
279
|
+
readonly maximumAttempts: number;
|
|
256
280
|
private readonly opts;
|
|
257
281
|
constructor(opts?: LlmClientOptions);
|
|
258
282
|
call(req: LlmCallRequest, per?: LlmClientOptions): Promise<LlmCallResult>;
|
|
@@ -262,4 +286,4 @@ declare class LlmClient {
|
|
|
262
286
|
}>;
|
|
263
287
|
}
|
|
264
288
|
|
|
265
|
-
export { type
|
|
289
|
+
export { type LlmCallMetadata as L, type LlmClientOptions as a, type LlmRouteRequirements as b, type LlmCallRequest as c, type LlmCallResult as d, type LlmUsage as e, LlmCallError as f, LlmClient as g, type LlmMessage as h, LlmResponseError as i, LlmRouteAssertionError as j, assertLlmRoute as k, backoffMs as l, callLlm as m, callLlmJson as n, costReceiptFromLlm as o, costReceiptFromLlmError as p, isTransientLlmError as q, maximumChargeForLlmRequest as r, probeLlm as s, stripFencedJson as t };
|
|
@@ -1,15 +1,16 @@
|
|
|
1
|
-
export { C as CalibrationBin, a as CalibrationOptions, b as CalibrationPair, c as CalibrationReport, d as CorrelationResult, e as CorrelationStudyOptions, f as CorrelationStudyResult, E as EvalMetricSpec, O as OutcomePair, g as calibrationCurve, h as calibrationFromPairs, i as correlationStudy } from '../calibration-
|
|
1
|
+
export { C as CalibrationBin, a as CalibrationOptions, b as CalibrationPair, c as CalibrationReport, d as CorrelationResult, e as CorrelationStudyOptions, f as CorrelationStudyResult, E as EvalMetricSpec, O as OutcomePair, g as calibrationCurve, h as calibrationFromPairs, i as correlationStudy } from '../calibration-C8MTS7cw.js';
|
|
2
2
|
export { D as DeploymentOutcome, F as FileSystemOutcomeStore, a as FileSystemOutcomeStoreOptions, I as InMemoryOutcomeStore, O as OutcomeFilter, b as OutcomeStore } from '../outcome-store-rnXLEqSn.js';
|
|
3
|
-
export { R as RubricOutcomePair, a as RubricPredictiveValidityInput, b as RubricPredictiveValidityReport, c as RubricRanking, r as rubricPredictiveValidity } from '../rubric-predictive-validity-
|
|
3
|
+
export { R as RubricOutcomePair, a as RubricPredictiveValidityInput, b as RubricPredictiveValidityReport, c as RubricRanking, r as rubricPredictiveValidity } from '../rubric-predictive-validity-p49lLVrE.js';
|
|
4
4
|
import { C as ContinuousAgreement, a as CalibrationResult, b as ContinuousCalibrationResult, c as CandidateScore, G as GoldenItem } from '../judge-calibration-7C-IDmKr.js';
|
|
5
5
|
import { S as SeriesConvergenceOptions, a as SeriesConvergenceResult } from '../series-convergence-D5OWMBg6.js';
|
|
6
|
-
import { C as CorpusAgreementReport } from '../statistics-
|
|
7
|
-
import '../store-
|
|
8
|
-
import '../schema-
|
|
9
|
-
import '../run-record-
|
|
6
|
+
import { C as CorpusAgreementReport } from '../statistics-KUnG73jH.js';
|
|
7
|
+
import '../store-DGqD0Pyo.js';
|
|
8
|
+
import '../schema-B3Q3l9Z_.js';
|
|
9
|
+
import '../run-record-BDH49H2E.js';
|
|
10
10
|
import '@tangle-network/agent-interface';
|
|
11
11
|
import '../errors-oeQrLqXC.js';
|
|
12
|
-
import '../types-
|
|
12
|
+
import '../types-BkfcQnxV.js';
|
|
13
|
+
import '../cost-ledger-DWy3XdJc.js';
|
|
13
14
|
import '@tangle-network/tcloud';
|
|
14
15
|
|
|
15
16
|
/**
|
package/dist/meta-eval/index.js
CHANGED
|
@@ -1,9 +1,16 @@
|
|
|
1
|
-
import { J as JudgeScore } from '../types-
|
|
1
|
+
import { J as JudgeScore } from '../types-BSw1rOUB.js';
|
|
2
2
|
import { AgentProfile } from '@tangle-network/agent-interface';
|
|
3
3
|
import { M as MatrixResult } from '../types-BUxNaJ8c.js';
|
|
4
|
-
import '../
|
|
4
|
+
import '../policy-edit-wG9uFEFm.js';
|
|
5
|
+
import '../run-record-BDH49H2E.js';
|
|
5
6
|
import '../errors-oeQrLqXC.js';
|
|
6
|
-
import '../schema-
|
|
7
|
+
import '../schema-B3Q3l9Z_.js';
|
|
8
|
+
import '../store-C1YxJDEK.js';
|
|
9
|
+
import '../types-BkfcQnxV.js';
|
|
10
|
+
import '../cost-ledger-DWy3XdJc.js';
|
|
11
|
+
import '@tangle-network/tcloud';
|
|
12
|
+
import '../llm-client-qoDd18Qz.js';
|
|
13
|
+
import '../raw-provider-sink-C46HDghv.js';
|
|
7
14
|
import '../verdict-C9MlYujm.js';
|
|
8
15
|
|
|
9
16
|
interface MultishotMessage {
|
package/dist/openapi.json
CHANGED
|
@@ -2,7 +2,7 @@
|
|
|
2
2
|
"openapi": "3.1.0",
|
|
3
3
|
"info": {
|
|
4
4
|
"title": "@tangle-network/agent-eval — wire protocol",
|
|
5
|
-
"version": "0.
|
|
5
|
+
"version": "0.117.0",
|
|
6
6
|
"description": "HTTP and stdio RPC interface to agent-eval. The TypeScript runtime is the source of truth; this spec is the contract that cross-language clients (Python, Rust, Go) generate from.\n\nWire-protocol version: 1.0.0. Bumps on breaking changes to request/response schemas.",
|
|
7
7
|
"contact": {
|
|
8
8
|
"name": "Tangle Network",
|
|
@@ -1,9 +1,9 @@
|
|
|
1
|
-
import { h as BudgetSpec, R as Run, T as ToolSpan } from '../schema-
|
|
2
|
-
import { T as TraceStore, R as RunFilter } from '../store-
|
|
3
|
-
export { a as FailureCluster, F as FailureClusterReport, f as failureClusterView } from '../failure-cluster-
|
|
4
|
-
import { T as TrajectoryStep, B as BaselineOptions, a as BaselineReport } from '../baseline-
|
|
5
|
-
export { c as computeToolUseMetrics } from '../baseline-
|
|
6
|
-
import { l as llmSpans } from '../query-
|
|
1
|
+
import { h as BudgetSpec, R as Run, T as ToolSpan } from '../schema-B3Q3l9Z_.js';
|
|
2
|
+
import { T as TraceStore, R as RunFilter } from '../store-DGqD0Pyo.js';
|
|
3
|
+
export { a as FailureCluster, F as FailureClusterReport, f as failureClusterView } from '../failure-cluster-DOAcSJ87.js';
|
|
4
|
+
import { T as TrajectoryStep, B as BaselineOptions, a as BaselineReport } from '../baseline-DKq3gJpP.js';
|
|
5
|
+
export { c as computeToolUseMetrics } from '../baseline-DKq3gJpP.js';
|
|
6
|
+
import { l as llmSpans } from '../query-CF7PG61p.js';
|
|
7
7
|
|
|
8
8
|
/**
|
|
9
9
|
* BudgetBreachView — aggregates breach events across the corpus.
|
|
@@ -121,8 +121,11 @@ interface StuckLoopFinding {
|
|
|
121
121
|
runId: string;
|
|
122
122
|
toolName: string;
|
|
123
123
|
argHash: string;
|
|
124
|
+
/** Calls in this episode's densest qualifying interval, not the whole-run total. */
|
|
124
125
|
occurrences: number;
|
|
125
126
|
spanIds: string[];
|
|
127
|
+
/** Nearest agent ancestor, or the direct parent when ancestry is incomplete. */
|
|
128
|
+
scopeSpanId?: string;
|
|
126
129
|
/** Milliseconds between first and last call in the loop. */
|
|
127
130
|
windowMs: number;
|
|
128
131
|
}
|
|
@@ -134,6 +137,13 @@ interface StuckLoopReport {
|
|
|
134
137
|
interface StuckLoopOptions {
|
|
135
138
|
/** Minimum call count to flag a loop (default 3). */
|
|
136
139
|
minOccurrences?: number;
|
|
140
|
+
/** Maximum time between the first and last repeated call (default 60 seconds). */
|
|
141
|
+
maxWindowMs?: number;
|
|
142
|
+
/**
|
|
143
|
+
* Maximum other tool calls allowed between adjacent repeats (default 0).
|
|
144
|
+
* Set to 1 to detect alternating patterns such as A,B,A,B,A.
|
|
145
|
+
*/
|
|
146
|
+
maxInterveningToolCalls?: number;
|
|
137
147
|
/** Filter to a specific runId; omit to scan the entire corpus. */
|
|
138
148
|
runId?: string;
|
|
139
149
|
}
|
package/dist/pipelines/index.js
CHANGED
|
@@ -4,7 +4,10 @@ import {
|
|
|
4
4
|
classifyFailure,
|
|
5
5
|
compareToBaseline,
|
|
6
6
|
computeToolUseMetrics
|
|
7
|
-
} from "../chunk-
|
|
7
|
+
} from "../chunk-ODVOOEWQ.js";
|
|
8
|
+
import {
|
|
9
|
+
executionTrackByLane
|
|
10
|
+
} from "../chunk-HHWE3POT.js";
|
|
8
11
|
import {
|
|
9
12
|
interRaterReliability,
|
|
10
13
|
pearsonR
|
|
@@ -12,10 +15,12 @@ import {
|
|
|
12
15
|
import {
|
|
13
16
|
aggregateLlm,
|
|
14
17
|
argHash,
|
|
18
|
+
hasCapturedToolArgs,
|
|
19
|
+
isToolSpan,
|
|
15
20
|
llmSpans,
|
|
16
21
|
runFailureClass,
|
|
17
22
|
toolSpans
|
|
18
|
-
} from "../chunk-
|
|
23
|
+
} from "../chunk-LQUTGLOZ.js";
|
|
19
24
|
import "../chunk-ONWEPEDO.js";
|
|
20
25
|
import "../chunk-PZ5AY32C.js";
|
|
21
26
|
|
|
@@ -80,7 +85,7 @@ async function failureClusterView(store, options = {}) {
|
|
|
80
85
|
const trig = spans.find((s) => s.spanId === cls.triggerSpanId);
|
|
81
86
|
if (trig?.kind === "tool") {
|
|
82
87
|
toolName = trig.toolName;
|
|
83
|
-
argPrefix = argHash(trig.args).slice(0, 16);
|
|
88
|
+
if (hasCapturedToolArgs(trig)) argPrefix = argHash(trig.args).slice(0, 16);
|
|
84
89
|
} else if (trig?.kind === "judge") {
|
|
85
90
|
dimension = trig.dimension;
|
|
86
91
|
}
|
|
@@ -90,7 +95,7 @@ async function failureClusterView(store, options = {}) {
|
|
|
90
95
|
const errored = ts.filter((t) => t.status === "error").pop();
|
|
91
96
|
if (errored) {
|
|
92
97
|
toolName = errored.toolName;
|
|
93
|
-
argPrefix = argHash(errored.args).slice(0, 16);
|
|
98
|
+
if (hasCapturedToolArgs(errored)) argPrefix = argHash(errored.args).slice(0, 16);
|
|
94
99
|
}
|
|
95
100
|
}
|
|
96
101
|
if (!dimension) {
|
|
@@ -296,33 +301,97 @@ function defaultExtract(metric) {
|
|
|
296
301
|
}
|
|
297
302
|
|
|
298
303
|
// src/pipelines/stuck-loop.ts
|
|
304
|
+
var DEFAULT_MAX_WINDOW_MS = 6e4;
|
|
305
|
+
var DEFAULT_MAX_INTERVENING_TOOL_CALLS = 0;
|
|
299
306
|
async function stuckLoopView(store, options = {}) {
|
|
300
307
|
const minOccurrences = options.minOccurrences ?? 3;
|
|
308
|
+
const maxWindowMs = options.maxWindowMs ?? DEFAULT_MAX_WINDOW_MS;
|
|
309
|
+
const maxInterveningToolCalls = options.maxInterveningToolCalls ?? DEFAULT_MAX_INTERVENING_TOOL_CALLS;
|
|
310
|
+
if (!Number.isInteger(minOccurrences) || minOccurrences < 1) {
|
|
311
|
+
throw new RangeError("minOccurrences must be a positive integer");
|
|
312
|
+
}
|
|
313
|
+
if (!Number.isFinite(maxWindowMs) || maxWindowMs < 0) {
|
|
314
|
+
throw new RangeError("maxWindowMs must be a finite non-negative number");
|
|
315
|
+
}
|
|
316
|
+
if (!Number.isInteger(maxInterveningToolCalls) || maxInterveningToolCalls < 0) {
|
|
317
|
+
throw new RangeError("maxInterveningToolCalls must be a non-negative integer");
|
|
318
|
+
}
|
|
301
319
|
const runs = options.runId ? [{ runId: options.runId }] : (await store.listRuns()).map((r) => ({ runId: r.runId }));
|
|
302
320
|
const findings = [];
|
|
303
321
|
for (const { runId } of runs) {
|
|
304
|
-
const
|
|
322
|
+
const spans = await store.spans({ runId });
|
|
323
|
+
const spansById = new Map(spans.map((span) => [span.spanId, span]));
|
|
324
|
+
const scopedTools = spans.filter(isToolSpan).map((span, sourceIndex) => ({ span, sourceIndex })).sort((a, b) => a.span.startedAt - b.span.startedAt || a.sourceIndex - b.sourceIndex).map(({ span }) => ({ span, ...executionScope(span, spansById) }));
|
|
325
|
+
const trackByLane = executionTrackByLane(
|
|
326
|
+
scopedTools.map((call) => {
|
|
327
|
+
const direct = call.laneSpanId === null || call.laneSpanId === call.scopeSpanId;
|
|
328
|
+
const timed = direct ? call.span : call.laneSpanId ? spansById.get(call.laneSpanId) : void 0;
|
|
329
|
+
return {
|
|
330
|
+
key: executionKey(call),
|
|
331
|
+
scopeKey: JSON.stringify(call.scopeSpanId),
|
|
332
|
+
start: timed?.startedAt ?? null,
|
|
333
|
+
end: timed?.endedAt ?? null
|
|
334
|
+
};
|
|
335
|
+
})
|
|
336
|
+
);
|
|
337
|
+
const nextToolIndexByTrack = /* @__PURE__ */ new Map();
|
|
338
|
+
const orderedTools = scopedTools.map((call) => {
|
|
339
|
+
const trackId = trackByLane.get(executionKey(call));
|
|
340
|
+
const toolCallIndex = nextToolIndexByTrack.get(trackId) ?? 0;
|
|
341
|
+
nextToolIndexByTrack.set(trackId, toolCallIndex + 1);
|
|
342
|
+
return { ...call, toolCallIndex, trackId };
|
|
343
|
+
});
|
|
305
344
|
const byKey = /* @__PURE__ */ new Map();
|
|
306
|
-
for (const
|
|
307
|
-
|
|
308
|
-
const
|
|
309
|
-
const
|
|
310
|
-
bucket.
|
|
345
|
+
for (const call of orderedTools) {
|
|
346
|
+
if (!hasCapturedToolArgs(call.span)) continue;
|
|
347
|
+
const h = argHash(call.span.args);
|
|
348
|
+
const key = JSON.stringify([call.trackId, call.span.toolName, h]);
|
|
349
|
+
const bucket = byKey.get(key) ?? {
|
|
350
|
+
calls: [],
|
|
351
|
+
argHash: h,
|
|
352
|
+
toolName: call.span.toolName,
|
|
353
|
+
scopeSpanId: call.scopeSpanId
|
|
354
|
+
};
|
|
355
|
+
bucket.calls.push(call);
|
|
311
356
|
byKey.set(key, bucket);
|
|
312
357
|
}
|
|
313
|
-
for (const {
|
|
314
|
-
if (
|
|
315
|
-
|
|
316
|
-
|
|
317
|
-
|
|
318
|
-
|
|
319
|
-
|
|
320
|
-
|
|
321
|
-
|
|
322
|
-
|
|
323
|
-
|
|
324
|
-
|
|
325
|
-
|
|
358
|
+
for (const { calls, argHash: h, toolName, scopeSpanId } of byKey.values()) {
|
|
359
|
+
if (calls.length < minOccurrences) continue;
|
|
360
|
+
let episodeStart = 0;
|
|
361
|
+
for (let episodeEnd = 1; episodeEnd <= calls.length; episodeEnd += 1) {
|
|
362
|
+
const previous = calls[episodeEnd - 1];
|
|
363
|
+
const next = calls[episodeEnd];
|
|
364
|
+
const episodeEnded = next === void 0 || next.span.startedAt - previous.span.startedAt > maxWindowMs || next.toolCallIndex - previous.toolCallIndex - 1 > maxInterveningToolCalls || !callsAreSerial(previous, next);
|
|
365
|
+
if (!episodeEnded) continue;
|
|
366
|
+
const episode = calls.slice(episodeStart, episodeEnd);
|
|
367
|
+
let left = 0;
|
|
368
|
+
let bestStart = 0;
|
|
369
|
+
let bestEnd = -1;
|
|
370
|
+
for (let right = 0; right < episode.length; right += 1) {
|
|
371
|
+
while (episode[right].span.startedAt - episode[left].span.startedAt > maxWindowMs) {
|
|
372
|
+
left += 1;
|
|
373
|
+
}
|
|
374
|
+
if (right - left > bestEnd - bestStart) {
|
|
375
|
+
bestStart = left;
|
|
376
|
+
bestEnd = right;
|
|
377
|
+
}
|
|
378
|
+
}
|
|
379
|
+
if (bestEnd - bestStart + 1 >= minOccurrences) {
|
|
380
|
+
const loop = episode.slice(bestStart, bestEnd + 1);
|
|
381
|
+
const first = loop[0].span.startedAt;
|
|
382
|
+
const last = loop[loop.length - 1].span.startedAt;
|
|
383
|
+
findings.push({
|
|
384
|
+
runId,
|
|
385
|
+
toolName,
|
|
386
|
+
argHash: h,
|
|
387
|
+
occurrences: loop.length,
|
|
388
|
+
spanIds: loop.map((call) => call.span.spanId),
|
|
389
|
+
...scopeSpanId ? { scopeSpanId } : {},
|
|
390
|
+
windowMs: last - first
|
|
391
|
+
});
|
|
392
|
+
}
|
|
393
|
+
episodeStart = episodeEnd;
|
|
394
|
+
}
|
|
326
395
|
}
|
|
327
396
|
}
|
|
328
397
|
const affectedRuns = new Set(findings.map((f) => f.runId));
|
|
@@ -332,6 +401,33 @@ async function stuckLoopView(store, options = {}) {
|
|
|
332
401
|
totalRuns: runs.length
|
|
333
402
|
};
|
|
334
403
|
}
|
|
404
|
+
function laneKey(call) {
|
|
405
|
+
return JSON.stringify([call.scopeSpanId, call.laneSpanId]);
|
|
406
|
+
}
|
|
407
|
+
function executionKey(call) {
|
|
408
|
+
return call.laneSpanId === null || call.laneSpanId === call.scopeSpanId ? JSON.stringify([call.scopeSpanId, call.span.spanId]) : laneKey(call);
|
|
409
|
+
}
|
|
410
|
+
function executionScope(span, spansById) {
|
|
411
|
+
const directParent = span.parentSpanId;
|
|
412
|
+
if (!directParent) return { scopeSpanId: null, laneSpanId: null };
|
|
413
|
+
let currentId = directParent;
|
|
414
|
+
let laneSpanId = null;
|
|
415
|
+
const seen = /* @__PURE__ */ new Set();
|
|
416
|
+
while (currentId && !seen.has(currentId)) {
|
|
417
|
+
seen.add(currentId);
|
|
418
|
+
const current = spansById.get(currentId);
|
|
419
|
+
if (!current) return { scopeSpanId: directParent, laneSpanId: directParent };
|
|
420
|
+
if (current.kind === "agent") {
|
|
421
|
+
return { scopeSpanId: current.spanId, laneSpanId: laneSpanId ?? current.spanId };
|
|
422
|
+
}
|
|
423
|
+
laneSpanId = current.spanId;
|
|
424
|
+
currentId = current.parentSpanId;
|
|
425
|
+
}
|
|
426
|
+
return { scopeSpanId: directParent, laneSpanId: directParent };
|
|
427
|
+
}
|
|
428
|
+
function callsAreSerial(previous, next) {
|
|
429
|
+
return previous.span.endedAt !== void 0 && previous.span.endedAt <= next.span.startedAt;
|
|
430
|
+
}
|
|
335
431
|
|
|
336
432
|
// src/pipelines/tool-waste.ts
|
|
337
433
|
async function toolWasteView(store, options = {}) {
|