@tangle-network/agent-eval 0.116.0 → 0.117.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +31 -0
- package/dist/analyst/index.d.ts +18 -11
- package/dist/analyst/index.js +10 -7
- package/dist/analyst/index.js.map +1 -1
- package/dist/{analyst-CFBc14Wc.d.ts → analyst-C8HHvfJp.d.ts} +1 -1
- package/dist/{analyze-runs-0rz_m29H.d.ts → analyze-runs--2x39HZ7.d.ts} +3 -3
- package/dist/{baseline-DsNteOgR.d.ts → baseline-DKq3gJpP.d.ts} +6 -3
- package/dist/belief-state/index.d.ts +6 -6
- package/dist/belief-state/index.js +1 -1
- package/dist/benchmarks/index.d.ts +11 -8
- package/dist/benchmarks/index.js +11 -10
- package/dist/builder-eval/index.d.ts +4 -4
- package/dist/builder-eval/index.js +1 -1
- package/dist/{calibration-Dz8TQV4y.d.ts → calibration-C8MTS7cw.d.ts} +2 -2
- package/dist/campaign/index.d.ts +53 -29
- package/dist/campaign/index.js +18 -13
- package/dist/chunk-3YYRZDON.js +45 -0
- package/dist/chunk-3YYRZDON.js.map +1 -0
- package/dist/{chunk-RPDDVKI7.js → chunk-4JLWXDYA.js} +2 -2
- package/dist/{chunk-NBSS5NDZ.js → chunk-CCZIVI3F.js} +54 -115
- package/dist/chunk-CCZIVI3F.js.map +1 -0
- package/dist/{chunk-J6P6PK2R.js → chunk-FQNLDL4D.js} +3 -3
- package/dist/{chunk-ONM6PEAE.js → chunk-GQCZRZ7L.js} +2 -2
- package/dist/chunk-HHWE3POT.js +94 -0
- package/dist/chunk-HHWE3POT.js.map +1 -0
- package/dist/{chunk-3274WNK7.js → chunk-HQPHZGL6.js} +687 -44
- package/dist/chunk-HQPHZGL6.js.map +1 -0
- package/dist/{chunk-FAOEFFRT.js → chunk-IDZTTFRR.js} +390 -78
- package/dist/chunk-IDZTTFRR.js.map +1 -0
- package/dist/{chunk-3LXTCTWL.js → chunk-JSDVRFAP.js} +2 -2
- package/dist/{chunk-MHNQWM4I.js → chunk-LQUTGLOZ.js} +5 -1
- package/dist/chunk-LQUTGLOZ.js.map +1 -0
- package/dist/{chunk-4D5RVB3W.js → chunk-LTVG32KX.js} +30 -5
- package/dist/chunk-LTVG32KX.js.map +1 -0
- package/dist/{chunk-CIUOICJT.js → chunk-MGEHEHSN.js} +62 -15
- package/dist/chunk-MGEHEHSN.js.map +1 -0
- package/dist/{chunk-GY4SYVPJ.js → chunk-NJC7U437.js} +97 -25
- package/dist/chunk-NJC7U437.js.map +1 -0
- package/dist/{chunk-NYFUT3B3.js → chunk-ODVOOEWQ.js} +31 -10
- package/dist/chunk-ODVOOEWQ.js.map +1 -0
- package/dist/{chunk-LNQEP766.js → chunk-S2F4J57L.js} +44 -4
- package/dist/chunk-S2F4J57L.js.map +1 -0
- package/dist/chunk-VCTY3W6J.js +798 -0
- package/dist/chunk-VCTY3W6J.js.map +1 -0
- package/dist/chunk-VF3XSYTI.js +545 -0
- package/dist/chunk-VF3XSYTI.js.map +1 -0
- package/dist/{chunk-TLDB7WRY.js → chunk-YZPO4UHR.js} +28 -31
- package/dist/chunk-YZPO4UHR.js.map +1 -0
- package/dist/{chunk-GSW3OBHK.js → chunk-ZUXV7UWZ.js} +350 -724
- package/dist/chunk-ZUXV7UWZ.js.map +1 -0
- package/dist/cli.js +4 -2
- package/dist/cli.js.map +1 -1
- package/dist/{code-agent-session-CdxteG0y.d.ts → code-agent-session-CjZsVd19.d.ts} +1 -1
- package/dist/contract/index.d.ts +43 -29
- package/dist/contract/index.js +56 -19
- package/dist/contract/index.js.map +1 -1
- package/dist/{control-DbcDxouY.d.ts → control-6vuGfmDH.d.ts} +5 -5
- package/dist/control.d.ts +6 -6
- package/dist/cost-ledger-DWy3XdJc.d.ts +183 -0
- package/dist/{default-registry-DDfv22MQ.d.ts → default-registry-DaK8b3fv.d.ts} +2 -2
- package/dist/{emitter-BRchAAAx.d.ts → emitter-CjD7vUwv.d.ts} +2 -2
- package/dist/{failure-cluster-C48PiReX.d.ts → failure-cluster-DOAcSJ87.d.ts} +2 -2
- package/dist/{feedback-trajectory-pDcz1lQ1.d.ts → feedback-trajectory-BUnM58xL.d.ts} +3 -3
- package/dist/fuzz.d.ts +8 -16
- package/dist/fuzz.js +72 -42
- package/dist/fuzz.js.map +1 -1
- package/dist/{gepa-CQelRtuC.d.ts → gepa-eESocoDi.d.ts} +56 -6
- package/dist/hosted/index.d.ts +13 -10
- package/dist/{index-DbCXJfZ1.d.ts → index-PdX4VnPA.d.ts} +3 -3
- package/dist/index.d.ts +102 -57
- package/dist/index.js +328 -235
- package/dist/index.js.map +1 -1
- package/dist/{insight-report-oMVxDTxl.d.ts → insight-report-DY4nDW9Q.d.ts} +1 -1
- package/dist/{integrity-C6PZ73iC.d.ts → integrity-DqlBiLyK.d.ts} +2 -2
- package/dist/{kind-factory-DWOvXjR_.d.ts → kind-factory-ClZmO25A.d.ts} +2 -2
- package/dist/llm-client-qoDd18Qz.d.ts +289 -0
- package/dist/meta-eval/index.d.ts +8 -7
- package/dist/meta-eval/index.js +1 -1
- package/dist/multishot/index.d.ts +9 -6
- package/dist/openapi.json +1 -1
- package/dist/pipelines/index.d.ts +16 -6
- package/dist/pipelines/index.js +119 -23
- package/dist/pipelines/index.js.map +1 -1
- package/dist/{policy-edit-Clb2v6Oa.d.ts → policy-edit-wG9uFEFm.d.ts} +13 -266
- package/dist/{pre-registration--vU0mMtD.d.ts → pre-registration-BWQhJ3vz.d.ts} +24 -5
- package/dist/{provenance-BbVagC68.d.ts → provenance-DpjwyseI.d.ts} +6 -6
- package/dist/{query-Ck190MOd.d.ts → query-CF7PG61p.d.ts} +5 -3
- package/dist/raw-provider-sink-C46HDghv.d.ts +132 -0
- package/dist/{release-report-CamNDe90.d.ts → release-report-C8G2i5Xi.d.ts} +2 -2
- package/dist/reporting.d.ts +10 -9
- package/dist/{researcher-Dwbo_Fxx.d.ts → researcher-C8XyxQsu.d.ts} +8 -8
- package/dist/rl.d.ts +18 -15
- package/dist/rl.js +2 -2
- package/dist/{rubric-predictive-validity-BIdf9h4R.d.ts → rubric-predictive-validity-p49lLVrE.d.ts} +1 -1
- package/dist/{run-campaign-UADIM77S.js → run-campaign-IM26A6PD.js} +4 -2
- package/dist/{run-record-CZmcpWPo.d.ts → run-record-BDH49H2E.d.ts} +1 -1
- package/dist/{runtime-trajectory-CC0jx9ql.d.ts → runtime-trajectory-DGBIUt4B.d.ts} +1 -1
- package/dist/{schema-SGWcK9wa.d.ts → schema-B3Q3l9Z_.d.ts} +2 -0
- package/dist/{semantic-concept-judge-CKjePUMh.d.ts → semantic-concept-judge-CXnPEJbf.d.ts} +24 -6
- package/dist/{statistics-oUbOJe-S.d.ts → statistics-KUnG73jH.d.ts} +1 -1
- package/dist/{storage-Dw_f7WMt.d.ts → storage-DrX3v_5B.d.ts} +12 -1
- package/dist/{store-9cAScOcb.d.ts → store-C1YxJDEK.d.ts} +1 -132
- package/dist/{store-BsVi7ncX.d.ts → store-DGqD0Pyo.d.ts} +1 -1
- package/dist/storyboard/index.d.ts +1 -1
- package/dist/{summary-report-DTNgQycC.d.ts → summary-report-C5bKFfm-.d.ts} +2 -2
- package/dist/{test-graded-scenario-mzYBKspu.d.ts → test-graded-scenario-B0ybnPY7.d.ts} +3 -3
- package/dist/traces.d.ts +25 -14
- package/dist/traces.js +16 -4
- package/dist/{types-Ca_63YSD.d.ts → types-BSw1rOUB.d.ts} +41 -39
- package/dist/{types-C7DGg5ex.d.ts → types-BkfcQnxV.d.ts} +15 -0
- package/dist/wire/index.d.ts +28 -19
- package/dist/wire/index.js +4 -2
- package/docs/distributed-driver.md +1 -1
- package/package.json +3 -3
- package/dist/chunk-3274WNK7.js.map +0 -1
- package/dist/chunk-4D5RVB3W.js.map +0 -1
- package/dist/chunk-7GKEAIAD.js +0 -205
- package/dist/chunk-7GKEAIAD.js.map +0 -1
- package/dist/chunk-CIUOICJT.js.map +0 -1
- package/dist/chunk-FAOEFFRT.js.map +0 -1
- package/dist/chunk-GSW3OBHK.js.map +0 -1
- package/dist/chunk-GY4SYVPJ.js.map +0 -1
- package/dist/chunk-LNQEP766.js.map +0 -1
- package/dist/chunk-MHNQWM4I.js.map +0 -1
- package/dist/chunk-MPHTT5HE.js +0 -74
- package/dist/chunk-MPHTT5HE.js.map +0 -1
- package/dist/chunk-NBSS5NDZ.js.map +0 -1
- package/dist/chunk-NYFUT3B3.js.map +0 -1
- package/dist/chunk-TLDB7WRY.js.map +0 -1
- package/dist/cost-ledger-DuSqlw5B.d.ts +0 -113
- /package/dist/{chunk-RPDDVKI7.js.map → chunk-4JLWXDYA.js.map} +0 -0
- /package/dist/{chunk-J6P6PK2R.js.map → chunk-FQNLDL4D.js.map} +0 -0
- /package/dist/{chunk-ONM6PEAE.js.map → chunk-GQCZRZ7L.js.map} +0 -0
- /package/dist/{chunk-3LXTCTWL.js.map → chunk-JSDVRFAP.js.map} +0 -0
- /package/dist/{run-campaign-UADIM77S.js.map → run-campaign-IM26A6PD.js.map} +0 -0
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import { G as GainDistributionBin, P as ParetoFigureSpec } from './summary-report-
|
|
1
|
+
import { G as GainDistributionBin, P as ParetoFigureSpec } from './summary-report-C5bKFfm-.js';
|
|
2
2
|
import { C as ContinuousAgreement } from './judge-calibration-7C-IDmKr.js';
|
|
3
3
|
|
|
4
4
|
/**
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
import { C as CaptureIntegrityError } from './errors-oeQrLqXC.js';
|
|
2
|
-
import { R as RawProviderSink } from './
|
|
3
|
-
import { T as TraceStore } from './store-
|
|
2
|
+
import { R as RawProviderSink } from './raw-provider-sink-C46HDghv.js';
|
|
3
|
+
import { T as TraceStore } from './store-DGqD0Pyo.js';
|
|
4
4
|
|
|
5
5
|
/**
|
|
6
6
|
* Run-completion integrity check — at end of run, verify the expected event
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
import { AxAIService, AxFunction } from '@ax-llm/ax';
|
|
2
|
-
import { T as TraceAnalysisStore } from './store-
|
|
2
|
+
import { T as TraceAnalysisStore } from './store-C1YxJDEK.js';
|
|
3
3
|
import { z } from 'zod';
|
|
4
|
-
import {
|
|
4
|
+
import { g as AnalystCost, a as AnalystContext, A as Analyst } from './policy-edit-wG9uFEFm.js';
|
|
5
5
|
|
|
6
6
|
/**
|
|
7
7
|
* Typed Ax output for analyst findings.
|
|
@@ -0,0 +1,289 @@
|
|
|
1
|
+
import { c as CostReceiptInput, M as MaximumCharge } from './cost-ledger-DWy3XdJc.js';
|
|
2
|
+
import { A as AgentEvalError, C as CaptureIntegrityError } from './errors-oeQrLqXC.js';
|
|
3
|
+
import { R as RawProviderSink, P as ProviderRedactor } from './raw-provider-sink-C46HDghv.js';
|
|
4
|
+
|
|
5
|
+
/**
|
|
6
|
+
* LLM client with graceful degrade.
|
|
7
|
+
*
|
|
8
|
+
* OpenAI-compatible `/v1/chat/completions` client with:
|
|
9
|
+
* - Exponential-backoff retry on 429 + 5xx gateway errors (502/503/504).
|
|
10
|
+
* - Retry on transient network errors (fetch failed, AbortError, ECONNRESET).
|
|
11
|
+
* - Graceful json_schema → json_object degrade on 400 with schema-reject body.
|
|
12
|
+
* - Fenced-JSON stripping (```json ... ```) for models that wrap structured output.
|
|
13
|
+
* - Configurable base URL + api key / bearer, works with LiteLLM proxies, OpenAI
|
|
14
|
+
* directly, cli-bridge subscriptions, and any router that speaks the spec.
|
|
15
|
+
*
|
|
16
|
+
* Usage:
|
|
17
|
+
* const { value, result } = await callLlmJson<MyType>(
|
|
18
|
+
* { model: 'gpt-4o', messages: [...], jsonSchema: { name: 'x', schema: {...} } },
|
|
19
|
+
* { baseUrl: 'https://router.tangle.tools/v1', apiKey: process.env.KEY },
|
|
20
|
+
* )
|
|
21
|
+
*
|
|
22
|
+
* This is THE llm-calling seam for agent-eval primitives that need structured
|
|
23
|
+
* output (semantic concept judge, reviewer directives, critic scores). Primitives
|
|
24
|
+
* that need free-form text use `callLlm` and parse output themselves.
|
|
25
|
+
*/
|
|
26
|
+
|
|
27
|
+
interface LlmMessage {
|
|
28
|
+
role: 'system' | 'user' | 'assistant';
|
|
29
|
+
/**
|
|
30
|
+
* Either a plain text content string OR a multimodal content array
|
|
31
|
+
* (text + image_url parts) for vision-capable models.
|
|
32
|
+
*/
|
|
33
|
+
content: string | Array<{
|
|
34
|
+
type: 'text';
|
|
35
|
+
text: string;
|
|
36
|
+
} | {
|
|
37
|
+
type: 'image_url';
|
|
38
|
+
image_url: {
|
|
39
|
+
url: string;
|
|
40
|
+
detail?: 'auto' | 'low' | 'high';
|
|
41
|
+
};
|
|
42
|
+
}>;
|
|
43
|
+
}
|
|
44
|
+
interface LlmCallRequest {
|
|
45
|
+
model: string;
|
|
46
|
+
messages: LlmMessage[];
|
|
47
|
+
/** Optional JSON-mode response format (response_format: json_object). */
|
|
48
|
+
jsonMode?: boolean;
|
|
49
|
+
/** Optional structured output via JSON Schema. Falls back to json_object on 400. */
|
|
50
|
+
jsonSchema?: {
|
|
51
|
+
name: string;
|
|
52
|
+
schema: Record<string, unknown>;
|
|
53
|
+
};
|
|
54
|
+
temperature?: number;
|
|
55
|
+
maxTokens?: number;
|
|
56
|
+
/** Per-call timeout, default 300s. */
|
|
57
|
+
timeoutMs?: number;
|
|
58
|
+
}
|
|
59
|
+
/** Conservative priced bound for the exact text request sent to a provider.
|
|
60
|
+
* Returns undefined when output or multimodal input is not bounded, causing a
|
|
61
|
+
* capped CostLedger to reject the call before execution. */
|
|
62
|
+
declare function maximumChargeForLlmRequest(request: Pick<LlmCallRequest, 'model' | 'messages' | 'jsonSchema' | 'maxTokens'>, options?: LlmClientOptions): MaximumCharge | undefined;
|
|
63
|
+
interface LlmUsage {
|
|
64
|
+
promptTokens: number;
|
|
65
|
+
completionTokens: number;
|
|
66
|
+
totalTokens: number;
|
|
67
|
+
/** False when the provider omitted or malformed prompt/completion usage. */
|
|
68
|
+
captured?: boolean;
|
|
69
|
+
/** Proxies populate this when prompt caching is on. */
|
|
70
|
+
cachedPromptTokens?: number;
|
|
71
|
+
}
|
|
72
|
+
interface LlmCallResult {
|
|
73
|
+
/** The text content of the first choice. Empty string if none. */
|
|
74
|
+
content: string;
|
|
75
|
+
usage: LlmUsage;
|
|
76
|
+
/**
|
|
77
|
+
* Cost in USD. Pulled from proxy's `_response_cost` field when present;
|
|
78
|
+
* `null` when neither the proxy nor the caller can derive it.
|
|
79
|
+
*/
|
|
80
|
+
costUsd: number | null;
|
|
81
|
+
/** Model name actually used (echoed from response). */
|
|
82
|
+
model: string;
|
|
83
|
+
/** Wall-clock duration of the HTTP call (last attempt, if retried). */
|
|
84
|
+
durationMs: number;
|
|
85
|
+
/**
|
|
86
|
+
* `finish_reason` echoed from the first choice (`stop`, `length`,
|
|
87
|
+
* `content_filter`, `tool_calls`, ...). `null` when the provider omits it.
|
|
88
|
+
* Exposed so a free-form `callLlm` caller CAN detect a truncated answer
|
|
89
|
+
* (`length`) instead of treating a cut-off completion as complete. Note:
|
|
90
|
+
* `callLlm` does not itself reject on it — acting on this signal is the
|
|
91
|
+
* caller's responsibility (in-repo free-form drivers do not yet enforce it).
|
|
92
|
+
*/
|
|
93
|
+
finishReason?: string | null;
|
|
94
|
+
/**
|
|
95
|
+
* True when `content.trim()` is empty. An empty completion is a silent zero
|
|
96
|
+
* for free-form `callLlm` callers; this flag is the signal a caller can
|
|
97
|
+
* inspect to fail loud rather than proceed on an empty string. `callLlm`
|
|
98
|
+
* surfaces it but does not throw on it.
|
|
99
|
+
*/
|
|
100
|
+
contentEmpty?: boolean;
|
|
101
|
+
/** Raw response body. */
|
|
102
|
+
raw: Record<string, unknown>;
|
|
103
|
+
}
|
|
104
|
+
type LlmCallMetadata = Pick<LlmCallResult, 'usage' | 'costUsd' | 'model' | 'durationMs'>;
|
|
105
|
+
/** Convert a provider result into the canonical paid-call receipt input. */
|
|
106
|
+
declare function costReceiptFromLlm(result: LlmCallResult): CostReceiptInput;
|
|
107
|
+
/** Structured-response failures retain their completed provider receipt. */
|
|
108
|
+
declare function costReceiptFromLlmError(error: Error): CostReceiptInput | undefined;
|
|
109
|
+
declare class LlmCallError extends AgentEvalError {
|
|
110
|
+
readonly status: number;
|
|
111
|
+
readonly body: string;
|
|
112
|
+
readonly model: string;
|
|
113
|
+
constructor(message: string, status: number, body: string, model: string);
|
|
114
|
+
}
|
|
115
|
+
/** A provider response completed and incurred measurable usage, but its content
|
|
116
|
+
* could not satisfy the caller's response contract. The response envelope is
|
|
117
|
+
* retained so accounting can commit the receipt before the error propagates. */
|
|
118
|
+
declare class LlmResponseError extends AgentEvalError {
|
|
119
|
+
readonly result: LlmCallResult;
|
|
120
|
+
constructor(message: string, result: LlmCallResult, options?: {
|
|
121
|
+
cause?: unknown;
|
|
122
|
+
});
|
|
123
|
+
}
|
|
124
|
+
interface LlmClientOptions {
|
|
125
|
+
/** Base URL (without trailing slash). Must end at the `/v1` prefix. */
|
|
126
|
+
baseUrl?: string;
|
|
127
|
+
/** Bearer token — either `apiKey` or `bearer` populates `Authorization: Bearer ...`. */
|
|
128
|
+
apiKey?: string;
|
|
129
|
+
bearer?: string;
|
|
130
|
+
/** Override for the `Authorization` header (e.g. `X-Auth: ...`). Takes precedence over apiKey/bearer. */
|
|
131
|
+
authHeader?: {
|
|
132
|
+
name: string;
|
|
133
|
+
value: string;
|
|
134
|
+
};
|
|
135
|
+
/** Stable provider idempotency key, reused across retries of this logical call. */
|
|
136
|
+
idempotencyKey?: string;
|
|
137
|
+
/** Default timeout in ms. Per-call can override. */
|
|
138
|
+
defaultTimeoutMs?: number;
|
|
139
|
+
/**
|
|
140
|
+
* Caller-supplied abort signal — e.g. a campaign-wide cancel. Linked to
|
|
141
|
+
* each attempt's per-attempt timeout controller, so aborting it cancels
|
|
142
|
+
* the in-flight fetch. A caller abort is FATAL: it is not retried even
|
|
143
|
+
* though an AbortError otherwise matches the transient patterns.
|
|
144
|
+
*/
|
|
145
|
+
signal?: AbortSignal;
|
|
146
|
+
/**
|
|
147
|
+
* Cross-attempt wall-clock budget in ms, measured from the first attempt.
|
|
148
|
+
* Before launching each attempt the loop checks the remaining budget and
|
|
149
|
+
* stops retrying once it is exhausted, rather than waiting the full
|
|
150
|
+
* per-attempt timeout on every retry. Bounds total time independent of
|
|
151
|
+
* total attempts × `timeoutMs`.
|
|
152
|
+
*/
|
|
153
|
+
deadlineMs?: number;
|
|
154
|
+
/** Total provider attempts. Legacy option name; default 3 (1 initial + 2 retries). */
|
|
155
|
+
maxRetries?: number;
|
|
156
|
+
/** Fetch implementation — defaults to global `fetch`. Override for custom transport (e.g. tests). */
|
|
157
|
+
fetch?: typeof fetch;
|
|
158
|
+
/**
|
|
159
|
+
* Optional raw HTTP capture sink. When provided, every request, response,
|
|
160
|
+
* and error (across all retry attempts) is recorded to the sink, with auth
|
|
161
|
+
* headers and credential-shaped body fields redacted by default. This is
|
|
162
|
+
* the layer-1 forensics primitive: structured `LlmSpan`s record intent,
|
|
163
|
+
* raw events record what actually crossed the wire.
|
|
164
|
+
*/
|
|
165
|
+
rawSink?: RawProviderSink;
|
|
166
|
+
/**
|
|
167
|
+
* Logical provider id attached to raw events. When omitted, derived from
|
|
168
|
+
* `baseUrl` via `providerFromBaseUrl`.
|
|
169
|
+
*/
|
|
170
|
+
provider?: string;
|
|
171
|
+
/** Trace context attached to raw events; populated by emitter-aware callers. */
|
|
172
|
+
traceContext?: {
|
|
173
|
+
runId?: string;
|
|
174
|
+
spanId?: string;
|
|
175
|
+
};
|
|
176
|
+
/** Override the redaction strategy for this call. Defaults to `defaultProviderRedactor`. */
|
|
177
|
+
redactor?: ProviderRedactor;
|
|
178
|
+
}
|
|
179
|
+
/**
|
|
180
|
+
* True when an error is a transient transport/network fault worth retrying,
|
|
181
|
+
* as opposed to a deterministic failure (4xx schema reject, JSON parse) that
|
|
182
|
+
* a retry cannot fix. Inspects `LlmCallError.status`, then the error's
|
|
183
|
+
* name/message/code, then recurses into `error.cause` — undici nests the
|
|
184
|
+
* real socket fault one or more levels under `.cause`.
|
|
185
|
+
*
|
|
186
|
+
* This is THE retry classifier for the package: `callLlm` and
|
|
187
|
+
* `withJudgeRetry` both route through it, so a connection-class error is
|
|
188
|
+
* treated identically whether it surfaces in the HTTP client or a
|
|
189
|
+
* TCloud-backed judge.
|
|
190
|
+
*/
|
|
191
|
+
declare function isTransientLlmError(err: unknown): boolean;
|
|
192
|
+
/** Exponential backoff: 500ms, 1s, 2s, 4s, ... capped at 16s. Attempt is 0-indexed. */
|
|
193
|
+
declare function backoffMs(attempt: number): number;
|
|
194
|
+
/**
|
|
195
|
+
* Strip a ```json / ``` code fence if the model emitted one.
|
|
196
|
+
* Idempotent for naked JSON. Some models (claude-code via router, certain
|
|
197
|
+
* deepseek models) wrap output even under json_object.
|
|
198
|
+
*/
|
|
199
|
+
declare function stripFencedJson(raw: string): string;
|
|
200
|
+
/**
|
|
201
|
+
* Low-level call. Returns raw content + usage + cost. Retries on transient
|
|
202
|
+
* failures; does NOT degrade schema here — callers that want graceful
|
|
203
|
+
* degrade use `callLlmJson`.
|
|
204
|
+
*/
|
|
205
|
+
declare function callLlm(req: LlmCallRequest, opts?: LlmClientOptions): Promise<LlmCallResult>;
|
|
206
|
+
/**
|
|
207
|
+
* Structured-output call. Returns parsed JSON plus the raw result envelope.
|
|
208
|
+
* Degrades `jsonSchema` → `jsonMode` on a 400 that names the schema param —
|
|
209
|
+
* critical for deepseek-v3/v4, kimi-k2.6, and other models that don't accept
|
|
210
|
+
* the `response_format.json_schema` shape but DO accept `json_object`.
|
|
211
|
+
*/
|
|
212
|
+
declare function callLlmJson<T = unknown>(req: LlmCallRequest, opts?: LlmClientOptions): Promise<{
|
|
213
|
+
value: T;
|
|
214
|
+
result: LlmCallResult;
|
|
215
|
+
}>;
|
|
216
|
+
type LlmRouteAssertionReason = 'no_explicit_base_url' | 'base_url_blocked' | 'base_url_not_allowed' | 'no_auth' | 'wrong_provider';
|
|
217
|
+
declare class LlmRouteAssertionError extends CaptureIntegrityError {
|
|
218
|
+
readonly reason: LlmRouteAssertionReason;
|
|
219
|
+
readonly baseUrl: string;
|
|
220
|
+
constructor(message: string, reason: LlmRouteAssertionReason, baseUrl: string);
|
|
221
|
+
}
|
|
222
|
+
interface LlmRouteRequirements {
|
|
223
|
+
/**
|
|
224
|
+
* Throw if `opts.baseUrl` is undefined, i.e. the call would fall back to
|
|
225
|
+
* `DEFAULT_BASE_URL`. Set this for evaluation runs where silently using
|
|
226
|
+
* the public/free-tier router is a defect — the launch reviewer needs to
|
|
227
|
+
* know exactly which provider answered.
|
|
228
|
+
*/
|
|
229
|
+
requireExplicitBaseUrl?: boolean;
|
|
230
|
+
/**
|
|
231
|
+
* Allowlist of acceptable base URLs. Strings match by prefix
|
|
232
|
+
* (case-insensitive); RegExps test against the full base URL.
|
|
233
|
+
*/
|
|
234
|
+
allowedBaseUrls?: Array<string | RegExp>;
|
|
235
|
+
/** Blocklist that takes precedence over `allowedBaseUrls`. */
|
|
236
|
+
blockedBaseUrls?: Array<string | RegExp>;
|
|
237
|
+
/** Throw if no auth header / api key is configured. */
|
|
238
|
+
requireAuth?: boolean;
|
|
239
|
+
/**
|
|
240
|
+
* Logical provider id the configured `baseUrl` is expected to match (via
|
|
241
|
+
* `providerFromBaseUrl`). Mainly useful when paired with `requireExplicitBaseUrl`.
|
|
242
|
+
*/
|
|
243
|
+
expectedProvider?: string;
|
|
244
|
+
}
|
|
245
|
+
/**
|
|
246
|
+
* Fail-loud assertion that the configured LLM client points at the route
|
|
247
|
+
* the caller intends. Designed for the matrix-runner preflight: invoke
|
|
248
|
+
* once before any LLM call to catch misconfiguration before a sweep burns
|
|
249
|
+
* dollars on the wrong provider.
|
|
250
|
+
*
|
|
251
|
+
* Throws `LlmRouteAssertionError`. Pure — no I/O — so it's safe to call
|
|
252
|
+
* from constructors and CI gates.
|
|
253
|
+
*/
|
|
254
|
+
declare function assertLlmRoute(opts: LlmClientOptions, req?: LlmRouteRequirements): void;
|
|
255
|
+
/**
|
|
256
|
+
* Probe whether a model is reachable. Returns latency + null error on
|
|
257
|
+
* success; `ok=false` + error message on any failure (HTTP, timeout,
|
|
258
|
+
* network, parse). Designed for sweep preflights — fail loud at the
|
|
259
|
+
* boundary before burning a 30-leaf run on a misconfigured router.
|
|
260
|
+
*
|
|
261
|
+
* Sends a tiny `ping` message with `maxTokens=64`. Reasoning models
|
|
262
|
+
* (glm-5.1, deepseek-v4) can burn the entire budget on internal reasoning
|
|
263
|
+
* for short prompts, so don't tighten this further. We don't validate
|
|
264
|
+
* content; HTTP 200 means reachable.
|
|
265
|
+
*/
|
|
266
|
+
declare function probeLlm(model: string, opts?: LlmClientOptions & {
|
|
267
|
+
timeoutMs?: number;
|
|
268
|
+
}): Promise<{
|
|
269
|
+
ok: boolean;
|
|
270
|
+
latencyMs: number;
|
|
271
|
+
error: string | null;
|
|
272
|
+
}>;
|
|
273
|
+
/**
|
|
274
|
+
* Stateful client — construct once with defaults, call many times.
|
|
275
|
+
* Thin wrapper around the free functions; exists for callers that want
|
|
276
|
+
* to inject a single configured instance into multiple primitives.
|
|
277
|
+
*/
|
|
278
|
+
declare class LlmClient {
|
|
279
|
+
readonly maximumAttempts: number;
|
|
280
|
+
private readonly opts;
|
|
281
|
+
constructor(opts?: LlmClientOptions);
|
|
282
|
+
call(req: LlmCallRequest, per?: LlmClientOptions): Promise<LlmCallResult>;
|
|
283
|
+
callJson<T = unknown>(req: LlmCallRequest, per?: LlmClientOptions): Promise<{
|
|
284
|
+
value: T;
|
|
285
|
+
result: LlmCallResult;
|
|
286
|
+
}>;
|
|
287
|
+
}
|
|
288
|
+
|
|
289
|
+
export { type LlmCallMetadata as L, type LlmClientOptions as a, type LlmRouteRequirements as b, type LlmCallRequest as c, type LlmCallResult as d, type LlmUsage as e, LlmCallError as f, LlmClient as g, type LlmMessage as h, LlmResponseError as i, LlmRouteAssertionError as j, assertLlmRoute as k, backoffMs as l, callLlm as m, callLlmJson as n, costReceiptFromLlm as o, costReceiptFromLlmError as p, isTransientLlmError as q, maximumChargeForLlmRequest as r, probeLlm as s, stripFencedJson as t };
|
|
@@ -1,15 +1,16 @@
|
|
|
1
|
-
export { C as CalibrationBin, a as CalibrationOptions, b as CalibrationPair, c as CalibrationReport, d as CorrelationResult, e as CorrelationStudyOptions, f as CorrelationStudyResult, E as EvalMetricSpec, O as OutcomePair, g as calibrationCurve, h as calibrationFromPairs, i as correlationStudy } from '../calibration-
|
|
1
|
+
export { C as CalibrationBin, a as CalibrationOptions, b as CalibrationPair, c as CalibrationReport, d as CorrelationResult, e as CorrelationStudyOptions, f as CorrelationStudyResult, E as EvalMetricSpec, O as OutcomePair, g as calibrationCurve, h as calibrationFromPairs, i as correlationStudy } from '../calibration-C8MTS7cw.js';
|
|
2
2
|
export { D as DeploymentOutcome, F as FileSystemOutcomeStore, a as FileSystemOutcomeStoreOptions, I as InMemoryOutcomeStore, O as OutcomeFilter, b as OutcomeStore } from '../outcome-store-rnXLEqSn.js';
|
|
3
|
-
export { R as RubricOutcomePair, a as RubricPredictiveValidityInput, b as RubricPredictiveValidityReport, c as RubricRanking, r as rubricPredictiveValidity } from '../rubric-predictive-validity-
|
|
3
|
+
export { R as RubricOutcomePair, a as RubricPredictiveValidityInput, b as RubricPredictiveValidityReport, c as RubricRanking, r as rubricPredictiveValidity } from '../rubric-predictive-validity-p49lLVrE.js';
|
|
4
4
|
import { C as ContinuousAgreement, a as CalibrationResult, b as ContinuousCalibrationResult, c as CandidateScore, G as GoldenItem } from '../judge-calibration-7C-IDmKr.js';
|
|
5
5
|
import { S as SeriesConvergenceOptions, a as SeriesConvergenceResult } from '../series-convergence-D5OWMBg6.js';
|
|
6
|
-
import { C as CorpusAgreementReport } from '../statistics-
|
|
7
|
-
import '../store-
|
|
8
|
-
import '../schema-
|
|
9
|
-
import '../run-record-
|
|
6
|
+
import { C as CorpusAgreementReport } from '../statistics-KUnG73jH.js';
|
|
7
|
+
import '../store-DGqD0Pyo.js';
|
|
8
|
+
import '../schema-B3Q3l9Z_.js';
|
|
9
|
+
import '../run-record-BDH49H2E.js';
|
|
10
10
|
import '@tangle-network/agent-interface';
|
|
11
11
|
import '../errors-oeQrLqXC.js';
|
|
12
|
-
import '../types-
|
|
12
|
+
import '../types-BkfcQnxV.js';
|
|
13
|
+
import '../cost-ledger-DWy3XdJc.js';
|
|
13
14
|
import '@tangle-network/tcloud';
|
|
14
15
|
|
|
15
16
|
/**
|
package/dist/meta-eval/index.js
CHANGED
|
@@ -1,13 +1,16 @@
|
|
|
1
|
-
import { J as JudgeScore } from '../types-
|
|
1
|
+
import { J as JudgeScore } from '../types-BSw1rOUB.js';
|
|
2
2
|
import { AgentProfile } from '@tangle-network/agent-interface';
|
|
3
3
|
import { M as MatrixResult } from '../types-BUxNaJ8c.js';
|
|
4
|
-
import '../policy-edit-
|
|
5
|
-
import '../run-record-
|
|
4
|
+
import '../policy-edit-wG9uFEFm.js';
|
|
5
|
+
import '../run-record-BDH49H2E.js';
|
|
6
6
|
import '../errors-oeQrLqXC.js';
|
|
7
|
-
import '../schema-
|
|
8
|
-
import '../store-
|
|
9
|
-
import '../types-
|
|
7
|
+
import '../schema-B3Q3l9Z_.js';
|
|
8
|
+
import '../store-C1YxJDEK.js';
|
|
9
|
+
import '../types-BkfcQnxV.js';
|
|
10
|
+
import '../cost-ledger-DWy3XdJc.js';
|
|
10
11
|
import '@tangle-network/tcloud';
|
|
12
|
+
import '../llm-client-qoDd18Qz.js';
|
|
13
|
+
import '../raw-provider-sink-C46HDghv.js';
|
|
11
14
|
import '../verdict-C9MlYujm.js';
|
|
12
15
|
|
|
13
16
|
interface MultishotMessage {
|
package/dist/openapi.json
CHANGED
|
@@ -2,7 +2,7 @@
|
|
|
2
2
|
"openapi": "3.1.0",
|
|
3
3
|
"info": {
|
|
4
4
|
"title": "@tangle-network/agent-eval — wire protocol",
|
|
5
|
-
"version": "0.
|
|
5
|
+
"version": "0.117.0",
|
|
6
6
|
"description": "HTTP and stdio RPC interface to agent-eval. The TypeScript runtime is the source of truth; this spec is the contract that cross-language clients (Python, Rust, Go) generate from.\n\nWire-protocol version: 1.0.0. Bumps on breaking changes to request/response schemas.",
|
|
7
7
|
"contact": {
|
|
8
8
|
"name": "Tangle Network",
|
|
@@ -1,9 +1,9 @@
|
|
|
1
|
-
import { h as BudgetSpec, R as Run, T as ToolSpan } from '../schema-
|
|
2
|
-
import { T as TraceStore, R as RunFilter } from '../store-
|
|
3
|
-
export { a as FailureCluster, F as FailureClusterReport, f as failureClusterView } from '../failure-cluster-
|
|
4
|
-
import { T as TrajectoryStep, B as BaselineOptions, a as BaselineReport } from '../baseline-
|
|
5
|
-
export { c as computeToolUseMetrics } from '../baseline-
|
|
6
|
-
import { l as llmSpans } from '../query-
|
|
1
|
+
import { h as BudgetSpec, R as Run, T as ToolSpan } from '../schema-B3Q3l9Z_.js';
|
|
2
|
+
import { T as TraceStore, R as RunFilter } from '../store-DGqD0Pyo.js';
|
|
3
|
+
export { a as FailureCluster, F as FailureClusterReport, f as failureClusterView } from '../failure-cluster-DOAcSJ87.js';
|
|
4
|
+
import { T as TrajectoryStep, B as BaselineOptions, a as BaselineReport } from '../baseline-DKq3gJpP.js';
|
|
5
|
+
export { c as computeToolUseMetrics } from '../baseline-DKq3gJpP.js';
|
|
6
|
+
import { l as llmSpans } from '../query-CF7PG61p.js';
|
|
7
7
|
|
|
8
8
|
/**
|
|
9
9
|
* BudgetBreachView — aggregates breach events across the corpus.
|
|
@@ -121,8 +121,11 @@ interface StuckLoopFinding {
|
|
|
121
121
|
runId: string;
|
|
122
122
|
toolName: string;
|
|
123
123
|
argHash: string;
|
|
124
|
+
/** Calls in this episode's densest qualifying interval, not the whole-run total. */
|
|
124
125
|
occurrences: number;
|
|
125
126
|
spanIds: string[];
|
|
127
|
+
/** Nearest agent ancestor, or the direct parent when ancestry is incomplete. */
|
|
128
|
+
scopeSpanId?: string;
|
|
126
129
|
/** Milliseconds between first and last call in the loop. */
|
|
127
130
|
windowMs: number;
|
|
128
131
|
}
|
|
@@ -134,6 +137,13 @@ interface StuckLoopReport {
|
|
|
134
137
|
interface StuckLoopOptions {
|
|
135
138
|
/** Minimum call count to flag a loop (default 3). */
|
|
136
139
|
minOccurrences?: number;
|
|
140
|
+
/** Maximum time between the first and last repeated call (default 60 seconds). */
|
|
141
|
+
maxWindowMs?: number;
|
|
142
|
+
/**
|
|
143
|
+
* Maximum other tool calls allowed between adjacent repeats (default 0).
|
|
144
|
+
* Set to 1 to detect alternating patterns such as A,B,A,B,A.
|
|
145
|
+
*/
|
|
146
|
+
maxInterveningToolCalls?: number;
|
|
137
147
|
/** Filter to a specific runId; omit to scan the entire corpus. */
|
|
138
148
|
runId?: string;
|
|
139
149
|
}
|
package/dist/pipelines/index.js
CHANGED
|
@@ -4,7 +4,10 @@ import {
|
|
|
4
4
|
classifyFailure,
|
|
5
5
|
compareToBaseline,
|
|
6
6
|
computeToolUseMetrics
|
|
7
|
-
} from "../chunk-
|
|
7
|
+
} from "../chunk-ODVOOEWQ.js";
|
|
8
|
+
import {
|
|
9
|
+
executionTrackByLane
|
|
10
|
+
} from "../chunk-HHWE3POT.js";
|
|
8
11
|
import {
|
|
9
12
|
interRaterReliability,
|
|
10
13
|
pearsonR
|
|
@@ -12,10 +15,12 @@ import {
|
|
|
12
15
|
import {
|
|
13
16
|
aggregateLlm,
|
|
14
17
|
argHash,
|
|
18
|
+
hasCapturedToolArgs,
|
|
19
|
+
isToolSpan,
|
|
15
20
|
llmSpans,
|
|
16
21
|
runFailureClass,
|
|
17
22
|
toolSpans
|
|
18
|
-
} from "../chunk-
|
|
23
|
+
} from "../chunk-LQUTGLOZ.js";
|
|
19
24
|
import "../chunk-ONWEPEDO.js";
|
|
20
25
|
import "../chunk-PZ5AY32C.js";
|
|
21
26
|
|
|
@@ -80,7 +85,7 @@ async function failureClusterView(store, options = {}) {
|
|
|
80
85
|
const trig = spans.find((s) => s.spanId === cls.triggerSpanId);
|
|
81
86
|
if (trig?.kind === "tool") {
|
|
82
87
|
toolName = trig.toolName;
|
|
83
|
-
argPrefix = argHash(trig.args).slice(0, 16);
|
|
88
|
+
if (hasCapturedToolArgs(trig)) argPrefix = argHash(trig.args).slice(0, 16);
|
|
84
89
|
} else if (trig?.kind === "judge") {
|
|
85
90
|
dimension = trig.dimension;
|
|
86
91
|
}
|
|
@@ -90,7 +95,7 @@ async function failureClusterView(store, options = {}) {
|
|
|
90
95
|
const errored = ts.filter((t) => t.status === "error").pop();
|
|
91
96
|
if (errored) {
|
|
92
97
|
toolName = errored.toolName;
|
|
93
|
-
argPrefix = argHash(errored.args).slice(0, 16);
|
|
98
|
+
if (hasCapturedToolArgs(errored)) argPrefix = argHash(errored.args).slice(0, 16);
|
|
94
99
|
}
|
|
95
100
|
}
|
|
96
101
|
if (!dimension) {
|
|
@@ -296,33 +301,97 @@ function defaultExtract(metric) {
|
|
|
296
301
|
}
|
|
297
302
|
|
|
298
303
|
// src/pipelines/stuck-loop.ts
|
|
304
|
+
var DEFAULT_MAX_WINDOW_MS = 6e4;
|
|
305
|
+
var DEFAULT_MAX_INTERVENING_TOOL_CALLS = 0;
|
|
299
306
|
async function stuckLoopView(store, options = {}) {
|
|
300
307
|
const minOccurrences = options.minOccurrences ?? 3;
|
|
308
|
+
const maxWindowMs = options.maxWindowMs ?? DEFAULT_MAX_WINDOW_MS;
|
|
309
|
+
const maxInterveningToolCalls = options.maxInterveningToolCalls ?? DEFAULT_MAX_INTERVENING_TOOL_CALLS;
|
|
310
|
+
if (!Number.isInteger(minOccurrences) || minOccurrences < 1) {
|
|
311
|
+
throw new RangeError("minOccurrences must be a positive integer");
|
|
312
|
+
}
|
|
313
|
+
if (!Number.isFinite(maxWindowMs) || maxWindowMs < 0) {
|
|
314
|
+
throw new RangeError("maxWindowMs must be a finite non-negative number");
|
|
315
|
+
}
|
|
316
|
+
if (!Number.isInteger(maxInterveningToolCalls) || maxInterveningToolCalls < 0) {
|
|
317
|
+
throw new RangeError("maxInterveningToolCalls must be a non-negative integer");
|
|
318
|
+
}
|
|
301
319
|
const runs = options.runId ? [{ runId: options.runId }] : (await store.listRuns()).map((r) => ({ runId: r.runId }));
|
|
302
320
|
const findings = [];
|
|
303
321
|
for (const { runId } of runs) {
|
|
304
|
-
const
|
|
322
|
+
const spans = await store.spans({ runId });
|
|
323
|
+
const spansById = new Map(spans.map((span) => [span.spanId, span]));
|
|
324
|
+
const scopedTools = spans.filter(isToolSpan).map((span, sourceIndex) => ({ span, sourceIndex })).sort((a, b) => a.span.startedAt - b.span.startedAt || a.sourceIndex - b.sourceIndex).map(({ span }) => ({ span, ...executionScope(span, spansById) }));
|
|
325
|
+
const trackByLane = executionTrackByLane(
|
|
326
|
+
scopedTools.map((call) => {
|
|
327
|
+
const direct = call.laneSpanId === null || call.laneSpanId === call.scopeSpanId;
|
|
328
|
+
const timed = direct ? call.span : call.laneSpanId ? spansById.get(call.laneSpanId) : void 0;
|
|
329
|
+
return {
|
|
330
|
+
key: executionKey(call),
|
|
331
|
+
scopeKey: JSON.stringify(call.scopeSpanId),
|
|
332
|
+
start: timed?.startedAt ?? null,
|
|
333
|
+
end: timed?.endedAt ?? null
|
|
334
|
+
};
|
|
335
|
+
})
|
|
336
|
+
);
|
|
337
|
+
const nextToolIndexByTrack = /* @__PURE__ */ new Map();
|
|
338
|
+
const orderedTools = scopedTools.map((call) => {
|
|
339
|
+
const trackId = trackByLane.get(executionKey(call));
|
|
340
|
+
const toolCallIndex = nextToolIndexByTrack.get(trackId) ?? 0;
|
|
341
|
+
nextToolIndexByTrack.set(trackId, toolCallIndex + 1);
|
|
342
|
+
return { ...call, toolCallIndex, trackId };
|
|
343
|
+
});
|
|
305
344
|
const byKey = /* @__PURE__ */ new Map();
|
|
306
|
-
for (const
|
|
307
|
-
|
|
308
|
-
const
|
|
309
|
-
const
|
|
310
|
-
bucket.
|
|
345
|
+
for (const call of orderedTools) {
|
|
346
|
+
if (!hasCapturedToolArgs(call.span)) continue;
|
|
347
|
+
const h = argHash(call.span.args);
|
|
348
|
+
const key = JSON.stringify([call.trackId, call.span.toolName, h]);
|
|
349
|
+
const bucket = byKey.get(key) ?? {
|
|
350
|
+
calls: [],
|
|
351
|
+
argHash: h,
|
|
352
|
+
toolName: call.span.toolName,
|
|
353
|
+
scopeSpanId: call.scopeSpanId
|
|
354
|
+
};
|
|
355
|
+
bucket.calls.push(call);
|
|
311
356
|
byKey.set(key, bucket);
|
|
312
357
|
}
|
|
313
|
-
for (const {
|
|
314
|
-
if (
|
|
315
|
-
|
|
316
|
-
|
|
317
|
-
|
|
318
|
-
|
|
319
|
-
|
|
320
|
-
|
|
321
|
-
|
|
322
|
-
|
|
323
|
-
|
|
324
|
-
|
|
325
|
-
|
|
358
|
+
for (const { calls, argHash: h, toolName, scopeSpanId } of byKey.values()) {
|
|
359
|
+
if (calls.length < minOccurrences) continue;
|
|
360
|
+
let episodeStart = 0;
|
|
361
|
+
for (let episodeEnd = 1; episodeEnd <= calls.length; episodeEnd += 1) {
|
|
362
|
+
const previous = calls[episodeEnd - 1];
|
|
363
|
+
const next = calls[episodeEnd];
|
|
364
|
+
const episodeEnded = next === void 0 || next.span.startedAt - previous.span.startedAt > maxWindowMs || next.toolCallIndex - previous.toolCallIndex - 1 > maxInterveningToolCalls || !callsAreSerial(previous, next);
|
|
365
|
+
if (!episodeEnded) continue;
|
|
366
|
+
const episode = calls.slice(episodeStart, episodeEnd);
|
|
367
|
+
let left = 0;
|
|
368
|
+
let bestStart = 0;
|
|
369
|
+
let bestEnd = -1;
|
|
370
|
+
for (let right = 0; right < episode.length; right += 1) {
|
|
371
|
+
while (episode[right].span.startedAt - episode[left].span.startedAt > maxWindowMs) {
|
|
372
|
+
left += 1;
|
|
373
|
+
}
|
|
374
|
+
if (right - left > bestEnd - bestStart) {
|
|
375
|
+
bestStart = left;
|
|
376
|
+
bestEnd = right;
|
|
377
|
+
}
|
|
378
|
+
}
|
|
379
|
+
if (bestEnd - bestStart + 1 >= minOccurrences) {
|
|
380
|
+
const loop = episode.slice(bestStart, bestEnd + 1);
|
|
381
|
+
const first = loop[0].span.startedAt;
|
|
382
|
+
const last = loop[loop.length - 1].span.startedAt;
|
|
383
|
+
findings.push({
|
|
384
|
+
runId,
|
|
385
|
+
toolName,
|
|
386
|
+
argHash: h,
|
|
387
|
+
occurrences: loop.length,
|
|
388
|
+
spanIds: loop.map((call) => call.span.spanId),
|
|
389
|
+
...scopeSpanId ? { scopeSpanId } : {},
|
|
390
|
+
windowMs: last - first
|
|
391
|
+
});
|
|
392
|
+
}
|
|
393
|
+
episodeStart = episodeEnd;
|
|
394
|
+
}
|
|
326
395
|
}
|
|
327
396
|
}
|
|
328
397
|
const affectedRuns = new Set(findings.map((f) => f.runId));
|
|
@@ -332,6 +401,33 @@ async function stuckLoopView(store, options = {}) {
|
|
|
332
401
|
totalRuns: runs.length
|
|
333
402
|
};
|
|
334
403
|
}
|
|
404
|
+
function laneKey(call) {
|
|
405
|
+
return JSON.stringify([call.scopeSpanId, call.laneSpanId]);
|
|
406
|
+
}
|
|
407
|
+
function executionKey(call) {
|
|
408
|
+
return call.laneSpanId === null || call.laneSpanId === call.scopeSpanId ? JSON.stringify([call.scopeSpanId, call.span.spanId]) : laneKey(call);
|
|
409
|
+
}
|
|
410
|
+
function executionScope(span, spansById) {
|
|
411
|
+
const directParent = span.parentSpanId;
|
|
412
|
+
if (!directParent) return { scopeSpanId: null, laneSpanId: null };
|
|
413
|
+
let currentId = directParent;
|
|
414
|
+
let laneSpanId = null;
|
|
415
|
+
const seen = /* @__PURE__ */ new Set();
|
|
416
|
+
while (currentId && !seen.has(currentId)) {
|
|
417
|
+
seen.add(currentId);
|
|
418
|
+
const current = spansById.get(currentId);
|
|
419
|
+
if (!current) return { scopeSpanId: directParent, laneSpanId: directParent };
|
|
420
|
+
if (current.kind === "agent") {
|
|
421
|
+
return { scopeSpanId: current.spanId, laneSpanId: laneSpanId ?? current.spanId };
|
|
422
|
+
}
|
|
423
|
+
laneSpanId = current.spanId;
|
|
424
|
+
currentId = current.parentSpanId;
|
|
425
|
+
}
|
|
426
|
+
return { scopeSpanId: directParent, laneSpanId: directParent };
|
|
427
|
+
}
|
|
428
|
+
function callsAreSerial(previous, next) {
|
|
429
|
+
return previous.span.endedAt !== void 0 && previous.span.endedAt <= next.span.startedAt;
|
|
430
|
+
}
|
|
335
431
|
|
|
336
432
|
// src/pipelines/tool-waste.ts
|
|
337
433
|
async function toolWasteView(store, options = {}) {
|