@tangle-network/agent-eval 0.116.0 → 0.117.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +31 -0
- package/dist/analyst/index.d.ts +18 -11
- package/dist/analyst/index.js +10 -7
- package/dist/analyst/index.js.map +1 -1
- package/dist/{analyst-CFBc14Wc.d.ts → analyst-C8HHvfJp.d.ts} +1 -1
- package/dist/{analyze-runs-0rz_m29H.d.ts → analyze-runs--2x39HZ7.d.ts} +3 -3
- package/dist/{baseline-DsNteOgR.d.ts → baseline-DKq3gJpP.d.ts} +6 -3
- package/dist/belief-state/index.d.ts +6 -6
- package/dist/belief-state/index.js +1 -1
- package/dist/benchmarks/index.d.ts +11 -8
- package/dist/benchmarks/index.js +11 -10
- package/dist/builder-eval/index.d.ts +4 -4
- package/dist/builder-eval/index.js +1 -1
- package/dist/{calibration-Dz8TQV4y.d.ts → calibration-C8MTS7cw.d.ts} +2 -2
- package/dist/campaign/index.d.ts +53 -29
- package/dist/campaign/index.js +18 -13
- package/dist/chunk-3YYRZDON.js +45 -0
- package/dist/chunk-3YYRZDON.js.map +1 -0
- package/dist/{chunk-RPDDVKI7.js → chunk-4JLWXDYA.js} +2 -2
- package/dist/{chunk-NBSS5NDZ.js → chunk-CCZIVI3F.js} +54 -115
- package/dist/chunk-CCZIVI3F.js.map +1 -0
- package/dist/{chunk-J6P6PK2R.js → chunk-FQNLDL4D.js} +3 -3
- package/dist/{chunk-ONM6PEAE.js → chunk-GQCZRZ7L.js} +2 -2
- package/dist/chunk-HHWE3POT.js +94 -0
- package/dist/chunk-HHWE3POT.js.map +1 -0
- package/dist/{chunk-3274WNK7.js → chunk-HQPHZGL6.js} +687 -44
- package/dist/chunk-HQPHZGL6.js.map +1 -0
- package/dist/{chunk-FAOEFFRT.js → chunk-IDZTTFRR.js} +390 -78
- package/dist/chunk-IDZTTFRR.js.map +1 -0
- package/dist/{chunk-3LXTCTWL.js → chunk-JSDVRFAP.js} +2 -2
- package/dist/{chunk-MHNQWM4I.js → chunk-LQUTGLOZ.js} +5 -1
- package/dist/chunk-LQUTGLOZ.js.map +1 -0
- package/dist/{chunk-4D5RVB3W.js → chunk-LTVG32KX.js} +30 -5
- package/dist/chunk-LTVG32KX.js.map +1 -0
- package/dist/{chunk-CIUOICJT.js → chunk-MGEHEHSN.js} +62 -15
- package/dist/chunk-MGEHEHSN.js.map +1 -0
- package/dist/{chunk-GY4SYVPJ.js → chunk-NJC7U437.js} +97 -25
- package/dist/chunk-NJC7U437.js.map +1 -0
- package/dist/{chunk-NYFUT3B3.js → chunk-ODVOOEWQ.js} +31 -10
- package/dist/chunk-ODVOOEWQ.js.map +1 -0
- package/dist/{chunk-LNQEP766.js → chunk-S2F4J57L.js} +44 -4
- package/dist/chunk-S2F4J57L.js.map +1 -0
- package/dist/chunk-VCTY3W6J.js +798 -0
- package/dist/chunk-VCTY3W6J.js.map +1 -0
- package/dist/chunk-VF3XSYTI.js +545 -0
- package/dist/chunk-VF3XSYTI.js.map +1 -0
- package/dist/{chunk-TLDB7WRY.js → chunk-YZPO4UHR.js} +28 -31
- package/dist/chunk-YZPO4UHR.js.map +1 -0
- package/dist/{chunk-GSW3OBHK.js → chunk-ZUXV7UWZ.js} +350 -724
- package/dist/chunk-ZUXV7UWZ.js.map +1 -0
- package/dist/cli.js +4 -2
- package/dist/cli.js.map +1 -1
- package/dist/{code-agent-session-CdxteG0y.d.ts → code-agent-session-CjZsVd19.d.ts} +1 -1
- package/dist/contract/index.d.ts +43 -29
- package/dist/contract/index.js +56 -19
- package/dist/contract/index.js.map +1 -1
- package/dist/{control-DbcDxouY.d.ts → control-6vuGfmDH.d.ts} +5 -5
- package/dist/control.d.ts +6 -6
- package/dist/cost-ledger-DWy3XdJc.d.ts +183 -0
- package/dist/{default-registry-DDfv22MQ.d.ts → default-registry-DaK8b3fv.d.ts} +2 -2
- package/dist/{emitter-BRchAAAx.d.ts → emitter-CjD7vUwv.d.ts} +2 -2
- package/dist/{failure-cluster-C48PiReX.d.ts → failure-cluster-DOAcSJ87.d.ts} +2 -2
- package/dist/{feedback-trajectory-pDcz1lQ1.d.ts → feedback-trajectory-BUnM58xL.d.ts} +3 -3
- package/dist/fuzz.d.ts +8 -16
- package/dist/fuzz.js +72 -42
- package/dist/fuzz.js.map +1 -1
- package/dist/{gepa-CQelRtuC.d.ts → gepa-eESocoDi.d.ts} +56 -6
- package/dist/hosted/index.d.ts +13 -10
- package/dist/{index-DbCXJfZ1.d.ts → index-PdX4VnPA.d.ts} +3 -3
- package/dist/index.d.ts +102 -57
- package/dist/index.js +328 -235
- package/dist/index.js.map +1 -1
- package/dist/{insight-report-oMVxDTxl.d.ts → insight-report-DY4nDW9Q.d.ts} +1 -1
- package/dist/{integrity-C6PZ73iC.d.ts → integrity-DqlBiLyK.d.ts} +2 -2
- package/dist/{kind-factory-DWOvXjR_.d.ts → kind-factory-ClZmO25A.d.ts} +2 -2
- package/dist/llm-client-qoDd18Qz.d.ts +289 -0
- package/dist/meta-eval/index.d.ts +8 -7
- package/dist/meta-eval/index.js +1 -1
- package/dist/multishot/index.d.ts +9 -6
- package/dist/openapi.json +1 -1
- package/dist/pipelines/index.d.ts +16 -6
- package/dist/pipelines/index.js +119 -23
- package/dist/pipelines/index.js.map +1 -1
- package/dist/{policy-edit-Clb2v6Oa.d.ts → policy-edit-wG9uFEFm.d.ts} +13 -266
- package/dist/{pre-registration--vU0mMtD.d.ts → pre-registration-BWQhJ3vz.d.ts} +24 -5
- package/dist/{provenance-BbVagC68.d.ts → provenance-DpjwyseI.d.ts} +6 -6
- package/dist/{query-Ck190MOd.d.ts → query-CF7PG61p.d.ts} +5 -3
- package/dist/raw-provider-sink-C46HDghv.d.ts +132 -0
- package/dist/{release-report-CamNDe90.d.ts → release-report-C8G2i5Xi.d.ts} +2 -2
- package/dist/reporting.d.ts +10 -9
- package/dist/{researcher-Dwbo_Fxx.d.ts → researcher-C8XyxQsu.d.ts} +8 -8
- package/dist/rl.d.ts +18 -15
- package/dist/rl.js +2 -2
- package/dist/{rubric-predictive-validity-BIdf9h4R.d.ts → rubric-predictive-validity-p49lLVrE.d.ts} +1 -1
- package/dist/{run-campaign-UADIM77S.js → run-campaign-IM26A6PD.js} +4 -2
- package/dist/{run-record-CZmcpWPo.d.ts → run-record-BDH49H2E.d.ts} +1 -1
- package/dist/{runtime-trajectory-CC0jx9ql.d.ts → runtime-trajectory-DGBIUt4B.d.ts} +1 -1
- package/dist/{schema-SGWcK9wa.d.ts → schema-B3Q3l9Z_.d.ts} +2 -0
- package/dist/{semantic-concept-judge-CKjePUMh.d.ts → semantic-concept-judge-CXnPEJbf.d.ts} +24 -6
- package/dist/{statistics-oUbOJe-S.d.ts → statistics-KUnG73jH.d.ts} +1 -1
- package/dist/{storage-Dw_f7WMt.d.ts → storage-DrX3v_5B.d.ts} +12 -1
- package/dist/{store-9cAScOcb.d.ts → store-C1YxJDEK.d.ts} +1 -132
- package/dist/{store-BsVi7ncX.d.ts → store-DGqD0Pyo.d.ts} +1 -1
- package/dist/storyboard/index.d.ts +1 -1
- package/dist/{summary-report-DTNgQycC.d.ts → summary-report-C5bKFfm-.d.ts} +2 -2
- package/dist/{test-graded-scenario-mzYBKspu.d.ts → test-graded-scenario-B0ybnPY7.d.ts} +3 -3
- package/dist/traces.d.ts +25 -14
- package/dist/traces.js +16 -4
- package/dist/{types-Ca_63YSD.d.ts → types-BSw1rOUB.d.ts} +41 -39
- package/dist/{types-C7DGg5ex.d.ts → types-BkfcQnxV.d.ts} +15 -0
- package/dist/wire/index.d.ts +28 -19
- package/dist/wire/index.js +4 -2
- package/docs/distributed-driver.md +1 -1
- package/package.json +3 -3
- package/dist/chunk-3274WNK7.js.map +0 -1
- package/dist/chunk-4D5RVB3W.js.map +0 -1
- package/dist/chunk-7GKEAIAD.js +0 -205
- package/dist/chunk-7GKEAIAD.js.map +0 -1
- package/dist/chunk-CIUOICJT.js.map +0 -1
- package/dist/chunk-FAOEFFRT.js.map +0 -1
- package/dist/chunk-GSW3OBHK.js.map +0 -1
- package/dist/chunk-GY4SYVPJ.js.map +0 -1
- package/dist/chunk-LNQEP766.js.map +0 -1
- package/dist/chunk-MHNQWM4I.js.map +0 -1
- package/dist/chunk-MPHTT5HE.js +0 -74
- package/dist/chunk-MPHTT5HE.js.map +0 -1
- package/dist/chunk-NBSS5NDZ.js.map +0 -1
- package/dist/chunk-NYFUT3B3.js.map +0 -1
- package/dist/chunk-TLDB7WRY.js.map +0 -1
- package/dist/cost-ledger-DuSqlw5B.d.ts +0 -113
- /package/dist/{chunk-RPDDVKI7.js.map → chunk-4JLWXDYA.js.map} +0 -0
- /package/dist/{chunk-J6P6PK2R.js.map → chunk-FQNLDL4D.js.map} +0 -0
- /package/dist/{chunk-ONM6PEAE.js.map → chunk-GQCZRZ7L.js.map} +0 -0
- /package/dist/{chunk-3LXTCTWL.js.map → chunk-JSDVRFAP.js.map} +0 -0
- /package/dist/{run-campaign-UADIM77S.js.map → run-campaign-IM26A6PD.js.map} +0 -0
package/dist/traces.d.ts
CHANGED
|
@@ -1,21 +1,24 @@
|
|
|
1
1
|
import { N as NotFoundError, R as ReplayError } from './errors-oeQrLqXC.js';
|
|
2
|
-
import {
|
|
3
|
-
export {
|
|
4
|
-
import { a as RunCompleteHookContext, R as RunCompleteHook } from './emitter-
|
|
5
|
-
export { S as SpanHandle, T as TraceEmitter, b as TraceEmitterOptions, l as llmSpanFromProvider } from './emitter-
|
|
6
|
-
export { b as RunIntegrityError, R as RunIntegrityExpectations, c as RunIntegrityIssue, d as RunIntegrityIssueCode, a as RunIntegrityReport, e as assertRunCaptured, t as throwIfRunIncomplete } from './integrity-
|
|
7
|
-
import { T as TraceStore } from './store-
|
|
8
|
-
export { E as EventFilter, F as FileSystemTraceStore, a as FileSystemTraceStoreOptions, I as InMemoryTraceStore, R as RunFilter, S as SpanFilter } from './store-
|
|
9
|
-
|
|
10
|
-
|
|
11
|
-
export {
|
|
12
|
-
import { A as AnalyzeTracesOptions, b as AnalyzeTracesResult } from './analyst-
|
|
13
|
-
export { a as AnalyzeTracesInput, c as AnalyzeTracesTurnSnapshot, d as analyzeTraces } from './analyst-
|
|
14
|
-
import {
|
|
2
|
+
import { P as ProviderRedactor, R as RawProviderSink, d as RawProviderEvent } from './raw-provider-sink-C46HDghv.js';
|
|
3
|
+
export { F as FileSystemRawProviderSink, a as FileSystemRawProviderSinkOptions, I as InMemoryRawProviderSink, b as InMemoryRawProviderSinkOptions, N as NoopRawProviderSink, c as RawProviderDirection, e as RawProviderSinkFilter, f as defaultProviderRedactor, p as providerFromBaseUrl } from './raw-provider-sink-C46HDghv.js';
|
|
4
|
+
import { a as RunCompleteHookContext, R as RunCompleteHook } from './emitter-CjD7vUwv.js';
|
|
5
|
+
export { S as SpanHandle, T as TraceEmitter, b as TraceEmitterOptions, l as llmSpanFromProvider } from './emitter-CjD7vUwv.js';
|
|
6
|
+
export { b as RunIntegrityError, R as RunIntegrityExpectations, c as RunIntegrityIssue, d as RunIntegrityIssueCode, a as RunIntegrityReport, e as assertRunCaptured, t as throwIfRunIncomplete } from './integrity-DqlBiLyK.js';
|
|
7
|
+
import { T as TraceStore } from './store-DGqD0Pyo.js';
|
|
8
|
+
export { E as EventFilter, F as FileSystemTraceStore, a as FileSystemTraceStoreOptions, I as InMemoryTraceStore, R as RunFilter, S as SpanFilter } from './store-DGqD0Pyo.js';
|
|
9
|
+
import { T as ToolSpan, R as Run } from './schema-B3Q3l9Z_.js';
|
|
10
|
+
export { A as Artifact, B as BudgetLedgerEntry, h as BudgetSpec, E as EventKind, i as FAILURE_CLASSES, F as FailureClass, G as GenericSpan, J as JudgeSpan, L as LlmSpan, M as Message, c as RetrievalSpan, g as RunLayer, a as RunOutcome, f as RunStatus, d as SandboxSpan, S as Span, j as SpanBase, b as SpanKind, k as SpanStatus, l as TRACE_SCHEMA_VERSION, e as TraceEvent, m as isJudgeSpan, n as isLlmSpan, o as isRetrievalSpan, p as isSandboxSpan, q as isToolSpan } from './schema-B3Q3l9Z_.js';
|
|
11
|
+
export { a as aggregateLlm, b as argHash, g as groupBy, h as hasCapturedToolArgs, j as judgeSpans, l as llmSpans, r as runFailureClass, c as runsForScenario, t as toolSpans } from './query-CF7PG61p.js';
|
|
12
|
+
import { A as AnalyzeTracesOptions, b as AnalyzeTracesResult } from './analyst-C8HHvfJp.js';
|
|
13
|
+
export { a as AnalyzeTracesInput, c as AnalyzeTracesTurnSnapshot, d as analyzeTraces } from './analyst-C8HHvfJp.js';
|
|
14
|
+
import { h as TraceAnalystSpanKind, i as TraceAnalystSpanStatus, T as TraceAnalysisStore, g as TraceAnalystFilters, b as DatasetOverview, Q as QueryTracesPage, l as ViewTraceResult, V as ViewSpansResult, c as SearchTraceResult, S as SearchSpanResult } from './store-C1YxJDEK.js';
|
|
15
|
+
export { D as DEFAULT_TRACE_ANALYST_BUDGETS, E as ErrorCluster, d as SpanMatchRecord, e as TRACE_ANALYST_TRUNCATION_MARKER_PREFIX, f as TraceAnalystByteBudgets, a as TraceAnalystSpan, j as TraceAnalystTraceSummary, k as ViewTraceOversized } from './store-C1YxJDEK.js';
|
|
16
|
+
import { a as RunSplitTag, c as RunTokenUsage, R as RunRecord } from './run-record-BDH49H2E.js';
|
|
15
17
|
import { AxFunction } from '@ax-llm/ax';
|
|
16
18
|
import '@tangle-network/agent-interface';
|
|
17
19
|
|
|
18
20
|
/** Canonical OpenInference-over-OTLP attribute vocabulary used at the trace boundary. */
|
|
21
|
+
|
|
19
22
|
declare const OPENINFERENCE_SPAN_KIND = "openinference.span.kind";
|
|
20
23
|
declare const LLM_MODEL_NAME = "llm.model_name";
|
|
21
24
|
declare const LLM_INPUT_TOKENS = "llm.token_count.prompt";
|
|
@@ -23,6 +26,10 @@ declare const LLM_OUTPUT_TOKENS = "llm.token_count.completion";
|
|
|
23
26
|
declare const LLM_CACHED_TOKENS = "llm.token_count.prompt_cache_hit";
|
|
24
27
|
declare const LLM_COST_USD = "llm.cost_usd";
|
|
25
28
|
declare const TOOL_NAME = "tool.name";
|
|
29
|
+
declare const TOOL_ARGS_CAPTURED = "tool.args_captured";
|
|
30
|
+
declare const TOOL_LATENCY_MS = "tool.latency_ms";
|
|
31
|
+
declare const INPUT_VALUE = "input.value";
|
|
32
|
+
declare const OUTPUT_VALUE = "output.value";
|
|
26
33
|
declare const SPAN_KIND_ATTR_KEYS: readonly ["openinference.span.kind", "inference.observation_kind"];
|
|
27
34
|
declare const LLM_MODEL_ATTR_KEYS: readonly ["llm.model_name", "inference.llm.model_name", "llm.model", "gen_ai.request.model", "gen_ai.response.model"];
|
|
28
35
|
declare const LLM_INPUT_TOKEN_ATTR_KEYS: readonly ["llm.token_count.prompt", "inference.llm.input_tokens", "llm.input_tokens", "gen_ai.usage.input_tokens", "gen_ai.usage.prompt_tokens"];
|
|
@@ -30,6 +37,8 @@ declare const LLM_OUTPUT_TOKEN_ATTR_KEYS: readonly ["llm.token_count.completion"
|
|
|
30
37
|
declare const LLM_CACHED_TOKEN_ATTR_KEYS: readonly ["llm.token_count.prompt_cache_hit", "inference.llm.cached_tokens", "llm.cached_tokens", "gen_ai.usage.cached_tokens"];
|
|
31
38
|
declare const LLM_COST_ATTR_KEYS: readonly ["llm.cost_usd", "inference.llm.cost.total", "llm.cost.total", "gen_ai.usage.cost"];
|
|
32
39
|
declare const TOOL_NAME_ATTR_KEYS: readonly ["tool.name", "inference.tool.name"];
|
|
40
|
+
type ToolSpanOtlpInput = Pick<ToolSpan, 'toolName' | 'args' | 'argsCaptured' | 'result' | 'latencyMs'>;
|
|
41
|
+
declare function applyToolSpanOtlpAttributes(attributes: Record<string, unknown>, span: ToolSpanOtlpInput): void;
|
|
33
42
|
declare function traceSpanKindToOpenInferenceKind(kind: string): string;
|
|
34
43
|
|
|
35
44
|
/**
|
|
@@ -708,6 +717,7 @@ declare function captureFetchToRawSink(fetch: typeof globalThis.fetch, sink: Raw
|
|
|
708
717
|
* or when the batch fills. No @opentelemetry SDK dependency — minimal
|
|
709
718
|
* OTLP/JSON serializer (~120 LOC) using the existing otel.ts helpers.
|
|
710
719
|
*/
|
|
720
|
+
|
|
711
721
|
interface OtelExportConfig {
|
|
712
722
|
/** OTLP endpoint. Reads OTEL_EXPORTER_OTLP_ENDPOINT env by default. */
|
|
713
723
|
endpoint?: string;
|
|
@@ -744,6 +754,7 @@ interface ExportableSpan {
|
|
|
744
754
|
inputTokens?: number;
|
|
745
755
|
outputTokens?: number;
|
|
746
756
|
costUsd?: number;
|
|
757
|
+
tool?: ToolSpanOtlpInput;
|
|
747
758
|
attributes?: Record<string, unknown>;
|
|
748
759
|
}
|
|
749
760
|
/**
|
|
@@ -1014,4 +1025,4 @@ declare function iterateRawCalls(sink: RawProviderSink, filter?: {
|
|
|
1014
1025
|
spanId?: string;
|
|
1015
1026
|
}): AsyncGenerator<ReplayCacheEntry>;
|
|
1016
1027
|
|
|
1017
|
-
export { AnalyzeTracesOptions, AnalyzeTracesResult, type CaptureFetchContext, type CaptureFetchOptions, DEFAULT_REDACTION_RULES, DatasetOverview, type ExportableSpan, type ExtractedUsage, type FlattenOtlpOptions, LLM_CACHED_TOKENS, LLM_CACHED_TOKEN_ATTR_KEYS, LLM_COST_ATTR_KEYS, LLM_COST_USD, LLM_INPUT_TOKENS, LLM_INPUT_TOKEN_ATTR_KEYS, LLM_MODEL_ATTR_KEYS, LLM_MODEL_NAME, LLM_OUTPUT_TOKENS, LLM_OUTPUT_TOKEN_ATTR_KEYS, OPENINFERENCE_SPAN_KIND, OTEL_AGENT_EVAL_SCOPE, type OtelExportConfig, type OtelExporter, type OtlpExport, OtlpFileTraceStore, type OtlpFileTraceStoreOptions, type OtlpFlatLine, type OtlpResourceSpans, type OtlpSpan, type OtlpToRunRecordsOptions, type OtlpTraceRunRecord, type ProjectedOtlpSpan, ProviderRedactor, QueryTracesPage, REDACTION_VERSION, RawProviderEvent, RawProviderSink, type RedactionReport, type RedactionRule, ReplayCache, type ReplayCacheEntry, ReplayCacheMissError, type ReplayCacheStats, type ReplayFetchOptions, Run, RunCompleteHook, RunCompleteHookContext, SPAN_KIND_ATTR_KEYS, SearchSpanResult, SearchTraceResult, SpanNotFoundError, TOOL_NAME, TOOL_NAME_ATTR_KEYS, TRACE_ANALYST_ACTOR_DESCRIPTION, TRACE_ANALYST_ACTOR_DESCRIPTION_VERSION, TRACE_ANALYST_SUBAGENT_DESCRIPTION, type TraceAggregate, TraceAnalysisStore, TraceAnalystFilters, type TraceAnalystHookOptions, TraceAnalystSpanKind, TraceAnalystSpanStatus, TraceFileMissingError, type TraceInsightContext, type TraceInsightFinding, type TraceInsightPanelRole, type TraceInsightPromptInput, type TraceInsightQualityGate, type TraceInsightQuestion, type TraceInsightReadiness, type TraceInsightSuite, type TraceInsightTask, TraceNotFoundError, TraceStore, type TraceStoreSource, type TraceStoreToOtlpOptions, type TracesToOtlpResult, ViewSpansResult, ViewTraceResult, asNumber, asString, buildTraceAnalystTools, buildTraceInsightContext, buildTraceInsightPrompt, captureFetchToRawSink, convertTraceStoresToOtlp, createOtelExporter, createOtelTracingStore, createReplayFetch, defaultTraceInsightPanel, describeTraceInsightScope, domainEvidencePattern, exportRunAsOtlp, extractOtlpAttributes, extractUsage, extractUsageFromResponse, extractUsageFromSse, firstNumberAttr, firstStringAttr, flattenOtlpExportToNdjson, inferDomainKeywords, inferOtlpKind, iterateRawCalls, otelRunCompleteHook, otlpToRunRecords, otlpToTraceRunRecords, planTraceInsightQuestions, projectOtlpFlatLine, readOtlpStatus, redactString, redactValue, scoreTraceInsightReadiness, stringField, tokenizeDomainWords, traceAnalystFunctionGroup, traceAnalystOnRunComplete, traceSpanKindToOpenInferenceKind };
|
|
1028
|
+
export { AnalyzeTracesOptions, AnalyzeTracesResult, type CaptureFetchContext, type CaptureFetchOptions, DEFAULT_REDACTION_RULES, DatasetOverview, type ExportableSpan, type ExtractedUsage, type FlattenOtlpOptions, INPUT_VALUE, LLM_CACHED_TOKENS, LLM_CACHED_TOKEN_ATTR_KEYS, LLM_COST_ATTR_KEYS, LLM_COST_USD, LLM_INPUT_TOKENS, LLM_INPUT_TOKEN_ATTR_KEYS, LLM_MODEL_ATTR_KEYS, LLM_MODEL_NAME, LLM_OUTPUT_TOKENS, LLM_OUTPUT_TOKEN_ATTR_KEYS, OPENINFERENCE_SPAN_KIND, OTEL_AGENT_EVAL_SCOPE, OUTPUT_VALUE, type OtelExportConfig, type OtelExporter, type OtlpExport, OtlpFileTraceStore, type OtlpFileTraceStoreOptions, type OtlpFlatLine, type OtlpResourceSpans, type OtlpSpan, type OtlpToRunRecordsOptions, type OtlpTraceRunRecord, type ProjectedOtlpSpan, ProviderRedactor, QueryTracesPage, REDACTION_VERSION, RawProviderEvent, RawProviderSink, type RedactionReport, type RedactionRule, ReplayCache, type ReplayCacheEntry, ReplayCacheMissError, type ReplayCacheStats, type ReplayFetchOptions, Run, RunCompleteHook, RunCompleteHookContext, SPAN_KIND_ATTR_KEYS, SearchSpanResult, SearchTraceResult, SpanNotFoundError, TOOL_ARGS_CAPTURED, TOOL_LATENCY_MS, TOOL_NAME, TOOL_NAME_ATTR_KEYS, TRACE_ANALYST_ACTOR_DESCRIPTION, TRACE_ANALYST_ACTOR_DESCRIPTION_VERSION, TRACE_ANALYST_SUBAGENT_DESCRIPTION, ToolSpan, type ToolSpanOtlpInput, type TraceAggregate, TraceAnalysisStore, TraceAnalystFilters, type TraceAnalystHookOptions, TraceAnalystSpanKind, TraceAnalystSpanStatus, TraceFileMissingError, type TraceInsightContext, type TraceInsightFinding, type TraceInsightPanelRole, type TraceInsightPromptInput, type TraceInsightQualityGate, type TraceInsightQuestion, type TraceInsightReadiness, type TraceInsightSuite, type TraceInsightTask, TraceNotFoundError, TraceStore, type TraceStoreSource, type TraceStoreToOtlpOptions, type TracesToOtlpResult, ViewSpansResult, ViewTraceResult, applyToolSpanOtlpAttributes, asNumber, asString, buildTraceAnalystTools, buildTraceInsightContext, buildTraceInsightPrompt, captureFetchToRawSink, convertTraceStoresToOtlp, createOtelExporter, createOtelTracingStore, createReplayFetch, defaultTraceInsightPanel, describeTraceInsightScope, domainEvidencePattern, exportRunAsOtlp, extractOtlpAttributes, extractUsage, extractUsageFromResponse, extractUsageFromSse, firstNumberAttr, firstStringAttr, flattenOtlpExportToNdjson, inferDomainKeywords, inferOtlpKind, iterateRawCalls, otelRunCompleteHook, otlpToRunRecords, otlpToTraceRunRecords, planTraceInsightQuestions, projectOtlpFlatLine, readOtlpStatus, redactString, redactValue, scoreTraceInsightReadiness, stringField, tokenizeDomainWords, traceAnalystFunctionGroup, traceAnalystOnRunComplete, traceSpanKindToOpenInferenceKind };
|
package/dist/traces.js
CHANGED
|
@@ -28,13 +28,14 @@ import {
|
|
|
28
28
|
scoreTraceInsightReadiness,
|
|
29
29
|
tokenizeDomainWords,
|
|
30
30
|
traceAnalystOnRunComplete
|
|
31
|
-
} from "./chunk-
|
|
31
|
+
} from "./chunk-YZPO4UHR.js";
|
|
32
32
|
import {
|
|
33
33
|
FAILURE_CLASSES,
|
|
34
34
|
TRACE_SCHEMA_VERSION,
|
|
35
35
|
aggregateLlm,
|
|
36
36
|
argHash,
|
|
37
37
|
groupBy,
|
|
38
|
+
hasCapturedToolArgs,
|
|
38
39
|
isJudgeSpan,
|
|
39
40
|
isLlmSpan,
|
|
40
41
|
isRetrievalSpan,
|
|
@@ -45,13 +46,13 @@ import {
|
|
|
45
46
|
runFailureClass,
|
|
46
47
|
runsForScenario,
|
|
47
48
|
toolSpans
|
|
48
|
-
} from "./chunk-
|
|
49
|
+
} from "./chunk-LQUTGLOZ.js";
|
|
49
50
|
import {
|
|
50
51
|
TRACE_ANALYST_ACTOR_DESCRIPTION,
|
|
51
52
|
TRACE_ANALYST_ACTOR_DESCRIPTION_VERSION,
|
|
52
53
|
TRACE_ANALYST_SUBAGENT_DESCRIPTION,
|
|
53
54
|
analyzeTraces
|
|
54
|
-
} from "./chunk-
|
|
55
|
+
} from "./chunk-4JLWXDYA.js";
|
|
55
56
|
import {
|
|
56
57
|
DEFAULT_REDACTION_RULES,
|
|
57
58
|
REDACTION_VERSION,
|
|
@@ -60,6 +61,7 @@ import {
|
|
|
60
61
|
} from "./chunk-GGE4NNQT.js";
|
|
61
62
|
import {
|
|
62
63
|
DEFAULT_TRACE_ANALYST_BUDGETS,
|
|
64
|
+
INPUT_VALUE,
|
|
63
65
|
LLM_CACHED_TOKENS,
|
|
64
66
|
LLM_CACHED_TOKEN_ATTR_KEYS,
|
|
65
67
|
LLM_COST_ATTR_KEYS,
|
|
@@ -71,14 +73,18 @@ import {
|
|
|
71
73
|
LLM_OUTPUT_TOKENS,
|
|
72
74
|
LLM_OUTPUT_TOKEN_ATTR_KEYS,
|
|
73
75
|
OPENINFERENCE_SPAN_KIND,
|
|
76
|
+
OUTPUT_VALUE,
|
|
74
77
|
OtlpFileTraceStore,
|
|
75
78
|
SPAN_KIND_ATTR_KEYS,
|
|
76
79
|
SpanNotFoundError,
|
|
80
|
+
TOOL_ARGS_CAPTURED,
|
|
81
|
+
TOOL_LATENCY_MS,
|
|
77
82
|
TOOL_NAME,
|
|
78
83
|
TOOL_NAME_ATTR_KEYS,
|
|
79
84
|
TRACE_ANALYST_TRUNCATION_MARKER_PREFIX,
|
|
80
85
|
TraceFileMissingError,
|
|
81
86
|
TraceNotFoundError,
|
|
87
|
+
applyToolSpanOtlpAttributes,
|
|
82
88
|
asNumber,
|
|
83
89
|
asString,
|
|
84
90
|
buildTraceAnalystTools,
|
|
@@ -91,7 +97,7 @@ import {
|
|
|
91
97
|
stringField,
|
|
92
98
|
traceAnalystFunctionGroup,
|
|
93
99
|
traceSpanKindToOpenInferenceKind
|
|
94
|
-
} from "./chunk-
|
|
100
|
+
} from "./chunk-S2F4J57L.js";
|
|
95
101
|
import {
|
|
96
102
|
RunIntegrityError,
|
|
97
103
|
assertRunCaptured,
|
|
@@ -119,6 +125,7 @@ export {
|
|
|
119
125
|
FAILURE_CLASSES,
|
|
120
126
|
FileSystemRawProviderSink,
|
|
121
127
|
FileSystemTraceStore,
|
|
128
|
+
INPUT_VALUE,
|
|
122
129
|
InMemoryRawProviderSink,
|
|
123
130
|
InMemoryTraceStore,
|
|
124
131
|
LLM_CACHED_TOKENS,
|
|
@@ -134,6 +141,7 @@ export {
|
|
|
134
141
|
NoopRawProviderSink,
|
|
135
142
|
OPENINFERENCE_SPAN_KIND,
|
|
136
143
|
OTEL_AGENT_EVAL_SCOPE,
|
|
144
|
+
OUTPUT_VALUE,
|
|
137
145
|
OtlpFileTraceStore,
|
|
138
146
|
REDACTION_VERSION,
|
|
139
147
|
ReplayCache,
|
|
@@ -141,6 +149,8 @@ export {
|
|
|
141
149
|
RunIntegrityError,
|
|
142
150
|
SPAN_KIND_ATTR_KEYS,
|
|
143
151
|
SpanNotFoundError,
|
|
152
|
+
TOOL_ARGS_CAPTURED,
|
|
153
|
+
TOOL_LATENCY_MS,
|
|
144
154
|
TOOL_NAME,
|
|
145
155
|
TOOL_NAME_ATTR_KEYS,
|
|
146
156
|
TRACE_ANALYST_ACTOR_DESCRIPTION,
|
|
@@ -153,6 +163,7 @@ export {
|
|
|
153
163
|
TraceNotFoundError,
|
|
154
164
|
aggregateLlm,
|
|
155
165
|
analyzeTraces,
|
|
166
|
+
applyToolSpanOtlpAttributes,
|
|
156
167
|
argHash,
|
|
157
168
|
asNumber,
|
|
158
169
|
asString,
|
|
@@ -178,6 +189,7 @@ export {
|
|
|
178
189
|
firstStringAttr,
|
|
179
190
|
flattenOtlpExportToNdjson,
|
|
180
191
|
groupBy,
|
|
192
|
+
hasCapturedToolArgs,
|
|
181
193
|
inferDomainKeywords,
|
|
182
194
|
inferOtlpKind,
|
|
183
195
|
isJudgeSpan,
|
|
@@ -1,5 +1,7 @@
|
|
|
1
|
-
import { P as PolicyEditCandidateRecord } from './policy-edit-
|
|
2
|
-
import {
|
|
1
|
+
import { P as PolicyEditCandidateRecord } from './policy-edit-wG9uFEFm.js';
|
|
2
|
+
import { R as RunPaidCallInput, a as CostChannel, P as PaidCallResult, C as CostLedger, b as CostLedgerSummary } from './cost-ledger-DWy3XdJc.js';
|
|
3
|
+
import { L as LlmCallMetadata } from './llm-client-qoDd18Qz.js';
|
|
4
|
+
import { c as RunTokenUsage } from './run-record-BDH49H2E.js';
|
|
3
5
|
|
|
4
6
|
/**
|
|
5
7
|
* Pass A substrate types — `runCampaign` is the one primitive every
|
|
@@ -83,12 +85,20 @@ interface JudgeDimension {
|
|
|
83
85
|
interface JudgeConfig<TArtifact, TScenario extends Scenario = Scenario> {
|
|
84
86
|
name: string;
|
|
85
87
|
dimensions: JudgeDimension[];
|
|
88
|
+
/** Stable scoring revision used by campaign resume and verdict caches.
|
|
89
|
+
* Built-in judges derive this from their prompt, model, and rubric. Custom
|
|
90
|
+
* judges should set it when closure state can change without changing code. */
|
|
91
|
+
judgeVersion?: string;
|
|
86
92
|
/** Score one artifact. Throw on failure — a thrown judge is recorded as a
|
|
87
93
|
* failed cell, never silently folded into a zero. */
|
|
88
94
|
score(input: {
|
|
89
95
|
artifact: TArtifact;
|
|
90
96
|
scenario: TScenario;
|
|
91
97
|
signal: AbortSignal;
|
|
98
|
+
/** Shared run spend account and receipt attribution phase. */
|
|
99
|
+
costLedger?: CostLedger;
|
|
100
|
+
costPhase?: string;
|
|
101
|
+
costTags?: Record<string, string>;
|
|
92
102
|
}): JudgeScore | Promise<JudgeScore>;
|
|
93
103
|
appliesTo?: (scenario: TScenario) => boolean;
|
|
94
104
|
}
|
|
@@ -105,6 +115,8 @@ interface JudgeScore {
|
|
|
105
115
|
dimensions: Record<string, number>;
|
|
106
116
|
composite: number;
|
|
107
117
|
notes: string;
|
|
118
|
+
/** Provider metadata for display and diagnostics; accounting uses CostLedger receipts. */
|
|
119
|
+
llmCall?: LlmCallMetadata;
|
|
108
120
|
/** Set when the judge itself failed (call error, unparseable output).
|
|
109
121
|
* `composite`/`dimensions` carry no signal — aggregators MUST exclude
|
|
110
122
|
* failed scores from means instead of folding them into zeros. */
|
|
@@ -270,6 +282,9 @@ interface ProposeContext<TFindings = unknown> {
|
|
|
270
282
|
* scenarios) into a merged candidate. Proposers doing pure single-parent
|
|
271
283
|
* reflection may ignore it. See {@link ParetoParent}. */
|
|
272
284
|
paretoParents?: ParetoParent[];
|
|
285
|
+
/** Shared run spend account and receipt attribution phase. */
|
|
286
|
+
costLedger?: CostLedger;
|
|
287
|
+
costPhase?: string;
|
|
273
288
|
/** FIREWALL (non-negotiable): the held-out judge is write-only — its verdicts
|
|
274
289
|
* score the chosen output and gate promotion, and are NEVER an input to
|
|
275
290
|
* proposal/steering (else the optimizer games the acceptance axis = an
|
|
@@ -351,6 +366,9 @@ interface GateContext<TArtifact, TScenario extends Scenario> {
|
|
|
351
366
|
candidate: number;
|
|
352
367
|
baseline: number;
|
|
353
368
|
};
|
|
369
|
+
/** Shared run spend account and receipt attribution phase. */
|
|
370
|
+
costLedger?: CostLedger;
|
|
371
|
+
costPhase?: string;
|
|
354
372
|
signal: AbortSignal;
|
|
355
373
|
}
|
|
356
374
|
interface GateResult {
|
|
@@ -389,35 +407,15 @@ interface CampaignArtifactWriter {
|
|
|
389
407
|
* backend-integrity guard with ONE source of truth — a field added to
|
|
390
408
|
* `RunTokenUsage` is a compile error here, not a silent drift. */
|
|
391
409
|
type CampaignTokenUsage = RunTokenUsage;
|
|
392
|
-
/** Cell-scoped
|
|
393
|
-
*
|
|
394
|
-
* tokens
|
|
395
|
-
*
|
|
396
|
-
* neither yields a `{cost:0, tokens:0}` cell, which the backend-integrity
|
|
397
|
-
* guard (`assertRealBackend`) correctly reads as a stub. Also use `observe`
|
|
398
|
-
* for non-LLM spend (sandbox time, tool costs). */
|
|
410
|
+
/** Cell-scoped paid-call entry point. The dispatch places every paid operation
|
|
411
|
+
* inside `runPaidCall`; the returned provider result supplies one receipt with
|
|
412
|
+
* cost, tokens, and resolved model. Calls made outside this method are not
|
|
413
|
+
* admitted or captured. */
|
|
399
414
|
interface CampaignCostMeter {
|
|
400
|
-
|
|
401
|
-
|
|
402
|
-
|
|
403
|
-
|
|
404
|
-
* `tokenUsage`, so a cell that never reports tokens reads as a stub. Any
|
|
405
|
-
* dispatch that calls an LLM MUST report its usage. */
|
|
406
|
-
observeTokens(usage: CampaignTokenUsage): void;
|
|
407
|
-
/** Record the concrete model the backend RESOLVED this cell to at runtime.
|
|
408
|
-
* The substrate cannot see the LLM call, so it cannot know which model a
|
|
409
|
-
* vendor-locked harness actually served — only the dispatch, reading the
|
|
410
|
-
* backend's usage/terminal events, can. A dispatch whose profile declares a
|
|
411
|
-
* runtime-resolved model (the `HARNESS_NATIVE_MODEL` sentinel) MUST report
|
|
412
|
-
* the resolved, snapshot-bearing id here so the RunRecord pins a real model
|
|
413
|
-
* instead of the sentinel. Last write wins (a cell issues one logical run);
|
|
414
|
-
* optional because most dispatches declare a concrete model up front. */
|
|
415
|
-
observeModel?(model: string): void;
|
|
416
|
-
current(): number;
|
|
417
|
-
/** Accumulated token usage for this cell (zeros if never observed). */
|
|
418
|
-
tokens(): CampaignTokenUsage;
|
|
419
|
-
/** The runtime-resolved model reported via `observeModel`, if any. */
|
|
420
|
-
resolvedModel?(): string | undefined;
|
|
415
|
+
/** The only paid-call path. Returns a typed result; callers must inspect it. */
|
|
416
|
+
runPaidCall<T>(input: Omit<RunPaidCallInput<T>, 'channel' | 'phase' | 'tags'> & {
|
|
417
|
+
channel?: CostChannel;
|
|
418
|
+
}): Promise<PaidCallResult<T>>;
|
|
421
419
|
}
|
|
422
420
|
/** Source tag — required on every store write. Used by the
|
|
423
421
|
* default training-source filter (production-trace samples NOT used as
|
|
@@ -510,15 +508,16 @@ interface CampaignCellResult<TArtifact> {
|
|
|
510
508
|
artifact: TArtifact;
|
|
511
509
|
judgeScores: Record<string, JudgeScore>;
|
|
512
510
|
costUsd: number;
|
|
513
|
-
/**
|
|
514
|
-
|
|
515
|
-
|
|
511
|
+
/** True when at least one priced receipt used the model table instead of a provider bill. */
|
|
512
|
+
costEstimated?: boolean;
|
|
513
|
+
/** Exact durable receipts required to reuse this cached result. */
|
|
514
|
+
costCallIds?: string[];
|
|
515
|
+
/** Agent-call token usage committed by `ctx.cost.runPaidCall`.
|
|
516
|
+
* `{ input: 0, output: 0 }` when no paid agent call was recorded. */
|
|
516
517
|
tokenUsage: CampaignTokenUsage;
|
|
517
|
-
/**
|
|
518
|
-
*
|
|
519
|
-
*
|
|
520
|
-
* need to. Consumed by `buildRunRecord` to pin the real model when the
|
|
521
|
-
* declared model is the `HARNESS_NATIVE_MODEL` sentinel. */
|
|
518
|
+
/** Concrete model from the latest committed agent receipt. Consumed by
|
|
519
|
+
* `buildRunRecord` to pin the model when the declared profile uses a
|
|
520
|
+
* runtime-resolved sentinel. */
|
|
522
521
|
resolvedModel?: string;
|
|
523
522
|
durationMs: number;
|
|
524
523
|
seed: number;
|
|
@@ -599,6 +598,9 @@ interface GenerationCandidate {
|
|
|
599
598
|
interface CampaignAggregates {
|
|
600
599
|
byJudge: Record<string, JudgeAggregate>;
|
|
601
600
|
byScenario: Record<string, ScenarioAggregate>;
|
|
601
|
+
/** Canonical campaign accounting, including worker and judge calls. */
|
|
602
|
+
cost: CostLedgerSummary;
|
|
603
|
+
/** Compatibility alias of `cost.totalCostUsd`. */
|
|
602
604
|
totalCostUsd: number;
|
|
603
605
|
cellsExecuted: number;
|
|
604
606
|
cellsSkipped: number;
|
|
@@ -629,4 +631,4 @@ interface CampaignResult<TArtifact = unknown, TScenario extends Scenario = Scena
|
|
|
629
631
|
scenarios: Array<Pick<TScenario, 'id' | 'kind'>>;
|
|
630
632
|
}
|
|
631
633
|
|
|
632
|
-
export { type JudgeAggregate as A, type ScenarioAggregate as B, type CampaignResult as C, type DispatchContext as D, type ScoredSurfaceOutcome as E, isProposedCandidate as F, type
|
|
634
|
+
export { type JudgeAggregate as A, type ScenarioAggregate as B, type CampaignResult as C, type DispatchContext as D, type ScoredSurfaceOutcome as E, isProposedCandidate as F, type Gate as G, labelTrustRank as H, type JudgeScore as J, type LabeledScenarioStore as L, type MutableSurface as M, type OptimizationProposer as O, type ParetoParent as P, type RedactionStatus as R, type Scenario as S, type TraceSpan as T, type JudgeDimension as a, type JudgeConfig as b, type SurfaceProposer as c, type GateDecision as d, type CampaignAggregates as e, type CampaignArtifactWriter as f, type CampaignCellResult as g, type CampaignCostMeter as h, type CampaignTraceWriter as i, type CodeSurface as j, type DispatchFn as k, type GateContext as l, type GateResult as m, type GenerationCandidate as n, type GenerationRecord as o, type Mutator as p, type OptimizerConfig as q, type SessionScript as r, type LabeledScenarioWrite as s, type LabeledScenarioSampleArgs as t, type LabeledScenarioRecord as u, type LabelTrust as v, type ProposedCandidate as w, type ProposeContext as x, type LabeledScenarioSource as y, type CampaignTokenUsage as z };
|
|
@@ -1,3 +1,4 @@
|
|
|
1
|
+
import { C as CostLedger, b as CostLedgerSummary } from './cost-ledger-DWy3XdJc.js';
|
|
1
2
|
import { TCloud } from '@tangle-network/tcloud';
|
|
2
3
|
|
|
3
4
|
interface Scenario {
|
|
@@ -50,6 +51,8 @@ interface ScenarioResult {
|
|
|
50
51
|
overallScore: number;
|
|
51
52
|
totalDurationMs: number;
|
|
52
53
|
artifacts: CollectedArtifacts;
|
|
54
|
+
/** Agent and judge spend attributed to this scenario. */
|
|
55
|
+
cost?: CostLedgerSummary;
|
|
53
56
|
}
|
|
54
57
|
interface TurnResult {
|
|
55
58
|
turnIndex: number;
|
|
@@ -96,6 +99,7 @@ interface BenchmarkReport {
|
|
|
96
99
|
promptVersion: string;
|
|
97
100
|
scenarioCount: number;
|
|
98
101
|
results: ScenarioResult[];
|
|
102
|
+
cost?: CostLedgerSummary;
|
|
99
103
|
summary: {
|
|
100
104
|
overallAvg: number;
|
|
101
105
|
byPersona: Record<string, {
|
|
@@ -266,11 +270,22 @@ interface BenchmarkRunnerConfig {
|
|
|
266
270
|
passThreshold?: number;
|
|
267
271
|
generation?: number;
|
|
268
272
|
promptVersion?: string;
|
|
273
|
+
/** Shared ledger for agent and judge calls made by the benchmark. */
|
|
274
|
+
costLedger?: CostLedger;
|
|
275
|
+
/** Exact maximum provider attempts configured on the supplied TCloud client. */
|
|
276
|
+
tcloudMaximumAttempts?: number;
|
|
269
277
|
}
|
|
270
278
|
interface JudgeInput {
|
|
271
279
|
scenario: Scenario;
|
|
272
280
|
turns: TurnResult[];
|
|
273
281
|
artifacts: CollectedArtifacts;
|
|
282
|
+
/** Shared ledger for paid built-in judges. Direct calls default to an uncapped ledger. */
|
|
283
|
+
costLedger?: CostLedger;
|
|
284
|
+
costPhase?: string;
|
|
285
|
+
costTags?: Record<string, string>;
|
|
286
|
+
signal?: AbortSignal;
|
|
287
|
+
/** Exact maximum provider attempts configured on the supplied TCloud client. */
|
|
288
|
+
tcloudMaximumAttempts?: number;
|
|
274
289
|
}
|
|
275
290
|
type JudgeFn = (tc: TCloud, input: JudgeInput) => Promise<JudgeScore[]>;
|
|
276
291
|
|
package/dist/wire/index.d.ts
CHANGED
|
@@ -1,14 +1,17 @@
|
|
|
1
|
-
import {
|
|
2
|
-
import {
|
|
1
|
+
import { C as CostLedger } from '../cost-ledger-DWy3XdJc.js';
|
|
2
|
+
import { F as FeedbackTrajectoryStore } from '../feedback-trajectory-BUnM58xL.js';
|
|
3
|
+
import { a as LlmClientOptions } from '../llm-client-qoDd18Qz.js';
|
|
4
|
+
import { T as TraceStore } from '../store-DGqD0Pyo.js';
|
|
3
5
|
import { z } from 'zod';
|
|
4
6
|
import { OpenAPIObject } from 'openapi3-ts/oas31';
|
|
5
7
|
import * as hono_types from 'hono/types';
|
|
6
8
|
import { ServerType } from '@hono/node-server';
|
|
7
9
|
import { Hono } from 'hono';
|
|
8
|
-
import '../emitter-BRchAAAx.js';
|
|
9
|
-
import '../schema-SGWcK9wa.js';
|
|
10
|
-
import '../dataset-NENEzRgk.js';
|
|
11
10
|
import '../errors-oeQrLqXC.js';
|
|
11
|
+
import '../emitter-CjD7vUwv.js';
|
|
12
|
+
import '../schema-B3Q3l9Z_.js';
|
|
13
|
+
import '../dataset-NENEzRgk.js';
|
|
14
|
+
import '../raw-provider-sink-C46HDghv.js';
|
|
12
15
|
|
|
13
16
|
declare const RubricDimensionSchema: z.ZodObject<{
|
|
14
17
|
id: z.ZodString;
|
|
@@ -122,9 +125,9 @@ declare const TraceEventSchema: z.ZodObject<{
|
|
|
122
125
|
runId: z.ZodString;
|
|
123
126
|
spanId: z.ZodOptional<z.ZodString>;
|
|
124
127
|
kind: z.ZodEnum<{
|
|
125
|
-
policy_violation: "policy_violation";
|
|
126
|
-
custom: "custom";
|
|
127
128
|
error: "error";
|
|
129
|
+
custom: "custom";
|
|
130
|
+
policy_violation: "policy_violation";
|
|
128
131
|
log: "log";
|
|
129
132
|
budget_decrement: "budget_decrement";
|
|
130
133
|
budget_breach: "budget_breach";
|
|
@@ -140,9 +143,9 @@ declare const TracesIngestRequestSchema: z.ZodObject<{
|
|
|
140
143
|
runId: z.ZodString;
|
|
141
144
|
spanId: z.ZodOptional<z.ZodString>;
|
|
142
145
|
kind: z.ZodEnum<{
|
|
143
|
-
policy_violation: "policy_violation";
|
|
144
|
-
custom: "custom";
|
|
145
146
|
error: "error";
|
|
147
|
+
custom: "custom";
|
|
148
|
+
policy_violation: "policy_violation";
|
|
146
149
|
log: "log";
|
|
147
150
|
budget_decrement: "budget_decrement";
|
|
148
151
|
budget_breach: "budget_breach";
|
|
@@ -165,8 +168,8 @@ declare const FeedbackLabelSchema: z.ZodObject<{
|
|
|
165
168
|
id: z.ZodOptional<z.ZodString>;
|
|
166
169
|
source: z.ZodEnum<{
|
|
167
170
|
judge: "judge";
|
|
168
|
-
system: "system";
|
|
169
171
|
user: "user";
|
|
172
|
+
system: "system";
|
|
170
173
|
policy: "policy";
|
|
171
174
|
environment: "environment";
|
|
172
175
|
metric: "metric";
|
|
@@ -198,10 +201,10 @@ declare const FeedbackAttemptSchema: z.ZodObject<{
|
|
|
198
201
|
id: z.ZodString;
|
|
199
202
|
stepIndex: z.ZodNumber;
|
|
200
203
|
artifactType: z.ZodEnum<{
|
|
201
|
-
action: "action";
|
|
202
|
-
decision: "decision";
|
|
203
204
|
text: "text";
|
|
204
205
|
code: "code";
|
|
206
|
+
action: "action";
|
|
207
|
+
decision: "decision";
|
|
205
208
|
plan: "plan";
|
|
206
209
|
research: "research";
|
|
207
210
|
ui: "ui";
|
|
@@ -226,8 +229,8 @@ declare const FeedbackAttemptSchema: z.ZodObject<{
|
|
|
226
229
|
id: z.ZodOptional<z.ZodString>;
|
|
227
230
|
source: z.ZodEnum<{
|
|
228
231
|
judge: "judge";
|
|
229
|
-
system: "system";
|
|
230
232
|
user: "user";
|
|
233
|
+
system: "system";
|
|
231
234
|
policy: "policy";
|
|
232
235
|
environment: "environment";
|
|
233
236
|
metric: "metric";
|
|
@@ -270,10 +273,10 @@ declare const FeedbackTrajectorySchema: z.ZodObject<{
|
|
|
270
273
|
id: z.ZodString;
|
|
271
274
|
stepIndex: z.ZodNumber;
|
|
272
275
|
artifactType: z.ZodEnum<{
|
|
273
|
-
action: "action";
|
|
274
|
-
decision: "decision";
|
|
275
276
|
text: "text";
|
|
276
277
|
code: "code";
|
|
278
|
+
action: "action";
|
|
279
|
+
decision: "decision";
|
|
277
280
|
plan: "plan";
|
|
278
281
|
research: "research";
|
|
279
282
|
ui: "ui";
|
|
@@ -298,8 +301,8 @@ declare const FeedbackTrajectorySchema: z.ZodObject<{
|
|
|
298
301
|
id: z.ZodOptional<z.ZodString>;
|
|
299
302
|
source: z.ZodEnum<{
|
|
300
303
|
judge: "judge";
|
|
301
|
-
system: "system";
|
|
302
304
|
user: "user";
|
|
305
|
+
system: "system";
|
|
303
306
|
policy: "policy";
|
|
304
307
|
environment: "environment";
|
|
305
308
|
metric: "metric";
|
|
@@ -334,8 +337,8 @@ declare const FeedbackTrajectorySchema: z.ZodObject<{
|
|
|
334
337
|
id: z.ZodOptional<z.ZodString>;
|
|
335
338
|
source: z.ZodEnum<{
|
|
336
339
|
judge: "judge";
|
|
337
|
-
system: "system";
|
|
338
340
|
user: "user";
|
|
341
|
+
system: "system";
|
|
339
342
|
policy: "policy";
|
|
340
343
|
environment: "environment";
|
|
341
344
|
metric: "metric";
|
|
@@ -438,7 +441,13 @@ declare class WireError extends Error {
|
|
|
438
441
|
readonly details?: unknown | undefined;
|
|
439
442
|
constructor(code: string, message: string, status?: number, details?: unknown | undefined);
|
|
440
443
|
}
|
|
441
|
-
|
|
444
|
+
interface HandleJudgeOptions {
|
|
445
|
+
costLedger?: CostLedger;
|
|
446
|
+
costPhase?: string;
|
|
447
|
+
llm?: LlmClientOptions;
|
|
448
|
+
signal?: AbortSignal;
|
|
449
|
+
}
|
|
450
|
+
declare function handleJudge(req: JudgeRequest, options?: HandleJudgeOptions): Promise<JudgeResult>;
|
|
442
451
|
declare function handleListRubrics(): ListRubricsResponse;
|
|
443
452
|
declare function handleVersion(): VersionResponse;
|
|
444
453
|
/**
|
|
@@ -569,4 +578,4 @@ interface StartedServer {
|
|
|
569
578
|
*/
|
|
570
579
|
declare function startServerAsync(opts?: ServeOptions): Promise<StartedServer>;
|
|
571
580
|
|
|
572
|
-
export { BUILTIN_RUBRICS, type ErrorResponse, ErrorResponseSchema, type FailureMode, FailureModeSchema, FeedbackAttemptSchema, type FeedbackIngestResponse, FeedbackIngestResponseSchema, FeedbackLabelSchema, type FeedbackTrajectory, FeedbackTrajectorySchema, HealthResponseSchema, type IngestionStores, type JudgeRequest, JudgeRequestSchema, type JudgeResult, JudgeResultSchema, type ListRubricsResponse, ListRubricsResponseSchema, type Rubric, type RubricDimension, RubricDimensionSchema, type RubricInfo, RubricInfoSchema, RubricSchema, type ServeOptions, type StartedServer, type TraceEvent, TraceEventSchema, type TracesIngestRequest, TracesIngestRequestSchema, type TracesIngestResponse, TracesIngestResponseSchema, type VersionResponse, VersionResponseSchema, WIRE_VERSION, WireError, buildOpenApi, createApp, dispatchRpc, getBuiltinRubric, handleFeedbackIngest, handleJudge, handleListRubrics, handleTracesIngest, handleVersion, hashRubric, listBuiltinRubrics, runRpcBatch, runRpcOnce, startServer, startServerAsync };
|
|
581
|
+
export { BUILTIN_RUBRICS, type ErrorResponse, ErrorResponseSchema, type FailureMode, FailureModeSchema, FeedbackAttemptSchema, type FeedbackIngestResponse, FeedbackIngestResponseSchema, FeedbackLabelSchema, type FeedbackTrajectory, FeedbackTrajectorySchema, type HandleJudgeOptions, HealthResponseSchema, type IngestionStores, type JudgeRequest, JudgeRequestSchema, type JudgeResult, JudgeResultSchema, type ListRubricsResponse, ListRubricsResponseSchema, type Rubric, type RubricDimension, RubricDimensionSchema, type RubricInfo, RubricInfoSchema, RubricSchema, type ServeOptions, type StartedServer, type TraceEvent, TraceEventSchema, type TracesIngestRequest, TracesIngestRequestSchema, type TracesIngestResponse, TracesIngestResponseSchema, type VersionResponse, VersionResponseSchema, WIRE_VERSION, WireError, buildOpenApi, createApp, dispatchRpc, getBuiltinRubric, handleFeedbackIngest, handleJudge, handleListRubrics, handleTracesIngest, handleVersion, hashRubric, listBuiltinRubrics, runRpcBatch, runRpcOnce, startServer, startServerAsync };
|
package/dist/wire/index.js
CHANGED
|
@@ -34,8 +34,10 @@ import {
|
|
|
34
34
|
runRpcOnce,
|
|
35
35
|
startServer,
|
|
36
36
|
startServerAsync
|
|
37
|
-
} from "../chunk-
|
|
38
|
-
import "../chunk-
|
|
37
|
+
} from "../chunk-LTVG32KX.js";
|
|
38
|
+
import "../chunk-NJC7U437.js";
|
|
39
|
+
import "../chunk-VCTY3W6J.js";
|
|
40
|
+
import "../chunk-VI2UW6B6.js";
|
|
39
41
|
import "../chunk-PC4UYEBM.js";
|
|
40
42
|
import "../chunk-ONWEPEDO.js";
|
|
41
43
|
import "../chunk-PZ5AY32C.js";
|
|
@@ -135,7 +135,7 @@ round-robin, region-affinity from a previous run, scheduling table).
|
|
|
135
135
|
| **Auth** | Bearer token on `Authorization`; pluggable via `auth: string \| () => string \| Promise<string>` for rotation/refresh. |
|
|
136
136
|
| **Payload size** | Server enforces `maxBodyBytes` (default 10 MB). |
|
|
137
137
|
| **Traces** | Both ends emit OTel — if both point at the same OTLP collector, you get a unified trace per cell. See `docs/adapters-observability.md`. |
|
|
138
|
-
| **Cost** | Worker's `ctx.cost.
|
|
138
|
+
| **Cost** | Worker's `ctx.cost.runPaidCall(...)` writes durable receipts in the worker process. Roll up those receipts server-side and attach them to worker telemetry; they are not forwarded to the coordinator automatically. |
|
|
139
139
|
|
|
140
140
|
## Running the reference example
|
|
141
141
|
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@tangle-network/agent-eval",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.117.0",
|
|
4
4
|
"description": "Evaluate and improve AI agents from runs, traces, judges, and feedback. Compare candidates, cluster failures, measure lift, and gate releases.",
|
|
5
5
|
"homepage": "https://github.com/tangle-network/agent-eval#readme",
|
|
6
6
|
"repository": {
|
|
@@ -148,7 +148,7 @@
|
|
|
148
148
|
"@hono/node-server": "^2.0.0",
|
|
149
149
|
"@tangle-network/agent-interface": "^0.22.0",
|
|
150
150
|
"@tangle-network/tcloud": "^0.4.14",
|
|
151
|
-
"hono": "^4.12.
|
|
151
|
+
"hono": "^4.12.30",
|
|
152
152
|
"zod": "^4.3.6"
|
|
153
153
|
},
|
|
154
154
|
"devDependencies": {
|
|
@@ -165,7 +165,7 @@
|
|
|
165
165
|
"minimumReleaseAge": 4320,
|
|
166
166
|
"overrides": {
|
|
167
167
|
"postcss@<8.5.10": "^8.5.10",
|
|
168
|
-
"ws@>=8.0.0 <8.
|
|
168
|
+
"ws@>=8.0.0 <8.21.0": "^8.21.0"
|
|
169
169
|
}
|
|
170
170
|
},
|
|
171
171
|
"engines": {
|