@tangle-network/agent-eval 0.116.0 → 0.117.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (135) hide show
  1. package/CHANGELOG.md +31 -0
  2. package/dist/analyst/index.d.ts +18 -11
  3. package/dist/analyst/index.js +10 -7
  4. package/dist/analyst/index.js.map +1 -1
  5. package/dist/{analyst-CFBc14Wc.d.ts → analyst-C8HHvfJp.d.ts} +1 -1
  6. package/dist/{analyze-runs-0rz_m29H.d.ts → analyze-runs--2x39HZ7.d.ts} +3 -3
  7. package/dist/{baseline-DsNteOgR.d.ts → baseline-DKq3gJpP.d.ts} +6 -3
  8. package/dist/belief-state/index.d.ts +6 -6
  9. package/dist/belief-state/index.js +1 -1
  10. package/dist/benchmarks/index.d.ts +11 -8
  11. package/dist/benchmarks/index.js +11 -10
  12. package/dist/builder-eval/index.d.ts +4 -4
  13. package/dist/builder-eval/index.js +1 -1
  14. package/dist/{calibration-Dz8TQV4y.d.ts → calibration-C8MTS7cw.d.ts} +2 -2
  15. package/dist/campaign/index.d.ts +53 -29
  16. package/dist/campaign/index.js +18 -13
  17. package/dist/chunk-3YYRZDON.js +45 -0
  18. package/dist/chunk-3YYRZDON.js.map +1 -0
  19. package/dist/{chunk-RPDDVKI7.js → chunk-4JLWXDYA.js} +2 -2
  20. package/dist/{chunk-NBSS5NDZ.js → chunk-CCZIVI3F.js} +54 -115
  21. package/dist/chunk-CCZIVI3F.js.map +1 -0
  22. package/dist/{chunk-J6P6PK2R.js → chunk-FQNLDL4D.js} +3 -3
  23. package/dist/{chunk-ONM6PEAE.js → chunk-GQCZRZ7L.js} +2 -2
  24. package/dist/chunk-HHWE3POT.js +94 -0
  25. package/dist/chunk-HHWE3POT.js.map +1 -0
  26. package/dist/{chunk-3274WNK7.js → chunk-HQPHZGL6.js} +687 -44
  27. package/dist/chunk-HQPHZGL6.js.map +1 -0
  28. package/dist/{chunk-FAOEFFRT.js → chunk-IDZTTFRR.js} +390 -78
  29. package/dist/chunk-IDZTTFRR.js.map +1 -0
  30. package/dist/{chunk-3LXTCTWL.js → chunk-JSDVRFAP.js} +2 -2
  31. package/dist/{chunk-MHNQWM4I.js → chunk-LQUTGLOZ.js} +5 -1
  32. package/dist/chunk-LQUTGLOZ.js.map +1 -0
  33. package/dist/{chunk-4D5RVB3W.js → chunk-LTVG32KX.js} +30 -5
  34. package/dist/chunk-LTVG32KX.js.map +1 -0
  35. package/dist/{chunk-CIUOICJT.js → chunk-MGEHEHSN.js} +62 -15
  36. package/dist/chunk-MGEHEHSN.js.map +1 -0
  37. package/dist/{chunk-GY4SYVPJ.js → chunk-NJC7U437.js} +97 -25
  38. package/dist/chunk-NJC7U437.js.map +1 -0
  39. package/dist/{chunk-NYFUT3B3.js → chunk-ODVOOEWQ.js} +31 -10
  40. package/dist/chunk-ODVOOEWQ.js.map +1 -0
  41. package/dist/{chunk-LNQEP766.js → chunk-S2F4J57L.js} +44 -4
  42. package/dist/chunk-S2F4J57L.js.map +1 -0
  43. package/dist/chunk-VCTY3W6J.js +798 -0
  44. package/dist/chunk-VCTY3W6J.js.map +1 -0
  45. package/dist/chunk-VF3XSYTI.js +545 -0
  46. package/dist/chunk-VF3XSYTI.js.map +1 -0
  47. package/dist/{chunk-TLDB7WRY.js → chunk-YZPO4UHR.js} +28 -31
  48. package/dist/chunk-YZPO4UHR.js.map +1 -0
  49. package/dist/{chunk-GSW3OBHK.js → chunk-ZUXV7UWZ.js} +350 -724
  50. package/dist/chunk-ZUXV7UWZ.js.map +1 -0
  51. package/dist/cli.js +4 -2
  52. package/dist/cli.js.map +1 -1
  53. package/dist/{code-agent-session-CdxteG0y.d.ts → code-agent-session-CjZsVd19.d.ts} +1 -1
  54. package/dist/contract/index.d.ts +43 -29
  55. package/dist/contract/index.js +56 -19
  56. package/dist/contract/index.js.map +1 -1
  57. package/dist/{control-DbcDxouY.d.ts → control-6vuGfmDH.d.ts} +5 -5
  58. package/dist/control.d.ts +6 -6
  59. package/dist/cost-ledger-DWy3XdJc.d.ts +183 -0
  60. package/dist/{default-registry-DDfv22MQ.d.ts → default-registry-DaK8b3fv.d.ts} +2 -2
  61. package/dist/{emitter-BRchAAAx.d.ts → emitter-CjD7vUwv.d.ts} +2 -2
  62. package/dist/{failure-cluster-C48PiReX.d.ts → failure-cluster-DOAcSJ87.d.ts} +2 -2
  63. package/dist/{feedback-trajectory-pDcz1lQ1.d.ts → feedback-trajectory-BUnM58xL.d.ts} +3 -3
  64. package/dist/fuzz.d.ts +8 -16
  65. package/dist/fuzz.js +72 -42
  66. package/dist/fuzz.js.map +1 -1
  67. package/dist/{gepa-CQelRtuC.d.ts → gepa-eESocoDi.d.ts} +56 -6
  68. package/dist/hosted/index.d.ts +13 -10
  69. package/dist/{index-DbCXJfZ1.d.ts → index-PdX4VnPA.d.ts} +3 -3
  70. package/dist/index.d.ts +102 -57
  71. package/dist/index.js +328 -235
  72. package/dist/index.js.map +1 -1
  73. package/dist/{insight-report-oMVxDTxl.d.ts → insight-report-DY4nDW9Q.d.ts} +1 -1
  74. package/dist/{integrity-C6PZ73iC.d.ts → integrity-DqlBiLyK.d.ts} +2 -2
  75. package/dist/{kind-factory-DWOvXjR_.d.ts → kind-factory-ClZmO25A.d.ts} +2 -2
  76. package/dist/llm-client-qoDd18Qz.d.ts +289 -0
  77. package/dist/meta-eval/index.d.ts +8 -7
  78. package/dist/meta-eval/index.js +1 -1
  79. package/dist/multishot/index.d.ts +9 -6
  80. package/dist/openapi.json +1 -1
  81. package/dist/pipelines/index.d.ts +16 -6
  82. package/dist/pipelines/index.js +119 -23
  83. package/dist/pipelines/index.js.map +1 -1
  84. package/dist/{policy-edit-Clb2v6Oa.d.ts → policy-edit-wG9uFEFm.d.ts} +13 -266
  85. package/dist/{pre-registration--vU0mMtD.d.ts → pre-registration-BWQhJ3vz.d.ts} +24 -5
  86. package/dist/{provenance-BbVagC68.d.ts → provenance-DpjwyseI.d.ts} +6 -6
  87. package/dist/{query-Ck190MOd.d.ts → query-CF7PG61p.d.ts} +5 -3
  88. package/dist/raw-provider-sink-C46HDghv.d.ts +132 -0
  89. package/dist/{release-report-CamNDe90.d.ts → release-report-C8G2i5Xi.d.ts} +2 -2
  90. package/dist/reporting.d.ts +10 -9
  91. package/dist/{researcher-Dwbo_Fxx.d.ts → researcher-C8XyxQsu.d.ts} +8 -8
  92. package/dist/rl.d.ts +18 -15
  93. package/dist/rl.js +2 -2
  94. package/dist/{rubric-predictive-validity-BIdf9h4R.d.ts → rubric-predictive-validity-p49lLVrE.d.ts} +1 -1
  95. package/dist/{run-campaign-UADIM77S.js → run-campaign-IM26A6PD.js} +4 -2
  96. package/dist/{run-record-CZmcpWPo.d.ts → run-record-BDH49H2E.d.ts} +1 -1
  97. package/dist/{runtime-trajectory-CC0jx9ql.d.ts → runtime-trajectory-DGBIUt4B.d.ts} +1 -1
  98. package/dist/{schema-SGWcK9wa.d.ts → schema-B3Q3l9Z_.d.ts} +2 -0
  99. package/dist/{semantic-concept-judge-CKjePUMh.d.ts → semantic-concept-judge-CXnPEJbf.d.ts} +24 -6
  100. package/dist/{statistics-oUbOJe-S.d.ts → statistics-KUnG73jH.d.ts} +1 -1
  101. package/dist/{storage-Dw_f7WMt.d.ts → storage-DrX3v_5B.d.ts} +12 -1
  102. package/dist/{store-9cAScOcb.d.ts → store-C1YxJDEK.d.ts} +1 -132
  103. package/dist/{store-BsVi7ncX.d.ts → store-DGqD0Pyo.d.ts} +1 -1
  104. package/dist/storyboard/index.d.ts +1 -1
  105. package/dist/{summary-report-DTNgQycC.d.ts → summary-report-C5bKFfm-.d.ts} +2 -2
  106. package/dist/{test-graded-scenario-mzYBKspu.d.ts → test-graded-scenario-B0ybnPY7.d.ts} +3 -3
  107. package/dist/traces.d.ts +25 -14
  108. package/dist/traces.js +16 -4
  109. package/dist/{types-Ca_63YSD.d.ts → types-BSw1rOUB.d.ts} +41 -39
  110. package/dist/{types-C7DGg5ex.d.ts → types-BkfcQnxV.d.ts} +15 -0
  111. package/dist/wire/index.d.ts +28 -19
  112. package/dist/wire/index.js +4 -2
  113. package/docs/distributed-driver.md +1 -1
  114. package/package.json +3 -3
  115. package/dist/chunk-3274WNK7.js.map +0 -1
  116. package/dist/chunk-4D5RVB3W.js.map +0 -1
  117. package/dist/chunk-7GKEAIAD.js +0 -205
  118. package/dist/chunk-7GKEAIAD.js.map +0 -1
  119. package/dist/chunk-CIUOICJT.js.map +0 -1
  120. package/dist/chunk-FAOEFFRT.js.map +0 -1
  121. package/dist/chunk-GSW3OBHK.js.map +0 -1
  122. package/dist/chunk-GY4SYVPJ.js.map +0 -1
  123. package/dist/chunk-LNQEP766.js.map +0 -1
  124. package/dist/chunk-MHNQWM4I.js.map +0 -1
  125. package/dist/chunk-MPHTT5HE.js +0 -74
  126. package/dist/chunk-MPHTT5HE.js.map +0 -1
  127. package/dist/chunk-NBSS5NDZ.js.map +0 -1
  128. package/dist/chunk-NYFUT3B3.js.map +0 -1
  129. package/dist/chunk-TLDB7WRY.js.map +0 -1
  130. package/dist/cost-ledger-DuSqlw5B.d.ts +0 -113
  131. /package/dist/{chunk-RPDDVKI7.js.map → chunk-4JLWXDYA.js.map} +0 -0
  132. /package/dist/{chunk-J6P6PK2R.js.map → chunk-FQNLDL4D.js.map} +0 -0
  133. /package/dist/{chunk-ONM6PEAE.js.map → chunk-GQCZRZ7L.js.map} +0 -0
  134. /package/dist/{chunk-3LXTCTWL.js.map → chunk-JSDVRFAP.js.map} +0 -0
  135. /package/dist/{run-campaign-UADIM77S.js.map → run-campaign-IM26A6PD.js.map} +0 -0
package/dist/traces.d.ts CHANGED
@@ -1,21 +1,24 @@
1
1
  import { N as NotFoundError, R as ReplayError } from './errors-oeQrLqXC.js';
2
- import { m as TraceAnalystSpanKind, n as TraceAnalystSpanStatus, T as TraceAnalysisStore, l as TraceAnalystFilters, b as DatasetOverview, Q as QueryTracesPage, q as ViewTraceResult, V as ViewSpansResult, h as SearchTraceResult, S as SearchSpanResult, P as ProviderRedactor, R as RawProviderSink, f as RawProviderEvent } from './store-9cAScOcb.js';
3
- export { D as DEFAULT_TRACE_ANALYST_BUDGETS, E as ErrorCluster, F as FileSystemRawProviderSink, c as FileSystemRawProviderSinkOptions, I as InMemoryRawProviderSink, d as InMemoryRawProviderSinkOptions, N as NoopRawProviderSink, e as RawProviderDirection, g as RawProviderSinkFilter, i as SpanMatchRecord, j as TRACE_ANALYST_TRUNCATION_MARKER_PREFIX, k as TraceAnalystByteBudgets, a as TraceAnalystSpan, o as TraceAnalystTraceSummary, p as ViewTraceOversized, r as defaultProviderRedactor, s as providerFromBaseUrl } from './store-9cAScOcb.js';
4
- import { a as RunCompleteHookContext, R as RunCompleteHook } from './emitter-BRchAAAx.js';
5
- export { S as SpanHandle, T as TraceEmitter, b as TraceEmitterOptions, l as llmSpanFromProvider } from './emitter-BRchAAAx.js';
6
- export { b as RunIntegrityError, R as RunIntegrityExpectations, c as RunIntegrityIssue, d as RunIntegrityIssueCode, a as RunIntegrityReport, e as assertRunCaptured, t as throwIfRunIncomplete } from './integrity-C6PZ73iC.js';
7
- import { T as TraceStore } from './store-BsVi7ncX.js';
8
- export { E as EventFilter, F as FileSystemTraceStore, a as FileSystemTraceStoreOptions, I as InMemoryTraceStore, R as RunFilter, S as SpanFilter } from './store-BsVi7ncX.js';
9
- export { a as aggregateLlm, b as argHash, g as groupBy, j as judgeSpans, l as llmSpans, r as runFailureClass, c as runsForScenario, t as toolSpans } from './query-Ck190MOd.js';
10
- import { R as Run } from './schema-SGWcK9wa.js';
11
- export { A as Artifact, B as BudgetLedgerEntry, h as BudgetSpec, E as EventKind, i as FAILURE_CLASSES, F as FailureClass, G as GenericSpan, J as JudgeSpan, L as LlmSpan, M as Message, c as RetrievalSpan, g as RunLayer, a as RunOutcome, f as RunStatus, d as SandboxSpan, S as Span, j as SpanBase, b as SpanKind, k as SpanStatus, l as TRACE_SCHEMA_VERSION, T as ToolSpan, e as TraceEvent, m as isJudgeSpan, n as isLlmSpan, o as isRetrievalSpan, p as isSandboxSpan, q as isToolSpan } from './schema-SGWcK9wa.js';
12
- import { A as AnalyzeTracesOptions, b as AnalyzeTracesResult } from './analyst-CFBc14Wc.js';
13
- export { a as AnalyzeTracesInput, c as AnalyzeTracesTurnSnapshot, d as analyzeTraces } from './analyst-CFBc14Wc.js';
14
- import { a as RunSplitTag, c as RunTokenUsage, R as RunRecord } from './run-record-CZmcpWPo.js';
2
+ import { P as ProviderRedactor, R as RawProviderSink, d as RawProviderEvent } from './raw-provider-sink-C46HDghv.js';
3
+ export { F as FileSystemRawProviderSink, a as FileSystemRawProviderSinkOptions, I as InMemoryRawProviderSink, b as InMemoryRawProviderSinkOptions, N as NoopRawProviderSink, c as RawProviderDirection, e as RawProviderSinkFilter, f as defaultProviderRedactor, p as providerFromBaseUrl } from './raw-provider-sink-C46HDghv.js';
4
+ import { a as RunCompleteHookContext, R as RunCompleteHook } from './emitter-CjD7vUwv.js';
5
+ export { S as SpanHandle, T as TraceEmitter, b as TraceEmitterOptions, l as llmSpanFromProvider } from './emitter-CjD7vUwv.js';
6
+ export { b as RunIntegrityError, R as RunIntegrityExpectations, c as RunIntegrityIssue, d as RunIntegrityIssueCode, a as RunIntegrityReport, e as assertRunCaptured, t as throwIfRunIncomplete } from './integrity-DqlBiLyK.js';
7
+ import { T as TraceStore } from './store-DGqD0Pyo.js';
8
+ export { E as EventFilter, F as FileSystemTraceStore, a as FileSystemTraceStoreOptions, I as InMemoryTraceStore, R as RunFilter, S as SpanFilter } from './store-DGqD0Pyo.js';
9
+ import { T as ToolSpan, R as Run } from './schema-B3Q3l9Z_.js';
10
+ export { A as Artifact, B as BudgetLedgerEntry, h as BudgetSpec, E as EventKind, i as FAILURE_CLASSES, F as FailureClass, G as GenericSpan, J as JudgeSpan, L as LlmSpan, M as Message, c as RetrievalSpan, g as RunLayer, a as RunOutcome, f as RunStatus, d as SandboxSpan, S as Span, j as SpanBase, b as SpanKind, k as SpanStatus, l as TRACE_SCHEMA_VERSION, e as TraceEvent, m as isJudgeSpan, n as isLlmSpan, o as isRetrievalSpan, p as isSandboxSpan, q as isToolSpan } from './schema-B3Q3l9Z_.js';
11
+ export { a as aggregateLlm, b as argHash, g as groupBy, h as hasCapturedToolArgs, j as judgeSpans, l as llmSpans, r as runFailureClass, c as runsForScenario, t as toolSpans } from './query-CF7PG61p.js';
12
+ import { A as AnalyzeTracesOptions, b as AnalyzeTracesResult } from './analyst-C8HHvfJp.js';
13
+ export { a as AnalyzeTracesInput, c as AnalyzeTracesTurnSnapshot, d as analyzeTraces } from './analyst-C8HHvfJp.js';
14
+ import { h as TraceAnalystSpanKind, i as TraceAnalystSpanStatus, T as TraceAnalysisStore, g as TraceAnalystFilters, b as DatasetOverview, Q as QueryTracesPage, l as ViewTraceResult, V as ViewSpansResult, c as SearchTraceResult, S as SearchSpanResult } from './store-C1YxJDEK.js';
15
+ export { D as DEFAULT_TRACE_ANALYST_BUDGETS, E as ErrorCluster, d as SpanMatchRecord, e as TRACE_ANALYST_TRUNCATION_MARKER_PREFIX, f as TraceAnalystByteBudgets, a as TraceAnalystSpan, j as TraceAnalystTraceSummary, k as ViewTraceOversized } from './store-C1YxJDEK.js';
16
+ import { a as RunSplitTag, c as RunTokenUsage, R as RunRecord } from './run-record-BDH49H2E.js';
15
17
  import { AxFunction } from '@ax-llm/ax';
16
18
  import '@tangle-network/agent-interface';
17
19
 
18
20
  /** Canonical OpenInference-over-OTLP attribute vocabulary used at the trace boundary. */
21
+
19
22
  declare const OPENINFERENCE_SPAN_KIND = "openinference.span.kind";
20
23
  declare const LLM_MODEL_NAME = "llm.model_name";
21
24
  declare const LLM_INPUT_TOKENS = "llm.token_count.prompt";
@@ -23,6 +26,10 @@ declare const LLM_OUTPUT_TOKENS = "llm.token_count.completion";
23
26
  declare const LLM_CACHED_TOKENS = "llm.token_count.prompt_cache_hit";
24
27
  declare const LLM_COST_USD = "llm.cost_usd";
25
28
  declare const TOOL_NAME = "tool.name";
29
+ declare const TOOL_ARGS_CAPTURED = "tool.args_captured";
30
+ declare const TOOL_LATENCY_MS = "tool.latency_ms";
31
+ declare const INPUT_VALUE = "input.value";
32
+ declare const OUTPUT_VALUE = "output.value";
26
33
  declare const SPAN_KIND_ATTR_KEYS: readonly ["openinference.span.kind", "inference.observation_kind"];
27
34
  declare const LLM_MODEL_ATTR_KEYS: readonly ["llm.model_name", "inference.llm.model_name", "llm.model", "gen_ai.request.model", "gen_ai.response.model"];
28
35
  declare const LLM_INPUT_TOKEN_ATTR_KEYS: readonly ["llm.token_count.prompt", "inference.llm.input_tokens", "llm.input_tokens", "gen_ai.usage.input_tokens", "gen_ai.usage.prompt_tokens"];
@@ -30,6 +37,8 @@ declare const LLM_OUTPUT_TOKEN_ATTR_KEYS: readonly ["llm.token_count.completion"
30
37
  declare const LLM_CACHED_TOKEN_ATTR_KEYS: readonly ["llm.token_count.prompt_cache_hit", "inference.llm.cached_tokens", "llm.cached_tokens", "gen_ai.usage.cached_tokens"];
31
38
  declare const LLM_COST_ATTR_KEYS: readonly ["llm.cost_usd", "inference.llm.cost.total", "llm.cost.total", "gen_ai.usage.cost"];
32
39
  declare const TOOL_NAME_ATTR_KEYS: readonly ["tool.name", "inference.tool.name"];
40
+ type ToolSpanOtlpInput = Pick<ToolSpan, 'toolName' | 'args' | 'argsCaptured' | 'result' | 'latencyMs'>;
41
+ declare function applyToolSpanOtlpAttributes(attributes: Record<string, unknown>, span: ToolSpanOtlpInput): void;
33
42
  declare function traceSpanKindToOpenInferenceKind(kind: string): string;
34
43
 
35
44
  /**
@@ -708,6 +717,7 @@ declare function captureFetchToRawSink(fetch: typeof globalThis.fetch, sink: Raw
708
717
  * or when the batch fills. No @opentelemetry SDK dependency — minimal
709
718
  * OTLP/JSON serializer (~120 LOC) using the existing otel.ts helpers.
710
719
  */
720
+
711
721
  interface OtelExportConfig {
712
722
  /** OTLP endpoint. Reads OTEL_EXPORTER_OTLP_ENDPOINT env by default. */
713
723
  endpoint?: string;
@@ -744,6 +754,7 @@ interface ExportableSpan {
744
754
  inputTokens?: number;
745
755
  outputTokens?: number;
746
756
  costUsd?: number;
757
+ tool?: ToolSpanOtlpInput;
747
758
  attributes?: Record<string, unknown>;
748
759
  }
749
760
  /**
@@ -1014,4 +1025,4 @@ declare function iterateRawCalls(sink: RawProviderSink, filter?: {
1014
1025
  spanId?: string;
1015
1026
  }): AsyncGenerator<ReplayCacheEntry>;
1016
1027
 
1017
- export { AnalyzeTracesOptions, AnalyzeTracesResult, type CaptureFetchContext, type CaptureFetchOptions, DEFAULT_REDACTION_RULES, DatasetOverview, type ExportableSpan, type ExtractedUsage, type FlattenOtlpOptions, LLM_CACHED_TOKENS, LLM_CACHED_TOKEN_ATTR_KEYS, LLM_COST_ATTR_KEYS, LLM_COST_USD, LLM_INPUT_TOKENS, LLM_INPUT_TOKEN_ATTR_KEYS, LLM_MODEL_ATTR_KEYS, LLM_MODEL_NAME, LLM_OUTPUT_TOKENS, LLM_OUTPUT_TOKEN_ATTR_KEYS, OPENINFERENCE_SPAN_KIND, OTEL_AGENT_EVAL_SCOPE, type OtelExportConfig, type OtelExporter, type OtlpExport, OtlpFileTraceStore, type OtlpFileTraceStoreOptions, type OtlpFlatLine, type OtlpResourceSpans, type OtlpSpan, type OtlpToRunRecordsOptions, type OtlpTraceRunRecord, type ProjectedOtlpSpan, ProviderRedactor, QueryTracesPage, REDACTION_VERSION, RawProviderEvent, RawProviderSink, type RedactionReport, type RedactionRule, ReplayCache, type ReplayCacheEntry, ReplayCacheMissError, type ReplayCacheStats, type ReplayFetchOptions, Run, RunCompleteHook, RunCompleteHookContext, SPAN_KIND_ATTR_KEYS, SearchSpanResult, SearchTraceResult, SpanNotFoundError, TOOL_NAME, TOOL_NAME_ATTR_KEYS, TRACE_ANALYST_ACTOR_DESCRIPTION, TRACE_ANALYST_ACTOR_DESCRIPTION_VERSION, TRACE_ANALYST_SUBAGENT_DESCRIPTION, type TraceAggregate, TraceAnalysisStore, TraceAnalystFilters, type TraceAnalystHookOptions, TraceAnalystSpanKind, TraceAnalystSpanStatus, TraceFileMissingError, type TraceInsightContext, type TraceInsightFinding, type TraceInsightPanelRole, type TraceInsightPromptInput, type TraceInsightQualityGate, type TraceInsightQuestion, type TraceInsightReadiness, type TraceInsightSuite, type TraceInsightTask, TraceNotFoundError, TraceStore, type TraceStoreSource, type TraceStoreToOtlpOptions, type TracesToOtlpResult, ViewSpansResult, ViewTraceResult, asNumber, asString, buildTraceAnalystTools, buildTraceInsightContext, buildTraceInsightPrompt, captureFetchToRawSink, convertTraceStoresToOtlp, createOtelExporter, createOtelTracingStore, createReplayFetch, defaultTraceInsightPanel, describeTraceInsightScope, domainEvidencePattern, exportRunAsOtlp, extractOtlpAttributes, extractUsage, extractUsageFromResponse, extractUsageFromSse, firstNumberAttr, firstStringAttr, flattenOtlpExportToNdjson, inferDomainKeywords, inferOtlpKind, iterateRawCalls, otelRunCompleteHook, otlpToRunRecords, otlpToTraceRunRecords, planTraceInsightQuestions, projectOtlpFlatLine, readOtlpStatus, redactString, redactValue, scoreTraceInsightReadiness, stringField, tokenizeDomainWords, traceAnalystFunctionGroup, traceAnalystOnRunComplete, traceSpanKindToOpenInferenceKind };
1028
+ export { AnalyzeTracesOptions, AnalyzeTracesResult, type CaptureFetchContext, type CaptureFetchOptions, DEFAULT_REDACTION_RULES, DatasetOverview, type ExportableSpan, type ExtractedUsage, type FlattenOtlpOptions, INPUT_VALUE, LLM_CACHED_TOKENS, LLM_CACHED_TOKEN_ATTR_KEYS, LLM_COST_ATTR_KEYS, LLM_COST_USD, LLM_INPUT_TOKENS, LLM_INPUT_TOKEN_ATTR_KEYS, LLM_MODEL_ATTR_KEYS, LLM_MODEL_NAME, LLM_OUTPUT_TOKENS, LLM_OUTPUT_TOKEN_ATTR_KEYS, OPENINFERENCE_SPAN_KIND, OTEL_AGENT_EVAL_SCOPE, OUTPUT_VALUE, type OtelExportConfig, type OtelExporter, type OtlpExport, OtlpFileTraceStore, type OtlpFileTraceStoreOptions, type OtlpFlatLine, type OtlpResourceSpans, type OtlpSpan, type OtlpToRunRecordsOptions, type OtlpTraceRunRecord, type ProjectedOtlpSpan, ProviderRedactor, QueryTracesPage, REDACTION_VERSION, RawProviderEvent, RawProviderSink, type RedactionReport, type RedactionRule, ReplayCache, type ReplayCacheEntry, ReplayCacheMissError, type ReplayCacheStats, type ReplayFetchOptions, Run, RunCompleteHook, RunCompleteHookContext, SPAN_KIND_ATTR_KEYS, SearchSpanResult, SearchTraceResult, SpanNotFoundError, TOOL_ARGS_CAPTURED, TOOL_LATENCY_MS, TOOL_NAME, TOOL_NAME_ATTR_KEYS, TRACE_ANALYST_ACTOR_DESCRIPTION, TRACE_ANALYST_ACTOR_DESCRIPTION_VERSION, TRACE_ANALYST_SUBAGENT_DESCRIPTION, ToolSpan, type ToolSpanOtlpInput, type TraceAggregate, TraceAnalysisStore, TraceAnalystFilters, type TraceAnalystHookOptions, TraceAnalystSpanKind, TraceAnalystSpanStatus, TraceFileMissingError, type TraceInsightContext, type TraceInsightFinding, type TraceInsightPanelRole, type TraceInsightPromptInput, type TraceInsightQualityGate, type TraceInsightQuestion, type TraceInsightReadiness, type TraceInsightSuite, type TraceInsightTask, TraceNotFoundError, TraceStore, type TraceStoreSource, type TraceStoreToOtlpOptions, type TracesToOtlpResult, ViewSpansResult, ViewTraceResult, applyToolSpanOtlpAttributes, asNumber, asString, buildTraceAnalystTools, buildTraceInsightContext, buildTraceInsightPrompt, captureFetchToRawSink, convertTraceStoresToOtlp, createOtelExporter, createOtelTracingStore, createReplayFetch, defaultTraceInsightPanel, describeTraceInsightScope, domainEvidencePattern, exportRunAsOtlp, extractOtlpAttributes, extractUsage, extractUsageFromResponse, extractUsageFromSse, firstNumberAttr, firstStringAttr, flattenOtlpExportToNdjson, inferDomainKeywords, inferOtlpKind, iterateRawCalls, otelRunCompleteHook, otlpToRunRecords, otlpToTraceRunRecords, planTraceInsightQuestions, projectOtlpFlatLine, readOtlpStatus, redactString, redactValue, scoreTraceInsightReadiness, stringField, tokenizeDomainWords, traceAnalystFunctionGroup, traceAnalystOnRunComplete, traceSpanKindToOpenInferenceKind };
package/dist/traces.js CHANGED
@@ -28,13 +28,14 @@ import {
28
28
  scoreTraceInsightReadiness,
29
29
  tokenizeDomainWords,
30
30
  traceAnalystOnRunComplete
31
- } from "./chunk-TLDB7WRY.js";
31
+ } from "./chunk-YZPO4UHR.js";
32
32
  import {
33
33
  FAILURE_CLASSES,
34
34
  TRACE_SCHEMA_VERSION,
35
35
  aggregateLlm,
36
36
  argHash,
37
37
  groupBy,
38
+ hasCapturedToolArgs,
38
39
  isJudgeSpan,
39
40
  isLlmSpan,
40
41
  isRetrievalSpan,
@@ -45,13 +46,13 @@ import {
45
46
  runFailureClass,
46
47
  runsForScenario,
47
48
  toolSpans
48
- } from "./chunk-MHNQWM4I.js";
49
+ } from "./chunk-LQUTGLOZ.js";
49
50
  import {
50
51
  TRACE_ANALYST_ACTOR_DESCRIPTION,
51
52
  TRACE_ANALYST_ACTOR_DESCRIPTION_VERSION,
52
53
  TRACE_ANALYST_SUBAGENT_DESCRIPTION,
53
54
  analyzeTraces
54
- } from "./chunk-RPDDVKI7.js";
55
+ } from "./chunk-4JLWXDYA.js";
55
56
  import {
56
57
  DEFAULT_REDACTION_RULES,
57
58
  REDACTION_VERSION,
@@ -60,6 +61,7 @@ import {
60
61
  } from "./chunk-GGE4NNQT.js";
61
62
  import {
62
63
  DEFAULT_TRACE_ANALYST_BUDGETS,
64
+ INPUT_VALUE,
63
65
  LLM_CACHED_TOKENS,
64
66
  LLM_CACHED_TOKEN_ATTR_KEYS,
65
67
  LLM_COST_ATTR_KEYS,
@@ -71,14 +73,18 @@ import {
71
73
  LLM_OUTPUT_TOKENS,
72
74
  LLM_OUTPUT_TOKEN_ATTR_KEYS,
73
75
  OPENINFERENCE_SPAN_KIND,
76
+ OUTPUT_VALUE,
74
77
  OtlpFileTraceStore,
75
78
  SPAN_KIND_ATTR_KEYS,
76
79
  SpanNotFoundError,
80
+ TOOL_ARGS_CAPTURED,
81
+ TOOL_LATENCY_MS,
77
82
  TOOL_NAME,
78
83
  TOOL_NAME_ATTR_KEYS,
79
84
  TRACE_ANALYST_TRUNCATION_MARKER_PREFIX,
80
85
  TraceFileMissingError,
81
86
  TraceNotFoundError,
87
+ applyToolSpanOtlpAttributes,
82
88
  asNumber,
83
89
  asString,
84
90
  buildTraceAnalystTools,
@@ -91,7 +97,7 @@ import {
91
97
  stringField,
92
98
  traceAnalystFunctionGroup,
93
99
  traceSpanKindToOpenInferenceKind
94
- } from "./chunk-LNQEP766.js";
100
+ } from "./chunk-S2F4J57L.js";
95
101
  import {
96
102
  RunIntegrityError,
97
103
  assertRunCaptured,
@@ -119,6 +125,7 @@ export {
119
125
  FAILURE_CLASSES,
120
126
  FileSystemRawProviderSink,
121
127
  FileSystemTraceStore,
128
+ INPUT_VALUE,
122
129
  InMemoryRawProviderSink,
123
130
  InMemoryTraceStore,
124
131
  LLM_CACHED_TOKENS,
@@ -134,6 +141,7 @@ export {
134
141
  NoopRawProviderSink,
135
142
  OPENINFERENCE_SPAN_KIND,
136
143
  OTEL_AGENT_EVAL_SCOPE,
144
+ OUTPUT_VALUE,
137
145
  OtlpFileTraceStore,
138
146
  REDACTION_VERSION,
139
147
  ReplayCache,
@@ -141,6 +149,8 @@ export {
141
149
  RunIntegrityError,
142
150
  SPAN_KIND_ATTR_KEYS,
143
151
  SpanNotFoundError,
152
+ TOOL_ARGS_CAPTURED,
153
+ TOOL_LATENCY_MS,
144
154
  TOOL_NAME,
145
155
  TOOL_NAME_ATTR_KEYS,
146
156
  TRACE_ANALYST_ACTOR_DESCRIPTION,
@@ -153,6 +163,7 @@ export {
153
163
  TraceNotFoundError,
154
164
  aggregateLlm,
155
165
  analyzeTraces,
166
+ applyToolSpanOtlpAttributes,
156
167
  argHash,
157
168
  asNumber,
158
169
  asString,
@@ -178,6 +189,7 @@ export {
178
189
  firstStringAttr,
179
190
  flattenOtlpExportToNdjson,
180
191
  groupBy,
192
+ hasCapturedToolArgs,
181
193
  inferDomainKeywords,
182
194
  inferOtlpKind,
183
195
  isJudgeSpan,
@@ -1,5 +1,7 @@
1
- import { P as PolicyEditCandidateRecord } from './policy-edit-Clb2v6Oa.js';
2
- import { c as RunTokenUsage } from './run-record-CZmcpWPo.js';
1
+ import { P as PolicyEditCandidateRecord } from './policy-edit-wG9uFEFm.js';
2
+ import { R as RunPaidCallInput, a as CostChannel, P as PaidCallResult, C as CostLedger, b as CostLedgerSummary } from './cost-ledger-DWy3XdJc.js';
3
+ import { L as LlmCallMetadata } from './llm-client-qoDd18Qz.js';
4
+ import { c as RunTokenUsage } from './run-record-BDH49H2E.js';
3
5
 
4
6
  /**
5
7
  * Pass A substrate types — `runCampaign` is the one primitive every
@@ -83,12 +85,20 @@ interface JudgeDimension {
83
85
  interface JudgeConfig<TArtifact, TScenario extends Scenario = Scenario> {
84
86
  name: string;
85
87
  dimensions: JudgeDimension[];
88
+ /** Stable scoring revision used by campaign resume and verdict caches.
89
+ * Built-in judges derive this from their prompt, model, and rubric. Custom
90
+ * judges should set it when closure state can change without changing code. */
91
+ judgeVersion?: string;
86
92
  /** Score one artifact. Throw on failure — a thrown judge is recorded as a
87
93
  * failed cell, never silently folded into a zero. */
88
94
  score(input: {
89
95
  artifact: TArtifact;
90
96
  scenario: TScenario;
91
97
  signal: AbortSignal;
98
+ /** Shared run spend account and receipt attribution phase. */
99
+ costLedger?: CostLedger;
100
+ costPhase?: string;
101
+ costTags?: Record<string, string>;
92
102
  }): JudgeScore | Promise<JudgeScore>;
93
103
  appliesTo?: (scenario: TScenario) => boolean;
94
104
  }
@@ -105,6 +115,8 @@ interface JudgeScore {
105
115
  dimensions: Record<string, number>;
106
116
  composite: number;
107
117
  notes: string;
118
+ /** Provider metadata for display and diagnostics; accounting uses CostLedger receipts. */
119
+ llmCall?: LlmCallMetadata;
108
120
  /** Set when the judge itself failed (call error, unparseable output).
109
121
  * `composite`/`dimensions` carry no signal — aggregators MUST exclude
110
122
  * failed scores from means instead of folding them into zeros. */
@@ -270,6 +282,9 @@ interface ProposeContext<TFindings = unknown> {
270
282
  * scenarios) into a merged candidate. Proposers doing pure single-parent
271
283
  * reflection may ignore it. See {@link ParetoParent}. */
272
284
  paretoParents?: ParetoParent[];
285
+ /** Shared run spend account and receipt attribution phase. */
286
+ costLedger?: CostLedger;
287
+ costPhase?: string;
273
288
  /** FIREWALL (non-negotiable): the held-out judge is write-only — its verdicts
274
289
  * score the chosen output and gate promotion, and are NEVER an input to
275
290
  * proposal/steering (else the optimizer games the acceptance axis = an
@@ -351,6 +366,9 @@ interface GateContext<TArtifact, TScenario extends Scenario> {
351
366
  candidate: number;
352
367
  baseline: number;
353
368
  };
369
+ /** Shared run spend account and receipt attribution phase. */
370
+ costLedger?: CostLedger;
371
+ costPhase?: string;
354
372
  signal: AbortSignal;
355
373
  }
356
374
  interface GateResult {
@@ -389,35 +407,15 @@ interface CampaignArtifactWriter {
389
407
  * backend-integrity guard with ONE source of truth — a field added to
390
408
  * `RunTokenUsage` is a compile error here, not a silent drift. */
391
409
  type CampaignTokenUsage = RunTokenUsage;
392
- /** Cell-scoped cost meter. NOTHING is captured automatically
393
- * the substrate does not intercept the LLM call, so it cannot see cost or
394
- * tokens unless the dispatch reports them. Every LLM cost MUST be reported via
395
- * `observe` and every token count via `observeTokens`; a dispatch that reports
396
- * neither yields a `{cost:0, tokens:0}` cell, which the backend-integrity
397
- * guard (`assertRealBackend`) correctly reads as a stub. Also use `observe`
398
- * for non-LLM spend (sandbox time, tool costs). */
410
+ /** Cell-scoped paid-call entry point. The dispatch places every paid operation
411
+ * inside `runPaidCall`; the returned provider result supplies one receipt with
412
+ * cost, tokens, and resolved model. Calls made outside this method are not
413
+ * admitted or captured. */
399
414
  interface CampaignCostMeter {
400
- observe(amountUsd: number, source: string): void;
401
- /** Record LLM token usage for this cell; accumulates across calls. A cell
402
- * has `costUsd` but no token counts unless the dispatch reports them here —
403
- * and the backend-integrity guard (`assertRealBackend`) keys on
404
- * `tokenUsage`, so a cell that never reports tokens reads as a stub. Any
405
- * dispatch that calls an LLM MUST report its usage. */
406
- observeTokens(usage: CampaignTokenUsage): void;
407
- /** Record the concrete model the backend RESOLVED this cell to at runtime.
408
- * The substrate cannot see the LLM call, so it cannot know which model a
409
- * vendor-locked harness actually served — only the dispatch, reading the
410
- * backend's usage/terminal events, can. A dispatch whose profile declares a
411
- * runtime-resolved model (the `HARNESS_NATIVE_MODEL` sentinel) MUST report
412
- * the resolved, snapshot-bearing id here so the RunRecord pins a real model
413
- * instead of the sentinel. Last write wins (a cell issues one logical run);
414
- * optional because most dispatches declare a concrete model up front. */
415
- observeModel?(model: string): void;
416
- current(): number;
417
- /** Accumulated token usage for this cell (zeros if never observed). */
418
- tokens(): CampaignTokenUsage;
419
- /** The runtime-resolved model reported via `observeModel`, if any. */
420
- resolvedModel?(): string | undefined;
415
+ /** The only paid-call path. Returns a typed result; callers must inspect it. */
416
+ runPaidCall<T>(input: Omit<RunPaidCallInput<T>, 'channel' | 'phase' | 'tags'> & {
417
+ channel?: CostChannel;
418
+ }): Promise<PaidCallResult<T>>;
421
419
  }
422
420
  /** Source tag — required on every store write. Used by the
423
421
  * default training-source filter (production-trace samples NOT used as
@@ -510,15 +508,16 @@ interface CampaignCellResult<TArtifact> {
510
508
  artifact: TArtifact;
511
509
  judgeScores: Record<string, JudgeScore>;
512
510
  costUsd: number;
513
- /** LLM token usage the dispatch reported via `ctx.cost.observeTokens`.
514
- * `{ input: 0, output: 0 }` when the dispatch reported none — which the
515
- * backend-integrity guard reads as a stub. */
511
+ /** True when at least one priced receipt used the model table instead of a provider bill. */
512
+ costEstimated?: boolean;
513
+ /** Exact durable receipts required to reuse this cached result. */
514
+ costCallIds?: string[];
515
+ /** Agent-call token usage committed by `ctx.cost.runPaidCall`.
516
+ * `{ input: 0, output: 0 }` when no paid agent call was recorded. */
516
517
  tokenUsage: CampaignTokenUsage;
517
- /** The concrete model the backend resolved this cell to at runtime, reported
518
- * by the dispatch via `ctx.cost.observeModel`. Set only when the dispatch
519
- * reported it — a profile that declares a concrete model up front has no
520
- * need to. Consumed by `buildRunRecord` to pin the real model when the
521
- * declared model is the `HARNESS_NATIVE_MODEL` sentinel. */
518
+ /** Concrete model from the latest committed agent receipt. Consumed by
519
+ * `buildRunRecord` to pin the model when the declared profile uses a
520
+ * runtime-resolved sentinel. */
522
521
  resolvedModel?: string;
523
522
  durationMs: number;
524
523
  seed: number;
@@ -599,6 +598,9 @@ interface GenerationCandidate {
599
598
  interface CampaignAggregates {
600
599
  byJudge: Record<string, JudgeAggregate>;
601
600
  byScenario: Record<string, ScenarioAggregate>;
601
+ /** Canonical campaign accounting, including worker and judge calls. */
602
+ cost: CostLedgerSummary;
603
+ /** Compatibility alias of `cost.totalCostUsd`. */
602
604
  totalCostUsd: number;
603
605
  cellsExecuted: number;
604
606
  cellsSkipped: number;
@@ -629,4 +631,4 @@ interface CampaignResult<TArtifact = unknown, TScenario extends Scenario = Scena
629
631
  scenarios: Array<Pick<TScenario, 'id' | 'kind'>>;
630
632
  }
631
633
 
632
- export { type JudgeAggregate as A, type ScenarioAggregate as B, type CampaignResult as C, type DispatchContext as D, type ScoredSurfaceOutcome as E, isProposedCandidate as F, type GateResult as G, labelTrustRank as H, type JudgeScore as J, type LabeledScenarioStore as L, type MutableSurface as M, type OptimizationProposer as O, type ParetoParent as P, type RedactionStatus as R, type Scenario as S, type TraceSpan as T, type JudgeDimension as a, type JudgeConfig as b, type DispatchFn as c, type CampaignTraceWriter as d, type GenerationRecord as e, type SurfaceProposer as f, type Gate as g, type GateDecision as h, type CampaignAggregates as i, type CampaignArtifactWriter as j, type CampaignCellResult as k, type CampaignCostMeter as l, type CodeSurface as m, type GateContext as n, type GenerationCandidate as o, type Mutator as p, type OptimizerConfig as q, type SessionScript as r, type LabeledScenarioWrite as s, type LabeledScenarioSampleArgs as t, type LabeledScenarioRecord as u, type LabelTrust as v, type ProposedCandidate as w, type ProposeContext as x, type LabeledScenarioSource as y, type CampaignTokenUsage as z };
634
+ export { type JudgeAggregate as A, type ScenarioAggregate as B, type CampaignResult as C, type DispatchContext as D, type ScoredSurfaceOutcome as E, isProposedCandidate as F, type Gate as G, labelTrustRank as H, type JudgeScore as J, type LabeledScenarioStore as L, type MutableSurface as M, type OptimizationProposer as O, type ParetoParent as P, type RedactionStatus as R, type Scenario as S, type TraceSpan as T, type JudgeDimension as a, type JudgeConfig as b, type SurfaceProposer as c, type GateDecision as d, type CampaignAggregates as e, type CampaignArtifactWriter as f, type CampaignCellResult as g, type CampaignCostMeter as h, type CampaignTraceWriter as i, type CodeSurface as j, type DispatchFn as k, type GateContext as l, type GateResult as m, type GenerationCandidate as n, type GenerationRecord as o, type Mutator as p, type OptimizerConfig as q, type SessionScript as r, type LabeledScenarioWrite as s, type LabeledScenarioSampleArgs as t, type LabeledScenarioRecord as u, type LabelTrust as v, type ProposedCandidate as w, type ProposeContext as x, type LabeledScenarioSource as y, type CampaignTokenUsage as z };
@@ -1,3 +1,4 @@
1
+ import { C as CostLedger, b as CostLedgerSummary } from './cost-ledger-DWy3XdJc.js';
1
2
  import { TCloud } from '@tangle-network/tcloud';
2
3
 
3
4
  interface Scenario {
@@ -50,6 +51,8 @@ interface ScenarioResult {
50
51
  overallScore: number;
51
52
  totalDurationMs: number;
52
53
  artifacts: CollectedArtifacts;
54
+ /** Agent and judge spend attributed to this scenario. */
55
+ cost?: CostLedgerSummary;
53
56
  }
54
57
  interface TurnResult {
55
58
  turnIndex: number;
@@ -96,6 +99,7 @@ interface BenchmarkReport {
96
99
  promptVersion: string;
97
100
  scenarioCount: number;
98
101
  results: ScenarioResult[];
102
+ cost?: CostLedgerSummary;
99
103
  summary: {
100
104
  overallAvg: number;
101
105
  byPersona: Record<string, {
@@ -266,11 +270,22 @@ interface BenchmarkRunnerConfig {
266
270
  passThreshold?: number;
267
271
  generation?: number;
268
272
  promptVersion?: string;
273
+ /** Shared ledger for agent and judge calls made by the benchmark. */
274
+ costLedger?: CostLedger;
275
+ /** Exact maximum provider attempts configured on the supplied TCloud client. */
276
+ tcloudMaximumAttempts?: number;
269
277
  }
270
278
  interface JudgeInput {
271
279
  scenario: Scenario;
272
280
  turns: TurnResult[];
273
281
  artifacts: CollectedArtifacts;
282
+ /** Shared ledger for paid built-in judges. Direct calls default to an uncapped ledger. */
283
+ costLedger?: CostLedger;
284
+ costPhase?: string;
285
+ costTags?: Record<string, string>;
286
+ signal?: AbortSignal;
287
+ /** Exact maximum provider attempts configured on the supplied TCloud client. */
288
+ tcloudMaximumAttempts?: number;
274
289
  }
275
290
  type JudgeFn = (tc: TCloud, input: JudgeInput) => Promise<JudgeScore[]>;
276
291
 
@@ -1,14 +1,17 @@
1
- import { F as FeedbackTrajectoryStore } from '../feedback-trajectory-pDcz1lQ1.js';
2
- import { T as TraceStore } from '../store-BsVi7ncX.js';
1
+ import { C as CostLedger } from '../cost-ledger-DWy3XdJc.js';
2
+ import { F as FeedbackTrajectoryStore } from '../feedback-trajectory-BUnM58xL.js';
3
+ import { a as LlmClientOptions } from '../llm-client-qoDd18Qz.js';
4
+ import { T as TraceStore } from '../store-DGqD0Pyo.js';
3
5
  import { z } from 'zod';
4
6
  import { OpenAPIObject } from 'openapi3-ts/oas31';
5
7
  import * as hono_types from 'hono/types';
6
8
  import { ServerType } from '@hono/node-server';
7
9
  import { Hono } from 'hono';
8
- import '../emitter-BRchAAAx.js';
9
- import '../schema-SGWcK9wa.js';
10
- import '../dataset-NENEzRgk.js';
11
10
  import '../errors-oeQrLqXC.js';
11
+ import '../emitter-CjD7vUwv.js';
12
+ import '../schema-B3Q3l9Z_.js';
13
+ import '../dataset-NENEzRgk.js';
14
+ import '../raw-provider-sink-C46HDghv.js';
12
15
 
13
16
  declare const RubricDimensionSchema: z.ZodObject<{
14
17
  id: z.ZodString;
@@ -122,9 +125,9 @@ declare const TraceEventSchema: z.ZodObject<{
122
125
  runId: z.ZodString;
123
126
  spanId: z.ZodOptional<z.ZodString>;
124
127
  kind: z.ZodEnum<{
125
- policy_violation: "policy_violation";
126
- custom: "custom";
127
128
  error: "error";
129
+ custom: "custom";
130
+ policy_violation: "policy_violation";
128
131
  log: "log";
129
132
  budget_decrement: "budget_decrement";
130
133
  budget_breach: "budget_breach";
@@ -140,9 +143,9 @@ declare const TracesIngestRequestSchema: z.ZodObject<{
140
143
  runId: z.ZodString;
141
144
  spanId: z.ZodOptional<z.ZodString>;
142
145
  kind: z.ZodEnum<{
143
- policy_violation: "policy_violation";
144
- custom: "custom";
145
146
  error: "error";
147
+ custom: "custom";
148
+ policy_violation: "policy_violation";
146
149
  log: "log";
147
150
  budget_decrement: "budget_decrement";
148
151
  budget_breach: "budget_breach";
@@ -165,8 +168,8 @@ declare const FeedbackLabelSchema: z.ZodObject<{
165
168
  id: z.ZodOptional<z.ZodString>;
166
169
  source: z.ZodEnum<{
167
170
  judge: "judge";
168
- system: "system";
169
171
  user: "user";
172
+ system: "system";
170
173
  policy: "policy";
171
174
  environment: "environment";
172
175
  metric: "metric";
@@ -198,10 +201,10 @@ declare const FeedbackAttemptSchema: z.ZodObject<{
198
201
  id: z.ZodString;
199
202
  stepIndex: z.ZodNumber;
200
203
  artifactType: z.ZodEnum<{
201
- action: "action";
202
- decision: "decision";
203
204
  text: "text";
204
205
  code: "code";
206
+ action: "action";
207
+ decision: "decision";
205
208
  plan: "plan";
206
209
  research: "research";
207
210
  ui: "ui";
@@ -226,8 +229,8 @@ declare const FeedbackAttemptSchema: z.ZodObject<{
226
229
  id: z.ZodOptional<z.ZodString>;
227
230
  source: z.ZodEnum<{
228
231
  judge: "judge";
229
- system: "system";
230
232
  user: "user";
233
+ system: "system";
231
234
  policy: "policy";
232
235
  environment: "environment";
233
236
  metric: "metric";
@@ -270,10 +273,10 @@ declare const FeedbackTrajectorySchema: z.ZodObject<{
270
273
  id: z.ZodString;
271
274
  stepIndex: z.ZodNumber;
272
275
  artifactType: z.ZodEnum<{
273
- action: "action";
274
- decision: "decision";
275
276
  text: "text";
276
277
  code: "code";
278
+ action: "action";
279
+ decision: "decision";
277
280
  plan: "plan";
278
281
  research: "research";
279
282
  ui: "ui";
@@ -298,8 +301,8 @@ declare const FeedbackTrajectorySchema: z.ZodObject<{
298
301
  id: z.ZodOptional<z.ZodString>;
299
302
  source: z.ZodEnum<{
300
303
  judge: "judge";
301
- system: "system";
302
304
  user: "user";
305
+ system: "system";
303
306
  policy: "policy";
304
307
  environment: "environment";
305
308
  metric: "metric";
@@ -334,8 +337,8 @@ declare const FeedbackTrajectorySchema: z.ZodObject<{
334
337
  id: z.ZodOptional<z.ZodString>;
335
338
  source: z.ZodEnum<{
336
339
  judge: "judge";
337
- system: "system";
338
340
  user: "user";
341
+ system: "system";
339
342
  policy: "policy";
340
343
  environment: "environment";
341
344
  metric: "metric";
@@ -438,7 +441,13 @@ declare class WireError extends Error {
438
441
  readonly details?: unknown | undefined;
439
442
  constructor(code: string, message: string, status?: number, details?: unknown | undefined);
440
443
  }
441
- declare function handleJudge(req: JudgeRequest): Promise<JudgeResult>;
444
+ interface HandleJudgeOptions {
445
+ costLedger?: CostLedger;
446
+ costPhase?: string;
447
+ llm?: LlmClientOptions;
448
+ signal?: AbortSignal;
449
+ }
450
+ declare function handleJudge(req: JudgeRequest, options?: HandleJudgeOptions): Promise<JudgeResult>;
442
451
  declare function handleListRubrics(): ListRubricsResponse;
443
452
  declare function handleVersion(): VersionResponse;
444
453
  /**
@@ -569,4 +578,4 @@ interface StartedServer {
569
578
  */
570
579
  declare function startServerAsync(opts?: ServeOptions): Promise<StartedServer>;
571
580
 
572
- export { BUILTIN_RUBRICS, type ErrorResponse, ErrorResponseSchema, type FailureMode, FailureModeSchema, FeedbackAttemptSchema, type FeedbackIngestResponse, FeedbackIngestResponseSchema, FeedbackLabelSchema, type FeedbackTrajectory, FeedbackTrajectorySchema, HealthResponseSchema, type IngestionStores, type JudgeRequest, JudgeRequestSchema, type JudgeResult, JudgeResultSchema, type ListRubricsResponse, ListRubricsResponseSchema, type Rubric, type RubricDimension, RubricDimensionSchema, type RubricInfo, RubricInfoSchema, RubricSchema, type ServeOptions, type StartedServer, type TraceEvent, TraceEventSchema, type TracesIngestRequest, TracesIngestRequestSchema, type TracesIngestResponse, TracesIngestResponseSchema, type VersionResponse, VersionResponseSchema, WIRE_VERSION, WireError, buildOpenApi, createApp, dispatchRpc, getBuiltinRubric, handleFeedbackIngest, handleJudge, handleListRubrics, handleTracesIngest, handleVersion, hashRubric, listBuiltinRubrics, runRpcBatch, runRpcOnce, startServer, startServerAsync };
581
+ export { BUILTIN_RUBRICS, type ErrorResponse, ErrorResponseSchema, type FailureMode, FailureModeSchema, FeedbackAttemptSchema, type FeedbackIngestResponse, FeedbackIngestResponseSchema, FeedbackLabelSchema, type FeedbackTrajectory, FeedbackTrajectorySchema, type HandleJudgeOptions, HealthResponseSchema, type IngestionStores, type JudgeRequest, JudgeRequestSchema, type JudgeResult, JudgeResultSchema, type ListRubricsResponse, ListRubricsResponseSchema, type Rubric, type RubricDimension, RubricDimensionSchema, type RubricInfo, RubricInfoSchema, RubricSchema, type ServeOptions, type StartedServer, type TraceEvent, TraceEventSchema, type TracesIngestRequest, TracesIngestRequestSchema, type TracesIngestResponse, TracesIngestResponseSchema, type VersionResponse, VersionResponseSchema, WIRE_VERSION, WireError, buildOpenApi, createApp, dispatchRpc, getBuiltinRubric, handleFeedbackIngest, handleJudge, handleListRubrics, handleTracesIngest, handleVersion, hashRubric, listBuiltinRubrics, runRpcBatch, runRpcOnce, startServer, startServerAsync };
@@ -34,8 +34,10 @@ import {
34
34
  runRpcOnce,
35
35
  startServer,
36
36
  startServerAsync
37
- } from "../chunk-4D5RVB3W.js";
38
- import "../chunk-GY4SYVPJ.js";
37
+ } from "../chunk-LTVG32KX.js";
38
+ import "../chunk-NJC7U437.js";
39
+ import "../chunk-VCTY3W6J.js";
40
+ import "../chunk-VI2UW6B6.js";
39
41
  import "../chunk-PC4UYEBM.js";
40
42
  import "../chunk-ONWEPEDO.js";
41
43
  import "../chunk-PZ5AY32C.js";
@@ -135,7 +135,7 @@ round-robin, region-affinity from a previous run, scheduling table).
135
135
  | **Auth** | Bearer token on `Authorization`; pluggable via `auth: string \| () => string \| Promise<string>` for rotation/refresh. |
136
136
  | **Payload size** | Server enforces `maxBodyBytes` (default 10 MB). |
137
137
  | **Traces** | Both ends emit OTel — if both point at the same OTLP collector, you get a unified trace per cell. See `docs/adapters-observability.md`. |
138
- | **Cost** | Worker's `ctx.cost.observe(usd, source)` is local to the worker process. Roll up server-side and attach to your worker-side telemetry; we don't (yet) forward cost back to the coordinator. Tracked as follow-up. |
138
+ | **Cost** | Worker's `ctx.cost.runPaidCall(...)` writes durable receipts in the worker process. Roll up those receipts server-side and attach them to worker telemetry; they are not forwarded to the coordinator automatically. |
139
139
 
140
140
  ## Running the reference example
141
141
 
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@tangle-network/agent-eval",
3
- "version": "0.116.0",
3
+ "version": "0.117.0",
4
4
  "description": "Evaluate and improve AI agents from runs, traces, judges, and feedback. Compare candidates, cluster failures, measure lift, and gate releases.",
5
5
  "homepage": "https://github.com/tangle-network/agent-eval#readme",
6
6
  "repository": {
@@ -148,7 +148,7 @@
148
148
  "@hono/node-server": "^2.0.0",
149
149
  "@tangle-network/agent-interface": "^0.22.0",
150
150
  "@tangle-network/tcloud": "^0.4.14",
151
- "hono": "^4.12.16",
151
+ "hono": "^4.12.30",
152
152
  "zod": "^4.3.6"
153
153
  },
154
154
  "devDependencies": {
@@ -165,7 +165,7 @@
165
165
  "minimumReleaseAge": 4320,
166
166
  "overrides": {
167
167
  "postcss@<8.5.10": "^8.5.10",
168
- "ws@>=8.0.0 <8.20.1": "^8.20.1"
168
+ "ws@>=8.0.0 <8.21.0": "^8.21.0"
169
169
  }
170
170
  },
171
171
  "engines": {