@tangle-network/agent-eval 0.173.2 → 0.174.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +40 -0
- package/dist/{proposal-findings-bko3GGy-.js → abort-signal-CtzAM_sJ.js} +11 -11
- package/dist/abort-signal-CtzAM_sJ.js.map +1 -0
- package/dist/adapters/http.d.ts +2 -2
- package/dist/agent-profile-_xPxqVJt.d.ts +488 -0
- package/dist/agent-profile-_xPxqVJt.d.ts.map +1 -0
- package/dist/analyst/index.d.ts +7 -9
- package/dist/analyst/index.d.ts.map +1 -1
- package/dist/analyst/index.js +8 -8
- package/dist/{benchmark-C4wk_Sjr.js → benchmark-DQKzykkO.js} +2 -2
- package/dist/{benchmark-C4wk_Sjr.js.map → benchmark-DQKzykkO.js.map} +1 -1
- package/dist/{benchmark-command-CS6gVHVq.js → benchmark-command-mZIlR-ra.js} +13 -13
- package/dist/{benchmark-command-CS6gVHVq.js.map → benchmark-command-mZIlR-ra.js.map} +1 -1
- package/dist/benchmarks/index.d.ts +3 -4
- package/dist/benchmarks/index.d.ts.map +1 -1
- package/dist/benchmarks/index.js +3 -3
- package/dist/campaign/index.d.ts +5 -9
- package/dist/campaign/index.js +7 -7
- package/dist/{campaign-B3kPMU8S.js → campaign-BzMSCejE.js} +8 -8
- package/dist/{campaign-B3kPMU8S.js.map → campaign-BzMSCejE.js.map} +1 -1
- package/dist/cli.js +1 -1
- package/dist/{client-DlqdbM7n.d.ts → client-vyYQg3bm.d.ts} +2 -2
- package/dist/{client-DlqdbM7n.d.ts.map → client-vyYQg3bm.d.ts.map} +1 -1
- package/dist/contract/index.d.ts +9 -10
- package/dist/contract/index.js +8 -8
- package/dist/{default-registry-B0bKikCb.js → default-registry-CrAp0pYq.js} +4 -4
- package/dist/{default-registry-B0bKikCb.js.map → default-registry-CrAp0pYq.js.map} +1 -1
- package/dist/{default-registry-BKwc8bN5.d.ts → default-registry-FfNzaUHV.d.ts} +3 -3
- package/dist/{default-registry-BKwc8bN5.d.ts.map → default-registry-FfNzaUHV.d.ts.map} +1 -1
- package/dist/{define-agent-eval-CY6qdlGV.d.ts → define-agent-eval-V1jQyCDR.d.ts} +102 -11
- package/dist/define-agent-eval-V1jQyCDR.d.ts.map +1 -0
- package/dist/{define-agent-eval-8h3lXXee.js → define-agent-eval-ox5McL6e.js} +331 -144
- package/dist/define-agent-eval-ox5McL6e.js.map +1 -0
- package/dist/{dspy-rlm-engine-CF0t2ITD.js → dspy-rlm-engine-Caz2pl4L.js} +3 -3
- package/dist/{dspy-rlm-engine-CF0t2ITD.js.map → dspy-rlm-engine-Caz2pl4L.js.map} +1 -1
- package/dist/{engine-DhFir3Ys.d.ts → engine-CvW_I72-.d.ts} +2 -2
- package/dist/{engine-DhFir3Ys.d.ts.map → engine-CvW_I72-.d.ts.map} +1 -1
- package/dist/experiment/index.d.ts +1 -4
- package/dist/experiment/index.d.ts.map +1 -1
- package/dist/{external-optimizer-process-BwITA9Jp.js → external-optimizer-process-CxnFL1hd.js} +2 -2
- package/dist/{external-optimizer-process-BwITA9Jp.js.map → external-optimizer-process-CxnFL1hd.js.map} +1 -1
- package/dist/{external-optimizer-subprocess-wBWeoG6A.js → external-optimizer-subprocess-CQi27uEI.js} +2 -2
- package/dist/{external-optimizer-subprocess-wBWeoG6A.js.map → external-optimizer-subprocess-CQi27uEI.js.map} +1 -1
- package/dist/fuzz.js +1 -1
- package/dist/fuzz.js.map +1 -1
- package/dist/hosted/index.d.ts +1 -1
- package/dist/{index-BQqOjerE.d.ts → index-BTrx5s8m.d.ts} +8 -9
- package/dist/index-BTrx5s8m.d.ts.map +1 -0
- package/dist/{index-D0Db5X-4.d.ts → index-Bn-nlnSV.d.ts} +4 -4
- package/dist/{index-D0Db5X-4.d.ts.map → index-Bn-nlnSV.d.ts.map} +1 -1
- package/dist/index-DKXuBPXf.d.ts +3840 -0
- package/dist/index-DKXuBPXf.d.ts.map +1 -0
- package/dist/index.d.ts +11 -13
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +10 -10
- package/dist/{kind-factory-gP6lDySe.js → kind-factory-BLvL-E44.js} +2 -2
- package/dist/{kind-factory-gP6lDySe.js.map → kind-factory-BLvL-E44.js.map} +1 -1
- package/dist/{llm-judge-BfqMFo4h.js → llm-judge-DmNaBrXB.js} +2541 -2435
- package/dist/llm-judge-DmNaBrXB.js.map +1 -0
- package/dist/{matrix-DGu8KhSs.d.ts → matrix-CJtXz1ky.d.ts} +2 -2
- package/dist/{matrix-DGu8KhSs.d.ts.map → matrix-CJtXz1ky.d.ts.map} +1 -1
- package/dist/multishot/golden/index.d.ts +1 -1
- package/dist/multishot/index.d.ts +2 -2
- package/dist/openapi.json +1 -1
- package/dist/{produced-state-D91uDvQw.js → produced-state-B8mw6zj9.js} +2 -2
- package/dist/{produced-state-D91uDvQw.js.map → produced-state-B8mw6zj9.js.map} +1 -1
- package/dist/rl.d.ts +1 -1
- package/dist/rl.d.ts.map +1 -1
- package/dist/rl.js.map +1 -1
- package/dist/{semantic-concept-judge-Dok7_35a.js → semantic-concept-judge-E3s_fEjB.js} +3 -3
- package/dist/{semantic-concept-judge-Dok7_35a.js.map → semantic-concept-judge-E3s_fEjB.js.map} +1 -1
- package/dist/{skillopt-optimization-method-DDw3v3gA.js → skillopt-optimization-method-f7399oGb.js} +5 -5
- package/dist/{skillopt-optimization-method-DDw3v3gA.js.map → skillopt-optimization-method-f7399oGb.js.map} +1 -1
- package/dist/statistical-heldout-Cqb73yE9.d.ts +1127 -0
- package/dist/statistical-heldout-Cqb73yE9.d.ts.map +1 -0
- package/dist/{store-otlp-Dow0pk_5.js → store-otlp-DV_H2HDu.js} +2 -2
- package/dist/{store-otlp-Dow0pk_5.js.map → store-otlp-DV_H2HDu.js.map} +1 -1
- package/dist/{store-tool-spans-CCZNsihA.d.ts → store-tool-spans-4o55ABER.d.ts} +3 -3
- package/dist/{store-tool-spans-CCZNsihA.d.ts.map → store-tool-spans-4o55ABER.d.ts.map} +1 -1
- package/dist/{store-tool-spans-CeNj_m2L.js → store-tool-spans-B9tjys_h.js} +3 -3
- package/dist/{store-tool-spans-CeNj_m2L.js.map → store-tool-spans-B9tjys_h.js.map} +1 -1
- package/dist/supervisor-run/index.d.ts.map +1 -1
- package/dist/supervisor-run/index.js +25 -7
- package/dist/supervisor-run/index.js.map +1 -1
- package/dist/{task-failure-attributes-CZjZeBsY.js → task-failure-attributes-CUy9mkIY.js} +2 -2
- package/dist/{task-failure-attributes-CZjZeBsY.js.map → task-failure-attributes-CUy9mkIY.js.map} +1 -1
- package/dist/{tool-groups-Cp4Xdzrp.d.ts → tool-groups-DAe1t6zb.d.ts} +2 -2
- package/dist/tool-groups-DAe1t6zb.d.ts.map +1 -0
- package/dist/trace-repair/index.d.ts +1 -1
- package/dist/traces.d.ts +2 -2
- package/dist/traces.js +4 -4
- package/dist/{types-Ba5UQyVD.d.ts → types-BJz2CPTM.d.ts} +2 -2
- package/dist/{types-Ba5UQyVD.d.ts.map → types-BJz2CPTM.d.ts.map} +1 -1
- package/dist/{types-CiWITkGo.js → types-DQ0e2E7y.js} +2 -2
- package/dist/types-DQ0e2E7y.js.map +1 -0
- package/dist/{types-BDV4PiMR.d.ts → types-Dd1ejaeI.d.ts} +2 -2
- package/dist/{types-BDV4PiMR.d.ts.map → types-Dd1ejaeI.d.ts.map} +1 -1
- package/docs/campaign-proposers.md +42 -0
- package/package.json +1 -1
- package/dist/agent-profile-B9_GGsG8.d.ts +0 -84
- package/dist/agent-profile-B9_GGsG8.d.ts.map +0 -1
- package/dist/backend-integrity-CeuTgqsd.d.ts +0 -280
- package/dist/backend-integrity-CeuTgqsd.d.ts.map +0 -1
- package/dist/benchmark-BjLGkfnN.d.ts +0 -236
- package/dist/benchmark-BjLGkfnN.d.ts.map +0 -1
- package/dist/define-agent-eval-8h3lXXee.js.map +0 -1
- package/dist/define-agent-eval-CY6qdlGV.d.ts.map +0 -1
- package/dist/external-optimizer-contracts-CQCpyrIL.d.ts +0 -172
- package/dist/external-optimizer-contracts-CQCpyrIL.d.ts.map +0 -1
- package/dist/heldout-gate-Df5hsqmm.d.ts +0 -453
- package/dist/heldout-gate-Df5hsqmm.d.ts.map +0 -1
- package/dist/index-BQqOjerE.d.ts.map +0 -1
- package/dist/index-CFDffsKz.d.ts +0 -1135
- package/dist/index-CFDffsKz.d.ts.map +0 -1
- package/dist/llm-judge-BfqMFo4h.js.map +0 -1
- package/dist/power-preflight-Ptse_Kq7.d.ts +0 -117
- package/dist/power-preflight-Ptse_Kq7.d.ts.map +0 -1
- package/dist/pre-registration-BoI4ucR3.d.ts +0 -592
- package/dist/pre-registration-BoI4ucR3.d.ts.map +0 -1
- package/dist/promotion-policy-CvMda3kU.d.ts +0 -134
- package/dist/promotion-policy-CvMda3kU.d.ts.map +0 -1
- package/dist/proposal-findings-bko3GGy-.js.map +0 -1
- package/dist/provenance-CRY67X50.d.ts +0 -1995
- package/dist/provenance-CRY67X50.d.ts.map +0 -1
- package/dist/statistical-heldout-DTyB_6-1.d.ts +0 -295
- package/dist/statistical-heldout-DTyB_6-1.d.ts.map +0 -1
- package/dist/tool-groups-Cp4Xdzrp.d.ts.map +0 -1
- package/dist/types-CiWITkGo.js.map +0 -1
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
import { s as ValidationError } from "./errors-Dngq5h35.js";
|
|
2
2
|
import { t as FAILURE_CLASSES } from "./schema-CSf6qWgZ.js";
|
|
3
|
-
import "./kind-factory-
|
|
3
|
+
import "./kind-factory-BLvL-E44.js";
|
|
4
4
|
import { LLM_CACHED_TOKEN_ATTR_KEYS, LLM_CACHE_WRITE_TOKEN_ATTR_KEYS, LLM_COST_ATTR_KEYS, LLM_INPUT_TOKEN_ATTR_KEYS, LLM_OUTPUT_TOKEN_ATTR_KEYS, LLM_REASONING_TOKEN_ATTR_KEYS, RUN_COST_ATTR_KEYS } from "./trace-attributes.js";
|
|
5
5
|
//#region src/trace/error-classification.ts
|
|
6
6
|
/**
|
|
@@ -308,4 +308,4 @@ function readConsistentRootString(roots, key, context) {
|
|
|
308
308
|
//#endregion
|
|
309
309
|
export { summarizeTraceErrors as i, recordAggregateMeasurements as n, summarizeExecutionMeasurements as r, readTaskFailureLabels as t };
|
|
310
310
|
|
|
311
|
-
//# sourceMappingURL=task-failure-attributes-
|
|
311
|
+
//# sourceMappingURL=task-failure-attributes-CUy9mkIY.js.map
|
package/dist/{task-failure-attributes-CZjZeBsY.js.map → task-failure-attributes-CUy9mkIY.js.map}
RENAMED
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"task-failure-attributes-CZjZeBsY.js","names":[],"sources":["../src/trace/error-classification.ts","../src/trace/execution-measurements.ts","../src/trace/task-failure-attributes.ts"],"sourcesContent":["import type { OtlpSpanRole } from './otlp-attributes'\n\nexport type TraceErrorRole = OtlpSpanRole\n\nexport interface TraceErrorSignal {\n id: string\n parentId?: string\n role: TraceErrorRole\n error: boolean\n processRoot: boolean\n}\n\nexport interface TraceErrorSummary {\n total: number\n execution: number\n process: number\n guardrail: number\n evaluation: number\n propagated: number\n unclassified: number\n}\n\n/**\n * Classify errored spans without counting a propagated parent status as a\n * second execution failure.\n */\nexport function summarizeTraceErrors(signals: readonly TraceErrorSignal[]): TraceErrorSummary {\n const byId = new Map<string, TraceErrorSignal>()\n for (const signal of signals) {\n if (byId.has(signal.id)) {\n throw new Error(`summarizeTraceErrors: duplicate span id '${signal.id}'`)\n }\n byId.set(signal.id, signal)\n }\n\n const propagated = new Set<string>()\n for (const signal of signals) {\n if (!signal.error) continue\n const visited = new Set<string>()\n let parentId = signal.parentId\n while (parentId && !visited.has(parentId)) {\n visited.add(parentId)\n const parent = byId.get(parentId)\n if (!parent) break\n if (parent.error) propagated.add(parent.id)\n parentId = parent.parentId\n }\n }\n\n const summary: TraceErrorSummary = {\n total: 0,\n execution: 0,\n process: 0,\n guardrail: 0,\n evaluation: 0,\n propagated: 0,\n unclassified: 0,\n }\n\n for (const signal of signals) {\n if (!signal.error) continue\n summary.total += 1\n if (signal.role === 'GUARDRAIL') {\n summary.guardrail += 1\n } else if (signal.role === 'EVALUATOR') {\n summary.evaluation += 1\n } else if (signal.processRoot) {\n summary.process += 1\n } else if (propagated.has(signal.id)) {\n summary.propagated += 1\n } else if (\n signal.role === 'AGENT' ||\n signal.role === 'CHAIN' ||\n signal.role === 'LLM' ||\n signal.role === 'TOOL'\n ) {\n summary.execution += 1\n } else {\n summary.unclassified += 1\n }\n }\n\n return summary\n}\n","import type { RunTokenUsage } from '../run-record'\nimport {\n LLM_CACHE_WRITE_TOKEN_ATTR_KEYS,\n LLM_CACHED_TOKEN_ATTR_KEYS,\n LLM_COST_ATTR_KEYS,\n LLM_INPUT_TOKEN_ATTR_KEYS,\n LLM_OUTPUT_TOKEN_ATTR_KEYS,\n LLM_REASONING_TOKEN_ATTR_KEYS,\n RUN_COST_ATTR_KEYS,\n} from './otlp-attributes'\n\nexport interface ExecutionMeasurementSpan {\n id: string\n parentId?: string\n attributes: Record<string, unknown>\n modelCall: boolean\n aggregate: boolean\n}\n\nexport interface MeasurementCoverage {\n value?: number\n reportingCalls: number\n complete: boolean\n}\n\nexport interface ExecutionMeasurements {\n tokenUsage: RunTokenUsage\n modelCallCount: number\n callSpanIds: string[]\n cost: MeasurementCoverage\n aggregate?: {\n tokenUsage: RunTokenUsage\n costUsd?: number\n }\n}\n\nconst TOKEN_MEASUREMENT_KEY_GROUPS = [\n LLM_INPUT_TOKEN_ATTR_KEYS,\n LLM_OUTPUT_TOKEN_ATTR_KEYS,\n LLM_REASONING_TOKEN_ATTR_KEYS,\n LLM_CACHED_TOKEN_ATTR_KEYS,\n LLM_CACHE_WRITE_TOKEN_ATTR_KEYS,\n] as const\n\nconst EXECUTION_MEASUREMENT_KEY_GROUPS = [\n ...TOKEN_MEASUREMENT_KEY_GROUPS,\n LLM_COST_ATTR_KEYS,\n] as const\n\ninterface RetainedCallSummary {\n callCount: number\n measurements: Array<{\n total: number\n reportingCalls: number\n }>\n}\n\n/**\n * Reconcile execution measurements across nested telemetry wrappers.\n * A measured parent is used only when a descendant call does not report the\n * same field, so aggregate wrappers neither duplicate complete child data nor\n * erase complementary parent fields.\n */\nexport function summarizeExecutionMeasurements(\n spans: ExecutionMeasurementSpan[],\n): ExecutionMeasurements {\n const byId = new Map<string, ExecutionMeasurementSpan>()\n for (const span of spans) {\n if (byId.has(span.id)) {\n throw new Error(`summarizeExecutionMeasurements: duplicate span id \"${span.id}\"`)\n }\n byId.set(span.id, span)\n }\n const tokenMeasurementKeys = TOKEN_MEASUREMENT_KEY_GROUPS.flat()\n const candidates = spans.filter(\n (span) =>\n span.modelCall ||\n (!span.aggregate && readNumber(span.attributes, tokenMeasurementKeys) !== undefined),\n )\n const candidateIds = new Set(candidates.map((span) => span.id))\n const candidateChildren = new Map<string, ExecutionMeasurementSpan[]>()\n for (const candidate of candidates) {\n const parentId = nearestCandidateParent(candidate, byId, candidateIds)\n if (!parentId) continue\n const children = candidateChildren.get(parentId) ?? []\n children.push(candidate)\n candidateChildren.set(parentId, children)\n }\n const aggregateIds = classifyAggregateSpans(candidates, candidateChildren)\n const untypedRunCostIds = new Set(\n spans\n .filter(\n (span) =>\n !span.modelCall &&\n !span.aggregate &&\n !candidateIds.has(span.id) &&\n readNumber(span.attributes, RUN_COST_ATTR_KEYS) !== undefined,\n )\n .map((span) => span.id),\n )\n const aggregateSourceIds = new Set(\n spans\n .filter(\n (span) => span.aggregate || aggregateIds.has(span.id) || untypedRunCostIds.has(span.id),\n )\n .map((span) => span.id),\n )\n\n const calls = candidates.filter((span) => !aggregateIds.has(span.id))\n const input = reconcileMeasurement(calls, byId, aggregateSourceIds, LLM_INPUT_TOKEN_ATTR_KEYS)\n const reasoning = reconcileMeasurement(\n calls,\n byId,\n aggregateSourceIds,\n LLM_REASONING_TOKEN_ATTR_KEYS,\n )\n const output = reconcileMeasurement(\n calls,\n byId,\n aggregateSourceIds,\n LLM_OUTPUT_TOKEN_ATTR_KEYS,\n LLM_REASONING_TOKEN_ATTR_KEYS,\n )\n const cached = reconcileMeasurement(calls, byId, aggregateSourceIds, LLM_CACHED_TOKEN_ATTR_KEYS)\n const cacheWrite = reconcileMeasurement(\n calls,\n byId,\n aggregateSourceIds,\n LLM_CACHE_WRITE_TOKEN_ATTR_KEYS,\n )\n const aggregate = summarizeAggregateMeasurements(\n spans,\n byId,\n new Set([...aggregateIds, ...untypedRunCostIds]),\n )\n\n return {\n tokenUsage: {\n input: input.value ?? 0,\n output: Math.max(output.value ?? 0, reasoning.value ?? 0),\n ...(reasoning.value !== undefined ? { reasoning: reasoning.value } : {}),\n ...(cached.value !== undefined ? { cached: cached.value } : {}),\n ...(cacheWrite.value !== undefined ? { cacheWrite: cacheWrite.value } : {}),\n },\n modelCallCount: calls.length,\n callSpanIds: calls.map((span) => span.id),\n cost: reconcileMeasurement(calls, byId, aggregateSourceIds, LLM_COST_ATTR_KEYS),\n ...(aggregate ? { aggregate } : {}),\n }\n}\n\nexport function recordAggregateMeasurements(\n raw: Record<string, number>,\n aggregate: ExecutionMeasurements['aggregate'],\n): void {\n if (!aggregate) return\n raw.aggregate_prompt_tokens = aggregate.tokenUsage.input\n raw.aggregate_completion_tokens = aggregate.tokenUsage.output\n if (aggregate.tokenUsage.reasoning !== undefined)\n raw.aggregate_reasoning_tokens = aggregate.tokenUsage.reasoning\n if (aggregate.tokenUsage.cached !== undefined)\n raw.aggregate_cached_tokens = aggregate.tokenUsage.cached\n if (aggregate.tokenUsage.cacheWrite !== undefined)\n raw.aggregate_cache_write_tokens = aggregate.tokenUsage.cacheWrite\n if (aggregate.costUsd !== undefined) raw.aggregate_cost_usd = aggregate.costUsd\n}\n\nfunction summarizeAggregateMeasurements(\n spans: ExecutionMeasurementSpan[],\n byId: Map<string, ExecutionMeasurementSpan>,\n aggregateIds: Set<string>,\n): ExecutionMeasurements['aggregate'] {\n const aggregates = spans.filter((span) => span.aggregate || aggregateIds.has(span.id))\n const input = reconcileTopLevelMeasurement(aggregates, byId, LLM_INPUT_TOKEN_ATTR_KEYS)\n const output = reconcileTopLevelMeasurement(aggregates, byId, LLM_OUTPUT_TOKEN_ATTR_KEYS)\n const reasoning = reconcileTopLevelMeasurement(aggregates, byId, LLM_REASONING_TOKEN_ATTR_KEYS)\n const cached = reconcileTopLevelMeasurement(aggregates, byId, LLM_CACHED_TOKEN_ATTR_KEYS)\n const cacheWrite = reconcileTopLevelMeasurement(aggregates, byId, LLM_CACHE_WRITE_TOKEN_ATTR_KEYS)\n const costUsd = reconcileTopLevelMeasurement(aggregates, byId, LLM_COST_ATTR_KEYS)\n if (\n input === undefined &&\n output === undefined &&\n reasoning === undefined &&\n cached === undefined &&\n cacheWrite === undefined &&\n costUsd === undefined\n )\n return undefined\n return {\n tokenUsage: {\n input: input ?? 0,\n output: output ?? reasoning ?? 0,\n ...(reasoning !== undefined ? { reasoning } : {}),\n ...(cached !== undefined ? { cached } : {}),\n ...(cacheWrite !== undefined ? { cacheWrite } : {}),\n },\n ...(costUsd !== undefined ? { costUsd } : {}),\n }\n}\n\nfunction reconcileTopLevelMeasurement(\n spans: ExecutionMeasurementSpan[],\n byId: Map<string, ExecutionMeasurementSpan>,\n keys: readonly string[],\n): number | undefined {\n const selected = new Map<string, number>()\n for (const span of spans) {\n const value = readNumber(span.attributes, keys)\n if (value !== undefined) selected.set(span.id, value)\n }\n for (const spanId of [...selected.keys()]) {\n const span = byId.get(spanId)\n if (!span) continue\n if (ancestorIds(span, byId).some((ancestorId) => selected.has(ancestorId))) {\n selected.delete(spanId)\n }\n }\n return selected.size > 0\n ? [...selected.values()].reduce((total, value) => total + value, 0)\n : undefined\n}\n\nfunction reconcileMeasurement(\n calls: ExecutionMeasurementSpan[],\n byId: Map<string, ExecutionMeasurementSpan>,\n aggregateSourceIds: Set<string>,\n keys: readonly string[],\n fallbackKeys?: readonly string[],\n): MeasurementCoverage {\n const selected = new Map<string, number>()\n const callIds = new Set(calls.map((call) => call.id))\n let reportingCalls = 0\n\n for (const call of calls) {\n const primary = nearestMeasurement(call, byId, callIds, aggregateSourceIds, keys)\n const fallback = fallbackKeys\n ? nearestMeasurement(call, byId, callIds, aggregateSourceIds, fallbackKeys)\n : undefined\n const source =\n primary && fallback && primary.span.id === fallback.span.id\n ? { span: primary.span, value: Math.max(primary.value, fallback.value) }\n : (primary ?? fallback)\n if (!source) continue\n reportingCalls += 1\n selected.set(source.span.id, source.value)\n }\n\n for (const spanId of [...selected.keys()]) {\n const span = byId.get(spanId)\n if (!span) continue\n if (\n ancestorIds(span, byId).some(\n (ancestorId) => selected.has(ancestorId) && !callIds.has(ancestorId),\n )\n ) {\n selected.delete(spanId)\n }\n }\n\n return {\n ...(selected.size > 0\n ? { value: [...selected.values()].reduce((total, value) => total + value, 0) }\n : {}),\n reportingCalls,\n complete: calls.length > 0 && reportingCalls === calls.length,\n }\n}\n\nfunction nearestMeasurement(\n call: ExecutionMeasurementSpan,\n byId: Map<string, ExecutionMeasurementSpan>,\n callIds: Set<string>,\n aggregateSourceIds: Set<string>,\n keys: readonly string[],\n): { span: ExecutionMeasurementSpan; value: number } | undefined {\n let current: ExecutionMeasurementSpan | undefined = call\n const seen = new Set<string>()\n while (current && !seen.has(current.id)) {\n seen.add(current.id)\n const value = readNumber(current.attributes, keys)\n if (\n value !== undefined &&\n (current.id === call.id || (!callIds.has(current.id) && aggregateSourceIds.has(current.id)))\n ) {\n return { span: current, value }\n }\n current = current.parentId ? byId.get(current.parentId) : undefined\n }\n return undefined\n}\n\nfunction classifyAggregateSpans(\n candidates: ExecutionMeasurementSpan[],\n childrenById: Map<string, ExecutionMeasurementSpan[]>,\n): Set<string> {\n const aggregateIds = new Set<string>()\n const summaries = new Map<string, RetainedCallSummary>()\n const visiting = new Set<string>()\n\n const visit = (span: ExecutionMeasurementSpan): RetainedCallSummary => {\n const cached = summaries.get(span.id)\n if (cached) return cached\n if (visiting.has(span.id)) return emptyRetainedCallSummary()\n visiting.add(span.id)\n\n const descendants = emptyRetainedCallSummary()\n for (const child of childrenById.get(span.id) ?? []) {\n const childDescendants = visit(child)\n if (!aggregateIds.has(child.id)) addRetainedCall(descendants, child)\n addRetainedCallSummary(descendants, childDescendants)\n }\n\n if (\n descendants.callCount > 0 &&\n (!span.modelCall || hasCompatibleDescendantMeasurements(span, descendants))\n ) {\n aggregateIds.add(span.id)\n }\n\n visiting.delete(span.id)\n summaries.set(span.id, descendants)\n return descendants\n }\n\n for (const candidate of candidates) visit(candidate)\n return aggregateIds\n}\n\nfunction emptyRetainedCallSummary(): RetainedCallSummary {\n return {\n callCount: 0,\n measurements: EXECUTION_MEASUREMENT_KEY_GROUPS.map(() => ({\n total: 0,\n reportingCalls: 0,\n })),\n }\n}\n\nfunction addRetainedCall(summary: RetainedCallSummary, span: ExecutionMeasurementSpan): void {\n summary.callCount += 1\n for (let index = 0; index < EXECUTION_MEASUREMENT_KEY_GROUPS.length; index += 1) {\n const value = readNumber(span.attributes, EXECUTION_MEASUREMENT_KEY_GROUPS[index]!)\n if (value === undefined) continue\n const measurement = summary.measurements[index]!\n measurement.total += value\n measurement.reportingCalls += 1\n }\n}\n\nfunction addRetainedCallSummary(target: RetainedCallSummary, source: RetainedCallSummary): void {\n target.callCount += source.callCount\n for (let index = 0; index < target.measurements.length; index += 1) {\n const measurement = target.measurements[index]!\n const sourceMeasurement = source.measurements[index]!\n measurement.total += sourceMeasurement.total\n measurement.reportingCalls += sourceMeasurement.reportingCalls\n }\n}\n\nfunction hasCompatibleDescendantMeasurements(\n span: ExecutionMeasurementSpan,\n descendants: RetainedCallSummary,\n): boolean {\n let parentMeasurements = 0\n let descendantMeasurements = 0\n for (let index = 0; index < EXECUTION_MEASUREMENT_KEY_GROUPS.length; index += 1) {\n const keys = EXECUTION_MEASUREMENT_KEY_GROUPS[index]!\n const parentValue = readNumber(span.attributes, keys)\n const measurement = descendants.measurements[index]!\n if (parentValue !== undefined) parentMeasurements += 1\n if (measurement.reportingCalls > 0) descendantMeasurements += 1\n if (parentValue === undefined || measurement.reportingCalls === 0) continue\n if (measurement.reportingCalls !== descendants.callCount) continue\n if (Math.abs(parentValue - measurement.total) > 1e-12) return false\n }\n return parentMeasurements === 0 || descendantMeasurements > 0\n}\n\nfunction ancestorIds(\n span: ExecutionMeasurementSpan,\n byId: Map<string, ExecutionMeasurementSpan>,\n): string[] {\n const ids: string[] = []\n const seen = new Set<string>()\n let parentId = span.parentId\n while (parentId && !seen.has(parentId)) {\n ids.push(parentId)\n seen.add(parentId)\n parentId = byId.get(parentId)?.parentId\n }\n return ids\n}\n\nfunction nearestCandidateParent(\n span: ExecutionMeasurementSpan,\n byId: Map<string, ExecutionMeasurementSpan>,\n candidateIds: Set<string>,\n): string | undefined {\n const seen = new Set<string>()\n let parentId = span.parentId\n while (parentId && !seen.has(parentId)) {\n if (candidateIds.has(parentId)) return parentId\n seen.add(parentId)\n parentId = byId.get(parentId)?.parentId\n }\n return undefined\n}\n\nfunction readNumber(\n attributes: Record<string, unknown>,\n keys: readonly string[],\n): number | undefined {\n for (const key of keys) {\n const value = attributes[key]\n const parsed =\n typeof value === 'number'\n ? value\n : typeof value === 'string' && value.length > 0\n ? Number(value)\n : Number.NaN\n if (Number.isFinite(parsed) && parsed >= 0) return parsed\n }\n return undefined\n}\n","import { ValidationError } from '../errors'\nimport { FAILURE_CLASSES, type FailureClass } from './schema'\n\nconst TASK_FAILURE_CLASS_ATTR = 'tangle.task.failure_class'\nconst TASK_FAILURE_MODE_ATTR = 'tangle.task.failure_mode'\n\ninterface AttributeCarrier {\n attributes: Record<string, unknown>\n}\n\nexport type TaskFailureLabels =\n | { failureClass?: undefined; failureMode?: undefined }\n | { failureClass: 'success'; failureMode?: undefined }\n | { failureClass: Exclude<FailureClass, 'success'>; failureMode?: string }\n\nexport function readTaskFailureLabels(\n roots: readonly AttributeCarrier[],\n context: string,\n): TaskFailureLabels {\n const failureClass = readConsistentRootString(roots, TASK_FAILURE_CLASS_ATTR, context)\n const failureMode = readConsistentRootString(roots, TASK_FAILURE_MODE_ATTR, context)\n\n if (failureClass !== undefined && !FAILURE_CLASSES.includes(failureClass as FailureClass)) {\n throw new ValidationError(\n `${context}: ${TASK_FAILURE_CLASS_ATTR} must be one of ${FAILURE_CLASSES.join(', ')}`,\n )\n }\n if (failureMode !== undefined && (failureClass === undefined || failureClass === 'success')) {\n throw new ValidationError(\n `${context}: ${TASK_FAILURE_MODE_ATTR} requires a non-success ${TASK_FAILURE_CLASS_ATTR}`,\n )\n }\n\n if (failureClass === undefined) return {}\n if (failureClass === 'success') return { failureClass }\n return {\n failureClass: failureClass as Exclude<FailureClass, 'success'>,\n ...(failureMode ? { failureMode } : {}),\n }\n}\n\nfunction readConsistentRootString(\n roots: readonly AttributeCarrier[],\n key: string,\n context: string,\n): string | undefined {\n const values = new Set<string>()\n for (const root of roots) {\n if (!Object.hasOwn(root.attributes, key)) continue\n const value = root.attributes[key]\n if (typeof value !== 'string' || value.trim().length === 0) {\n throw new ValidationError(`${context}: ${key} must be a non-empty string`)\n }\n values.add(value)\n }\n\n if (values.size > 1) {\n throw new ValidationError(\n `${context}: conflicting ${key} values: ${[...values].sort().join(', ')}`,\n )\n }\n return values.values().next().value\n}\n"],"mappings":";;;;;;;;;AA0BA,SAAgB,qBAAqB,SAAyD;CAC5F,MAAM,uBAAO,IAAI,IAA8B;CAC/C,KAAK,MAAM,UAAU,SAAS;EAC5B,IAAI,KAAK,IAAI,OAAO,EAAE,GACpB,MAAM,IAAI,MAAM,4CAA4C,OAAO,GAAG,EAAE;EAE1E,KAAK,IAAI,OAAO,IAAI,MAAM;CAC5B;CAEA,MAAM,6BAAa,IAAI,IAAY;CACnC,KAAK,MAAM,UAAU,SAAS;EAC5B,IAAI,CAAC,OAAO,OAAO;EACnB,MAAM,0BAAU,IAAI,IAAY;EAChC,IAAI,WAAW,OAAO;EACtB,OAAO,YAAY,CAAC,QAAQ,IAAI,QAAQ,GAAG;GACzC,QAAQ,IAAI,QAAQ;GACpB,MAAM,SAAS,KAAK,IAAI,QAAQ;GAChC,IAAI,CAAC,QAAQ;GACb,IAAI,OAAO,OAAO,WAAW,IAAI,OAAO,EAAE;GAC1C,WAAW,OAAO;EACpB;CACF;CAEA,MAAM,UAA6B;EACjC,OAAO;EACP,WAAW;EACX,SAAS;EACT,WAAW;EACX,YAAY;EACZ,YAAY;EACZ,cAAc;CAChB;CAEA,KAAK,MAAM,UAAU,SAAS;EAC5B,IAAI,CAAC,OAAO,OAAO;EACnB,QAAQ,SAAS;EACjB,IAAI,OAAO,SAAS,aAClB,QAAQ,aAAa;OAChB,IAAI,OAAO,SAAS,aACzB,QAAQ,cAAc;OACjB,IAAI,OAAO,aAChB,QAAQ,WAAW;OACd,IAAI,WAAW,IAAI,OAAO,EAAE,GACjC,QAAQ,cAAc;OACjB,IACL,OAAO,SAAS,WAChB,OAAO,SAAS,WAChB,OAAO,SAAS,SAChB,OAAO,SAAS,QAEhB,QAAQ,aAAa;OAErB,QAAQ,gBAAgB;CAE5B;CAEA,OAAO;AACT;;;AC/CA,MAAM,+BAA+B;CACnC;CACA;CACA;CACA;CACA;AACF;AAEA,MAAM,mCAAmC,CACvC,GAAG,8BACH,kBACF;;;;;;;AAgBA,SAAgB,+BACd,OACuB;CACvB,MAAM,uBAAO,IAAI,IAAsC;CACvD,KAAK,MAAM,QAAQ,OAAO;EACxB,IAAI,KAAK,IAAI,KAAK,EAAE,GAClB,MAAM,IAAI,MAAM,sDAAsD,KAAK,GAAG,EAAE;EAElF,KAAK,IAAI,KAAK,IAAI,IAAI;CACxB;CACA,MAAM,uBAAuB,6BAA6B,KAAK;CAC/D,MAAM,aAAa,MAAM,QACtB,SACC,KAAK,aACJ,CAAC,KAAK,aAAa,WAAW,KAAK,YAAY,oBAAoB,MAAM,KAAA,CAC9E;CACA,MAAM,eAAe,IAAI,IAAI,WAAW,KAAK,SAAS,KAAK,EAAE,CAAC;CAC9D,MAAM,oCAAoB,IAAI,IAAwC;CACtE,KAAK,MAAM,aAAa,YAAY;EAClC,MAAM,WAAW,uBAAuB,WAAW,MAAM,YAAY;EACrE,IAAI,CAAC,UAAU;EACf,MAAM,WAAW,kBAAkB,IAAI,QAAQ,KAAK,CAAC;EACrD,SAAS,KAAK,SAAS;EACvB,kBAAkB,IAAI,UAAU,QAAQ;CAC1C;CACA,MAAM,eAAe,uBAAuB,YAAY,iBAAiB;CACzE,MAAM,oBAAoB,IAAI,IAC5B,MACG,QACE,SACC,CAAC,KAAK,aACN,CAAC,KAAK,aACN,CAAC,aAAa,IAAI,KAAK,EAAE,KACzB,WAAW,KAAK,YAAY,kBAAkB,MAAM,KAAA,CACxD,CAAC,CACA,KAAK,SAAS,KAAK,EAAE,CAC1B;CACA,MAAM,qBAAqB,IAAI,IAC7B,MACG,QACE,SAAS,KAAK,aAAa,aAAa,IAAI,KAAK,EAAE,KAAK,kBAAkB,IAAI,KAAK,EAAE,CACxF,CAAC,CACA,KAAK,SAAS,KAAK,EAAE,CAC1B;CAEA,MAAM,QAAQ,WAAW,QAAQ,SAAS,CAAC,aAAa,IAAI,KAAK,EAAE,CAAC;CACpE,MAAM,QAAQ,qBAAqB,OAAO,MAAM,oBAAoB,yBAAyB;CAC7F,MAAM,YAAY,qBAChB,OACA,MACA,oBACA,6BACF;CACA,MAAM,SAAS,qBACb,OACA,MACA,oBACA,4BACA,6BACF;CACA,MAAM,SAAS,qBAAqB,OAAO,MAAM,oBAAoB,0BAA0B;CAC/F,MAAM,aAAa,qBACjB,OACA,MACA,oBACA,+BACF;CACA,MAAM,YAAY,+BAChB,OACA,sBACA,IAAI,IAAI,CAAC,GAAG,cAAc,GAAG,iBAAiB,CAAC,CACjD;CAEA,OAAO;EACL,YAAY;GACV,OAAO,MAAM,SAAS;GACtB,QAAQ,KAAK,IAAI,OAAO,SAAS,GAAG,UAAU,SAAS,CAAC;GACxD,GAAI,UAAU,UAAU,KAAA,IAAY,EAAE,WAAW,UAAU,MAAM,IAAI,CAAC;GACtE,GAAI,OAAO,UAAU,KAAA,IAAY,EAAE,QAAQ,OAAO,MAAM,IAAI,CAAC;GAC7D,GAAI,WAAW,UAAU,KAAA,IAAY,EAAE,YAAY,WAAW,MAAM,IAAI,CAAC;EAC3E;EACA,gBAAgB,MAAM;EACtB,aAAa,MAAM,KAAK,SAAS,KAAK,EAAE;EACxC,MAAM,qBAAqB,OAAO,MAAM,oBAAoB,kBAAkB;EAC9E,GAAI,YAAY,EAAE,UAAU,IAAI,CAAC;CACnC;AACF;AAEA,SAAgB,4BACd,KACA,WACM;CACN,IAAI,CAAC,WAAW;CAChB,IAAI,0BAA0B,UAAU,WAAW;CACnD,IAAI,8BAA8B,UAAU,WAAW;CACvD,IAAI,UAAU,WAAW,cAAc,KAAA,GACrC,IAAI,6BAA6B,UAAU,WAAW;CACxD,IAAI,UAAU,WAAW,WAAW,KAAA,GAClC,IAAI,0BAA0B,UAAU,WAAW;CACrD,IAAI,UAAU,WAAW,eAAe,KAAA,GACtC,IAAI,+BAA+B,UAAU,WAAW;CAC1D,IAAI,UAAU,YAAY,KAAA,GAAW,IAAI,qBAAqB,UAAU;AAC1E;AAEA,SAAS,+BACP,OACA,MACA,cACoC;CACpC,MAAM,aAAa,MAAM,QAAQ,SAAS,KAAK,aAAa,aAAa,IAAI,KAAK,EAAE,CAAC;CACrF,MAAM,QAAQ,6BAA6B,YAAY,MAAM,yBAAyB;CACtF,MAAM,SAAS,6BAA6B,YAAY,MAAM,0BAA0B;CACxF,MAAM,YAAY,6BAA6B,YAAY,MAAM,6BAA6B;CAC9F,MAAM,SAAS,6BAA6B,YAAY,MAAM,0BAA0B;CACxF,MAAM,aAAa,6BAA6B,YAAY,MAAM,+BAA+B;CACjG,MAAM,UAAU,6BAA6B,YAAY,MAAM,kBAAkB;CACjF,IACE,UAAU,KAAA,KACV,WAAW,KAAA,KACX,cAAc,KAAA,KACd,WAAW,KAAA,KACX,eAAe,KAAA,KACf,YAAY,KAAA,GAEZ,OAAO,KAAA;CACT,OAAO;EACL,YAAY;GACV,OAAO,SAAS;GAChB,QAAQ,UAAU,aAAa;GAC/B,GAAI,cAAc,KAAA,IAAY,EAAE,UAAU,IAAI,CAAC;GAC/C,GAAI,WAAW,KAAA,IAAY,EAAE,OAAO,IAAI,CAAC;GACzC,GAAI,eAAe,KAAA,IAAY,EAAE,WAAW,IAAI,CAAC;EACnD;EACA,GAAI,YAAY,KAAA,IAAY,EAAE,QAAQ,IAAI,CAAC;CAC7C;AACF;AAEA,SAAS,6BACP,OACA,MACA,MACoB;CACpB,MAAM,2BAAW,IAAI,IAAoB;CACzC,KAAK,MAAM,QAAQ,OAAO;EACxB,MAAM,QAAQ,WAAW,KAAK,YAAY,IAAI;EAC9C,IAAI,UAAU,KAAA,GAAW,SAAS,IAAI,KAAK,IAAI,KAAK;CACtD;CACA,KAAK,MAAM,UAAU,CAAC,GAAG,SAAS,KAAK,CAAC,GAAG;EACzC,MAAM,OAAO,KAAK,IAAI,MAAM;EAC5B,IAAI,CAAC,MAAM;EACX,IAAI,YAAY,MAAM,IAAI,CAAC,CAAC,MAAM,eAAe,SAAS,IAAI,UAAU,CAAC,GACvE,SAAS,OAAO,MAAM;CAE1B;CACA,OAAO,SAAS,OAAO,IACnB,CAAC,GAAG,SAAS,OAAO,CAAC,CAAC,CAAC,QAAQ,OAAO,UAAU,QAAQ,OAAO,CAAC,IAChE,KAAA;AACN;AAEA,SAAS,qBACP,OACA,MACA,oBACA,MACA,cACqB;CACrB,MAAM,2BAAW,IAAI,IAAoB;CACzC,MAAM,UAAU,IAAI,IAAI,MAAM,KAAK,SAAS,KAAK,EAAE,CAAC;CACpD,IAAI,iBAAiB;CAErB,KAAK,MAAM,QAAQ,OAAO;EACxB,MAAM,UAAU,mBAAmB,MAAM,MAAM,SAAS,oBAAoB,IAAI;EAChF,MAAM,WAAW,eACb,mBAAmB,MAAM,MAAM,SAAS,oBAAoB,YAAY,IACxE,KAAA;EACJ,MAAM,SACJ,WAAW,YAAY,QAAQ,KAAK,OAAO,SAAS,KAAK,KACrD;GAAE,MAAM,QAAQ;GAAM,OAAO,KAAK,IAAI,QAAQ,OAAO,SAAS,KAAK;EAAE,IACpE,WAAW;EAClB,IAAI,CAAC,QAAQ;EACb,kBAAkB;EAClB,SAAS,IAAI,OAAO,KAAK,IAAI,OAAO,KAAK;CAC3C;CAEA,KAAK,MAAM,UAAU,CAAC,GAAG,SAAS,KAAK,CAAC,GAAG;EACzC,MAAM,OAAO,KAAK,IAAI,MAAM;EAC5B,IAAI,CAAC,MAAM;EACX,IACE,YAAY,MAAM,IAAI,CAAC,CAAC,MACrB,eAAe,SAAS,IAAI,UAAU,KAAK,CAAC,QAAQ,IAAI,UAAU,CACrE,GAEA,SAAS,OAAO,MAAM;CAE1B;CAEA,OAAO;EACL,GAAI,SAAS,OAAO,IAChB,EAAE,OAAO,CAAC,GAAG,SAAS,OAAO,CAAC,CAAC,CAAC,QAAQ,OAAO,UAAU,QAAQ,OAAO,CAAC,EAAE,IAC3E,CAAC;EACL;EACA,UAAU,MAAM,SAAS,KAAK,mBAAmB,MAAM;CACzD;AACF;AAEA,SAAS,mBACP,MACA,MACA,SACA,oBACA,MAC+D;CAC/D,IAAI,UAAgD;CACpD,MAAM,uBAAO,IAAI,IAAY;CAC7B,OAAO,WAAW,CAAC,KAAK,IAAI,QAAQ,EAAE,GAAG;EACvC,KAAK,IAAI,QAAQ,EAAE;EACnB,MAAM,QAAQ,WAAW,QAAQ,YAAY,IAAI;EACjD,IACE,UAAU,KAAA,MACT,QAAQ,OAAO,KAAK,MAAO,CAAC,QAAQ,IAAI,QAAQ,EAAE,KAAK,mBAAmB,IAAI,QAAQ,EAAE,IAEzF,OAAO;GAAE,MAAM;GAAS;EAAM;EAEhC,UAAU,QAAQ,WAAW,KAAK,IAAI,QAAQ,QAAQ,IAAI,KAAA;CAC5D;AAEF;AAEA,SAAS,uBACP,YACA,cACa;CACb,MAAM,+BAAe,IAAI,IAAY;CACrC,MAAM,4BAAY,IAAI,IAAiC;CACvD,MAAM,2BAAW,IAAI,IAAY;CAEjC,MAAM,SAAS,SAAwD;EACrE,MAAM,SAAS,UAAU,IAAI,KAAK,EAAE;EACpC,IAAI,QAAQ,OAAO;EACnB,IAAI,SAAS,IAAI,KAAK,EAAE,GAAG,OAAO,yBAAyB;EAC3D,SAAS,IAAI,KAAK,EAAE;EAEpB,MAAM,cAAc,yBAAyB;EAC7C,KAAK,MAAM,SAAS,aAAa,IAAI,KAAK,EAAE,KAAK,CAAC,GAAG;GACnD,MAAM,mBAAmB,MAAM,KAAK;GACpC,IAAI,CAAC,aAAa,IAAI,MAAM,EAAE,GAAG,gBAAgB,aAAa,KAAK;GACnE,uBAAuB,aAAa,gBAAgB;EACtD;EAEA,IACE,YAAY,YAAY,MACvB,CAAC,KAAK,aAAa,oCAAoC,MAAM,WAAW,IAEzE,aAAa,IAAI,KAAK,EAAE;EAG1B,SAAS,OAAO,KAAK,EAAE;EACvB,UAAU,IAAI,KAAK,IAAI,WAAW;EAClC,OAAO;CACT;CAEA,KAAK,MAAM,aAAa,YAAY,MAAM,SAAS;CACnD,OAAO;AACT;AAEA,SAAS,2BAAgD;CACvD,OAAO;EACL,WAAW;EACX,cAAc,iCAAiC,WAAW;GACxD,OAAO;GACP,gBAAgB;EAClB,EAAE;CACJ;AACF;AAEA,SAAS,gBAAgB,SAA8B,MAAsC;CAC3F,QAAQ,aAAa;CACrB,KAAK,IAAI,QAAQ,GAAG,QAAQ,iCAAiC,QAAQ,SAAS,GAAG;EAC/E,MAAM,QAAQ,WAAW,KAAK,YAAY,iCAAiC,MAAO;EAClF,IAAI,UAAU,KAAA,GAAW;EACzB,MAAM,cAAc,QAAQ,aAAa;EACzC,YAAY,SAAS;EACrB,YAAY,kBAAkB;CAChC;AACF;AAEA,SAAS,uBAAuB,QAA6B,QAAmC;CAC9F,OAAO,aAAa,OAAO;CAC3B,KAAK,IAAI,QAAQ,GAAG,QAAQ,OAAO,aAAa,QAAQ,SAAS,GAAG;EAClE,MAAM,cAAc,OAAO,aAAa;EACxC,MAAM,oBAAoB,OAAO,aAAa;EAC9C,YAAY,SAAS,kBAAkB;EACvC,YAAY,kBAAkB,kBAAkB;CAClD;AACF;AAEA,SAAS,oCACP,MACA,aACS;CACT,IAAI,qBAAqB;CACzB,IAAI,yBAAyB;CAC7B,KAAK,IAAI,QAAQ,GAAG,QAAQ,iCAAiC,QAAQ,SAAS,GAAG;EAC/E,MAAM,OAAO,iCAAiC;EAC9C,MAAM,cAAc,WAAW,KAAK,YAAY,IAAI;EACpD,MAAM,cAAc,YAAY,aAAa;EAC7C,IAAI,gBAAgB,KAAA,GAAW,sBAAsB;EACrD,IAAI,YAAY,iBAAiB,GAAG,0BAA0B;EAC9D,IAAI,gBAAgB,KAAA,KAAa,YAAY,mBAAmB,GAAG;EACnE,IAAI,YAAY,mBAAmB,YAAY,WAAW;EAC1D,IAAI,KAAK,IAAI,cAAc,YAAY,KAAK,IAAI,OAAO,OAAO;CAChE;CACA,OAAO,uBAAuB,KAAK,yBAAyB;AAC9D;AAEA,SAAS,YACP,MACA,MACU;CACV,MAAM,MAAgB,CAAC;CACvB,MAAM,uBAAO,IAAI,IAAY;CAC7B,IAAI,WAAW,KAAK;CACpB,OAAO,YAAY,CAAC,KAAK,IAAI,QAAQ,GAAG;EACtC,IAAI,KAAK,QAAQ;EACjB,KAAK,IAAI,QAAQ;EACjB,WAAW,KAAK,IAAI,QAAQ,CAAC,EAAE;CACjC;CACA,OAAO;AACT;AAEA,SAAS,uBACP,MACA,MACA,cACoB;CACpB,MAAM,uBAAO,IAAI,IAAY;CAC7B,IAAI,WAAW,KAAK;CACpB,OAAO,YAAY,CAAC,KAAK,IAAI,QAAQ,GAAG;EACtC,IAAI,aAAa,IAAI,QAAQ,GAAG,OAAO;EACvC,KAAK,IAAI,QAAQ;EACjB,WAAW,KAAK,IAAI,QAAQ,CAAC,EAAE;CACjC;AAEF;AAEA,SAAS,WACP,YACA,MACoB;CACpB,KAAK,MAAM,OAAO,MAAM;EACtB,MAAM,QAAQ,WAAW;EACzB,MAAM,SACJ,OAAO,UAAU,WACb,QACA,OAAO,UAAU,YAAY,MAAM,SAAS,IAC1C,OAAO,KAAK,IACZ;EACR,IAAI,OAAO,SAAS,MAAM,KAAK,UAAU,GAAG,OAAO;CACrD;AAEF;;;ACpaA,MAAM,0BAA0B;AAChC,MAAM,yBAAyB;AAW/B,SAAgB,sBACd,OACA,SACmB;CACnB,MAAM,eAAe,yBAAyB,OAAO,yBAAyB,OAAO;CACrF,MAAM,cAAc,yBAAyB,OAAO,wBAAwB,OAAO;CAEnF,IAAI,iBAAiB,KAAA,KAAa,CAAC,gBAAgB,SAAS,YAA4B,GACtF,MAAM,IAAI,gBACR,GAAG,QAAQ,IAAI,wBAAwB,kBAAkB,gBAAgB,KAAK,IAAI,GACpF;CAEF,IAAI,gBAAgB,KAAA,MAAc,iBAAiB,KAAA,KAAa,iBAAiB,YAC/E,MAAM,IAAI,gBACR,GAAG,QAAQ,IAAI,uBAAuB,0BAA0B,yBAClE;CAGF,IAAI,iBAAiB,KAAA,GAAW,OAAO,CAAC;CACxC,IAAI,iBAAiB,WAAW,OAAO,EAAE,aAAa;CACtD,OAAO;EACS;EACd,GAAI,cAAc,EAAE,YAAY,IAAI,CAAC;CACvC;AACF;AAEA,SAAS,yBACP,OACA,KACA,SACoB;CACpB,MAAM,yBAAS,IAAI,IAAY;CAC/B,KAAK,MAAM,QAAQ,OAAO;EACxB,IAAI,CAAC,OAAO,OAAO,KAAK,YAAY,GAAG,GAAG;EAC1C,MAAM,QAAQ,KAAK,WAAW;EAC9B,IAAI,OAAO,UAAU,YAAY,MAAM,KAAK,CAAC,CAAC,WAAW,GACvD,MAAM,IAAI,gBAAgB,GAAG,QAAQ,IAAI,IAAI,4BAA4B;EAE3E,OAAO,IAAI,KAAK;CAClB;CAEA,IAAI,OAAO,OAAO,GAChB,MAAM,IAAI,gBACR,GAAG,QAAQ,gBAAgB,IAAI,WAAW,CAAC,GAAG,MAAM,CAAC,CAAC,KAAK,CAAC,CAAC,KAAK,IAAI,GACxE;CAEF,OAAO,OAAO,OAAO,CAAC,CAAC,KAAK,CAAC,CAAC;AAChC"}
|
|
1
|
+
{"version":3,"file":"task-failure-attributes-CUy9mkIY.js","names":[],"sources":["../src/trace/error-classification.ts","../src/trace/execution-measurements.ts","../src/trace/task-failure-attributes.ts"],"sourcesContent":["import type { OtlpSpanRole } from './otlp-attributes'\n\nexport type TraceErrorRole = OtlpSpanRole\n\nexport interface TraceErrorSignal {\n id: string\n parentId?: string\n role: TraceErrorRole\n error: boolean\n processRoot: boolean\n}\n\nexport interface TraceErrorSummary {\n total: number\n execution: number\n process: number\n guardrail: number\n evaluation: number\n propagated: number\n unclassified: number\n}\n\n/**\n * Classify errored spans without counting a propagated parent status as a\n * second execution failure.\n */\nexport function summarizeTraceErrors(signals: readonly TraceErrorSignal[]): TraceErrorSummary {\n const byId = new Map<string, TraceErrorSignal>()\n for (const signal of signals) {\n if (byId.has(signal.id)) {\n throw new Error(`summarizeTraceErrors: duplicate span id '${signal.id}'`)\n }\n byId.set(signal.id, signal)\n }\n\n const propagated = new Set<string>()\n for (const signal of signals) {\n if (!signal.error) continue\n const visited = new Set<string>()\n let parentId = signal.parentId\n while (parentId && !visited.has(parentId)) {\n visited.add(parentId)\n const parent = byId.get(parentId)\n if (!parent) break\n if (parent.error) propagated.add(parent.id)\n parentId = parent.parentId\n }\n }\n\n const summary: TraceErrorSummary = {\n total: 0,\n execution: 0,\n process: 0,\n guardrail: 0,\n evaluation: 0,\n propagated: 0,\n unclassified: 0,\n }\n\n for (const signal of signals) {\n if (!signal.error) continue\n summary.total += 1\n if (signal.role === 'GUARDRAIL') {\n summary.guardrail += 1\n } else if (signal.role === 'EVALUATOR') {\n summary.evaluation += 1\n } else if (signal.processRoot) {\n summary.process += 1\n } else if (propagated.has(signal.id)) {\n summary.propagated += 1\n } else if (\n signal.role === 'AGENT' ||\n signal.role === 'CHAIN' ||\n signal.role === 'LLM' ||\n signal.role === 'TOOL'\n ) {\n summary.execution += 1\n } else {\n summary.unclassified += 1\n }\n }\n\n return summary\n}\n","import type { RunTokenUsage } from '../run-record'\nimport {\n LLM_CACHE_WRITE_TOKEN_ATTR_KEYS,\n LLM_CACHED_TOKEN_ATTR_KEYS,\n LLM_COST_ATTR_KEYS,\n LLM_INPUT_TOKEN_ATTR_KEYS,\n LLM_OUTPUT_TOKEN_ATTR_KEYS,\n LLM_REASONING_TOKEN_ATTR_KEYS,\n RUN_COST_ATTR_KEYS,\n} from './otlp-attributes'\n\nexport interface ExecutionMeasurementSpan {\n id: string\n parentId?: string\n attributes: Record<string, unknown>\n modelCall: boolean\n aggregate: boolean\n}\n\nexport interface MeasurementCoverage {\n value?: number\n reportingCalls: number\n complete: boolean\n}\n\nexport interface ExecutionMeasurements {\n tokenUsage: RunTokenUsage\n modelCallCount: number\n callSpanIds: string[]\n cost: MeasurementCoverage\n aggregate?: {\n tokenUsage: RunTokenUsage\n costUsd?: number\n }\n}\n\nconst TOKEN_MEASUREMENT_KEY_GROUPS = [\n LLM_INPUT_TOKEN_ATTR_KEYS,\n LLM_OUTPUT_TOKEN_ATTR_KEYS,\n LLM_REASONING_TOKEN_ATTR_KEYS,\n LLM_CACHED_TOKEN_ATTR_KEYS,\n LLM_CACHE_WRITE_TOKEN_ATTR_KEYS,\n] as const\n\nconst EXECUTION_MEASUREMENT_KEY_GROUPS = [\n ...TOKEN_MEASUREMENT_KEY_GROUPS,\n LLM_COST_ATTR_KEYS,\n] as const\n\ninterface RetainedCallSummary {\n callCount: number\n measurements: Array<{\n total: number\n reportingCalls: number\n }>\n}\n\n/**\n * Reconcile execution measurements across nested telemetry wrappers.\n * A measured parent is used only when a descendant call does not report the\n * same field, so aggregate wrappers neither duplicate complete child data nor\n * erase complementary parent fields.\n */\nexport function summarizeExecutionMeasurements(\n spans: ExecutionMeasurementSpan[],\n): ExecutionMeasurements {\n const byId = new Map<string, ExecutionMeasurementSpan>()\n for (const span of spans) {\n if (byId.has(span.id)) {\n throw new Error(`summarizeExecutionMeasurements: duplicate span id \"${span.id}\"`)\n }\n byId.set(span.id, span)\n }\n const tokenMeasurementKeys = TOKEN_MEASUREMENT_KEY_GROUPS.flat()\n const candidates = spans.filter(\n (span) =>\n span.modelCall ||\n (!span.aggregate && readNumber(span.attributes, tokenMeasurementKeys) !== undefined),\n )\n const candidateIds = new Set(candidates.map((span) => span.id))\n const candidateChildren = new Map<string, ExecutionMeasurementSpan[]>()\n for (const candidate of candidates) {\n const parentId = nearestCandidateParent(candidate, byId, candidateIds)\n if (!parentId) continue\n const children = candidateChildren.get(parentId) ?? []\n children.push(candidate)\n candidateChildren.set(parentId, children)\n }\n const aggregateIds = classifyAggregateSpans(candidates, candidateChildren)\n const untypedRunCostIds = new Set(\n spans\n .filter(\n (span) =>\n !span.modelCall &&\n !span.aggregate &&\n !candidateIds.has(span.id) &&\n readNumber(span.attributes, RUN_COST_ATTR_KEYS) !== undefined,\n )\n .map((span) => span.id),\n )\n const aggregateSourceIds = new Set(\n spans\n .filter(\n (span) => span.aggregate || aggregateIds.has(span.id) || untypedRunCostIds.has(span.id),\n )\n .map((span) => span.id),\n )\n\n const calls = candidates.filter((span) => !aggregateIds.has(span.id))\n const input = reconcileMeasurement(calls, byId, aggregateSourceIds, LLM_INPUT_TOKEN_ATTR_KEYS)\n const reasoning = reconcileMeasurement(\n calls,\n byId,\n aggregateSourceIds,\n LLM_REASONING_TOKEN_ATTR_KEYS,\n )\n const output = reconcileMeasurement(\n calls,\n byId,\n aggregateSourceIds,\n LLM_OUTPUT_TOKEN_ATTR_KEYS,\n LLM_REASONING_TOKEN_ATTR_KEYS,\n )\n const cached = reconcileMeasurement(calls, byId, aggregateSourceIds, LLM_CACHED_TOKEN_ATTR_KEYS)\n const cacheWrite = reconcileMeasurement(\n calls,\n byId,\n aggregateSourceIds,\n LLM_CACHE_WRITE_TOKEN_ATTR_KEYS,\n )\n const aggregate = summarizeAggregateMeasurements(\n spans,\n byId,\n new Set([...aggregateIds, ...untypedRunCostIds]),\n )\n\n return {\n tokenUsage: {\n input: input.value ?? 0,\n output: Math.max(output.value ?? 0, reasoning.value ?? 0),\n ...(reasoning.value !== undefined ? { reasoning: reasoning.value } : {}),\n ...(cached.value !== undefined ? { cached: cached.value } : {}),\n ...(cacheWrite.value !== undefined ? { cacheWrite: cacheWrite.value } : {}),\n },\n modelCallCount: calls.length,\n callSpanIds: calls.map((span) => span.id),\n cost: reconcileMeasurement(calls, byId, aggregateSourceIds, LLM_COST_ATTR_KEYS),\n ...(aggregate ? { aggregate } : {}),\n }\n}\n\nexport function recordAggregateMeasurements(\n raw: Record<string, number>,\n aggregate: ExecutionMeasurements['aggregate'],\n): void {\n if (!aggregate) return\n raw.aggregate_prompt_tokens = aggregate.tokenUsage.input\n raw.aggregate_completion_tokens = aggregate.tokenUsage.output\n if (aggregate.tokenUsage.reasoning !== undefined)\n raw.aggregate_reasoning_tokens = aggregate.tokenUsage.reasoning\n if (aggregate.tokenUsage.cached !== undefined)\n raw.aggregate_cached_tokens = aggregate.tokenUsage.cached\n if (aggregate.tokenUsage.cacheWrite !== undefined)\n raw.aggregate_cache_write_tokens = aggregate.tokenUsage.cacheWrite\n if (aggregate.costUsd !== undefined) raw.aggregate_cost_usd = aggregate.costUsd\n}\n\nfunction summarizeAggregateMeasurements(\n spans: ExecutionMeasurementSpan[],\n byId: Map<string, ExecutionMeasurementSpan>,\n aggregateIds: Set<string>,\n): ExecutionMeasurements['aggregate'] {\n const aggregates = spans.filter((span) => span.aggregate || aggregateIds.has(span.id))\n const input = reconcileTopLevelMeasurement(aggregates, byId, LLM_INPUT_TOKEN_ATTR_KEYS)\n const output = reconcileTopLevelMeasurement(aggregates, byId, LLM_OUTPUT_TOKEN_ATTR_KEYS)\n const reasoning = reconcileTopLevelMeasurement(aggregates, byId, LLM_REASONING_TOKEN_ATTR_KEYS)\n const cached = reconcileTopLevelMeasurement(aggregates, byId, LLM_CACHED_TOKEN_ATTR_KEYS)\n const cacheWrite = reconcileTopLevelMeasurement(aggregates, byId, LLM_CACHE_WRITE_TOKEN_ATTR_KEYS)\n const costUsd = reconcileTopLevelMeasurement(aggregates, byId, LLM_COST_ATTR_KEYS)\n if (\n input === undefined &&\n output === undefined &&\n reasoning === undefined &&\n cached === undefined &&\n cacheWrite === undefined &&\n costUsd === undefined\n )\n return undefined\n return {\n tokenUsage: {\n input: input ?? 0,\n output: output ?? reasoning ?? 0,\n ...(reasoning !== undefined ? { reasoning } : {}),\n ...(cached !== undefined ? { cached } : {}),\n ...(cacheWrite !== undefined ? { cacheWrite } : {}),\n },\n ...(costUsd !== undefined ? { costUsd } : {}),\n }\n}\n\nfunction reconcileTopLevelMeasurement(\n spans: ExecutionMeasurementSpan[],\n byId: Map<string, ExecutionMeasurementSpan>,\n keys: readonly string[],\n): number | undefined {\n const selected = new Map<string, number>()\n for (const span of spans) {\n const value = readNumber(span.attributes, keys)\n if (value !== undefined) selected.set(span.id, value)\n }\n for (const spanId of [...selected.keys()]) {\n const span = byId.get(spanId)\n if (!span) continue\n if (ancestorIds(span, byId).some((ancestorId) => selected.has(ancestorId))) {\n selected.delete(spanId)\n }\n }\n return selected.size > 0\n ? [...selected.values()].reduce((total, value) => total + value, 0)\n : undefined\n}\n\nfunction reconcileMeasurement(\n calls: ExecutionMeasurementSpan[],\n byId: Map<string, ExecutionMeasurementSpan>,\n aggregateSourceIds: Set<string>,\n keys: readonly string[],\n fallbackKeys?: readonly string[],\n): MeasurementCoverage {\n const selected = new Map<string, number>()\n const callIds = new Set(calls.map((call) => call.id))\n let reportingCalls = 0\n\n for (const call of calls) {\n const primary = nearestMeasurement(call, byId, callIds, aggregateSourceIds, keys)\n const fallback = fallbackKeys\n ? nearestMeasurement(call, byId, callIds, aggregateSourceIds, fallbackKeys)\n : undefined\n const source =\n primary && fallback && primary.span.id === fallback.span.id\n ? { span: primary.span, value: Math.max(primary.value, fallback.value) }\n : (primary ?? fallback)\n if (!source) continue\n reportingCalls += 1\n selected.set(source.span.id, source.value)\n }\n\n for (const spanId of [...selected.keys()]) {\n const span = byId.get(spanId)\n if (!span) continue\n if (\n ancestorIds(span, byId).some(\n (ancestorId) => selected.has(ancestorId) && !callIds.has(ancestorId),\n )\n ) {\n selected.delete(spanId)\n }\n }\n\n return {\n ...(selected.size > 0\n ? { value: [...selected.values()].reduce((total, value) => total + value, 0) }\n : {}),\n reportingCalls,\n complete: calls.length > 0 && reportingCalls === calls.length,\n }\n}\n\nfunction nearestMeasurement(\n call: ExecutionMeasurementSpan,\n byId: Map<string, ExecutionMeasurementSpan>,\n callIds: Set<string>,\n aggregateSourceIds: Set<string>,\n keys: readonly string[],\n): { span: ExecutionMeasurementSpan; value: number } | undefined {\n let current: ExecutionMeasurementSpan | undefined = call\n const seen = new Set<string>()\n while (current && !seen.has(current.id)) {\n seen.add(current.id)\n const value = readNumber(current.attributes, keys)\n if (\n value !== undefined &&\n (current.id === call.id || (!callIds.has(current.id) && aggregateSourceIds.has(current.id)))\n ) {\n return { span: current, value }\n }\n current = current.parentId ? byId.get(current.parentId) : undefined\n }\n return undefined\n}\n\nfunction classifyAggregateSpans(\n candidates: ExecutionMeasurementSpan[],\n childrenById: Map<string, ExecutionMeasurementSpan[]>,\n): Set<string> {\n const aggregateIds = new Set<string>()\n const summaries = new Map<string, RetainedCallSummary>()\n const visiting = new Set<string>()\n\n const visit = (span: ExecutionMeasurementSpan): RetainedCallSummary => {\n const cached = summaries.get(span.id)\n if (cached) return cached\n if (visiting.has(span.id)) return emptyRetainedCallSummary()\n visiting.add(span.id)\n\n const descendants = emptyRetainedCallSummary()\n for (const child of childrenById.get(span.id) ?? []) {\n const childDescendants = visit(child)\n if (!aggregateIds.has(child.id)) addRetainedCall(descendants, child)\n addRetainedCallSummary(descendants, childDescendants)\n }\n\n if (\n descendants.callCount > 0 &&\n (!span.modelCall || hasCompatibleDescendantMeasurements(span, descendants))\n ) {\n aggregateIds.add(span.id)\n }\n\n visiting.delete(span.id)\n summaries.set(span.id, descendants)\n return descendants\n }\n\n for (const candidate of candidates) visit(candidate)\n return aggregateIds\n}\n\nfunction emptyRetainedCallSummary(): RetainedCallSummary {\n return {\n callCount: 0,\n measurements: EXECUTION_MEASUREMENT_KEY_GROUPS.map(() => ({\n total: 0,\n reportingCalls: 0,\n })),\n }\n}\n\nfunction addRetainedCall(summary: RetainedCallSummary, span: ExecutionMeasurementSpan): void {\n summary.callCount += 1\n for (let index = 0; index < EXECUTION_MEASUREMENT_KEY_GROUPS.length; index += 1) {\n const value = readNumber(span.attributes, EXECUTION_MEASUREMENT_KEY_GROUPS[index]!)\n if (value === undefined) continue\n const measurement = summary.measurements[index]!\n measurement.total += value\n measurement.reportingCalls += 1\n }\n}\n\nfunction addRetainedCallSummary(target: RetainedCallSummary, source: RetainedCallSummary): void {\n target.callCount += source.callCount\n for (let index = 0; index < target.measurements.length; index += 1) {\n const measurement = target.measurements[index]!\n const sourceMeasurement = source.measurements[index]!\n measurement.total += sourceMeasurement.total\n measurement.reportingCalls += sourceMeasurement.reportingCalls\n }\n}\n\nfunction hasCompatibleDescendantMeasurements(\n span: ExecutionMeasurementSpan,\n descendants: RetainedCallSummary,\n): boolean {\n let parentMeasurements = 0\n let descendantMeasurements = 0\n for (let index = 0; index < EXECUTION_MEASUREMENT_KEY_GROUPS.length; index += 1) {\n const keys = EXECUTION_MEASUREMENT_KEY_GROUPS[index]!\n const parentValue = readNumber(span.attributes, keys)\n const measurement = descendants.measurements[index]!\n if (parentValue !== undefined) parentMeasurements += 1\n if (measurement.reportingCalls > 0) descendantMeasurements += 1\n if (parentValue === undefined || measurement.reportingCalls === 0) continue\n if (measurement.reportingCalls !== descendants.callCount) continue\n if (Math.abs(parentValue - measurement.total) > 1e-12) return false\n }\n return parentMeasurements === 0 || descendantMeasurements > 0\n}\n\nfunction ancestorIds(\n span: ExecutionMeasurementSpan,\n byId: Map<string, ExecutionMeasurementSpan>,\n): string[] {\n const ids: string[] = []\n const seen = new Set<string>()\n let parentId = span.parentId\n while (parentId && !seen.has(parentId)) {\n ids.push(parentId)\n seen.add(parentId)\n parentId = byId.get(parentId)?.parentId\n }\n return ids\n}\n\nfunction nearestCandidateParent(\n span: ExecutionMeasurementSpan,\n byId: Map<string, ExecutionMeasurementSpan>,\n candidateIds: Set<string>,\n): string | undefined {\n const seen = new Set<string>()\n let parentId = span.parentId\n while (parentId && !seen.has(parentId)) {\n if (candidateIds.has(parentId)) return parentId\n seen.add(parentId)\n parentId = byId.get(parentId)?.parentId\n }\n return undefined\n}\n\nfunction readNumber(\n attributes: Record<string, unknown>,\n keys: readonly string[],\n): number | undefined {\n for (const key of keys) {\n const value = attributes[key]\n const parsed =\n typeof value === 'number'\n ? value\n : typeof value === 'string' && value.length > 0\n ? Number(value)\n : Number.NaN\n if (Number.isFinite(parsed) && parsed >= 0) return parsed\n }\n return undefined\n}\n","import { ValidationError } from '../errors'\nimport { FAILURE_CLASSES, type FailureClass } from './schema'\n\nconst TASK_FAILURE_CLASS_ATTR = 'tangle.task.failure_class'\nconst TASK_FAILURE_MODE_ATTR = 'tangle.task.failure_mode'\n\ninterface AttributeCarrier {\n attributes: Record<string, unknown>\n}\n\nexport type TaskFailureLabels =\n | { failureClass?: undefined; failureMode?: undefined }\n | { failureClass: 'success'; failureMode?: undefined }\n | { failureClass: Exclude<FailureClass, 'success'>; failureMode?: string }\n\nexport function readTaskFailureLabels(\n roots: readonly AttributeCarrier[],\n context: string,\n): TaskFailureLabels {\n const failureClass = readConsistentRootString(roots, TASK_FAILURE_CLASS_ATTR, context)\n const failureMode = readConsistentRootString(roots, TASK_FAILURE_MODE_ATTR, context)\n\n if (failureClass !== undefined && !FAILURE_CLASSES.includes(failureClass as FailureClass)) {\n throw new ValidationError(\n `${context}: ${TASK_FAILURE_CLASS_ATTR} must be one of ${FAILURE_CLASSES.join(', ')}`,\n )\n }\n if (failureMode !== undefined && (failureClass === undefined || failureClass === 'success')) {\n throw new ValidationError(\n `${context}: ${TASK_FAILURE_MODE_ATTR} requires a non-success ${TASK_FAILURE_CLASS_ATTR}`,\n )\n }\n\n if (failureClass === undefined) return {}\n if (failureClass === 'success') return { failureClass }\n return {\n failureClass: failureClass as Exclude<FailureClass, 'success'>,\n ...(failureMode ? { failureMode } : {}),\n }\n}\n\nfunction readConsistentRootString(\n roots: readonly AttributeCarrier[],\n key: string,\n context: string,\n): string | undefined {\n const values = new Set<string>()\n for (const root of roots) {\n if (!Object.hasOwn(root.attributes, key)) continue\n const value = root.attributes[key]\n if (typeof value !== 'string' || value.trim().length === 0) {\n throw new ValidationError(`${context}: ${key} must be a non-empty string`)\n }\n values.add(value)\n }\n\n if (values.size > 1) {\n throw new ValidationError(\n `${context}: conflicting ${key} values: ${[...values].sort().join(', ')}`,\n )\n }\n return values.values().next().value\n}\n"],"mappings":";;;;;;;;;AA0BA,SAAgB,qBAAqB,SAAyD;CAC5F,MAAM,uBAAO,IAAI,IAA8B;CAC/C,KAAK,MAAM,UAAU,SAAS;EAC5B,IAAI,KAAK,IAAI,OAAO,EAAE,GACpB,MAAM,IAAI,MAAM,4CAA4C,OAAO,GAAG,EAAE;EAE1E,KAAK,IAAI,OAAO,IAAI,MAAM;CAC5B;CAEA,MAAM,6BAAa,IAAI,IAAY;CACnC,KAAK,MAAM,UAAU,SAAS;EAC5B,IAAI,CAAC,OAAO,OAAO;EACnB,MAAM,0BAAU,IAAI,IAAY;EAChC,IAAI,WAAW,OAAO;EACtB,OAAO,YAAY,CAAC,QAAQ,IAAI,QAAQ,GAAG;GACzC,QAAQ,IAAI,QAAQ;GACpB,MAAM,SAAS,KAAK,IAAI,QAAQ;GAChC,IAAI,CAAC,QAAQ;GACb,IAAI,OAAO,OAAO,WAAW,IAAI,OAAO,EAAE;GAC1C,WAAW,OAAO;EACpB;CACF;CAEA,MAAM,UAA6B;EACjC,OAAO;EACP,WAAW;EACX,SAAS;EACT,WAAW;EACX,YAAY;EACZ,YAAY;EACZ,cAAc;CAChB;CAEA,KAAK,MAAM,UAAU,SAAS;EAC5B,IAAI,CAAC,OAAO,OAAO;EACnB,QAAQ,SAAS;EACjB,IAAI,OAAO,SAAS,aAClB,QAAQ,aAAa;OAChB,IAAI,OAAO,SAAS,aACzB,QAAQ,cAAc;OACjB,IAAI,OAAO,aAChB,QAAQ,WAAW;OACd,IAAI,WAAW,IAAI,OAAO,EAAE,GACjC,QAAQ,cAAc;OACjB,IACL,OAAO,SAAS,WAChB,OAAO,SAAS,WAChB,OAAO,SAAS,SAChB,OAAO,SAAS,QAEhB,QAAQ,aAAa;OAErB,QAAQ,gBAAgB;CAE5B;CAEA,OAAO;AACT;;;AC/CA,MAAM,+BAA+B;CACnC;CACA;CACA;CACA;CACA;AACF;AAEA,MAAM,mCAAmC,CACvC,GAAG,8BACH,kBACF;;;;;;;AAgBA,SAAgB,+BACd,OACuB;CACvB,MAAM,uBAAO,IAAI,IAAsC;CACvD,KAAK,MAAM,QAAQ,OAAO;EACxB,IAAI,KAAK,IAAI,KAAK,EAAE,GAClB,MAAM,IAAI,MAAM,sDAAsD,KAAK,GAAG,EAAE;EAElF,KAAK,IAAI,KAAK,IAAI,IAAI;CACxB;CACA,MAAM,uBAAuB,6BAA6B,KAAK;CAC/D,MAAM,aAAa,MAAM,QACtB,SACC,KAAK,aACJ,CAAC,KAAK,aAAa,WAAW,KAAK,YAAY,oBAAoB,MAAM,KAAA,CAC9E;CACA,MAAM,eAAe,IAAI,IAAI,WAAW,KAAK,SAAS,KAAK,EAAE,CAAC;CAC9D,MAAM,oCAAoB,IAAI,IAAwC;CACtE,KAAK,MAAM,aAAa,YAAY;EAClC,MAAM,WAAW,uBAAuB,WAAW,MAAM,YAAY;EACrE,IAAI,CAAC,UAAU;EACf,MAAM,WAAW,kBAAkB,IAAI,QAAQ,KAAK,CAAC;EACrD,SAAS,KAAK,SAAS;EACvB,kBAAkB,IAAI,UAAU,QAAQ;CAC1C;CACA,MAAM,eAAe,uBAAuB,YAAY,iBAAiB;CACzE,MAAM,oBAAoB,IAAI,IAC5B,MACG,QACE,SACC,CAAC,KAAK,aACN,CAAC,KAAK,aACN,CAAC,aAAa,IAAI,KAAK,EAAE,KACzB,WAAW,KAAK,YAAY,kBAAkB,MAAM,KAAA,CACxD,CAAC,CACA,KAAK,SAAS,KAAK,EAAE,CAC1B;CACA,MAAM,qBAAqB,IAAI,IAC7B,MACG,QACE,SAAS,KAAK,aAAa,aAAa,IAAI,KAAK,EAAE,KAAK,kBAAkB,IAAI,KAAK,EAAE,CACxF,CAAC,CACA,KAAK,SAAS,KAAK,EAAE,CAC1B;CAEA,MAAM,QAAQ,WAAW,QAAQ,SAAS,CAAC,aAAa,IAAI,KAAK,EAAE,CAAC;CACpE,MAAM,QAAQ,qBAAqB,OAAO,MAAM,oBAAoB,yBAAyB;CAC7F,MAAM,YAAY,qBAChB,OACA,MACA,oBACA,6BACF;CACA,MAAM,SAAS,qBACb,OACA,MACA,oBACA,4BACA,6BACF;CACA,MAAM,SAAS,qBAAqB,OAAO,MAAM,oBAAoB,0BAA0B;CAC/F,MAAM,aAAa,qBACjB,OACA,MACA,oBACA,+BACF;CACA,MAAM,YAAY,+BAChB,OACA,sBACA,IAAI,IAAI,CAAC,GAAG,cAAc,GAAG,iBAAiB,CAAC,CACjD;CAEA,OAAO;EACL,YAAY;GACV,OAAO,MAAM,SAAS;GACtB,QAAQ,KAAK,IAAI,OAAO,SAAS,GAAG,UAAU,SAAS,CAAC;GACxD,GAAI,UAAU,UAAU,KAAA,IAAY,EAAE,WAAW,UAAU,MAAM,IAAI,CAAC;GACtE,GAAI,OAAO,UAAU,KAAA,IAAY,EAAE,QAAQ,OAAO,MAAM,IAAI,CAAC;GAC7D,GAAI,WAAW,UAAU,KAAA,IAAY,EAAE,YAAY,WAAW,MAAM,IAAI,CAAC;EAC3E;EACA,gBAAgB,MAAM;EACtB,aAAa,MAAM,KAAK,SAAS,KAAK,EAAE;EACxC,MAAM,qBAAqB,OAAO,MAAM,oBAAoB,kBAAkB;EAC9E,GAAI,YAAY,EAAE,UAAU,IAAI,CAAC;CACnC;AACF;AAEA,SAAgB,4BACd,KACA,WACM;CACN,IAAI,CAAC,WAAW;CAChB,IAAI,0BAA0B,UAAU,WAAW;CACnD,IAAI,8BAA8B,UAAU,WAAW;CACvD,IAAI,UAAU,WAAW,cAAc,KAAA,GACrC,IAAI,6BAA6B,UAAU,WAAW;CACxD,IAAI,UAAU,WAAW,WAAW,KAAA,GAClC,IAAI,0BAA0B,UAAU,WAAW;CACrD,IAAI,UAAU,WAAW,eAAe,KAAA,GACtC,IAAI,+BAA+B,UAAU,WAAW;CAC1D,IAAI,UAAU,YAAY,KAAA,GAAW,IAAI,qBAAqB,UAAU;AAC1E;AAEA,SAAS,+BACP,OACA,MACA,cACoC;CACpC,MAAM,aAAa,MAAM,QAAQ,SAAS,KAAK,aAAa,aAAa,IAAI,KAAK,EAAE,CAAC;CACrF,MAAM,QAAQ,6BAA6B,YAAY,MAAM,yBAAyB;CACtF,MAAM,SAAS,6BAA6B,YAAY,MAAM,0BAA0B;CACxF,MAAM,YAAY,6BAA6B,YAAY,MAAM,6BAA6B;CAC9F,MAAM,SAAS,6BAA6B,YAAY,MAAM,0BAA0B;CACxF,MAAM,aAAa,6BAA6B,YAAY,MAAM,+BAA+B;CACjG,MAAM,UAAU,6BAA6B,YAAY,MAAM,kBAAkB;CACjF,IACE,UAAU,KAAA,KACV,WAAW,KAAA,KACX,cAAc,KAAA,KACd,WAAW,KAAA,KACX,eAAe,KAAA,KACf,YAAY,KAAA,GAEZ,OAAO,KAAA;CACT,OAAO;EACL,YAAY;GACV,OAAO,SAAS;GAChB,QAAQ,UAAU,aAAa;GAC/B,GAAI,cAAc,KAAA,IAAY,EAAE,UAAU,IAAI,CAAC;GAC/C,GAAI,WAAW,KAAA,IAAY,EAAE,OAAO,IAAI,CAAC;GACzC,GAAI,eAAe,KAAA,IAAY,EAAE,WAAW,IAAI,CAAC;EACnD;EACA,GAAI,YAAY,KAAA,IAAY,EAAE,QAAQ,IAAI,CAAC;CAC7C;AACF;AAEA,SAAS,6BACP,OACA,MACA,MACoB;CACpB,MAAM,2BAAW,IAAI,IAAoB;CACzC,KAAK,MAAM,QAAQ,OAAO;EACxB,MAAM,QAAQ,WAAW,KAAK,YAAY,IAAI;EAC9C,IAAI,UAAU,KAAA,GAAW,SAAS,IAAI,KAAK,IAAI,KAAK;CACtD;CACA,KAAK,MAAM,UAAU,CAAC,GAAG,SAAS,KAAK,CAAC,GAAG;EACzC,MAAM,OAAO,KAAK,IAAI,MAAM;EAC5B,IAAI,CAAC,MAAM;EACX,IAAI,YAAY,MAAM,IAAI,CAAC,CAAC,MAAM,eAAe,SAAS,IAAI,UAAU,CAAC,GACvE,SAAS,OAAO,MAAM;CAE1B;CACA,OAAO,SAAS,OAAO,IACnB,CAAC,GAAG,SAAS,OAAO,CAAC,CAAC,CAAC,QAAQ,OAAO,UAAU,QAAQ,OAAO,CAAC,IAChE,KAAA;AACN;AAEA,SAAS,qBACP,OACA,MACA,oBACA,MACA,cACqB;CACrB,MAAM,2BAAW,IAAI,IAAoB;CACzC,MAAM,UAAU,IAAI,IAAI,MAAM,KAAK,SAAS,KAAK,EAAE,CAAC;CACpD,IAAI,iBAAiB;CAErB,KAAK,MAAM,QAAQ,OAAO;EACxB,MAAM,UAAU,mBAAmB,MAAM,MAAM,SAAS,oBAAoB,IAAI;EAChF,MAAM,WAAW,eACb,mBAAmB,MAAM,MAAM,SAAS,oBAAoB,YAAY,IACxE,KAAA;EACJ,MAAM,SACJ,WAAW,YAAY,QAAQ,KAAK,OAAO,SAAS,KAAK,KACrD;GAAE,MAAM,QAAQ;GAAM,OAAO,KAAK,IAAI,QAAQ,OAAO,SAAS,KAAK;EAAE,IACpE,WAAW;EAClB,IAAI,CAAC,QAAQ;EACb,kBAAkB;EAClB,SAAS,IAAI,OAAO,KAAK,IAAI,OAAO,KAAK;CAC3C;CAEA,KAAK,MAAM,UAAU,CAAC,GAAG,SAAS,KAAK,CAAC,GAAG;EACzC,MAAM,OAAO,KAAK,IAAI,MAAM;EAC5B,IAAI,CAAC,MAAM;EACX,IACE,YAAY,MAAM,IAAI,CAAC,CAAC,MACrB,eAAe,SAAS,IAAI,UAAU,KAAK,CAAC,QAAQ,IAAI,UAAU,CACrE,GAEA,SAAS,OAAO,MAAM;CAE1B;CAEA,OAAO;EACL,GAAI,SAAS,OAAO,IAChB,EAAE,OAAO,CAAC,GAAG,SAAS,OAAO,CAAC,CAAC,CAAC,QAAQ,OAAO,UAAU,QAAQ,OAAO,CAAC,EAAE,IAC3E,CAAC;EACL;EACA,UAAU,MAAM,SAAS,KAAK,mBAAmB,MAAM;CACzD;AACF;AAEA,SAAS,mBACP,MACA,MACA,SACA,oBACA,MAC+D;CAC/D,IAAI,UAAgD;CACpD,MAAM,uBAAO,IAAI,IAAY;CAC7B,OAAO,WAAW,CAAC,KAAK,IAAI,QAAQ,EAAE,GAAG;EACvC,KAAK,IAAI,QAAQ,EAAE;EACnB,MAAM,QAAQ,WAAW,QAAQ,YAAY,IAAI;EACjD,IACE,UAAU,KAAA,MACT,QAAQ,OAAO,KAAK,MAAO,CAAC,QAAQ,IAAI,QAAQ,EAAE,KAAK,mBAAmB,IAAI,QAAQ,EAAE,IAEzF,OAAO;GAAE,MAAM;GAAS;EAAM;EAEhC,UAAU,QAAQ,WAAW,KAAK,IAAI,QAAQ,QAAQ,IAAI,KAAA;CAC5D;AAEF;AAEA,SAAS,uBACP,YACA,cACa;CACb,MAAM,+BAAe,IAAI,IAAY;CACrC,MAAM,4BAAY,IAAI,IAAiC;CACvD,MAAM,2BAAW,IAAI,IAAY;CAEjC,MAAM,SAAS,SAAwD;EACrE,MAAM,SAAS,UAAU,IAAI,KAAK,EAAE;EACpC,IAAI,QAAQ,OAAO;EACnB,IAAI,SAAS,IAAI,KAAK,EAAE,GAAG,OAAO,yBAAyB;EAC3D,SAAS,IAAI,KAAK,EAAE;EAEpB,MAAM,cAAc,yBAAyB;EAC7C,KAAK,MAAM,SAAS,aAAa,IAAI,KAAK,EAAE,KAAK,CAAC,GAAG;GACnD,MAAM,mBAAmB,MAAM,KAAK;GACpC,IAAI,CAAC,aAAa,IAAI,MAAM,EAAE,GAAG,gBAAgB,aAAa,KAAK;GACnE,uBAAuB,aAAa,gBAAgB;EACtD;EAEA,IACE,YAAY,YAAY,MACvB,CAAC,KAAK,aAAa,oCAAoC,MAAM,WAAW,IAEzE,aAAa,IAAI,KAAK,EAAE;EAG1B,SAAS,OAAO,KAAK,EAAE;EACvB,UAAU,IAAI,KAAK,IAAI,WAAW;EAClC,OAAO;CACT;CAEA,KAAK,MAAM,aAAa,YAAY,MAAM,SAAS;CACnD,OAAO;AACT;AAEA,SAAS,2BAAgD;CACvD,OAAO;EACL,WAAW;EACX,cAAc,iCAAiC,WAAW;GACxD,OAAO;GACP,gBAAgB;EAClB,EAAE;CACJ;AACF;AAEA,SAAS,gBAAgB,SAA8B,MAAsC;CAC3F,QAAQ,aAAa;CACrB,KAAK,IAAI,QAAQ,GAAG,QAAQ,iCAAiC,QAAQ,SAAS,GAAG;EAC/E,MAAM,QAAQ,WAAW,KAAK,YAAY,iCAAiC,MAAO;EAClF,IAAI,UAAU,KAAA,GAAW;EACzB,MAAM,cAAc,QAAQ,aAAa;EACzC,YAAY,SAAS;EACrB,YAAY,kBAAkB;CAChC;AACF;AAEA,SAAS,uBAAuB,QAA6B,QAAmC;CAC9F,OAAO,aAAa,OAAO;CAC3B,KAAK,IAAI,QAAQ,GAAG,QAAQ,OAAO,aAAa,QAAQ,SAAS,GAAG;EAClE,MAAM,cAAc,OAAO,aAAa;EACxC,MAAM,oBAAoB,OAAO,aAAa;EAC9C,YAAY,SAAS,kBAAkB;EACvC,YAAY,kBAAkB,kBAAkB;CAClD;AACF;AAEA,SAAS,oCACP,MACA,aACS;CACT,IAAI,qBAAqB;CACzB,IAAI,yBAAyB;CAC7B,KAAK,IAAI,QAAQ,GAAG,QAAQ,iCAAiC,QAAQ,SAAS,GAAG;EAC/E,MAAM,OAAO,iCAAiC;EAC9C,MAAM,cAAc,WAAW,KAAK,YAAY,IAAI;EACpD,MAAM,cAAc,YAAY,aAAa;EAC7C,IAAI,gBAAgB,KAAA,GAAW,sBAAsB;EACrD,IAAI,YAAY,iBAAiB,GAAG,0BAA0B;EAC9D,IAAI,gBAAgB,KAAA,KAAa,YAAY,mBAAmB,GAAG;EACnE,IAAI,YAAY,mBAAmB,YAAY,WAAW;EAC1D,IAAI,KAAK,IAAI,cAAc,YAAY,KAAK,IAAI,OAAO,OAAO;CAChE;CACA,OAAO,uBAAuB,KAAK,yBAAyB;AAC9D;AAEA,SAAS,YACP,MACA,MACU;CACV,MAAM,MAAgB,CAAC;CACvB,MAAM,uBAAO,IAAI,IAAY;CAC7B,IAAI,WAAW,KAAK;CACpB,OAAO,YAAY,CAAC,KAAK,IAAI,QAAQ,GAAG;EACtC,IAAI,KAAK,QAAQ;EACjB,KAAK,IAAI,QAAQ;EACjB,WAAW,KAAK,IAAI,QAAQ,CAAC,EAAE;CACjC;CACA,OAAO;AACT;AAEA,SAAS,uBACP,MACA,MACA,cACoB;CACpB,MAAM,uBAAO,IAAI,IAAY;CAC7B,IAAI,WAAW,KAAK;CACpB,OAAO,YAAY,CAAC,KAAK,IAAI,QAAQ,GAAG;EACtC,IAAI,aAAa,IAAI,QAAQ,GAAG,OAAO;EACvC,KAAK,IAAI,QAAQ;EACjB,WAAW,KAAK,IAAI,QAAQ,CAAC,EAAE;CACjC;AAEF;AAEA,SAAS,WACP,YACA,MACoB;CACpB,KAAK,MAAM,OAAO,MAAM;EACtB,MAAM,QAAQ,WAAW;EACzB,MAAM,SACJ,OAAO,UAAU,WACb,QACA,OAAO,UAAU,YAAY,MAAM,SAAS,IAC1C,OAAO,KAAK,IACZ;EACR,IAAI,OAAO,SAAS,MAAM,KAAK,UAAU,GAAG,OAAO;CACrD;AAEF;;;ACpaA,MAAM,0BAA0B;AAChC,MAAM,yBAAyB;AAW/B,SAAgB,sBACd,OACA,SACmB;CACnB,MAAM,eAAe,yBAAyB,OAAO,yBAAyB,OAAO;CACrF,MAAM,cAAc,yBAAyB,OAAO,wBAAwB,OAAO;CAEnF,IAAI,iBAAiB,KAAA,KAAa,CAAC,gBAAgB,SAAS,YAA4B,GACtF,MAAM,IAAI,gBACR,GAAG,QAAQ,IAAI,wBAAwB,kBAAkB,gBAAgB,KAAK,IAAI,GACpF;CAEF,IAAI,gBAAgB,KAAA,MAAc,iBAAiB,KAAA,KAAa,iBAAiB,YAC/E,MAAM,IAAI,gBACR,GAAG,QAAQ,IAAI,uBAAuB,0BAA0B,yBAClE;CAGF,IAAI,iBAAiB,KAAA,GAAW,OAAO,CAAC;CACxC,IAAI,iBAAiB,WAAW,OAAO,EAAE,aAAa;CACtD,OAAO;EACS;EACd,GAAI,cAAc,EAAE,YAAY,IAAI,CAAC;CACvC;AACF;AAEA,SAAS,yBACP,OACA,KACA,SACoB;CACpB,MAAM,yBAAS,IAAI,IAAY;CAC/B,KAAK,MAAM,QAAQ,OAAO;EACxB,IAAI,CAAC,OAAO,OAAO,KAAK,YAAY,GAAG,GAAG;EAC1C,MAAM,QAAQ,KAAK,WAAW;EAC9B,IAAI,OAAO,UAAU,YAAY,MAAM,KAAK,CAAC,CAAC,WAAW,GACvD,MAAM,IAAI,gBAAgB,GAAG,QAAQ,IAAI,IAAI,4BAA4B;EAE3E,OAAO,IAAI,KAAK;CAClB;CAEA,IAAI,OAAO,OAAO,GAChB,MAAM,IAAI,gBACR,GAAG,QAAQ,gBAAgB,IAAI,WAAW,CAAC,GAAG,MAAM,CAAC,CAAC,KAAK,CAAC,CAAC,KAAK,IAAI,GACxE;CAEF,OAAO,OAAO,OAAO,CAAC,CAAC,KAAK,CAAC,CAAC;AAChC"}
|
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
import { w as TraceAnalysisStore } from "./types-DN2WdT5S.js";
|
|
2
|
-
import { h as TraceAnalysisToolDescriptor } from "./engine-
|
|
2
|
+
import { h as TraceAnalysisToolDescriptor } from "./engine-CvW_I72-.js";
|
|
3
3
|
//#region src/analyst/tool-groups.d.ts
|
|
4
4
|
/** Named tool sets. Kinds pass `tools: TRACE_TOOL_GROUPS.failureForensics` etc. */
|
|
5
5
|
type TraceToolGroupName =
|
|
@@ -25,4 +25,4 @@ type TraceToolGroupName =
|
|
|
25
25
|
declare function buildTraceToolsForGroup(group: TraceToolGroupName, store: TraceAnalysisStore): TraceAnalysisToolDescriptor[];
|
|
26
26
|
//#endregion
|
|
27
27
|
export { buildTraceToolsForGroup as n, TraceToolGroupName as t };
|
|
28
|
-
//# sourceMappingURL=tool-groups-
|
|
28
|
+
//# sourceMappingURL=tool-groups-DAe1t6zb.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"tool-groups-DAe1t6zb.d.ts","names":[],"sources":["../src/analyst/tool-groups.ts"],"mappings":";;;;KAmBY;;;;;;;;;;;;;;;;;;;;iBAgDI,wBACd,OAAO,oBACP,OAAO,qBACN"}
|
|
@@ -2,7 +2,7 @@ import { r as CaptureIntegrityError } from "../errors-DEE6u6ot.js";
|
|
|
2
2
|
import { b as CustomTokenPricing, c as CostLedgerHandle, p as CostProvenance } from "../cost-ledger-DbQdN3nO.js";
|
|
3
3
|
import { u as RunTokenUsage } from "../run-record-DTv1MdjK.js";
|
|
4
4
|
import { t as DefaultVerdict } from "../verdict-E4eRNf7-.js";
|
|
5
|
-
import { a as TraceAnalystLimits, n as TraceAnalysisEngine } from "../engine-
|
|
5
|
+
import { a as TraceAnalystLimits, n as TraceAnalysisEngine } from "../engine-CvW_I72-.js";
|
|
6
6
|
import { t as PrimeBridgeTransport } from "../prime-bridge-transport-6feEglLf.js";
|
|
7
7
|
import { a as RecordedTrajectoryStep, h as isRecordedTimeout } from "../steps-CiNVJry_.js";
|
|
8
8
|
//#region src/trace-repair/mini-swe-scaffold.d.ts
|
package/dist/traces.d.ts
CHANGED
|
@@ -2,10 +2,10 @@ import { c as ValidationError, o as LimitExceededError, r as CaptureIntegrityErr
|
|
|
2
2
|
import { C as TraceEvent, E as isToolSpan, S as ToolSpan, T as isLlmSpan, _ as Span, a as FAILURE_CLASSES, b as SpanStatus, c as JudgeSpan, d as RetrievalSpan, f as Run, g as SandboxSpan, h as RunStatus, i as EventKind, l as LlmSpan, m as RunOutcome, n as BudgetLedgerEntry, o as FailureClass, p as RunLayer, r as BudgetSpec, s as GenericSpan, t as Artifact, u as Message, v as SpanBase, w as isJudgeSpan, x as TRACE_SCHEMA_VERSION, y as SpanKind } from "./schema-CR5cpjQ3.js";
|
|
3
3
|
import { a as RunRecord, l as RunTerminalOutcome, s as RunSplitTag, u as RunTokenUsage } from "./run-record-DTv1MdjK.js";
|
|
4
4
|
import { A as SearchSpanResult, B as ViewSpansResult, C as TRACE_ANALYSIS_LIMITS, D as DatasetOverview, E as DEFAULT_TRACE_ANALYST_BUDGETS, F as TraceAnalystFilters, H as ViewTraceResult, I as TraceAnalystSpan, L as TraceAnalystSpanKind, M as SpanMatchRecord, N as TRACE_ANALYST_TRUNCATION_MARKER_PREFIX, O as ErrorCluster, P as TraceAnalystByteBudgets, R as TraceAnalystSpanStatus, S as BoundedTraceAnalysisStoreOptions, T as TraceAnalysisStoreContext, V as ViewTraceOversized, j as SearchTraceResult, k as QueryTracesPage, w as TraceAnalysisStore, z as TraceAnalystTraceSummary } from "./types-DN2WdT5S.js";
|
|
5
|
-
import { A as REDACTION_VERSION, B as exportRunAsOtlp, C as scoreTraceInsightReadiness, D as AnalyzeTracesResult, E as AnalyzeTracesOptions, F as OtlpFlatLine, G as ExtractedUsage, H as CaptureFetchOptions, I as OTEL_AGENT_EVAL_SCOPE, J as extractUsageFromResponse, K as SseUsageMode, L as OtlpExport, M as RedactionRule, N as redactString, O as analyzeTraces, P as redactValue, R as OtlpResourceSpans, S as planTraceInsightQuestions, T as AnalyzeTracesInput, U as captureFetchToRawSink, V as CaptureFetchContext, W as ExtractUsageFromSseOptions, X as createBoundedTraceAnalysisStore, Y as extractUsageFromSse, _ as buildTraceInsightPrompt, a as ToolSpansToTraceAnalysisStoreOptions, b as domainEvidencePattern, c as TraceInsightFinding, d as TraceInsightQualityGate, f as TraceInsightQuestion, g as buildTraceInsightContext, h as TraceInsightTask, i as OtlpFileTraceStoreOptions, j as RedactionReport, k as DEFAULT_REDACTION_RULES, l as TraceInsightPanelRole, m as TraceInsightSuite, n as toolSpansToTraceAnalysisStore, o as otlpTextToTraceAnalysisStore, p as TraceInsightReadiness, q as extractUsage, r as OtlpFileTraceStore, s as TraceInsightContext, t as ToolTraceMissingError, u as TraceInsightPromptInput, v as defaultTraceInsightPanel, w as tokenizeDomainWords, x as inferDomainKeywords, y as describeTraceInsightScope, z as OtlpSpan } from "./store-tool-spans-
|
|
5
|
+
import { A as REDACTION_VERSION, B as exportRunAsOtlp, C as scoreTraceInsightReadiness, D as AnalyzeTracesResult, E as AnalyzeTracesOptions, F as OtlpFlatLine, G as ExtractedUsage, H as CaptureFetchOptions, I as OTEL_AGENT_EVAL_SCOPE, J as extractUsageFromResponse, K as SseUsageMode, L as OtlpExport, M as RedactionRule, N as redactString, O as analyzeTraces, P as redactValue, R as OtlpResourceSpans, S as planTraceInsightQuestions, T as AnalyzeTracesInput, U as captureFetchToRawSink, V as CaptureFetchContext, W as ExtractUsageFromSseOptions, X as createBoundedTraceAnalysisStore, Y as extractUsageFromSse, _ as buildTraceInsightPrompt, a as ToolSpansToTraceAnalysisStoreOptions, b as domainEvidencePattern, c as TraceInsightFinding, d as TraceInsightQualityGate, f as TraceInsightQuestion, g as buildTraceInsightContext, h as TraceInsightTask, i as OtlpFileTraceStoreOptions, j as RedactionReport, k as DEFAULT_REDACTION_RULES, l as TraceInsightPanelRole, m as TraceInsightSuite, n as toolSpansToTraceAnalysisStore, o as otlpTextToTraceAnalysisStore, p as TraceInsightReadiness, q as extractUsage, r as OtlpFileTraceStore, s as TraceInsightContext, t as ToolTraceMissingError, u as TraceInsightPromptInput, v as defaultTraceInsightPanel, w as tokenizeDomainWords, x as inferDomainKeywords, y as describeTraceInsightScope, z as OtlpSpan } from "./store-tool-spans-4o55ABER.js";
|
|
6
6
|
import { G as NoopRawProviderSink, H as FileSystemRawProviderSinkOptions, J as RawProviderEvent, K as ProviderRedactor, Q as providerFromBaseUrl, U as InMemoryRawProviderSink, V as FileSystemRawProviderSink, W as InMemoryRawProviderSinkOptions, X as RawProviderSinkFilter, Y as RawProviderSink, Z as defaultProviderRedactor, q as RawProviderDirection } from "./types-gvRsyJLh.js";
|
|
7
7
|
import { a as RunFilter, i as InMemoryTraceStore, n as FileSystemTraceStore, o as SpanFilter, r as FileSystemTraceStoreOptions, s as TraceStore, t as EventFilter } from "./store-BErPvYBr.js";
|
|
8
|
-
import { _ as traceAnalystFunctionGroup, g as buildTraceAnalysisToolDescriptors, h as TraceAnalysisToolDescriptor, m as TRACE_ANALYST_TOOL_NAMESPACE, p as BuildTraceAnalysisToolsOptions } from "./engine-
|
|
8
|
+
import { _ as traceAnalystFunctionGroup, g as buildTraceAnalysisToolDescriptors, h as TraceAnalysisToolDescriptor, m as TRACE_ANALYST_TOOL_NAMESPACE, p as BuildTraceAnalysisToolsOptions } from "./engine-CvW_I72-.js";
|
|
9
9
|
import { a as TraceEmitterOptions, i as TraceEmitter, n as RunCompleteHookContext, r as SpanHandle, t as RunCompleteHook } from "./emitter-Cs0egaFd.js";
|
|
10
10
|
import { C as TOOL_LATENCY_MS, D as asNumber, E as applyLlmSpanOtlpAttributes, O as contextInputTokens, S as TOOL_ARGS_CAPTURED, T as TOOL_NAME_ATTR_KEYS, _ as LlmSpanOtlpInput, a as LLM_CACHE_WRITE_TOKEN_ATTR_KEYS, b as RUN_COST_ATTR_KEYS, c as LLM_COST_USD, d as LLM_MODEL_ATTR_KEYS, f as LLM_MODEL_NAME, g as LLM_REASONING_TOKEN_ATTR_KEYS, h as LLM_REASONING_TOKENS, i as LLM_CACHE_WRITE_TOKENS, k as firstNumberAttr, l as LLM_INPUT_TOKENS, m as LLM_OUTPUT_TOKEN_ATTR_KEYS, n as LLM_CACHED_TOKENS, o as LLM_CONTEXT_TOKENS, p as LLM_OUTPUT_TOKENS, r as LLM_CACHED_TOKEN_ATTR_KEYS, s as LLM_COST_ATTR_KEYS, t as INPUT_VALUE, u as LLM_INPUT_TOKEN_ATTR_KEYS, v as OPENINFERENCE_SPAN_KIND, w as TOOL_NAME, x as SPAN_KIND_ATTR_KEYS, y as OUTPUT_VALUE } from "./attribute-vocabulary-DLJ6303h.js";
|
|
11
11
|
import { a as RunIntegrityReport, i as RunIntegrityIssueCode, n as RunIntegrityExpectations, o as assertRunCaptured, r as RunIntegrityIssue, s as throwIfRunIncomplete, t as RunIntegrityError } from "./integrity-BKTcA-HP.js";
|
package/dist/traces.js
CHANGED
|
@@ -3,15 +3,15 @@ import { a as hasCapturedToolArgs, c as llmSpans, d as runsForScenario, f as too
|
|
|
3
3
|
import { c as validateRunRecord, i as modelHasSnapshot } from "./run-record-ZIsR9Fif.js";
|
|
4
4
|
import { i as redactValue, n as REDACTION_VERSION, r as redactString, t as DEFAULT_REDACTION_RULES } from "./redact-7Aq1ukl-.js";
|
|
5
5
|
import { t as TraceEmitter } from "./emitter-DeQHiDMm.js";
|
|
6
|
-
import { C as asString, D as projectOtlpFlatLine, E as firstStringAttr, G as classifyOtlpSpanRole, K as isOtlpModelCall, O as spanEpochMillis, S as TraceNotFoundError, T as extractOtlpAttributes, W as applyToolSpanOtlpAttributes, _ as TraceAnalysisStoreContractError, a as TRACE_ANALYST_TOOL_NAMESPACE, b as TraceFileMissingError, c as createBoundedTraceAnalysisStore, f as DEFAULT_TRACE_ANALYST_BUDGETS, g as TraceAnalysisLimitError, h as SpanNotFoundError, k as stringField, m as TRACE_ANALYSIS_LIMITS, o as buildTraceAnalysisToolDescriptors, p as TRACE_ANALYST_TRUNCATION_MARKER_PREFIX, q as traceSpanKindToOpenInferenceKind, s as traceAnalystFunctionGroup, v as TraceAnalysisValidationError, w as compareSpanTime, x as TraceFileTooLargeError, y as TraceFileMalformedError } from "./kind-factory-
|
|
6
|
+
import { C as asString, D as projectOtlpFlatLine, E as firstStringAttr, G as classifyOtlpSpanRole, K as isOtlpModelCall, O as spanEpochMillis, S as TraceNotFoundError, T as extractOtlpAttributes, W as applyToolSpanOtlpAttributes, _ as TraceAnalysisStoreContractError, a as TRACE_ANALYST_TOOL_NAMESPACE, b as TraceFileMissingError, c as createBoundedTraceAnalysisStore, f as DEFAULT_TRACE_ANALYST_BUDGETS, g as TraceAnalysisLimitError, h as SpanNotFoundError, k as stringField, m as TRACE_ANALYSIS_LIMITS, o as buildTraceAnalysisToolDescriptors, p as TRACE_ANALYST_TRUNCATION_MARKER_PREFIX, q as traceSpanKindToOpenInferenceKind, s as traceAnalystFunctionGroup, v as TraceAnalysisValidationError, w as compareSpanTime, x as TraceFileTooLargeError, y as TraceFileMalformedError } from "./kind-factory-BLvL-E44.js";
|
|
7
7
|
import { INPUT_VALUE, LLM_CACHED_TOKENS, LLM_CACHED_TOKEN_ATTR_KEYS, LLM_CACHE_WRITE_TOKENS, LLM_CACHE_WRITE_TOKEN_ATTR_KEYS, LLM_CONTEXT_TOKENS, LLM_COST_ATTR_KEYS, LLM_COST_USD, LLM_INPUT_TOKENS, LLM_INPUT_TOKEN_ATTR_KEYS, LLM_MODEL_ATTR_KEYS, LLM_MODEL_NAME, LLM_OUTPUT_TOKENS, LLM_OUTPUT_TOKEN_ATTR_KEYS, LLM_REASONING_TOKENS, LLM_REASONING_TOKEN_ATTR_KEYS, OPENINFERENCE_SPAN_KIND, OUTPUT_VALUE, RUN_COST_ATTR_KEYS, SPAN_KIND_ATTR_KEYS, TOOL_ARGS_CAPTURED, TOOL_LATENCY_MS, TOOL_NAME, TOOL_NAME_ATTR_KEYS, applyLlmSpanOtlpAttributes, asNumber, contextInputTokens, firstNumberAttr } from "./trace-attributes.js";
|
|
8
8
|
import { n as extractUsageFromResponse, r as extractUsageFromSse, t as extractUsage } from "./extract-usage-BrQ8mCLX.js";
|
|
9
9
|
import { a as providerFromBaseUrl, i as defaultProviderRedactor, n as InMemoryRawProviderSink, r as NoopRawProviderSink, t as FileSystemRawProviderSink } from "./raw-provider-sink-BQd7mzyT.js";
|
|
10
|
-
import { C as toOtlpAttributes, S as msToUnixNano, _ as TRACE_ANALYST_ACTOR_DESCRIPTION_VERSION, a as spanStatusToOtlp, b as spanIdForWire, c as defaultTraceInsightPanel, d as inferDomainKeywords, f as planTraceInsightQuestions, g as TRACE_ANALYST_ACTOR_DESCRIPTION, h as analyzeTraces, i as epochMillisToIso, l as describeTraceInsightScope, m as tokenizeDomainWords, n as toolSpansToTraceAnalysisStore, o as buildTraceInsightContext, p as scoreTraceInsightReadiness, r as createOtlpFlatLine, s as buildTraceInsightPrompt, t as ToolTraceMissingError, u as domainEvidencePattern, v as OTEL_AGENT_EVAL_SCOPE, w as captureFetchToRawSink, x as traceIdForWire, y as exportRunAsOtlp } from "./store-tool-spans-
|
|
10
|
+
import { C as toOtlpAttributes, S as msToUnixNano, _ as TRACE_ANALYST_ACTOR_DESCRIPTION_VERSION, a as spanStatusToOtlp, b as spanIdForWire, c as defaultTraceInsightPanel, d as inferDomainKeywords, f as planTraceInsightQuestions, g as TRACE_ANALYST_ACTOR_DESCRIPTION, h as analyzeTraces, i as epochMillisToIso, l as describeTraceInsightScope, m as tokenizeDomainWords, n as toolSpansToTraceAnalysisStore, o as buildTraceInsightContext, p as scoreTraceInsightReadiness, r as createOtlpFlatLine, s as buildTraceInsightPrompt, t as ToolTraceMissingError, u as domainEvidencePattern, v as OTEL_AGENT_EVAL_SCOPE, w as captureFetchToRawSink, x as traceIdForWire, y as exportRunAsOtlp } from "./store-tool-spans-B9tjys_h.js";
|
|
11
11
|
import { n as assertRunCaptured, r as throwIfRunIncomplete, t as RunIntegrityError } from "./integrity-Cy9WHAtb.js";
|
|
12
12
|
import { n as InMemoryTraceStore, t as FileSystemTraceStore } from "./store-DNe_Uv1Q.js";
|
|
13
|
-
import { i as otlpTextToTraceAnalysisStore, n as OtlpFileTraceStore } from "./store-otlp-
|
|
14
|
-
import { i as summarizeTraceErrors, n as recordAggregateMeasurements, r as summarizeExecutionMeasurements, t as readTaskFailureLabels } from "./task-failure-attributes-
|
|
13
|
+
import { i as otlpTextToTraceAnalysisStore, n as OtlpFileTraceStore } from "./store-otlp-DV_H2HDu.js";
|
|
14
|
+
import { i as summarizeTraceErrors, n as recordAggregateMeasurements, r as summarizeExecutionMeasurements, t as readTaskFailureLabels } from "./task-failure-attributes-CUy9mkIY.js";
|
|
15
15
|
import { readFileSync, readdirSync, statSync, writeFileSync } from "node:fs";
|
|
16
16
|
import { join } from "node:path";
|
|
17
17
|
//#region src/trace/otel-bridge.ts
|
|
@@ -571,7 +571,7 @@ interface GenerationCandidate {
|
|
|
571
571
|
surfaceHash: string;
|
|
572
572
|
/** Mean over complete task-quality scores, or null when none were produced. */
|
|
573
573
|
composite: number | null;
|
|
574
|
-
/**
|
|
574
|
+
/** Estimated interval for `composite`, or null when uncertainty was not estimated. */
|
|
575
575
|
ci95: [number, number] | null;
|
|
576
576
|
/** Exact surface this candidate mutated. */
|
|
577
577
|
parentSurfaceHash?: string;
|
|
@@ -670,4 +670,4 @@ interface CampaignResult<TArtifact = unknown, TScenario extends Scenario = Scena
|
|
|
670
670
|
}
|
|
671
671
|
//#endregion
|
|
672
672
|
export { LabeledScenarioWrite as A, ScoredSurfaceOutcome as B, JudgeDimension as C, LabeledScenarioSampleArgs as D, LabeledScenarioRecord as E, ProposeContext as F, labelTrustRank as G, SurfaceProposer as H, ProposedCandidate as I, RedactionStatus as L, OptimizerConfig as M, ParetoParent as N, LabeledScenarioSource as O, ProposalTrackContext as P, Scenario as R, JudgeConfig as S, LabelTrust as T, TraceSpan as U, SessionScript as V, isProposedCandidate as W, GateDecision as _, CampaignResult as a, GenerationRecord as b, CampaignTraceWriter as c, DispatchContext as d, DispatchFn as f, GateContribution as g, GateContext as h, CampaignCostMeter as i, MutableSurface as j, LabeledScenarioStore as k, CodeSurface as l, GateCheckStatus as m, CampaignArtifactWriter as n, CampaignScenarioIdentity as o, Gate as p, CampaignCellResult as r, CampaignTokenUsage as s, CampaignAggregates as t, ComponentSurface as u, GateResult as v, JudgeScore as w, JudgeAggregate as x, GenerationCandidate as y, ScenarioAggregate as z };
|
|
673
|
-
//# sourceMappingURL=types-
|
|
673
|
+
//# sourceMappingURL=types-BJz2CPTM.d.ts.map
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"types-
|
|
1
|
+
{"version":3,"file":"types-BJz2CPTM.d.ts","names":[],"sources":["../src/campaign/types.ts"],"mappings":";;;;;;;;UAiCiB;EACf;EACA;EACA;;;;;EAKA;;;UAIe,iCAAiC,KAAK;EACrD;;;;;UAMe;EACf;;EAEA;EACA;EACA;EACA;EACA,QAAQ;EACR,OAAO;EACP,WAAW;EACX,MAAM;;EAEN;;EAEA;;;;;;;;EAQA;;;;KAKU,WAAW,kBAAkB,UAAU,cACjD,UAAU,WACV,KAAK,oBACF,QAAQ;;;;UAOI,cAAc,WAAW;EACxC;EACA;EACA;;EAEA;;;EAGA,sBAAsB,UAAU,WAAW,sBAAsB,UAAU,cAAc;;UAK1E;;EAEf;;EAEA;;;;;;;;;UAUe,YAAY,WAAW,kBAAkB,WAAW;EACnE;EACA,YAAY;;;;EAIZ;;;EAGA,MAAM;IACJ,UAAU;IACV,UAAU;IACV,QAAQ;;IAER,aAAa;IACb;IACA,WAAW;MACT,aAAa,QAAQ;EACzB,aAAa,UAAU;;;;;;;;;;;UAYR;EACf,YAAY;EACZ;EACA;;EAEA,UAAU;;;;;;;EAOV;;;;EAIA,eAAe,eAAe;IAAQ;IAAe;;;;;EAIrD;;;EAGA;;EAEA;;EAEA,WAAW,eAAe;;;;;;;UAUX;WACN;;;WAGA;;WAEA;;WAEA;;WAEA;;WAEA;;WAEA;;;WAGA;aACE;aACA;aACA;;;WAGF;;;UAIM;WACN;WACA,YAAY,SAAS;;;;;;;;;;;KAYpB,0BAA0B,mBAAmB;;;;;;;UAQxC;EACf,SAAS;;EAET;;;;EAIA;;;;;;EAMA,cAAc,SAAS;;;;iBAKT,oBACd,OAAO,iBAAiB,oBACvB,SAAS;;;;;;;;;UAkBK;EACf,SAAS;EACT;;;EAGA,YAAY;;;EAGZ;;EAEA;EACA;EACA;;;UAIe;EACf;EACA;EACA;EACA;EACA;;;;;UAMe;;;EAGf;;EAEA;EACA;EACA;EACA,YAAY;;;;;EAKZ,WAAW;IAAQ;IAAoB;IAAmB;IAAgB;;EAC1E;IACE;IACA;;;;;UAMa,eAAe,YAAY;;;WAGjC,gBAAgB;WAChB,SAAS,cAAc;WACvB,UAAU,cAAc;;WAExB;WACA;WACA,QAAQ;;WAER,QAAQ;;;WAGR,kBAAkB;;;WAGlB,mBAAmB;;;;WAInB,gBAAgB;;;;WAIhB;;;;;;;WAOA,gBAAgB,cAAc;;WAE9B,aAAa;WACb;;;;;;;;;;;;;;UAeM,gBAAgB,YAAY;EAC3C;;;;;EAKA,QAAQ,KAAK,eAAe,aAAa,QAAQ,MAAM,iBAAiB;;;EAGxE,QAAQ;IAAQ,SAAS,cAAc;;IAAwB;IAAe;;;UAG/D;EACf;EACA;EACA,mBAAmB,qBAAqB;;UAGzB,wBAAwB;EACvC,UAAU;;;KAMA;;KAGA;UAEK;EACf;EACA,QAAQ;EACR;;UAGe,YAAY,WAAW,kBAAkB;EACxD,oBAAoB,YAAY;EAChC,oBAAoB,YAAY;;EAEhC,aAAa,YAAY,eAAe;;;;;EAKxC,sBAAsB,YAAY,eAAe;;;;;;;EAOjD,yBAAyB,YAAY,eAAe;;;EAGpD,uBAAuB,YAAY;EACnC,WAAW;EACX;IAAQ;IAAmB;;;EAE3B,aAAa;EACb;EACA,QAAQ;;UAGO;EACf,UAAU;EACV;EACA,mBAAmB;EACnB;;;UAIe,KAAK,qBAAqB,kBAAkB,WAAW;EACtE;EACA,OAAO,KAAK,YAAY,WAAW,aAAa,QAAQ;;;;UAOzC;EACf,KAAK,cAAc,aAAa,0BAA0B;EAC1D,SAAS;;UAGM;EACf,IAAI,aAAa;EACjB,aAAa,aAAa;;;;UAKX;EACf,MAAM,cAAc,kBAAkB,aAAa;EACnD,UAAU,cAAc,iBAAiB;;;;;;KAO/B,qBAAqB;;;;;UAMhB;;EAEf,YAAY,GACV,OAAO,KAAK,iBAAiB;IAC3B,UAAU;MAEX,QAAQ,eAAe;;;;;KAQhB;KAOA;;;;;;;;;;;;;;;KAgBA;;iBASI,eAAe,OAAO;;;;UAOrB,qBAAqB,kBAAkB,WAAW,UAAU;EAC3E,UAAU;EACV,UAAU;EACV,aAAa,eAAe;EAC5B,QAAQ;EACR;EACA;EACA,iBAAiB;;;;;EAKjB,aAAa;;EAEb;;UAGe,sBAAsB,kBAAkB,WAAW,UAAU,6BACpE,qBAAqB,WAAW;;EAExC;;;EAGA;;UAGe;EACf;;EAEA;;;;EAIA;EACA;IACE;IACA,SAAS,wBAAwB;IACjC;IACA;;;;;IAKA,WAAW;;;UAIE;EACf,QAAQ,OAAO,uBAAuB;EACtC,OAAO,MAAM,4BAA4B,QAAQ;EACjD,QAAQ;IACN;IACA;IACA,UAAU;;;IAGV,SAAS,OAAO;;;UAMH,mBAAmB;;;EAGlC;EACA;EACA;EACA;EACA;EACA,UAAU;EACV,aAAa,eAAe;;;EAG5B;;EAEA,gBAAgB;;EAEhB;;;EAGA,YAAY;;;EAGZ;;;EAGA;EACA;EACA;EACA;;;;EAIA;;EAEA;;EAEA;EACA;;UAGe;EACf;EACA;EACA;EACA;;;;;;EAMA,cAAc;;UAGC;EACf;EACA;EACA;;;EAGA,cAAc;;UAGC;EACf;EACA,YAAY;EACZ;;;;;;UAOe;EACf;;EAEA;;EAEA;;EAEA;;EAEA;;;EAGA;;;;EAIA;;;;EAIA;IACE;IACA;IACA,iBAAiB;MAAQ;MAAgB;;;;;EAI3C,YAAY;;;;;;;;;;EAUZ,WAAW;IAAQ;IAAoB;IAAmB;IAAgB;;;;EAG1E;;;;EAIA;;;;EAIA,cAAc,SAAS;;UAGR;EACf,SAAS,eAAe;EACxB,YAAY,eAAe;;EAE3B,MAAM;;EAEN;EACA;EACA;;EAEA;;EAEA;;EAEA;;EAEA;;UAGe,eAAe,qBAAqB,kBAAkB,WAAW;;EAEhF;;EAEA;EACA;;EAEA;EACA;EACA;EACA;EACA,OAAO,MAAM,mBAAmB;EAChC,YAAY;EACZ;IACE,aAAa;IACb;;EAEF,OAAO;EACP;EACA;EACA,iBAAiB;;;EAGjB,WAAW,MAAM,2BAA2B,KAAK"}
|
|
@@ -87,7 +87,7 @@ function assertNonNegativeFinite(value, field, context) {
|
|
|
87
87
|
//#region src/analyst/types.ts
|
|
88
88
|
/**
|
|
89
89
|
* Analyst contract — the missing orchestration layer over agent-eval's
|
|
90
|
-
* existing analyzers (analyzeTraces, MultiLayerVerifier,
|
|
90
|
+
* existing analyzers (analyzeTraces, MultiLayerVerifier,
|
|
91
91
|
* SemanticConceptJudge, JudgeFn, ...).
|
|
92
92
|
*
|
|
93
93
|
* Each existing primitive returns its own output shape. The Analyst
|
|
@@ -160,4 +160,4 @@ function makeProposalFinding(init) {
|
|
|
160
160
|
//#endregion
|
|
161
161
|
export { assertValidAnalystUsageReceipt as a, validateUsageSettlementTimeout as c, makeProposalFinding as i, deepFreezeCanonicalJson as l, computeFindingId as n, settleUsageReceiptFromCostLedger as o, makeFinding as r, usageReceiptFromCostLedger as s, ANALYST_SEVERITIES as t };
|
|
162
162
|
|
|
163
|
-
//# sourceMappingURL=types-
|
|
163
|
+
//# sourceMappingURL=types-DQ0e2E7y.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"types-DQ0e2E7y.js","names":[],"sources":["../src/ledger-core/deep-freeze.ts","../src/analyst/usage-receipt.ts","../src/analyst/types.ts"],"sourcesContent":["/** Freeze a detached canonical-JSON graph. Canonicalization has already ruled out cycles.\n *\n * Lives outside canonical.ts so the analyst-benchmark implementation digest,\n * which covers canonical.ts, stays bound to the published benchmark evidence. */\nexport function deepFreezeCanonicalJson<T>(value: T): T {\n if (value && typeof value === 'object' && !Object.isFrozen(value)) {\n Object.freeze(value)\n for (const nested of Object.values(value)) deepFreezeCanonicalJson(nested)\n }\n return value\n}\n","import type { CostChannel, CostLedgerFilter, CostLedgerHandle } from '../cost-ledger'\nimport type { AnalystUsageReceipt } from './types'\n\nexport const DEFAULT_USAGE_SETTLEMENT_TIMEOUT_MS = 5_000\n\n/** Convert one ledger channel's complete call set into one analyst receipt. */\nexport function usageReceiptFromCostLedger(\n ledger: CostLedgerHandle,\n filter: CostChannel | CostLedgerFilter = 'analyst',\n): AnalystUsageReceipt {\n const resolvedFilter = typeof filter === 'string' ? { channel: filter } : filter\n const summary = ledger.summary(resolvedFilter)\n const receipts = ledger.list(resolvedFilter)\n const hasReasoningUsage = receipts.some((receipt) => receipt.reasoningTokens !== undefined)\n const hasCacheWriteUsage = receipts.some((receipt) => receipt.cacheWriteTokens !== undefined)\n const cost = summary.costProvenance\n return {\n calls: summary.totalCalls + summary.pendingCalls,\n tokens: summary.usageComplete\n ? {\n input: summary.inputTokens,\n output: summary.outputTokens,\n ...(hasReasoningUsage ? { reasoning: summary.reasoningTokens ?? 0 } : {}),\n ...(summary.cachedTokens > 0 ? { cached: summary.cachedTokens } : {}),\n ...(hasCacheWriteUsage ? { cacheWrite: summary.cacheWriteTokens ?? 0 } : {}),\n }\n : null,\n cost,\n ...(cost.kind === 'uncaptured' ? { knownCostUsd: summary.totalCostUsd } : {}),\n }\n}\n\nexport interface SettledUsageReceipt {\n settled: boolean\n pendingCalls: number\n receipt: AnalystUsageReceipt\n}\n\n/** Wait a bounded time for late provider receipts, then take one immutable snapshot. */\nexport async function settleUsageReceiptFromCostLedger(\n ledger: CostLedgerHandle,\n options: CostLedgerFilter & { timeoutMs?: number } = {},\n): Promise<SettledUsageReceipt> {\n const { timeoutMs: requestedTimeoutMs, ...requestedFilter } = options\n const filter: CostLedgerFilter = {\n channel: requestedFilter.channel ?? 'analyst',\n ...(requestedFilter.phase === undefined ? {} : { phase: requestedFilter.phase }),\n ...(requestedFilter.tags === undefined ? {} : { tags: requestedFilter.tags }),\n }\n const timeoutMs = validateUsageSettlementTimeout(requestedTimeoutMs)\n const initial = ledger.summary(filter)\n const waitResult =\n initial.pendingCalls === 0\n ? true\n : ledger.waitForIdle\n ? await ledger.waitForIdle({ timeoutMs })\n : false\n const pendingCalls = ledger.summary(filter).pendingCalls\n return {\n settled: waitResult && pendingCalls === 0,\n pendingCalls,\n receipt: usageReceiptFromCostLedger(ledger, filter),\n }\n}\n\nexport function validateUsageSettlementTimeout(timeoutMs?: number): number {\n const resolved = timeoutMs ?? DEFAULT_USAGE_SETTLEMENT_TIMEOUT_MS\n if (!Number.isSafeInteger(resolved) || resolved < 0 || resolved > 2_147_483_647) {\n throw new TypeError(\n 'settlementTimeoutMs must be a non-negative safe integer no greater than 2147483647',\n )\n }\n return resolved\n}\n\nexport function assertValidAnalystUsageReceipt(\n receipt: AnalystUsageReceipt,\n context = 'AnalystContext.recordUsage',\n): void {\n if (receipt.calls !== null && (!Number.isSafeInteger(receipt.calls) || receipt.calls < 0)) {\n throw new Error(`${context}: calls must be a non-negative safe integer or null`)\n }\n if (receipt.tokens) {\n assertNonNegativeSafeInteger(receipt.tokens.input, 'tokens.input', context)\n assertNonNegativeSafeInteger(receipt.tokens.output, 'tokens.output', context)\n if (receipt.tokens.reasoning !== undefined) {\n assertNonNegativeSafeInteger(receipt.tokens.reasoning, 'tokens.reasoning', context)\n if (receipt.tokens.reasoning > receipt.tokens.output) {\n throw new Error(`${context}: tokens.reasoning must not exceed tokens.output`)\n }\n }\n if (receipt.tokens.cached !== undefined) {\n assertNonNegativeSafeInteger(receipt.tokens.cached, 'tokens.cached', context)\n }\n if (receipt.tokens.cacheWrite !== undefined) {\n assertNonNegativeSafeInteger(receipt.tokens.cacheWrite, 'tokens.cacheWrite', context)\n }\n }\n if (receipt.cost.kind !== 'uncaptured') {\n assertNonNegativeFinite(receipt.cost.usd, 'cost.usd', context)\n } else if (receipt.cost.usd !== null) {\n throw new Error(`${context}: uncaptured cost.usd must be null`)\n }\n if (receipt.knownCostUsd !== undefined) {\n assertNonNegativeFinite(receipt.knownCostUsd, 'knownCostUsd', context)\n }\n if (receipt.partialTokens) {\n const { input, output } = receipt.partialTokens\n if (receipt.tokens) {\n throw new Error(`${context}: partialTokens must be absent when tokens is complete`)\n }\n if (input === null && output === null) {\n throw new Error(`${context}: partialTokens must carry at least one reported side`)\n }\n if (input !== null) assertNonNegativeSafeInteger(input, 'partialTokens.input', context)\n if (output !== null) assertNonNegativeSafeInteger(output, 'partialTokens.output', context)\n }\n}\n\nfunction assertNonNegativeSafeInteger(value: number, field: string, context: string): void {\n if (!Number.isSafeInteger(value) || value < 0) {\n throw new Error(`${context}: ${field} must be a non-negative safe integer`)\n }\n}\n\nfunction assertNonNegativeFinite(value: number, field: string, context: string): void {\n if (!Number.isFinite(value) || value < 0) {\n throw new Error(`${context}: ${field} must be a non-negative finite number`)\n }\n}\n","/**\n * Analyst contract — the missing orchestration layer over agent-eval's\n * existing analyzers (analyzeTraces, MultiLayerVerifier,\n * SemanticConceptJudge, JudgeFn, ...).\n *\n * Each existing primitive returns its own output shape. The Analyst\n * contract is the single envelope every primitive lifts into, so a\n * registry can run N analysts against a run and a single renderer can\n * compose findings without knowing which analyzer produced them.\n *\n * The contract is intentionally domain-agnostic: nothing here knows\n * about code, voice, RAG, or any particular agent stack. Analysts\n * declare what INPUT KIND they need (a trace store, an artifact dir,\n * a RunRecord, a JudgeInput, or `custom`), and the registry routes\n * the matching input from `AnalystRunInputs`.\n */\n\nimport { createHash } from 'node:crypto'\nimport type { CostLedgerHandle } from '../cost-ledger'\nimport type { RunCostProvenance, RunRecord, RunTokenUsage } from '../run-record'\nimport type { TraceAnalysisStore } from '../trace-analyst/store'\nimport type { JudgeInput } from '../types'\nimport type { ChatClient } from './chat-client'\n\n/**\n * Unified envelope every analyst emits. Schema-versioned so renderers\n * and time-series diffs survive future field additions.\n */\nexport interface AnalystFinding {\n schema_version: '1.0.0'\n /**\n * Stable hash over identity-defining fields (analyst_id + canonical\n * claim + area + optional subject). Two findings from two runs that\n * \"are the same finding\" share this id — that's what `diffFindings`\n * uses to compute appeared/disappeared sets across runs.\n */\n finding_id: string\n analyst_id: string\n produced_at: string\n severity: AnalystSeverity\n /**\n * Coarse classification. Renderers group by this. Free-form so\n * domain-specific analysts can introduce categories without a\n * schema change ('agent-reasoning', 'verification', 'cost',\n * 'tool-use', 'safety', 'latency', 'data-quality', ...).\n */\n area: string\n claim: string\n rationale?: string\n evidence_refs: EvidenceRef[]\n recommended_action?: string\n validation_plan?: string\n /** 0..1 — the analyst's own confidence. Not calibrated across analysts. */\n confidence: number\n /**\n * Optional subject the finding is about — leaf id, agent id, request\n * id. Included in finding_id when present so per-subject findings\n * diff cleanly across runs.\n */\n subject?: string\n /** True when this finding was lifted from a judge result rather than observed\n * directly in a trace or artifact. Descriptive only: proposal access is\n * controlled by `ProposalFinding.proposal_origin`. */\n derived_from_judge?: boolean\n /** Analyst-private extras; renderers ignore unless they know the analyst. */\n metadata?: Record<string, unknown>\n}\n\n/** Finding severity. `AnalystSeverity` derives from this array, so the type\n * and the schema that validates a finding cannot name different levels. */\nexport const ANALYST_SEVERITIES = ['critical', 'high', 'medium', 'low', 'info'] as const\n\nexport type AnalystSeverity = (typeof ANALYST_SEVERITIES)[number]\n\n/** Data sources that candidate generation may intentionally learn from. */\nexport type ProposalFindingOrigin = 'search' | 'production'\n\n/** A finding explicitly admitted as candidate-generation input. */\nexport type ProposalFinding = AnalystFinding & {\n readonly proposal_origin: ProposalFindingOrigin\n}\n\nexport interface EvidenceRef {\n /**\n * Where the evidence lives. `span` and `event` refer to OTLP trace\n * elements; `artifact` to a file inside the run's artifact tree;\n * `finding` to another AnalystFinding (cross-analyst chaining);\n * `metric` to a named scalar reading the renderer knows how to read.\n */\n kind: 'span' | 'event' | 'artifact' | 'finding' | 'metric'\n uri: string\n excerpt?: string\n}\n\n// ── Analyst contract ─────────────────────────────────────────────────\n\n/**\n * The discriminator the registry uses to pass the right input.\n * `custom` is the escape hatch — analysts that need something else\n * (e.g. an embedding cache, a partner SDK handle) read it from\n * `AnalystRunInputs.custom[<analyst id>]`.\n */\nexport type AnalystInputKind =\n | 'trace-store'\n | 'artifact-dir'\n | 'run-record'\n | 'judge-input'\n | 'custom'\n\nexport interface AnalystCost {\n /** `deterministic` analysts MUST NOT call the LLM. */\n kind: 'deterministic' | 'llm'\n /** Optional declared upper bound; the registry can enforce a budget. */\n est_usd_per_run?: number\n /** Models the analyst expects to use (informational). */\n models?: string[]\n /** Maximum post-cancellation wait for provider usage. Model analysts default to 5 seconds. */\n settlement_timeout_ms?: number\n}\n\nexport interface AnalystRequirements {\n /** Min number of shots / samples the analyst needs to produce signal. */\n min_shots?: number\n /** Capabilities the runtime must supply (e.g. ['network', 'gpu']). */\n capabilities?: string[]\n}\n\n/**\n * What's passed to every analyst call. The registry resolves which\n * field the analyst's `inputKind` selects and asserts it's present.\n */\nexport interface AnalystRunInputs {\n traceStore?: TraceAnalysisStore\n artifactDir?: string\n runRecord?: RunRecord\n judgeInput?: JudgeInput\n /** Keyed by analyst id; populated by callers that registered custom analysts. */\n custom?: Record<string, unknown>\n}\n\nexport interface AnalystContext {\n runId: string\n /** Stable correlation id so logs from a single registry.run() share a tag. */\n correlationId: string\n /** Enforced wall-clock deadline (epoch ms). */\n deadlineMs?: number\n /** Per-analyst USD budget. Analysts MAY check before issuing LLM calls. */\n budgetUsd?: number\n /** Shared paid-call account when the analyst runs inside a larger campaign. */\n costLedger?: CostLedgerHandle\n /** Attribution phase used when writing to the shared paid-call account. */\n costPhase?: string\n /**\n * Shared chat client. Analysts that call an LLM go through this so\n * the operator picks transport (sandbox-sdk | router | cli-bridge |\n * direct-provider | mock) at the registry boundary without touching\n * analyst code.\n */\n chat?: ChatClient\n /**\n * Findings from a prior run the operator wants the analyst to see as\n * retrieval context. Kinds that take advantage of cross-run memory\n * (failure-mode \"I saw this cluster last run\", knowledge-gap \"the wiki\n * page I asked for is still missing\") render these into the actor's\n * working set. Filtering is the operator's job: pass the slice that\n * matches the analyst's id, or pass everything and let the kind\n * filter. Empty / absent means no cross-run context.\n */\n priorFindings?: ReadonlyArray<AnalystFinding>\n /**\n * Findings emitted by analysts that completed earlier in this registry run.\n * This is separate from `priorFindings`: upstream findings are dependency\n * context for the current pass, while prior findings are cross-run memory.\n * The registry populates this only when `RegistryRunOpts.chainFindings` is on.\n */\n upstreamFindings?: ReadonlyArray<AnalystFinding>\n /**\n * Report metered work independently of findings. This keeps an empty finding\n * set from erasing token/cost telemetry. Multiple receipts are accumulated.\n */\n recordUsage?: (receipt: AnalystUsageReceipt) => void\n /** Free-form runtime tags (env, host, op). Findings can echo these into metadata. */\n tags?: Record<string, string>\n /** Logger callback — analysts SHOULD prefer this over console.* for testability. */\n log?: (msg: string, fields?: Record<string, unknown>) => void\n /** Optional abort signal. Analysts SHOULD pass it through to LLM calls. */\n signal?: AbortSignal\n /**\n * Optional live-execution port. A runtime that owns a sandbox or checkout\n * fills it so an analyst can execute a bounded probe against the run's\n * produced state instead of reasoning about it from the trace alone. This\n * package defines only the port: no field here reaches for an agent loop,\n * and an absent probe means the analyst works from recorded evidence.\n */\n probe?: ExecutionProbe\n}\n\n// ── Live-execution port ─────────────────────────────────────────────\n\n/** One bounded command an analyst asks the probe to run. */\nexport interface ExecutionProbeRequest {\n command: string\n /** Working directory inside the probed environment. */\n cwd?: string\n /** Hard wall-clock deadline for this one execution. */\n timeoutMs: number\n /** Bytes of combined output retained; the prober truncates beyond it. */\n maxOutputBytes?: number\n signal?: AbortSignal\n}\n\n/**\n * Typed outcome of one probe execution. `succeeded: false` is a PROBE failure\n * (the environment could not run the command); a command that ran and exited\n * non-zero is a successful observation with a non-zero `exitCode`.\n */\nexport type ExecutionProbeOutcome =\n | {\n succeeded: true\n exitCode: number\n stdout: string\n stderr: string\n durationMs: number\n /** True when output was cut at `maxOutputBytes`. */\n truncated: boolean\n }\n | { succeeded: false; error: { class: string; message: string } }\n\n/**\n * The seam a runtime fills to let analysts observe produced state live.\n * Implementations own sandboxing, credentials, and cleanup; analysts only\n * submit bounded requests and read typed outcomes.\n */\nexport interface ExecutionProbe {\n /** One plain sentence naming what is being probed (e.g. a sandbox id). */\n readonly description: string\n execute(request: ExecutionProbeRequest): Promise<ExecutionProbeOutcome>\n}\n\n/**\n * The minimal contract. Concrete analysts can refine `TInput` so\n * implementations stay type-safe (e.g. a trace analyst's `TInput` is\n * `TraceAnalysisStore`); the registry passes the right field from\n * `AnalystRunInputs` based on `inputKind`.\n */\nexport interface Analyst<TInput = unknown> {\n /** Stable identifier — appears in finding_id, telemetry, and registry exclusion lists. */\n readonly id: string\n /** Human-readable. One sentence. */\n readonly description: string\n readonly inputKind: AnalystInputKind\n readonly cost: AnalystCost\n readonly requires?: AnalystRequirements\n /** Bump on breaking changes to claim wording or area so old finding_ids don't collide. */\n readonly version: string\n analyze(input: TInput, ctx: AnalystContext): Promise<AnalystFinding[]>\n}\n\n/** Metered work performed by one analyst call. */\nexport interface AnalystUsageReceipt {\n /** Number of model-usage records observed at the provider boundary. */\n calls: number | null\n /** Null when the provider did not return token accounting. */\n tokens: RunTokenUsage | null\n /** Observed, estimated, or explicitly uncaptured dollar cost. */\n cost: RunCostProvenance\n /** Known lower bound when one or more calls have uncaptured cost. */\n knownCostUsd?: number\n /**\n * Token counts the provider reported on only one side. Present exactly when\n * `tokens` is null and at least one side WAS reported: `RunTokenUsage` has no\n * nullable side, so a one-sided count cannot live in `tokens` without writing\n * a zero nobody measured. Read it as a lower bound, never as a total — the\n * field exists so a null `tokens` cannot hide a real count.\n */\n partialTokens?: { input: number | null; output: number | null }\n /**\n * True when the token counts were DERIVED by the transport (from character\n * lengths, say) rather than measured by the model provider. `cost.kind` is\n * `estimated` both for a rate estimate over exact tokens and for one over\n * derived tokens; this is the field that separates them.\n */\n tokensEstimated?: boolean\n}\n\n// ── finding_id stability ─────────────────────────────────────────────\n\n/**\n * Compute the stable finding_id from the identity-defining fields.\n * Default implementation hashes {analyst_id, area, subject, normalized claim}.\n * Analysts that emit findings whose claim text varies per run (timestamps,\n * counts) SHOULD either: (a) pass an explicit `id_basis` to fix the hash,\n * or (b) move the variable part into `rationale`/`metadata` and keep the\n * `claim` static.\n */\nexport function computeFindingId(input: {\n analyst_id: string\n area: string\n subject?: string\n claim: string\n /** Override the claim for hashing — use when the displayed claim has run-specific bits. */\n id_basis?: string\n}): string {\n const basis = JSON.stringify({\n a: input.analyst_id,\n r: input.area,\n s: input.subject ?? '',\n c: normalizeClaim(input.id_basis ?? input.claim),\n })\n return `f_${createHash('sha256').update(basis).digest('hex').slice(0, 20)}`\n}\n\nfunction normalizeClaim(c: string): string {\n // Lowercase, collapse whitespace, strip trailing punctuation. Goal:\n // \"Leaf X failed install\" and \"Leaf X failed install.\" hash the same.\n return c\n .toLowerCase()\n .replace(/\\s+/g, ' ')\n .replace(/[.!?;:,]+$/g, '')\n .trim()\n}\n\n/**\n * Convenience factory: produce a fully-formed AnalystFinding with the\n * id computed automatically. Analyst code stays terse.\n */\nexport function makeFinding(\n init: Omit<AnalystFinding, 'schema_version' | 'finding_id' | 'produced_at'> & {\n id_basis?: string\n produced_at?: string\n },\n): AnalystFinding {\n const { id_basis, produced_at, ...rest } = init\n return {\n schema_version: '1.0.0',\n finding_id: computeFindingId({\n analyst_id: rest.analyst_id,\n area: rest.area,\n subject: rest.subject,\n claim: rest.claim,\n id_basis,\n }),\n produced_at: produced_at ?? new Date().toISOString(),\n ...rest,\n }\n}\n\n/** Build a finding whose source is explicitly allowed during candidate generation. */\nexport function makeProposalFinding(\n init: Omit<ProposalFinding, 'schema_version' | 'finding_id' | 'produced_at'> & {\n id_basis?: string\n produced_at?: string\n },\n): ProposalFinding {\n const { proposal_origin, ...finding } = init\n return { ...makeFinding(finding), proposal_origin }\n}\n\n// ── Registry result envelope ────────────────────────────────────────\n\nexport interface AnalystRunSummary {\n analyst_id: string\n status: 'ok' | 'skipped' | 'failed'\n /** Why skipped — missing input, budget exceeded, capability unmet. */\n reason?: string\n findings_count: number\n latency_ms: number\n /** Additive model usage and cost provenance for this analyst. */\n usage: AnalystUsageReceipt\n /** When `status='failed'`: the error class + message, never the full stack. */\n error?: { class: string; message: string }\n}\n\nexport interface AnalystRunResult {\n run_id: string\n correlation_id: string\n started_at: string\n ended_at: string\n findings: AnalystFinding[]\n per_analyst: AnalystRunSummary[]\n /** Total LLM cost in USD across all analysts in this registry.run(). */\n total_cost_usd: number\n /**\n * Provenance for `total_cost_usd`. When uncaptured, the numeric field is only\n * the known subtotal and must not be treated as the run's total spend.\n */\n total_cost_provenance?: RunCostProvenance\n}\n\n// ── Streaming event envelope ────────────────────────────────────────\n\n/**\n * Events emitted by `AnalystRegistry.runStream(...)` in real time as\n * the registry executes. UIs subscribe via `for await (const ev of\n * registry.runStream(...))`; `registry.run(...)` is a thin collector\n * over the same stream, so the two surfaces share their invariants.\n *\n * Per-finding events are intentionally omitted — analyzers are batch\n * operations (a recursive engine returns the full `findings:json[]` at the\n * end of the responder), so streaming inside one analyst would only\n * emit partial JSON consumers can't render. The kind-completion event\n * is the right granularity; subscribers wanting per-finding rendering\n * iterate `event.findings` themselves.\n */\nexport type AnalystRunEvent =\n | {\n type: 'run-started'\n run_id: string\n correlation_id: string\n started_at: string\n /** The ordered list of analyst ids the registry will run. */\n analyst_ids: ReadonlyArray<string>\n }\n | {\n type: 'analyst-skipped'\n summary: AnalystRunSummary\n }\n | {\n type: 'analyst-started'\n analyst_id: string\n started_at: string\n }\n | {\n type: 'analyst-completed'\n /** `summary.status` is `'ok'` for clean completion or `'failed'` for thrown analysts. */\n summary: AnalystRunSummary\n findings: ReadonlyArray<AnalystFinding>\n }\n | {\n type: 'run-completed'\n result: AnalystRunResult\n }\n"],"mappings":";;;;;;AAIA,SAAgB,wBAA2B,OAAa;CACtD,IAAI,SAAS,OAAO,UAAU,YAAY,CAAC,OAAO,SAAS,KAAK,GAAG;EACjE,OAAO,OAAO,KAAK;EACnB,KAAK,MAAM,UAAU,OAAO,OAAO,KAAK,GAAG,wBAAwB,MAAM;CAC3E;CACA,OAAO;AACT;;ACJA,SAAgB,2BACd,QACA,SAAyC,WACpB;CACrB,MAAM,iBAAiB,OAAO,WAAW,WAAW,EAAE,SAAS,OAAO,IAAI;CAC1E,MAAM,UAAU,OAAO,QAAQ,cAAc;CAC7C,MAAM,WAAW,OAAO,KAAK,cAAc;CAC3C,MAAM,oBAAoB,SAAS,MAAM,YAAY,QAAQ,oBAAoB,KAAA,CAAS;CAC1F,MAAM,qBAAqB,SAAS,MAAM,YAAY,QAAQ,qBAAqB,KAAA,CAAS;CAC5F,MAAM,OAAO,QAAQ;CACrB,OAAO;EACL,OAAO,QAAQ,aAAa,QAAQ;EACpC,QAAQ,QAAQ,gBACZ;GACE,OAAO,QAAQ;GACf,QAAQ,QAAQ;GAChB,GAAI,oBAAoB,EAAE,WAAW,QAAQ,mBAAmB,EAAE,IAAI,CAAC;GACvE,GAAI,QAAQ,eAAe,IAAI,EAAE,QAAQ,QAAQ,aAAa,IAAI,CAAC;GACnE,GAAI,qBAAqB,EAAE,YAAY,QAAQ,oBAAoB,EAAE,IAAI,CAAC;EAC5E,IACA;EACJ;EACA,GAAI,KAAK,SAAS,eAAe,EAAE,cAAc,QAAQ,aAAa,IAAI,CAAC;CAC7E;AACF;;AASA,eAAsB,iCACpB,QACA,UAAqD,CAAC,GACxB;CAC9B,MAAM,EAAE,WAAW,oBAAoB,GAAG,oBAAoB;CAC9D,MAAM,SAA2B;EAC/B,SAAS,gBAAgB,WAAW;EACpC,GAAI,gBAAgB,UAAU,KAAA,IAAY,CAAC,IAAI,EAAE,OAAO,gBAAgB,MAAM;EAC9E,GAAI,gBAAgB,SAAS,KAAA,IAAY,CAAC,IAAI,EAAE,MAAM,gBAAgB,KAAK;CAC7E;CACA,MAAM,YAAY,+BAA+B,kBAAkB;CAEnE,MAAM,aADU,OAAO,QAAQ,MAEvB,CAAC,CAAC,iBAAiB,IACrB,OACA,OAAO,cACL,MAAM,OAAO,YAAY,EAAE,UAAU,CAAC,IACtC;CACR,MAAM,eAAe,OAAO,QAAQ,MAAM,CAAC,CAAC;CAC5C,OAAO;EACL,SAAS,cAAc,iBAAiB;EACxC;EACA,SAAS,2BAA2B,QAAQ,MAAM;CACpD;AACF;AAEA,SAAgB,+BAA+B,WAA4B;CACzE,MAAM,WAAW,aAAA;CACjB,IAAI,CAAC,OAAO,cAAc,QAAQ,KAAK,WAAW,KAAK,WAAW,YAChE,MAAM,IAAI,UACR,oFACF;CAEF,OAAO;AACT;AAEA,SAAgB,+BACd,SACA,UAAU,8BACJ;CACN,IAAI,QAAQ,UAAU,SAAS,CAAC,OAAO,cAAc,QAAQ,KAAK,KAAK,QAAQ,QAAQ,IACrF,MAAM,IAAI,MAAM,GAAG,QAAQ,oDAAoD;CAEjF,IAAI,QAAQ,QAAQ;EAClB,6BAA6B,QAAQ,OAAO,OAAO,gBAAgB,OAAO;EAC1E,6BAA6B,QAAQ,OAAO,QAAQ,iBAAiB,OAAO;EAC5E,IAAI,QAAQ,OAAO,cAAc,KAAA,GAAW;GAC1C,6BAA6B,QAAQ,OAAO,WAAW,oBAAoB,OAAO;GAClF,IAAI,QAAQ,OAAO,YAAY,QAAQ,OAAO,QAC5C,MAAM,IAAI,MAAM,GAAG,QAAQ,iDAAiD;EAEhF;EACA,IAAI,QAAQ,OAAO,WAAW,KAAA,GAC5B,6BAA6B,QAAQ,OAAO,QAAQ,iBAAiB,OAAO;EAE9E,IAAI,QAAQ,OAAO,eAAe,KAAA,GAChC,6BAA6B,QAAQ,OAAO,YAAY,qBAAqB,OAAO;CAExF;CACA,IAAI,QAAQ,KAAK,SAAS,cACxB,wBAAwB,QAAQ,KAAK,KAAK,YAAY,OAAO;MACxD,IAAI,QAAQ,KAAK,QAAQ,MAC9B,MAAM,IAAI,MAAM,GAAG,QAAQ,mCAAmC;CAEhE,IAAI,QAAQ,iBAAiB,KAAA,GAC3B,wBAAwB,QAAQ,cAAc,gBAAgB,OAAO;CAEvE,IAAI,QAAQ,eAAe;EACzB,MAAM,EAAE,OAAO,WAAW,QAAQ;EAClC,IAAI,QAAQ,QACV,MAAM,IAAI,MAAM,GAAG,QAAQ,uDAAuD;EAEpF,IAAI,UAAU,QAAQ,WAAW,MAC/B,MAAM,IAAI,MAAM,GAAG,QAAQ,sDAAsD;EAEnF,IAAI,UAAU,MAAM,6BAA6B,OAAO,uBAAuB,OAAO;EACtF,IAAI,WAAW,MAAM,6BAA6B,QAAQ,wBAAwB,OAAO;CAC3F;AACF;AAEA,SAAS,6BAA6B,OAAe,OAAe,SAAuB;CACzF,IAAI,CAAC,OAAO,cAAc,KAAK,KAAK,QAAQ,GAC1C,MAAM,IAAI,MAAM,GAAG,QAAQ,IAAI,MAAM,qCAAqC;AAE9E;AAEA,SAAS,wBAAwB,OAAe,OAAe,SAAuB;CACpF,IAAI,CAAC,OAAO,SAAS,KAAK,KAAK,QAAQ,GACrC,MAAM,IAAI,MAAM,GAAG,QAAQ,IAAI,MAAM,sCAAsC;AAE/E;;;;;;;;;;;;;;;;;;;;;AC3DA,MAAa,qBAAqB;CAAC;CAAY;CAAQ;CAAU;CAAO;AAAM;;;;;;;;;AAiO9E,SAAgB,iBAAiB,OAOtB;CACT,MAAM,QAAQ,KAAK,UAAU;EAC3B,GAAG,MAAM;EACT,GAAG,MAAM;EACT,GAAG,MAAM,WAAW;EACpB,GAAG,eAAe,MAAM,YAAY,MAAM,KAAK;CACjD,CAAC;CACD,OAAO,KAAK,WAAW,QAAQ,CAAC,CAAC,OAAO,KAAK,CAAC,CAAC,OAAO,KAAK,CAAC,CAAC,MAAM,GAAG,EAAE;AAC1E;AAEA,SAAS,eAAe,GAAmB;CAGzC,OAAO,EACJ,YAAY,CAAC,CACb,QAAQ,QAAQ,GAAG,CAAC,CACpB,QAAQ,eAAe,EAAE,CAAC,CAC1B,KAAK;AACV;;;;;AAMA,SAAgB,YACd,MAIgB;CAChB,MAAM,EAAE,UAAU,aAAa,GAAG,SAAS;CAC3C,OAAO;EACL,gBAAgB;EAChB,YAAY,iBAAiB;GAC3B,YAAY,KAAK;GACjB,MAAM,KAAK;GACX,SAAS,KAAK;GACd,OAAO,KAAK;GACZ;EACF,CAAC;EACD,aAAa,gCAAe,IAAI,KAAK,EAAA,CAAE,YAAY;EACnD,GAAG;CACL;AACF;;AAGA,SAAgB,oBACd,MAIiB;CACjB,MAAM,EAAE,iBAAiB,GAAG,YAAY;CACxC,OAAO;EAAE,GAAG,YAAY,OAAO;EAAG;CAAgB;AACpD"}
|
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
import { s as RunSplitTag } from "./run-record-DTv1MdjK.js";
|
|
2
|
-
import { R as Scenario, d as DispatchContext } from "./types-
|
|
2
|
+
import { R as Scenario, d as DispatchContext } from "./types-BJz2CPTM.js";
|
|
3
3
|
//#region src/benchmarks/types.d.ts
|
|
4
4
|
type BenchmarkTaskKind = 'retrieval' | 'rag-answer' | 'hallucination' | 'kb-improvement' | 'routing' | 'custom';
|
|
5
5
|
type BenchmarkFamily = 'beir' | 'mteb-retrieval' | 'msmarco' | 'trec-dl' | 'miracl' | 'lotte' | 'bright' | 'crag' | 'hotpotqa' | 'kilt' | 'ragtruth' | 'faithbench' | 'first-party' | 'custom';
|
|
@@ -90,4 +90,4 @@ declare const BENCHMARK_SPLIT_SEED = "agent-eval-v1";
|
|
|
90
90
|
declare function deterministicSplit(itemId: string, seed?: string): RunSplitTag;
|
|
91
91
|
//#endregion
|
|
92
92
|
export { BenchmarkFamily as a, BenchmarkSource as c, BenchmarkEvaluation as i, BenchmarkTaskKind as l, BenchmarkAdapter as n, BenchmarkResponder as o, BenchmarkDatasetItem as r, BenchmarkScenario as s, BENCHMARK_SPLIT_SEED as t, deterministicSplit as u };
|
|
93
|
-
//# sourceMappingURL=types-
|
|
93
|
+
//# sourceMappingURL=types-Dd1ejaeI.d.ts.map
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"types-
|
|
1
|
+
{"version":3,"file":"types-Dd1ejaeI.d.ts","names":[],"sources":["../src/benchmarks/types.ts"],"mappings":";;;KASY;KAQA;UAgBK,qBAAqB;;;EAGpC;;EAEA,SAAS;;EAET,QAAQ;;EAER,SAAS;;EAET,WAAW;;EAEX;;EAEA,SAAS;EACT,WAAW;;UAGI;;;;EAIf;;EAEA;;EAEA,aAAa;;;EAGb,KAAK;EACL;;UAGe;EACf;EACA;EACA;EACA;EACA;;;UAQe,iBAAiB,kBAAkB,oBAAoB;;EAEtE;EACA,SAAS;EACT,WAAW;EACX;EACA,SAAS;EACT;;;;;EAKA,YAAY,OAAO,cAAc,QAAQ,qBAAqB;;EAE9D,SAAS,MAAM,qBAAqB,WAAW,UAAU,YAAY,QAAQ;;;;EAI7E,YAAY,iBAAiB;;UAGd,kBAAkB,4BAA4B;EAC7D;EACA;EACA,QAAQ;EACR,UAAU;EACV,UAAU;EACV,MAAM,qBAAqB;;KAGjB,mBAAmB,oBAAoB,uBAAuB;EACxE,UAAU,kBAAkB;EAC5B,MAAM,qBAAqB;EAC3B,SAAS;MACL,QAAQ,aAAa;;;cAoBd;;;;;;;;;iBAUG,mBACd,gBACA,gBACC"}
|
|
@@ -20,6 +20,48 @@ Use it when one surface must get better.
|
|
|
20
20
|
Use it when two or more methods must be compared at equal budget.
|
|
21
21
|
Runnable versions: [`examples/self-improve-optimizer`](../examples/self-improve-optimizer/) and [`examples/compare-optimization-methods`](../examples/compare-optimization-methods/).
|
|
22
22
|
|
|
23
|
+
## Read An Improvement Result
|
|
24
|
+
|
|
25
|
+
`selfImprove({ method })` executes the complete method once and measures its selected surface on final cases.
|
|
26
|
+
The method may select the unchanged baseline; that result returns `gateDecision: 'hold'` and an empty diff.
|
|
27
|
+
Agent Eval does not score train and selection cases again or choose a different surface after the method finishes.
|
|
28
|
+
|
|
29
|
+
The result type has two modes:
|
|
30
|
+
|
|
31
|
+
| Mode | Result | Search evidence | Cost |
|
|
32
|
+
|---|---|---|---|
|
|
33
|
+
| `proposer` | `SelfImproveProposerResult` | Native `raw.generations`, `generationsExplored`, and optional `searchHistory` | Shared `cost` ledger summary |
|
|
34
|
+
| `method` | `SelfImproveMethodResult` | Actual `raw.method` and its optional `searchHistory` | Combined method and final `cost`; receipt breakdown in `ledgerCost` |
|
|
35
|
+
|
|
36
|
+
Both types are exported from the package root and `/contract`.
|
|
37
|
+
`SelfImproveResult` is their union; branch on `result.mode` before reading mode-specific fields.
|
|
38
|
+
Calls with a concrete `method` or `proposer` infer the corresponding result type.
|
|
39
|
+
Method mode has no native generation count or fabricated native search measurements.
|
|
40
|
+
Its durable `method-provenance.json` uses schema `tangle.method-improvement` and records partition, measurement, and cost-receipt digests.
|
|
41
|
+
Proposer mode retains `LoopProvenanceRecord`.
|
|
42
|
+
|
|
43
|
+
When method holdout is deferred, `baseline` and `winner.compositeMean` are `null`, `lift` is absent, and the decision is `hold`.
|
|
44
|
+
The selected surface remains available in `winner.surface`.
|
|
45
|
+
Method cost preserves the larger of reported search spend and newly recorded search receipts, then adds final measurements without counting receipts twice.
|
|
46
|
+
Underreported spending and incomplete receipts remain explicit; `raw.method.cost` retains the original report.
|
|
47
|
+
Inspect `cost.accountingComplete` and `cost.incompleteReasons` before treating the known subtotal as complete spending.
|
|
48
|
+
The shared dollar limit controls calls admitted through the cost ledger; arbitrary off-ledger callbacks must enforce their own spending limits.
|
|
49
|
+
|
|
50
|
+
Native generation records report `ci95: null` because search does not estimate candidate uncertainty.
|
|
51
|
+
Final comparisons retain their independently computed statistics.
|
|
52
|
+
Every final case and replica must have complete execution and judge results before comparison.
|
|
53
|
+
|
|
54
|
+
## Bind Cached Measurements To Their Evaluator
|
|
55
|
+
|
|
56
|
+
Candidate surface content is part of native search and final measurement identity.
|
|
57
|
+
Pass a stable `dispatchRef` for execution behavior outside that surface, such as the worker revision and tool configuration.
|
|
58
|
+
Change it when that behavior changes; function names cannot identify captured state.
|
|
59
|
+
Set `judgeVersion` when a judge's scoring behavior changes.
|
|
60
|
+
|
|
61
|
+
To reuse `premeasuredBaseline` in proposer mode, measure the same train cases, seed, replicas, execution revision, and judges.
|
|
62
|
+
The standalone campaign must use `dispatchRef: surfaceDispatchRef(baselineSurface, dispatchRef)` from `/campaign`.
|
|
63
|
+
Agent Eval refuses a prior baseline whose evaluator manifest differs.
|
|
64
|
+
|
|
23
65
|
## Adapt A Third-Party Text Optimizer
|
|
24
66
|
|
|
25
67
|
`externalTextOptimizationMethod()` is the general adapter for a package that already owns text or component search.
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@tangle-network/agent-eval",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.174.0",
|
|
4
4
|
"description": "Evaluate and improve AI agents from runs, traces, judges, and feedback. Compare candidates, cluster failures, measure lift, and gate releases.",
|
|
5
5
|
"homepage": "https://github.com/tangle-network/agent-eval#readme",
|
|
6
6
|
"repository": {
|
|
@@ -1,84 +0,0 @@
|
|
|
1
|
-
import { AgentProfile, AgentProfile as AgentProfile$1, HarnessType, HarnessType as HarnessType$1 } from "@tangle-network/agent-interface";
|
|
2
|
-
//#region src/agent-profile.d.ts
|
|
3
|
-
/**
|
|
4
|
-
* The agentic coding harnesses an eval sweeps by default — the ones we care about
|
|
5
|
-
* ranking. This is the SINGLE source of that list; consumers import it instead of
|
|
6
|
-
* re-declaring their own (a re-declared list is how the fleet drifts). Pass an
|
|
7
|
-
* explicit `harnesses` (e.g. `harnessTypeSchema.options` for literally every known
|
|
8
|
-
* harness) to widen beyond these.
|
|
9
|
-
*/
|
|
10
|
-
declare const CODING_HARNESSES: readonly HarnessType[];
|
|
11
|
-
interface ProfileAxisSpec {
|
|
12
|
-
/** The domain profile to sweep. Its prompt/tools/skills are held fixed; only the
|
|
13
|
-
* harness and model vary. `model.default` is the fallback model. */
|
|
14
|
-
base: AgentProfile;
|
|
15
|
-
/** Harnesses to cross. Default: {@link CODING_HARNESSES}. */
|
|
16
|
-
harnesses?: readonly HarnessType[];
|
|
17
|
-
/** Models to cross. Default: `[base.model.default]` — one model, i.e. today's
|
|
18
|
-
* single-model behaviour, so omitting this never changes an existing run. */
|
|
19
|
-
models?: readonly string[];
|
|
20
|
-
/** Force every (harness, model) pair verbatim, even ones the harness can't run —
|
|
21
|
-
* for deliberately testing failure modes. Default (false): SNAP instead — a
|
|
22
|
-
* vendor-locked harness runs only the swept models in its family, or its native
|
|
23
|
-
* default when it supports none, so no harness is dropped and none gets a
|
|
24
|
-
* guaranteed-failing foreign-model cell. */
|
|
25
|
-
keepIncompatible?: boolean;
|
|
26
|
-
}
|
|
27
|
-
/** Model sentinel for a vendor-locked harness that supports none of the swept models:
|
|
28
|
-
* it carries no provider prefix, so `harnessSupportsModel` accepts it and the harness
|
|
29
|
-
* resolves it to its own native default model at runtime (e.g. kimi-code → its Kimi
|
|
30
|
-
* model). Lets `expandProfileAxes` snap-instead-of-drop without a per-harness flagship
|
|
31
|
-
* table that would rot as router catalogs change. */
|
|
32
|
-
declare const HARNESS_NATIVE_MODEL = "default";
|
|
33
|
-
/**
|
|
34
|
-
* Expand a base profile across the harness × model matrix into the `AgentProfile[]`
|
|
35
|
-
* that `runProfileMatrix` / `selfImprove` score — the ONE place "which harnesses ×
|
|
36
|
-
* which models do we evaluate" lives, so no product hand-rolls its own harness list
|
|
37
|
-
* or column→profile mapping (the pattern that let those copies drift and silently
|
|
38
|
-
* break the harness pivot).
|
|
39
|
-
*
|
|
40
|
-
* Each cell clones `base`, sets the canonical top-level `harness` and `model.default`,
|
|
41
|
-
* and stamps `metadata.harness` + `metadata.harnessModel` for matrix grouping. Both
|
|
42
|
-
* metadata fields are hash-bearing, so every cell gets a distinct `agentProfileId` row
|
|
43
|
-
* and results join back by harness/model via {@link harnessAxisOf} with no
|
|
44
|
-
* hand-recomputed key. A vendor-locked harness snaps to its family's swept models — or
|
|
45
|
-
* its native default ({@link HARNESS_NATIVE_MODEL}) when it supports none — so every
|
|
46
|
-
* requested harness runs; `keepIncompatible` forces every pair verbatim.
|
|
47
|
-
*
|
|
48
|
-
* Omit `harnesses`/`models` to sweep the full default set — the "turn it on for
|
|
49
|
-
* everything we care about" switch, identical in shape whether one harness or all.
|
|
50
|
-
*/
|
|
51
|
-
declare function expandProfileAxes(spec: ProfileAxisSpec): AgentProfile[];
|
|
52
|
-
/**
|
|
53
|
-
* Read the (harness, model) a matrix cell ran under, off a profile or a result row's
|
|
54
|
-
* profile — the join-back for a `byHarness` pivot. Returns undefined when the profile
|
|
55
|
-
* wasn't produced by {@link expandProfileAxes}. Callers group `result.byProfile` by
|
|
56
|
-
* this instead of recomputing an id (recomputing the wrong key is what broke the pivot
|
|
57
|
-
* in the hand-rolled copies).
|
|
58
|
-
*/
|
|
59
|
-
declare function harnessAxisOf(profile: Pick<AgentProfile, 'metadata'>): {
|
|
60
|
-
harness: HarnessType;
|
|
61
|
-
model: string;
|
|
62
|
-
} | undefined;
|
|
63
|
-
/**
|
|
64
|
-
* Collision-resistant, path-safe, human-readable profile id for eval artifacts.
|
|
65
|
-
* Scorecard joins still use `agentProfileHash`; this id is for run ids, matrix
|
|
66
|
-
* keys, and directory names where two profiles must not collapse onto one row.
|
|
67
|
-
* The suffix is the first 64 bits of the behaviour hash, enough for ordinary
|
|
68
|
-
* eval matrices while keeping filenames readable.
|
|
69
|
-
*/
|
|
70
|
-
declare function agentProfileId(profile: AgentProfile): string;
|
|
71
|
-
/**
|
|
72
|
-
* Deterministic behaviour identity for the canonical
|
|
73
|
-
* `@tangle-network/agent-interface` AgentProfile.
|
|
74
|
-
*
|
|
75
|
-
* `name` and `description` are labels and do not affect the hash. Profile
|
|
76
|
-
* `version`, prompt, model hints, tools, resources, hooks, modes, permissions,
|
|
77
|
-
* and extensions do affect the hash. Resource array order is hash-bearing
|
|
78
|
-
* because mount order can change agent behaviour. Undefined fields are treated
|
|
79
|
-
* as absent; explicit `null` fields remain hash-bearing.
|
|
80
|
-
*/
|
|
81
|
-
declare function agentProfileHash(profile: AgentProfile): string;
|
|
82
|
-
//#endregion
|
|
83
|
-
export { ProfileAxisSpec as a, expandProfileAxes as c, HarnessType$1 as i, harnessAxisOf as l, CODING_HARNESSES as n, agentProfileHash as o, HARNESS_NATIVE_MODEL as r, agentProfileId as s, AgentProfile$1 as t };
|
|
84
|
-
//# sourceMappingURL=agent-profile-B9_GGsG8.d.ts.map
|
|
@@ -1 +0,0 @@
|
|
|
1
|
-
{"version":3,"file":"agent-profile-B9_GGsG8.d.ts","names":[],"sources":["../src/agent-profile.ts"],"mappings":";;;;;;;;;cAea,2BAA2B;UAOvB;;;EAGf,MAAM;;EAEN,qBAAqB;;;EAGrB;;;;;;EAMA;;;;;;;cAQW;;;;;;;;;;;;;;;;;;;iBAoBG,kBAAkB,MAAM,kBAAkB;;;;;;;;iBAwD1C,cACd,SAAS,KAAK;EACX,SAAS;EAAa;;;;;;;;;iBAiBX,eAAe,SAAS;;;;;;;;;;;iBAmDxB,iBAAiB,SAAS"}
|