@tangle-network/agent-eval 0.115.3 → 0.117.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +55 -0
- package/dist/analyst/index.d.ts +16 -11
- package/dist/analyst/index.js +33 -25
- package/dist/analyst/index.js.map +1 -1
- package/dist/{analyze-runs-BYHg6Irm.d.ts → analyze-runs--2x39HZ7.d.ts} +3 -3
- package/dist/{baseline-DsNteOgR.d.ts → baseline-DKq3gJpP.d.ts} +6 -3
- package/dist/belief-state/index.d.ts +6 -6
- package/dist/belief-state/index.js +1 -1
- package/dist/benchmarks/index.d.ts +12 -5
- package/dist/benchmarks/index.js +11 -10
- package/dist/builder-eval/index.d.ts +4 -4
- package/dist/builder-eval/index.js +1 -1
- package/dist/{calibration-Dz8TQV4y.d.ts → calibration-C8MTS7cw.d.ts} +2 -2
- package/dist/campaign/index.d.ts +247 -34
- package/dist/campaign/index.js +33 -13
- package/dist/chunk-3YYRZDON.js +45 -0
- package/dist/chunk-3YYRZDON.js.map +1 -0
- package/dist/{chunk-RPDDVKI7.js → chunk-4JLWXDYA.js} +2 -2
- package/dist/{chunk-WSBUZMBU.js → chunk-CCZIVI3F.js} +54 -115
- package/dist/chunk-CCZIVI3F.js.map +1 -0
- package/dist/{chunk-J6P6PK2R.js → chunk-FQNLDL4D.js} +3 -3
- package/dist/{chunk-ONM6PEAE.js → chunk-GQCZRZ7L.js} +2 -2
- package/dist/chunk-HHWE3POT.js +94 -0
- package/dist/chunk-HHWE3POT.js.map +1 -0
- package/dist/{chunk-ADYLPOSX.js → chunk-HQPHZGL6.js} +1112 -135
- package/dist/chunk-HQPHZGL6.js.map +1 -0
- package/dist/{chunk-FAOEFFRT.js → chunk-IDZTTFRR.js} +390 -78
- package/dist/chunk-IDZTTFRR.js.map +1 -0
- package/dist/{chunk-3LXTCTWL.js → chunk-JSDVRFAP.js} +2 -2
- package/dist/{chunk-MHNQWM4I.js → chunk-LQUTGLOZ.js} +5 -1
- package/dist/chunk-LQUTGLOZ.js.map +1 -0
- package/dist/{chunk-4D5RVB3W.js → chunk-LTVG32KX.js} +30 -5
- package/dist/chunk-LTVG32KX.js.map +1 -0
- package/dist/{chunk-5S5NJ63F.js → chunk-MGEHEHSN.js} +807 -15
- package/dist/chunk-MGEHEHSN.js.map +1 -0
- package/dist/{chunk-GY4SYVPJ.js → chunk-NJC7U437.js} +97 -25
- package/dist/chunk-NJC7U437.js.map +1 -0
- package/dist/{chunk-NYFUT3B3.js → chunk-ODVOOEWQ.js} +31 -10
- package/dist/chunk-ODVOOEWQ.js.map +1 -0
- package/dist/{chunk-LNQEP766.js → chunk-S2F4J57L.js} +44 -4
- package/dist/chunk-S2F4J57L.js.map +1 -0
- package/dist/chunk-VCTY3W6J.js +798 -0
- package/dist/chunk-VCTY3W6J.js.map +1 -0
- package/dist/chunk-VF3XSYTI.js +545 -0
- package/dist/chunk-VF3XSYTI.js.map +1 -0
- package/dist/{chunk-TLDB7WRY.js → chunk-YZPO4UHR.js} +28 -31
- package/dist/chunk-YZPO4UHR.js.map +1 -0
- package/dist/{chunk-KG4TD7EQ.js → chunk-ZUXV7UWZ.js} +1425 -697
- package/dist/chunk-ZUXV7UWZ.js.map +1 -0
- package/dist/cli.js +4 -2
- package/dist/cli.js.map +1 -1
- package/dist/{code-agent-session-D-g04tcy.d.ts → code-agent-session-CjZsVd19.d.ts} +1 -1
- package/dist/contract/index.d.ts +45 -31
- package/dist/contract/index.js +58 -19
- package/dist/contract/index.js.map +1 -1
- package/dist/{control-CcBiAEnn.d.ts → control-6vuGfmDH.d.ts} +5 -5
- package/dist/control.d.ts +6 -6
- package/dist/cost-ledger-DWy3XdJc.d.ts +183 -0
- package/dist/{default-registry-DltpYR5u.d.ts → default-registry-DaK8b3fv.d.ts} +2 -1
- package/dist/{emitter-BRchAAAx.d.ts → emitter-CjD7vUwv.d.ts} +2 -2
- package/dist/{failure-cluster-C48PiReX.d.ts → failure-cluster-DOAcSJ87.d.ts} +2 -2
- package/dist/{feedback-trajectory-pDcz1lQ1.d.ts → feedback-trajectory-BUnM58xL.d.ts} +3 -3
- package/dist/fuzz.d.ts +8 -16
- package/dist/fuzz.js +72 -42
- package/dist/fuzz.js.map +1 -1
- package/dist/{gepa-dne9JDPL.d.ts → gepa-eESocoDi.d.ts} +64 -12
- package/dist/hosted/index.d.ts +14 -7
- package/dist/{index-BTEpx9He.d.ts → index-PdX4VnPA.d.ts} +3 -3
- package/dist/index.d.ts +97 -55
- package/dist/index.js +343 -244
- package/dist/index.js.map +1 -1
- package/dist/{insight-report-IwwvqZZv.d.ts → insight-report-DY4nDW9Q.d.ts} +1 -1
- package/dist/{integrity-qemeBAyx.d.ts → integrity-DqlBiLyK.d.ts} +1 -1
- package/dist/kind-factory-ClZmO25A.d.ts +171 -0
- package/dist/{llm-client-DyqEH4jH.d.ts → llm-client-qoDd18Qz.d.ts} +27 -3
- package/dist/meta-eval/index.d.ts +8 -7
- package/dist/meta-eval/index.js +1 -1
- package/dist/multishot/index.d.ts +10 -3
- package/dist/openapi.json +1 -1
- package/dist/pipelines/index.d.ts +16 -6
- package/dist/pipelines/index.js +119 -23
- package/dist/pipelines/index.js.map +1 -1
- package/dist/{kind-factory-DcNg13sZ.d.ts → policy-edit-wG9uFEFm.d.ts} +114 -167
- package/dist/{pre-registration-D8h7ZxNL.d.ts → pre-registration-BWQhJ3vz.d.ts} +23 -4
- package/dist/{provenance-Bibyg1U9.d.ts → provenance-DpjwyseI.d.ts} +28 -16
- package/dist/{query-Ck190MOd.d.ts → query-CF7PG61p.d.ts} +5 -3
- package/dist/{release-report-CCtzajxP.d.ts → release-report-C8G2i5Xi.d.ts} +2 -2
- package/dist/reporting.d.ts +10 -9
- package/dist/{researcher-Dq-EtpbE.d.ts → researcher-C8XyxQsu.d.ts} +7 -7
- package/dist/rl.d.ts +17 -12
- package/dist/rl.js +2 -2
- package/dist/{rubric-predictive-validity-DYTLjGWu.d.ts → rubric-predictive-validity-p49lLVrE.d.ts} +1 -1
- package/dist/{run-campaign-UADIM77S.js → run-campaign-IM26A6PD.js} +4 -2
- package/dist/{run-record-B7RTi_ix.d.ts → run-record-BDH49H2E.d.ts} +2 -2
- package/dist/{runtime-trajectory-Dws7Kpgi.d.ts → runtime-trajectory-DGBIUt4B.d.ts} +1 -1
- package/dist/{schema-SGWcK9wa.d.ts → schema-B3Q3l9Z_.d.ts} +2 -0
- package/dist/{semantic-concept-judge-DxJmRkyJ.d.ts → semantic-concept-judge-CXnPEJbf.d.ts} +23 -5
- package/dist/{statistics-oUbOJe-S.d.ts → statistics-KUnG73jH.d.ts} +1 -1
- package/dist/{storage-Dw_f7WMt.d.ts → storage-DrX3v_5B.d.ts} +12 -1
- package/dist/{store-BsVi7ncX.d.ts → store-DGqD0Pyo.d.ts} +1 -1
- package/dist/storyboard/index.d.ts +1 -1
- package/dist/{summary-report-BJ5aNwZ1.d.ts → summary-report-C5bKFfm-.d.ts} +2 -2
- package/dist/{test-graded-scenario-mzYBKspu.d.ts → test-graded-scenario-B0ybnPY7.d.ts} +3 -3
- package/dist/traces.d.ts +19 -10
- package/dist/traces.js +16 -4
- package/dist/{types-C5gJrOVT.d.ts → types-BSw1rOUB.d.ts} +97 -38
- package/dist/{types-C7DGg5ex.d.ts → types-BkfcQnxV.d.ts} +15 -0
- package/dist/wire/index.d.ts +28 -19
- package/dist/wire/index.js +4 -2
- package/docs/design/loop-taxonomy.md +1 -2
- package/docs/distributed-driver.md +1 -1
- package/package.json +3 -3
- package/dist/chunk-4D5RVB3W.js.map +0 -1
- package/dist/chunk-5S5NJ63F.js.map +0 -1
- package/dist/chunk-ADYLPOSX.js.map +0 -1
- package/dist/chunk-FAOEFFRT.js.map +0 -1
- package/dist/chunk-GY4SYVPJ.js.map +0 -1
- package/dist/chunk-I6LVHOV3.js +0 -205
- package/dist/chunk-I6LVHOV3.js.map +0 -1
- package/dist/chunk-KG4TD7EQ.js.map +0 -1
- package/dist/chunk-LNQEP766.js.map +0 -1
- package/dist/chunk-MHNQWM4I.js.map +0 -1
- package/dist/chunk-NYFUT3B3.js.map +0 -1
- package/dist/chunk-QMXXSNC4.js +0 -761
- package/dist/chunk-QMXXSNC4.js.map +0 -1
- package/dist/chunk-TLDB7WRY.js.map +0 -1
- package/dist/chunk-WSBUZMBU.js.map +0 -1
- package/dist/cost-ledger-DuSqlw5B.d.ts +0 -113
- package/dist/policy-edit-RLn8GWof.d.ts +0 -103
- /package/dist/{chunk-RPDDVKI7.js.map → chunk-4JLWXDYA.js.map} +0 -0
- /package/dist/{chunk-J6P6PK2R.js.map → chunk-FQNLDL4D.js.map} +0 -0
- /package/dist/{chunk-ONM6PEAE.js.map → chunk-GQCZRZ7L.js.map} +0 -0
- /package/dist/{chunk-3LXTCTWL.js.map → chunk-JSDVRFAP.js.map} +0 -0
- /package/dist/{run-campaign-UADIM77S.js.map → run-campaign-IM26A6PD.js.map} +0 -0
|
@@ -1,9 +1,10 @@
|
|
|
1
|
-
import { S as Scenario,
|
|
2
|
-
import {
|
|
3
|
-
import { R as RunRecord } from './run-record-
|
|
4
|
-
import { a as PairedBootstrapResult } from './statistics-
|
|
1
|
+
import { S as Scenario, G as Gate, m as GateResult, l as GateContext, C as CampaignResult, p as Mutator, c as SurfaceProposer, M as MutableSurface, n as GenerationCandidate, d as GateDecision } from './types-BSw1rOUB.js';
|
|
2
|
+
import { n as RedTeamCase, D as Direction, h as RunCampaignOptions } from './gepa-eESocoDi.js';
|
|
3
|
+
import { R as RunRecord } from './run-record-BDH49H2E.js';
|
|
4
|
+
import { a as PairedBootstrapResult } from './statistics-KUnG73jH.js';
|
|
5
|
+
import { P as PolicyEditCandidateRecord } from './policy-edit-wG9uFEFm.js';
|
|
5
6
|
import { HostedClient, TraceSpanEvent } from './hosted/index.js';
|
|
6
|
-
import { C as CampaignStorage } from './storage-
|
|
7
|
+
import { C as CampaignStorage } from './storage-DrX3v_5B.js';
|
|
7
8
|
|
|
8
9
|
/**
|
|
9
10
|
* Compose multiple `Gate` implementations — every gate must pass for the
|
|
@@ -356,7 +357,7 @@ declare function evolutionaryProposer<TFindings = unknown>(opts: EvolutionaryPro
|
|
|
356
357
|
* Two artifacts, one source of truth:
|
|
357
358
|
*
|
|
358
359
|
* 1. `LoopProvenanceRecord` — a structured JSON record capturing every
|
|
359
|
-
* candidate (surfaceHash + label + rationale), its measured composite,
|
|
360
|
+
* candidate (surfaceHash + label + rationale + structured cause), its measured composite,
|
|
360
361
|
* the gate decision + reasons + delta, the held-out lift, the explicit
|
|
361
362
|
* baseline→candidate diff, and BACKEND PROVENANCE (the
|
|
362
363
|
* `assertRealBackend` verdict + worker call count + model). This is the
|
|
@@ -387,6 +388,18 @@ interface LoopProvenanceCandidate {
|
|
|
387
388
|
/** Proposer rationale — the "because Z". When the proposer returned a bare
|
|
388
389
|
* surface (blind mutator) this is absent. */
|
|
389
390
|
rationale?: string;
|
|
391
|
+
/** Exact validated cause when the proposer emitted a structured record. */
|
|
392
|
+
candidateRecord?: PolicyEditCandidateRecord;
|
|
393
|
+
/** Exact complete incumbent this candidate mutated. */
|
|
394
|
+
parentSurfaceHash: string;
|
|
395
|
+
/** Search-split composite of the exact parent. */
|
|
396
|
+
parentComposite: number;
|
|
397
|
+
/** Search-split composite change relative to the exact parent. */
|
|
398
|
+
observedDeltaFromParent?: number;
|
|
399
|
+
/** Whether the candidate completed every designed cell and could be selected. */
|
|
400
|
+
eligibleForPromotion: boolean;
|
|
401
|
+
/** Designed-denominator receipt retained even for incomplete candidates. */
|
|
402
|
+
coverage: NonNullable<GenerationCandidate['coverage']>;
|
|
390
403
|
/** Mean composite this candidate scored on the search split. */
|
|
391
404
|
composite: number;
|
|
392
405
|
/** Whether this candidate was promoted out of its generation. */
|
|
@@ -409,7 +422,7 @@ interface LoopProvenanceBackend {
|
|
|
409
422
|
* the bare hosted event) + backend provenance.
|
|
410
423
|
*/
|
|
411
424
|
interface LoopProvenanceRecord {
|
|
412
|
-
schema: 'tangle.loop-provenance.
|
|
425
|
+
schema: 'tangle.loop-provenance.v3';
|
|
413
426
|
runId: string;
|
|
414
427
|
runDir: string;
|
|
415
428
|
timestamp: string;
|
|
@@ -421,8 +434,10 @@ interface LoopProvenanceRecord {
|
|
|
421
434
|
winnerRationale?: string;
|
|
422
435
|
/** The explicit baseline→winner unified diff the gate decided on. */
|
|
423
436
|
diff: string;
|
|
424
|
-
/** Every candidate across every generation,
|
|
437
|
+
/** Every candidate across every generation, with its rationale and structured cause. */
|
|
425
438
|
candidates: LoopProvenanceCandidate[];
|
|
439
|
+
/** Baseline composite on the search split that generated the candidates. */
|
|
440
|
+
baselineSearchComposite: number;
|
|
426
441
|
/** The gate verdict — decision + reasons + contributing gates + delta. */
|
|
427
442
|
gate: {
|
|
428
443
|
decision: GateDecision;
|
|
@@ -453,18 +468,15 @@ interface BuildLoopProvenanceArgs<TArtifact, TScenario extends Scenario> {
|
|
|
453
468
|
winnerLabel?: string;
|
|
454
469
|
winnerRationale?: string;
|
|
455
470
|
diff: string;
|
|
471
|
+
/** Baseline composite on the search split, distinct from holdout scoring. */
|
|
472
|
+
baselineSearchComposite: number;
|
|
456
473
|
/** Per-generation candidate records straight off the loop result. */
|
|
457
474
|
generations: Array<{
|
|
458
475
|
generationIndex: number;
|
|
459
|
-
candidates:
|
|
460
|
-
surfaceHash: string;
|
|
461
|
-
composite: number;
|
|
462
|
-
label?: string;
|
|
463
|
-
rationale?: string;
|
|
464
|
-
}>;
|
|
476
|
+
candidates: GenerationCandidate[];
|
|
465
477
|
promoted: string[];
|
|
466
|
-
/** Surfaces measured this generation, keyed
|
|
467
|
-
*
|
|
478
|
+
/** Surfaces measured this generation, keyed by surface hash so the content
|
|
479
|
+
* hash can be computed and the loop identity rechecked from real bytes. */
|
|
468
480
|
surfaces: Array<{
|
|
469
481
|
surfaceHash: string;
|
|
470
482
|
surface: MutableSurface;
|
|
@@ -1,5 +1,5 @@
|
|
|
1
|
-
import { L as LlmSpan, J as JudgeSpan, R as Run, F as FailureClass
|
|
2
|
-
import { T as TraceStore } from './store-
|
|
1
|
+
import { L as LlmSpan, T as ToolSpan, J as JudgeSpan, R as Run, F as FailureClass } from './schema-B3Q3l9Z_.js';
|
|
2
|
+
import { T as TraceStore } from './store-DGqD0Pyo.js';
|
|
3
3
|
|
|
4
4
|
/**
|
|
5
5
|
* Typed query helpers over TraceStore.
|
|
@@ -19,6 +19,8 @@ declare function judgeSpans(store: TraceStore, runId?: string): Promise<JudgeSpa
|
|
|
19
19
|
declare function groupBy<T, K extends string | number>(items: T[], key: (t: T) => K): Map<K, T[]>;
|
|
20
20
|
/** Hash tool arguments to an orderless-key-stable string for de-duplication. */
|
|
21
21
|
declare function argHash(args: unknown): string;
|
|
22
|
+
/** Whether argument-based comparisons are valid for this tool call. */
|
|
23
|
+
declare function hasCapturedToolArgs(span: ToolSpan): boolean;
|
|
22
24
|
/** Sum an LLM-span array into aggregate token + cost. */
|
|
23
25
|
declare function aggregateLlm(spans: LlmSpan[]): {
|
|
24
26
|
inputTokens: number;
|
|
@@ -30,4 +32,4 @@ declare function aggregateLlm(spans: LlmSpan[]): {
|
|
|
30
32
|
/** Pick the outcome's failure class when present, else derive 'success' from run status. */
|
|
31
33
|
declare function runFailureClass(run: Run): FailureClass;
|
|
32
34
|
|
|
33
|
-
export { aggregateLlm as a, argHash as b, runsForScenario as c, groupBy as g, judgeSpans as j, llmSpans as l, runFailureClass as r, toolSpans as t };
|
|
35
|
+
export { aggregateLlm as a, argHash as b, runsForScenario as c, groupBy as g, hasCapturedToolArgs as h, judgeSpans as j, llmSpans as l, runFailureClass as r, toolSpans as t };
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
import { a as DatasetSplit, b as DatasetManifest, D as DatasetScenario } from './dataset-NENEzRgk.js';
|
|
2
|
-
import { m as GateDecision } from './summary-report-
|
|
3
|
-
import { R as RunRecord,
|
|
2
|
+
import { m as GateDecision } from './summary-report-C5bKFfm-.js';
|
|
3
|
+
import { R as RunRecord, a as RunSplitTag } from './run-record-BDH49H2E.js';
|
|
4
4
|
|
|
5
5
|
/**
|
|
6
6
|
* Release confidence gate.
|
package/dist/reporting.d.ts
CHANGED
|
@@ -1,16 +1,17 @@
|
|
|
1
|
-
export { R as RubricOutcomePair, a as RubricPredictiveValidityInput, b as RubricPredictiveValidityReport, c as RubricRanking, r as rubricPredictiveValidity } from './rubric-predictive-validity-
|
|
2
|
-
export { B as BootstrapOptions, a as BootstrapResult, J as JudgeReplayGateArgs, R as ReleaseConfidenceAxis, b as ReleaseConfidenceAxisName, c as ReleaseConfidenceInput, d as ReleaseConfidenceIssue, e as ReleaseConfidenceMetrics, f as ReleaseConfidenceScorecard, g as ReleaseConfidenceStatus, h as ReleaseConfidenceThresholds, i as ReleaseTraceEvidence, j as RenderReleaseReportOptions, V as Verdict, k as assertReleaseConfidence, l as bootstrapCi, m as evaluateReleaseConfidence, n as judgeReplayGate, r as renderReleaseReport } from './release-report-
|
|
1
|
+
export { R as RubricOutcomePair, a as RubricPredictiveValidityInput, b as RubricPredictiveValidityReport, c as RubricRanking, r as rubricPredictiveValidity } from './rubric-predictive-validity-p49lLVrE.js';
|
|
2
|
+
export { B as BootstrapOptions, a as BootstrapResult, J as JudgeReplayGateArgs, R as ReleaseConfidenceAxis, b as ReleaseConfidenceAxisName, c as ReleaseConfidenceInput, d as ReleaseConfidenceIssue, e as ReleaseConfidenceMetrics, f as ReleaseConfidenceScorecard, g as ReleaseConfidenceStatus, h as ReleaseConfidenceThresholds, i as ReleaseTraceEvidence, j as RenderReleaseReportOptions, V as Verdict, k as assertReleaseConfidence, l as bootstrapCi, m as evaluateReleaseConfidence, n as judgeReplayGate, r as renderReleaseReport } from './release-report-C8G2i5Xi.js';
|
|
3
3
|
export { I as InterimReleaseConfidence, a as InterimReleaseConfidenceInput, P as PairedEvalueOptions, b as PairedEvalueSequence, c as PairedEvalueStep, S as SequentialDecision, e as evaluateInterimReleaseConfidence, p as pairedEvalueSequence } from './sequential-5iSVfzl2.js';
|
|
4
|
-
export { P as PairedBootstrapOptions, a as PairedBootstrapResult, b as benjaminiHochberg, p as pairedBootstrap, w as wilcoxonSignedRank } from './statistics-
|
|
5
|
-
export { G as GainDistributionBin, a as GainDistributionFigureSpec, b as GainDistributionOptions, P as ParetoFigureSpec, c as ParetoPoint, R as RESEARCH_REPORT_HARD_PAIR_FLOOR, d as ResearchReport, e as ResearchReportCandidate, f as ResearchReportDecision, g as ResearchReportMethodology, h as ResearchReportOptions, i as ResearchReportRecommendation, S as SummaryTable, j as SummaryTableOptions, k as SummaryTableRow, l as gainHistogram, p as paretoChart, r as researchReport, s as summaryTable } from './summary-report-
|
|
6
|
-
import './run-record-
|
|
4
|
+
export { P as PairedBootstrapOptions, a as PairedBootstrapResult, b as benjaminiHochberg, p as pairedBootstrap, w as wilcoxonSignedRank } from './statistics-KUnG73jH.js';
|
|
5
|
+
export { G as GainDistributionBin, a as GainDistributionFigureSpec, b as GainDistributionOptions, P as ParetoFigureSpec, c as ParetoPoint, R as RESEARCH_REPORT_HARD_PAIR_FLOOR, d as ResearchReport, e as ResearchReportCandidate, f as ResearchReportDecision, g as ResearchReportMethodology, h as ResearchReportOptions, i as ResearchReportRecommendation, S as SummaryTable, j as SummaryTableOptions, k as SummaryTableRow, l as gainHistogram, p as paretoChart, r as researchReport, s as summaryTable } from './summary-report-C5bKFfm-.js';
|
|
6
|
+
import './run-record-BDH49H2E.js';
|
|
7
7
|
import '@tangle-network/agent-interface';
|
|
8
8
|
import './errors-oeQrLqXC.js';
|
|
9
|
-
import './schema-
|
|
9
|
+
import './schema-B3Q3l9Z_.js';
|
|
10
10
|
import './outcome-store-rnXLEqSn.js';
|
|
11
11
|
import './dataset-NENEzRgk.js';
|
|
12
12
|
import './judge-calibration-7C-IDmKr.js';
|
|
13
|
-
import './types-
|
|
13
|
+
import './types-BkfcQnxV.js';
|
|
14
|
+
import './cost-ledger-DWy3XdJc.js';
|
|
14
15
|
import '@tangle-network/tcloud';
|
|
15
|
-
import './failure-cluster-
|
|
16
|
-
import './store-
|
|
16
|
+
import './failure-cluster-DOAcSJ87.js';
|
|
17
|
+
import './store-DGqD0Pyo.js';
|
|
@@ -1,11 +1,11 @@
|
|
|
1
|
-
import {
|
|
2
|
-
import {
|
|
3
|
-
import { h as ResearchReportOptions, d as ResearchReport, m as GateDecision } from './summary-report-
|
|
4
|
-
import { T as TraceEmitter, R as RunCompleteHook } from './emitter-
|
|
5
|
-
import { R as RunIntegrityExpectations, a as RunIntegrityReport } from './integrity-
|
|
1
|
+
import { a as RunSplitTag, c as RunTokenUsage, d as RunJudgeMetadata, J as JudgeScoresRecord, A as AgentProfileCell, e as AgentProfileCellInput, R as RunRecord } from './run-record-BDH49H2E.js';
|
|
2
|
+
import { a as LlmClientOptions, b as LlmRouteRequirements } from './llm-client-qoDd18Qz.js';
|
|
3
|
+
import { h as ResearchReportOptions, d as ResearchReport, m as GateDecision } from './summary-report-C5bKFfm-.js';
|
|
4
|
+
import { T as TraceEmitter, R as RunCompleteHook } from './emitter-CjD7vUwv.js';
|
|
5
|
+
import { R as RunIntegrityExpectations, a as RunIntegrityReport } from './integrity-DqlBiLyK.js';
|
|
6
6
|
import { R as RawProviderSink } from './raw-provider-sink-C46HDghv.js';
|
|
7
|
-
import { F as FailureClass } from './schema-
|
|
8
|
-
import { T as TraceStore } from './store-
|
|
7
|
+
import { F as FailureClass } from './schema-B3Q3l9Z_.js';
|
|
8
|
+
import { T as TraceStore } from './store-DGqD0Pyo.js';
|
|
9
9
|
|
|
10
10
|
/**
|
|
11
11
|
* EvalCampaign — opinionated matrix runner that wires the four
|
package/dist/rl.d.ts
CHANGED
|
@@ -1,25 +1,30 @@
|
|
|
1
|
-
import { R as RunRecord,
|
|
1
|
+
import { R as RunRecord, a as RunSplitTag } from './run-record-BDH49H2E.js';
|
|
2
2
|
export { A as AdversarialMutation } from './adversarial-B7loGVVX.js';
|
|
3
|
-
import { S as Span } from './schema-
|
|
4
|
-
import { T as TraceStore } from './store-
|
|
3
|
+
import { S as Span } from './schema-B3Q3l9Z_.js';
|
|
4
|
+
import { T as TraceStore } from './store-DGqD0Pyo.js';
|
|
5
5
|
export { O as OffPolicyEstimate, a as OffPolicyOptions, b as OffPolicyTrajectory, d as doublyRobust, i as inverseProbabilityWeighting, o as offPolicyEstimateAll, s as selfNormalizedImportanceWeighting } from './off-policy-DiwuKKg7.js';
|
|
6
6
|
import { b as OutcomeStore } from './outcome-store-rnXLEqSn.js';
|
|
7
7
|
export { D as DeploymentOutcome, F as FileSystemOutcomeStore, a as FileSystemOutcomeStoreOptions, I as InMemoryOutcomeStore } from './outcome-store-rnXLEqSn.js';
|
|
8
|
-
import { b as RubricPredictiveValidityReport } from './rubric-predictive-validity-
|
|
9
|
-
import { R as Researcher, F as FailureMode, S as SteeringChange, E as ExperimentPlan, a as ExperimentResult, b as EvalCampaignResult, c as EvalCampaignOptions } from './researcher-
|
|
10
|
-
export { r as runEvalCampaign } from './researcher-
|
|
8
|
+
import { b as RubricPredictiveValidityReport } from './rubric-predictive-validity-p49lLVrE.js';
|
|
9
|
+
import { R as Researcher, F as FailureMode, S as SteeringChange, E as ExperimentPlan, a as ExperimentResult, b as EvalCampaignResult, c as EvalCampaignOptions } from './researcher-C8XyxQsu.js';
|
|
10
|
+
export { r as runEvalCampaign } from './researcher-C8XyxQsu.js';
|
|
11
11
|
import { a as VerificationReport } from './multi-layer-verifier-BsqKuLyN.js';
|
|
12
12
|
import { I as InterimReleaseConfidence } from './sequential-5iSVfzl2.js';
|
|
13
|
-
import { C as CampaignResult } from './types-
|
|
13
|
+
import { C as CampaignResult } from './types-BSw1rOUB.js';
|
|
14
14
|
import '@tangle-network/agent-interface';
|
|
15
15
|
import './errors-oeQrLqXC.js';
|
|
16
|
-
import './llm-client-
|
|
16
|
+
import './llm-client-qoDd18Qz.js';
|
|
17
|
+
import './cost-ledger-DWy3XdJc.js';
|
|
17
18
|
import './raw-provider-sink-C46HDghv.js';
|
|
18
|
-
import './summary-report-
|
|
19
|
-
import './failure-cluster-
|
|
20
|
-
import './emitter-
|
|
21
|
-
import './integrity-
|
|
19
|
+
import './summary-report-C5bKFfm-.js';
|
|
20
|
+
import './failure-cluster-DOAcSJ87.js';
|
|
21
|
+
import './emitter-CjD7vUwv.js';
|
|
22
|
+
import './integrity-DqlBiLyK.js';
|
|
22
23
|
import './verdict-C9MlYujm.js';
|
|
24
|
+
import './policy-edit-wG9uFEFm.js';
|
|
25
|
+
import './store-C1YxJDEK.js';
|
|
26
|
+
import './types-BkfcQnxV.js';
|
|
27
|
+
import '@tangle-network/tcloud';
|
|
23
28
|
|
|
24
29
|
/**
|
|
25
30
|
* Adaptive curriculum / active scenario selection.
|
package/dist/rl.js
CHANGED
|
@@ -10,7 +10,7 @@ import {
|
|
|
10
10
|
} from "./chunk-3RF76KTD.js";
|
|
11
11
|
import {
|
|
12
12
|
runEvalCampaign
|
|
13
|
-
} from "./chunk-
|
|
13
|
+
} from "./chunk-GQCZRZ7L.js";
|
|
14
14
|
import {
|
|
15
15
|
detectRewardHacking,
|
|
16
16
|
extractVerifiableReward,
|
|
@@ -38,7 +38,7 @@ import "./chunk-TVVP3ZZQ.js";
|
|
|
38
38
|
import "./chunk-5UF54T55.js";
|
|
39
39
|
import "./chunk-XJYR7XFV.js";
|
|
40
40
|
import "./chunk-VSMTAMNK.js";
|
|
41
|
-
import "./chunk-
|
|
41
|
+
import "./chunk-NJC7U437.js";
|
|
42
42
|
import "./chunk-PC4UYEBM.js";
|
|
43
43
|
import {
|
|
44
44
|
ValidationError
|
|
@@ -1,12 +1,14 @@
|
|
|
1
1
|
import {
|
|
2
2
|
planCampaignRun,
|
|
3
3
|
runCampaign
|
|
4
|
-
} from "./chunk-
|
|
4
|
+
} from "./chunk-IDZTTFRR.js";
|
|
5
5
|
import "./chunk-PJQFMIOX.js";
|
|
6
|
+
import "./chunk-VCTY3W6J.js";
|
|
7
|
+
import "./chunk-VI2UW6B6.js";
|
|
6
8
|
import "./chunk-ONWEPEDO.js";
|
|
7
9
|
import "./chunk-PZ5AY32C.js";
|
|
8
10
|
export {
|
|
9
11
|
planCampaignRun,
|
|
10
12
|
runCampaign
|
|
11
13
|
};
|
|
12
|
-
//# sourceMappingURL=run-campaign-
|
|
14
|
+
//# sourceMappingURL=run-campaign-IM26A6PD.js.map
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
import { AgentProfile } from '@tangle-network/agent-interface';
|
|
2
2
|
import { V as ValidationError } from './errors-oeQrLqXC.js';
|
|
3
|
-
import { F as FailureClass } from './schema-
|
|
3
|
+
import { F as FailureClass } from './schema-B3Q3l9Z_.js';
|
|
4
4
|
|
|
5
5
|
type AgentProfileCellSchemaVersion = 'agent-profile-cell/v1';
|
|
6
6
|
type AgentProfileJson = string | number | boolean | null | AgentProfileJson[] | {
|
|
@@ -357,4 +357,4 @@ declare function roundTripRunRecord(record: RunRecord): RunRecord;
|
|
|
357
357
|
*/
|
|
358
358
|
declare function modelHasSnapshot(model: string): boolean;
|
|
359
359
|
|
|
360
|
-
export { type AgentProfileCell as A, requireAgentProfileCell as B, resolveRunCostProvenance as C, roundTripRunRecord as D, toAgentProfileJson as E, validateAgentProfileCell as F, validateRunRecord as G, verifyAgentProfileCell as H, type JudgeScoresRecord as J, type RunRecord as R, type
|
|
360
|
+
export { type AgentProfileCell as A, requireAgentProfileCell as B, resolveRunCostProvenance as C, roundTripRunRecord as D, toAgentProfileJson as E, validateAgentProfileCell as F, validateRunRecord as G, verifyAgentProfileCell as H, type JudgeScoresRecord as J, type RunRecord as R, type RunSplitTag as a, type RunCostProvenance as b, type RunTokenUsage as c, type RunJudgeMetadata as d, type AgentProfileCellInput as e, type AgentProfileJson as f, AGENT_PROFILE_KINDS as g, type AgentInterfaceProfileLike as h, type AgentProfileCellSchemaVersion as i, AgentProfileCellValidationError as j, type AgentProfileDimensionValue as k, type AgentProfileHarness as l, type AgentProfileKind as m, type AgentProfileSource as n, type AgentProfileSourceInput as o, type RunOutcome as p, RunRecordValidationError as q, agentProfileCellHashMaterial as r, agentProfileCellKey as s, assertRunAgentProfileCell as t, buildAgentInterfaceProfileCell as u, buildAgentProfileCell as v, groupRunsByAgentProfileCell as w, isRunRecord as x, modelHasSnapshot as y, parseRunRecordSafe as z };
|
|
@@ -116,6 +116,8 @@ interface ToolSpan extends SpanBase {
|
|
|
116
116
|
kind: 'tool';
|
|
117
117
|
toolName: string;
|
|
118
118
|
args: unknown;
|
|
119
|
+
/** False when the source observed the call but did not capture its arguments. */
|
|
120
|
+
argsCaptured?: boolean;
|
|
119
121
|
result?: unknown;
|
|
120
122
|
latencyMs?: number;
|
|
121
123
|
}
|
|
@@ -1,10 +1,12 @@
|
|
|
1
1
|
import { AxAIService } from '@ax-llm/ax';
|
|
2
2
|
import { z } from 'zod';
|
|
3
|
-
import {
|
|
3
|
+
import { c as AnalystFinding, A as Analyst, a as AnalystContext } from './policy-edit-wG9uFEFm.js';
|
|
4
|
+
import { T as TraceAnalystKindSpec } from './kind-factory-ClZmO25A.js';
|
|
4
5
|
import { a as TraceAnalystSpan } from './store-C1YxJDEK.js';
|
|
5
|
-
import { R as Run, S as Span, e as TraceEvent, A as Artifact, B as BudgetLedgerEntry } from './schema-
|
|
6
|
-
import { T as TraceStore } from './store-
|
|
7
|
-
import {
|
|
6
|
+
import { R as Run, S as Span, e as TraceEvent, A as Artifact, B as BudgetLedgerEntry } from './schema-B3Q3l9Z_.js';
|
|
7
|
+
import { T as TraceStore } from './store-DGqD0Pyo.js';
|
|
8
|
+
import { C as CostLedger } from './cost-ledger-DWy3XdJc.js';
|
|
9
|
+
import { a as LlmClientOptions } from './llm-client-qoDd18Qz.js';
|
|
8
10
|
import { S as Severity } from './multi-layer-verifier-BsqKuLyN.js';
|
|
9
11
|
|
|
10
12
|
interface CreateAnalystAiConfig {
|
|
@@ -499,7 +501,12 @@ interface SuboptimalSignal {
|
|
|
499
501
|
evidence: Record<string, number | string | boolean>;
|
|
500
502
|
}
|
|
501
503
|
interface BehavioralMetrics {
|
|
504
|
+
/** The only trace represented by these metrics; null when spans are empty. */
|
|
505
|
+
traceId: string | null;
|
|
502
506
|
llmCallCount: number;
|
|
507
|
+
/** Causally serial LLM timelines. Parallel branches are never joined. */
|
|
508
|
+
tokenSequences: BehavioralTokenSequence[];
|
|
509
|
+
/** Token values from the longest serial timeline, retained for convenience. */
|
|
503
510
|
inputTokenTrajectory: number[];
|
|
504
511
|
outputTokenTrajectory: number[];
|
|
505
512
|
toolHistogram: Record<string, number>;
|
|
@@ -510,6 +517,12 @@ interface BehavioralMetrics {
|
|
|
510
517
|
hasSelfVerification: boolean;
|
|
511
518
|
signals: SuboptimalSignal[];
|
|
512
519
|
}
|
|
520
|
+
interface BehavioralTokenSequence {
|
|
521
|
+
scopeId: string;
|
|
522
|
+
spanIds: string[];
|
|
523
|
+
inputTokenTrajectory: Array<number | null>;
|
|
524
|
+
outputTokenTrajectory: Array<number | null>;
|
|
525
|
+
}
|
|
513
526
|
/**
|
|
514
527
|
* Reduce a span list to behavioral metrics + fired suboptimality signals.
|
|
515
528
|
* Pure + deterministic: same spans → same output, on any machine, no model.
|
|
@@ -672,6 +685,8 @@ interface SemanticConceptJudgeOptions {
|
|
|
672
685
|
model?: string;
|
|
673
686
|
/** Per-call timeout. Default 300s. */
|
|
674
687
|
timeoutMs?: number;
|
|
688
|
+
/** Provider-enforced output limit. Default 16000. */
|
|
689
|
+
maxTokens?: number;
|
|
675
690
|
/** Pipeline budget for the prompt (source blob truncation). Default 45000. */
|
|
676
691
|
maxSourceChars?: number;
|
|
677
692
|
/** Per-file cap before inclusion. Default 20000. */
|
|
@@ -680,6 +695,9 @@ interface SemanticConceptJudgeOptions {
|
|
|
680
695
|
maxHtmlChars?: number;
|
|
681
696
|
/** LlmClient config (baseUrl, apiKey, authHeader, …). */
|
|
682
697
|
llm?: LlmClientOptions;
|
|
698
|
+
costLedger?: CostLedger;
|
|
699
|
+
costPhase?: string;
|
|
700
|
+
signal?: AbortSignal;
|
|
683
701
|
/**
|
|
684
702
|
* Score aggregation strategy. Default `mean` — uniform average across
|
|
685
703
|
* concepts. Cross-vertical comparisons should use `complexity` to
|
|
@@ -702,4 +720,4 @@ declare function runSemanticConceptJudge(input: SemanticConceptJudgeInput, optio
|
|
|
702
720
|
*/
|
|
703
721
|
declare function createSemanticConceptJudge(options?: SemanticConceptJudgeOptions): (input: SemanticConceptJudgeInput) => Promise<SemanticConceptJudgeResult>;
|
|
704
722
|
|
|
705
|
-
export { type RunScore as A, type BehavioralMetrics as B, type CreateAnalystAiConfig as C, DEFAULT_TRACE_ANALYST_KINDS as D, type RunScoreWeights as E, FAILURE_MODE_KIND_SPEC as F, type
|
|
723
|
+
export { runSemanticConceptJudge as $, type RunScore as A, type BehavioralMetrics as B, type CreateAnalystAiConfig as C, DEFAULT_TRACE_ANALYST_KINDS as D, type RunScoreWeights as E, FAILURE_MODE_KIND_SPEC as F, type BehavioralTokenSequence as G, type ConceptComplexity as H, IMPROVEMENT_KIND_SPEC as I, type ConceptFinding as J, KIND_EXPECTED_SUBJECTS as K, type ConceptSpec as L, type ConceptWeightStrategy as M, DEFAULT_COMPLEXITY_WEIGHTS as N, DEFAULT_RUN_SCORE_WEIGHTS as O, type PersistedFinding as P, type RunCriticOptions as Q, RunCritic as R, type SemanticConceptJudgeOptions as S, SEMANTIC_CONCEPT_JUDGE_VERSION as T, type SemanticConceptJudgeResult as U, type SuboptimalCode as V, type SuboptimalSignal as W, aggregateRunScore as X, clamp01 as Y, computeTraceMetrics as Z, createSemanticConceptJudge as _, type RunTrace as a, type SemanticConceptJudgeInput as b, type DiffPolicy as c, FINDING_SUBJECT_GRAMMAR_PROMPT as d, FINDING_SUBJECT_KINDS as e, FINDING_SUBJECT_SYNTAX as f, type FindingSubject as g, type FindingSubjectKind as h, FindingSubjectStringSchema as i, type FindingsDiff as j, FindingsStore as k, KNOWLEDGE_GAP_KIND_SPEC as l, KNOWLEDGE_POISONING_KIND_SPEC as m, SKILL_USAGE_ANALYST as n, SkillUsageAnalyst as o, type SkillUsageRecord as p, type SkillUsageReport as q, type SkillUsageScanConfig as r, buildSkillUsageReport as s, createAnalystAi as t, defaultIsMaterial as u, diffFindings as v, emitSkillUsageFindings as w, findingSubjectGrammarPromptFor as x, parseFindingSubject as y, renderFindingSubject as z };
|
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
import { d as ContinuousAgreementOptions, C as ContinuousAgreement } from './judge-calibration-7C-IDmKr.js';
|
|
2
|
-
import { J as JudgeScore } from './types-
|
|
2
|
+
import { J as JudgeScore } from './types-BkfcQnxV.js';
|
|
3
3
|
|
|
4
4
|
/** Identity: dimensions already follow "higher = better" by prompt convention
|
|
5
5
|
* (inverted dims like hallucination are scored 10 = best at the source). */
|
|
@@ -1,3 +1,5 @@
|
|
|
1
|
+
import { C as CostLedger } from './cost-ledger-DWy3XdJc.js';
|
|
2
|
+
|
|
1
3
|
/**
|
|
2
4
|
* `CampaignStorage` — the filesystem seam `runCampaign` writes through
|
|
3
5
|
* (run/cell dirs, the resumability cache, per-cell artifacts, trace spans).
|
|
@@ -22,6 +24,9 @@ interface CampaignStorage {
|
|
|
22
24
|
read(path: string): string | undefined;
|
|
23
25
|
/** Write a file (string or bytes). Parent dir is assumed ensured. */
|
|
24
26
|
write(path: string, content: string | Uint8Array): void;
|
|
27
|
+
/** Append only when the current UTF-8 byte length matches `expectedBytes`.
|
|
28
|
+
* Returns the new length, or undefined when another writer won. */
|
|
29
|
+
append?(path: string, content: string, expectedBytes: number): number | undefined;
|
|
25
30
|
}
|
|
26
31
|
/** Node-filesystem storage — the default. Lazily requires `node:fs` so the
|
|
27
32
|
* module imports cleanly in non-Node runtimes (where the caller passes
|
|
@@ -35,5 +40,11 @@ declare function fsCampaignStorage(): CampaignStorage;
|
|
|
35
40
|
* live in a `Map` for the duration of the run; the `CampaignResult` is
|
|
36
41
|
* fully populated, but nothing is persisted to disk. */
|
|
37
42
|
declare function inMemoryCampaignStorage(): CampaignStorage;
|
|
43
|
+
/** Open the durable spend account stored beside a logical run. */
|
|
44
|
+
declare function createRunCostLedger(input: {
|
|
45
|
+
storage: CampaignStorage;
|
|
46
|
+
runDir: string;
|
|
47
|
+
costCeilingUsd?: number;
|
|
48
|
+
}): CostLedger;
|
|
38
49
|
|
|
39
|
-
export { type CampaignStorage as C, fsCampaignStorage as f, inMemoryCampaignStorage as i };
|
|
50
|
+
export { type CampaignStorage as C, createRunCostLedger as c, fsCampaignStorage as f, inMemoryCampaignStorage as i };
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import { R as Run, S as Span, e as TraceEvent, A as Artifact, B as BudgetLedgerEntry, f as RunStatus, g as RunLayer, b as SpanKind, E as EventKind } from './schema-
|
|
1
|
+
import { R as Run, S as Span, e as TraceEvent, A as Artifact, B as BudgetLedgerEntry, f as RunStatus, g as RunLayer, b as SpanKind, E as EventKind } from './schema-B3Q3l9Z_.js';
|
|
2
2
|
|
|
3
3
|
interface RunFilter {
|
|
4
4
|
scenarioId?: string;
|
|
@@ -1,5 +1,5 @@
|
|
|
1
|
-
import { R as RunRecord } from './run-record-
|
|
2
|
-
import { F as FailureClusterReport } from './failure-cluster-
|
|
1
|
+
import { R as RunRecord } from './run-record-BDH49H2E.js';
|
|
2
|
+
import { F as FailureClusterReport } from './failure-cluster-DOAcSJ87.js';
|
|
3
3
|
|
|
4
4
|
/**
|
|
5
5
|
* HeldOutGate — first-class held-out paired-delta promotion gate.
|
|
@@ -1,6 +1,6 @@
|
|
|
1
|
-
import { T as TraceEmitter } from './emitter-
|
|
2
|
-
import { R as Run, F as FailureClass } from './schema-
|
|
3
|
-
import { T as TraceStore } from './store-
|
|
1
|
+
import { T as TraceEmitter } from './emitter-CjD7vUwv.js';
|
|
2
|
+
import { R as Run, F as FailureClass } from './schema-B3Q3l9Z_.js';
|
|
3
|
+
import { T as TraceStore } from './store-DGqD0Pyo.js';
|
|
4
4
|
|
|
5
5
|
/**
|
|
6
6
|
* SandboxHarness — executes a scenario in an isolated environment and
|
package/dist/traces.d.ts
CHANGED
|
@@ -1,23 +1,24 @@
|
|
|
1
1
|
import { N as NotFoundError, R as ReplayError } from './errors-oeQrLqXC.js';
|
|
2
2
|
import { P as ProviderRedactor, R as RawProviderSink, d as RawProviderEvent } from './raw-provider-sink-C46HDghv.js';
|
|
3
3
|
export { F as FileSystemRawProviderSink, a as FileSystemRawProviderSinkOptions, I as InMemoryRawProviderSink, b as InMemoryRawProviderSinkOptions, N as NoopRawProviderSink, c as RawProviderDirection, e as RawProviderSinkFilter, f as defaultProviderRedactor, p as providerFromBaseUrl } from './raw-provider-sink-C46HDghv.js';
|
|
4
|
-
import { a as RunCompleteHookContext, R as RunCompleteHook } from './emitter-
|
|
5
|
-
export { S as SpanHandle, T as TraceEmitter, b as TraceEmitterOptions, l as llmSpanFromProvider } from './emitter-
|
|
6
|
-
export { b as RunIntegrityError, R as RunIntegrityExpectations, c as RunIntegrityIssue, d as RunIntegrityIssueCode, a as RunIntegrityReport, e as assertRunCaptured, t as throwIfRunIncomplete } from './integrity-
|
|
7
|
-
import { T as TraceStore } from './store-
|
|
8
|
-
export { E as EventFilter, F as FileSystemTraceStore, a as FileSystemTraceStoreOptions, I as InMemoryTraceStore, R as RunFilter, S as SpanFilter } from './store-
|
|
9
|
-
|
|
10
|
-
|
|
11
|
-
export {
|
|
4
|
+
import { a as RunCompleteHookContext, R as RunCompleteHook } from './emitter-CjD7vUwv.js';
|
|
5
|
+
export { S as SpanHandle, T as TraceEmitter, b as TraceEmitterOptions, l as llmSpanFromProvider } from './emitter-CjD7vUwv.js';
|
|
6
|
+
export { b as RunIntegrityError, R as RunIntegrityExpectations, c as RunIntegrityIssue, d as RunIntegrityIssueCode, a as RunIntegrityReport, e as assertRunCaptured, t as throwIfRunIncomplete } from './integrity-DqlBiLyK.js';
|
|
7
|
+
import { T as TraceStore } from './store-DGqD0Pyo.js';
|
|
8
|
+
export { E as EventFilter, F as FileSystemTraceStore, a as FileSystemTraceStoreOptions, I as InMemoryTraceStore, R as RunFilter, S as SpanFilter } from './store-DGqD0Pyo.js';
|
|
9
|
+
import { T as ToolSpan, R as Run } from './schema-B3Q3l9Z_.js';
|
|
10
|
+
export { A as Artifact, B as BudgetLedgerEntry, h as BudgetSpec, E as EventKind, i as FAILURE_CLASSES, F as FailureClass, G as GenericSpan, J as JudgeSpan, L as LlmSpan, M as Message, c as RetrievalSpan, g as RunLayer, a as RunOutcome, f as RunStatus, d as SandboxSpan, S as Span, j as SpanBase, b as SpanKind, k as SpanStatus, l as TRACE_SCHEMA_VERSION, e as TraceEvent, m as isJudgeSpan, n as isLlmSpan, o as isRetrievalSpan, p as isSandboxSpan, q as isToolSpan } from './schema-B3Q3l9Z_.js';
|
|
11
|
+
export { a as aggregateLlm, b as argHash, g as groupBy, h as hasCapturedToolArgs, j as judgeSpans, l as llmSpans, r as runFailureClass, c as runsForScenario, t as toolSpans } from './query-CF7PG61p.js';
|
|
12
12
|
import { A as AnalyzeTracesOptions, b as AnalyzeTracesResult } from './analyst-C8HHvfJp.js';
|
|
13
13
|
export { a as AnalyzeTracesInput, c as AnalyzeTracesTurnSnapshot, d as analyzeTraces } from './analyst-C8HHvfJp.js';
|
|
14
14
|
import { h as TraceAnalystSpanKind, i as TraceAnalystSpanStatus, T as TraceAnalysisStore, g as TraceAnalystFilters, b as DatasetOverview, Q as QueryTracesPage, l as ViewTraceResult, V as ViewSpansResult, c as SearchTraceResult, S as SearchSpanResult } from './store-C1YxJDEK.js';
|
|
15
15
|
export { D as DEFAULT_TRACE_ANALYST_BUDGETS, E as ErrorCluster, d as SpanMatchRecord, e as TRACE_ANALYST_TRUNCATION_MARKER_PREFIX, f as TraceAnalystByteBudgets, a as TraceAnalystSpan, j as TraceAnalystTraceSummary, k as ViewTraceOversized } from './store-C1YxJDEK.js';
|
|
16
|
-
import {
|
|
16
|
+
import { a as RunSplitTag, c as RunTokenUsage, R as RunRecord } from './run-record-BDH49H2E.js';
|
|
17
17
|
import { AxFunction } from '@ax-llm/ax';
|
|
18
18
|
import '@tangle-network/agent-interface';
|
|
19
19
|
|
|
20
20
|
/** Canonical OpenInference-over-OTLP attribute vocabulary used at the trace boundary. */
|
|
21
|
+
|
|
21
22
|
declare const OPENINFERENCE_SPAN_KIND = "openinference.span.kind";
|
|
22
23
|
declare const LLM_MODEL_NAME = "llm.model_name";
|
|
23
24
|
declare const LLM_INPUT_TOKENS = "llm.token_count.prompt";
|
|
@@ -25,6 +26,10 @@ declare const LLM_OUTPUT_TOKENS = "llm.token_count.completion";
|
|
|
25
26
|
declare const LLM_CACHED_TOKENS = "llm.token_count.prompt_cache_hit";
|
|
26
27
|
declare const LLM_COST_USD = "llm.cost_usd";
|
|
27
28
|
declare const TOOL_NAME = "tool.name";
|
|
29
|
+
declare const TOOL_ARGS_CAPTURED = "tool.args_captured";
|
|
30
|
+
declare const TOOL_LATENCY_MS = "tool.latency_ms";
|
|
31
|
+
declare const INPUT_VALUE = "input.value";
|
|
32
|
+
declare const OUTPUT_VALUE = "output.value";
|
|
28
33
|
declare const SPAN_KIND_ATTR_KEYS: readonly ["openinference.span.kind", "inference.observation_kind"];
|
|
29
34
|
declare const LLM_MODEL_ATTR_KEYS: readonly ["llm.model_name", "inference.llm.model_name", "llm.model", "gen_ai.request.model", "gen_ai.response.model"];
|
|
30
35
|
declare const LLM_INPUT_TOKEN_ATTR_KEYS: readonly ["llm.token_count.prompt", "inference.llm.input_tokens", "llm.input_tokens", "gen_ai.usage.input_tokens", "gen_ai.usage.prompt_tokens"];
|
|
@@ -32,6 +37,8 @@ declare const LLM_OUTPUT_TOKEN_ATTR_KEYS: readonly ["llm.token_count.completion"
|
|
|
32
37
|
declare const LLM_CACHED_TOKEN_ATTR_KEYS: readonly ["llm.token_count.prompt_cache_hit", "inference.llm.cached_tokens", "llm.cached_tokens", "gen_ai.usage.cached_tokens"];
|
|
33
38
|
declare const LLM_COST_ATTR_KEYS: readonly ["llm.cost_usd", "inference.llm.cost.total", "llm.cost.total", "gen_ai.usage.cost"];
|
|
34
39
|
declare const TOOL_NAME_ATTR_KEYS: readonly ["tool.name", "inference.tool.name"];
|
|
40
|
+
type ToolSpanOtlpInput = Pick<ToolSpan, 'toolName' | 'args' | 'argsCaptured' | 'result' | 'latencyMs'>;
|
|
41
|
+
declare function applyToolSpanOtlpAttributes(attributes: Record<string, unknown>, span: ToolSpanOtlpInput): void;
|
|
35
42
|
declare function traceSpanKindToOpenInferenceKind(kind: string): string;
|
|
36
43
|
|
|
37
44
|
/**
|
|
@@ -710,6 +717,7 @@ declare function captureFetchToRawSink(fetch: typeof globalThis.fetch, sink: Raw
|
|
|
710
717
|
* or when the batch fills. No @opentelemetry SDK dependency — minimal
|
|
711
718
|
* OTLP/JSON serializer (~120 LOC) using the existing otel.ts helpers.
|
|
712
719
|
*/
|
|
720
|
+
|
|
713
721
|
interface OtelExportConfig {
|
|
714
722
|
/** OTLP endpoint. Reads OTEL_EXPORTER_OTLP_ENDPOINT env by default. */
|
|
715
723
|
endpoint?: string;
|
|
@@ -746,6 +754,7 @@ interface ExportableSpan {
|
|
|
746
754
|
inputTokens?: number;
|
|
747
755
|
outputTokens?: number;
|
|
748
756
|
costUsd?: number;
|
|
757
|
+
tool?: ToolSpanOtlpInput;
|
|
749
758
|
attributes?: Record<string, unknown>;
|
|
750
759
|
}
|
|
751
760
|
/**
|
|
@@ -1016,4 +1025,4 @@ declare function iterateRawCalls(sink: RawProviderSink, filter?: {
|
|
|
1016
1025
|
spanId?: string;
|
|
1017
1026
|
}): AsyncGenerator<ReplayCacheEntry>;
|
|
1018
1027
|
|
|
1019
|
-
export { AnalyzeTracesOptions, AnalyzeTracesResult, type CaptureFetchContext, type CaptureFetchOptions, DEFAULT_REDACTION_RULES, DatasetOverview, type ExportableSpan, type ExtractedUsage, type FlattenOtlpOptions, LLM_CACHED_TOKENS, LLM_CACHED_TOKEN_ATTR_KEYS, LLM_COST_ATTR_KEYS, LLM_COST_USD, LLM_INPUT_TOKENS, LLM_INPUT_TOKEN_ATTR_KEYS, LLM_MODEL_ATTR_KEYS, LLM_MODEL_NAME, LLM_OUTPUT_TOKENS, LLM_OUTPUT_TOKEN_ATTR_KEYS, OPENINFERENCE_SPAN_KIND, OTEL_AGENT_EVAL_SCOPE, type OtelExportConfig, type OtelExporter, type OtlpExport, OtlpFileTraceStore, type OtlpFileTraceStoreOptions, type OtlpFlatLine, type OtlpResourceSpans, type OtlpSpan, type OtlpToRunRecordsOptions, type OtlpTraceRunRecord, type ProjectedOtlpSpan, ProviderRedactor, QueryTracesPage, REDACTION_VERSION, RawProviderEvent, RawProviderSink, type RedactionReport, type RedactionRule, ReplayCache, type ReplayCacheEntry, ReplayCacheMissError, type ReplayCacheStats, type ReplayFetchOptions, Run, RunCompleteHook, RunCompleteHookContext, SPAN_KIND_ATTR_KEYS, SearchSpanResult, SearchTraceResult, SpanNotFoundError, TOOL_NAME, TOOL_NAME_ATTR_KEYS, TRACE_ANALYST_ACTOR_DESCRIPTION, TRACE_ANALYST_ACTOR_DESCRIPTION_VERSION, TRACE_ANALYST_SUBAGENT_DESCRIPTION, type TraceAggregate, TraceAnalysisStore, TraceAnalystFilters, type TraceAnalystHookOptions, TraceAnalystSpanKind, TraceAnalystSpanStatus, TraceFileMissingError, type TraceInsightContext, type TraceInsightFinding, type TraceInsightPanelRole, type TraceInsightPromptInput, type TraceInsightQualityGate, type TraceInsightQuestion, type TraceInsightReadiness, type TraceInsightSuite, type TraceInsightTask, TraceNotFoundError, TraceStore, type TraceStoreSource, type TraceStoreToOtlpOptions, type TracesToOtlpResult, ViewSpansResult, ViewTraceResult, asNumber, asString, buildTraceAnalystTools, buildTraceInsightContext, buildTraceInsightPrompt, captureFetchToRawSink, convertTraceStoresToOtlp, createOtelExporter, createOtelTracingStore, createReplayFetch, defaultTraceInsightPanel, describeTraceInsightScope, domainEvidencePattern, exportRunAsOtlp, extractOtlpAttributes, extractUsage, extractUsageFromResponse, extractUsageFromSse, firstNumberAttr, firstStringAttr, flattenOtlpExportToNdjson, inferDomainKeywords, inferOtlpKind, iterateRawCalls, otelRunCompleteHook, otlpToRunRecords, otlpToTraceRunRecords, planTraceInsightQuestions, projectOtlpFlatLine, readOtlpStatus, redactString, redactValue, scoreTraceInsightReadiness, stringField, tokenizeDomainWords, traceAnalystFunctionGroup, traceAnalystOnRunComplete, traceSpanKindToOpenInferenceKind };
|
|
1028
|
+
export { AnalyzeTracesOptions, AnalyzeTracesResult, type CaptureFetchContext, type CaptureFetchOptions, DEFAULT_REDACTION_RULES, DatasetOverview, type ExportableSpan, type ExtractedUsage, type FlattenOtlpOptions, INPUT_VALUE, LLM_CACHED_TOKENS, LLM_CACHED_TOKEN_ATTR_KEYS, LLM_COST_ATTR_KEYS, LLM_COST_USD, LLM_INPUT_TOKENS, LLM_INPUT_TOKEN_ATTR_KEYS, LLM_MODEL_ATTR_KEYS, LLM_MODEL_NAME, LLM_OUTPUT_TOKENS, LLM_OUTPUT_TOKEN_ATTR_KEYS, OPENINFERENCE_SPAN_KIND, OTEL_AGENT_EVAL_SCOPE, OUTPUT_VALUE, type OtelExportConfig, type OtelExporter, type OtlpExport, OtlpFileTraceStore, type OtlpFileTraceStoreOptions, type OtlpFlatLine, type OtlpResourceSpans, type OtlpSpan, type OtlpToRunRecordsOptions, type OtlpTraceRunRecord, type ProjectedOtlpSpan, ProviderRedactor, QueryTracesPage, REDACTION_VERSION, RawProviderEvent, RawProviderSink, type RedactionReport, type RedactionRule, ReplayCache, type ReplayCacheEntry, ReplayCacheMissError, type ReplayCacheStats, type ReplayFetchOptions, Run, RunCompleteHook, RunCompleteHookContext, SPAN_KIND_ATTR_KEYS, SearchSpanResult, SearchTraceResult, SpanNotFoundError, TOOL_ARGS_CAPTURED, TOOL_LATENCY_MS, TOOL_NAME, TOOL_NAME_ATTR_KEYS, TRACE_ANALYST_ACTOR_DESCRIPTION, TRACE_ANALYST_ACTOR_DESCRIPTION_VERSION, TRACE_ANALYST_SUBAGENT_DESCRIPTION, ToolSpan, type ToolSpanOtlpInput, type TraceAggregate, TraceAnalysisStore, TraceAnalystFilters, type TraceAnalystHookOptions, TraceAnalystSpanKind, TraceAnalystSpanStatus, TraceFileMissingError, type TraceInsightContext, type TraceInsightFinding, type TraceInsightPanelRole, type TraceInsightPromptInput, type TraceInsightQualityGate, type TraceInsightQuestion, type TraceInsightReadiness, type TraceInsightSuite, type TraceInsightTask, TraceNotFoundError, TraceStore, type TraceStoreSource, type TraceStoreToOtlpOptions, type TracesToOtlpResult, ViewSpansResult, ViewTraceResult, applyToolSpanOtlpAttributes, asNumber, asString, buildTraceAnalystTools, buildTraceInsightContext, buildTraceInsightPrompt, captureFetchToRawSink, convertTraceStoresToOtlp, createOtelExporter, createOtelTracingStore, createReplayFetch, defaultTraceInsightPanel, describeTraceInsightScope, domainEvidencePattern, exportRunAsOtlp, extractOtlpAttributes, extractUsage, extractUsageFromResponse, extractUsageFromSse, firstNumberAttr, firstStringAttr, flattenOtlpExportToNdjson, inferDomainKeywords, inferOtlpKind, iterateRawCalls, otelRunCompleteHook, otlpToRunRecords, otlpToTraceRunRecords, planTraceInsightQuestions, projectOtlpFlatLine, readOtlpStatus, redactString, redactValue, scoreTraceInsightReadiness, stringField, tokenizeDomainWords, traceAnalystFunctionGroup, traceAnalystOnRunComplete, traceSpanKindToOpenInferenceKind };
|