@tangle-network/agent-eval 0.116.0 → 0.117.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +38 -0
- package/dist/analyst/index.d.ts +18 -11
- package/dist/analyst/index.js +10 -7
- package/dist/analyst/index.js.map +1 -1
- package/dist/{analyst-CFBc14Wc.d.ts → analyst-C8HHvfJp.d.ts} +1 -1
- package/dist/{analyze-runs-0rz_m29H.d.ts → analyze-runs--2x39HZ7.d.ts} +3 -3
- package/dist/{baseline-DsNteOgR.d.ts → baseline-DKq3gJpP.d.ts} +6 -3
- package/dist/belief-state/index.d.ts +6 -6
- package/dist/belief-state/index.js +1 -1
- package/dist/benchmarks/index.d.ts +11 -8
- package/dist/benchmarks/index.js +11 -10
- package/dist/builder-eval/index.d.ts +4 -4
- package/dist/builder-eval/index.js +1 -1
- package/dist/{calibration-Dz8TQV4y.d.ts → calibration-C8MTS7cw.d.ts} +2 -2
- package/dist/campaign/index.d.ts +54 -30
- package/dist/campaign/index.js +18 -13
- package/dist/chunk-3YYRZDON.js +45 -0
- package/dist/chunk-3YYRZDON.js.map +1 -0
- package/dist/{chunk-RPDDVKI7.js → chunk-4JLWXDYA.js} +2 -2
- package/dist/{chunk-NBSS5NDZ.js → chunk-CCZIVI3F.js} +54 -115
- package/dist/chunk-CCZIVI3F.js.map +1 -0
- package/dist/{chunk-J6P6PK2R.js → chunk-FQNLDL4D.js} +3 -3
- package/dist/{chunk-ONM6PEAE.js → chunk-GQCZRZ7L.js} +2 -2
- package/dist/chunk-HHWE3POT.js +94 -0
- package/dist/chunk-HHWE3POT.js.map +1 -0
- package/dist/{chunk-3274WNK7.js → chunk-HQPHZGL6.js} +687 -44
- package/dist/chunk-HQPHZGL6.js.map +1 -0
- package/dist/{chunk-FAOEFFRT.js → chunk-IDZTTFRR.js} +390 -78
- package/dist/chunk-IDZTTFRR.js.map +1 -0
- package/dist/{chunk-3LXTCTWL.js → chunk-JSDVRFAP.js} +2 -2
- package/dist/{chunk-GSW3OBHK.js → chunk-JSJZ4PJ6.js} +406 -726
- package/dist/chunk-JSJZ4PJ6.js.map +1 -0
- package/dist/{chunk-MHNQWM4I.js → chunk-LQUTGLOZ.js} +5 -1
- package/dist/chunk-LQUTGLOZ.js.map +1 -0
- package/dist/{chunk-4D5RVB3W.js → chunk-LTVG32KX.js} +30 -5
- package/dist/chunk-LTVG32KX.js.map +1 -0
- package/dist/{chunk-CIUOICJT.js → chunk-MGEHEHSN.js} +62 -15
- package/dist/chunk-MGEHEHSN.js.map +1 -0
- package/dist/{chunk-GY4SYVPJ.js → chunk-NJC7U437.js} +97 -25
- package/dist/chunk-NJC7U437.js.map +1 -0
- package/dist/{chunk-NYFUT3B3.js → chunk-ODVOOEWQ.js} +31 -10
- package/dist/chunk-ODVOOEWQ.js.map +1 -0
- package/dist/{chunk-LNQEP766.js → chunk-S2F4J57L.js} +44 -4
- package/dist/chunk-S2F4J57L.js.map +1 -0
- package/dist/chunk-VCTY3W6J.js +798 -0
- package/dist/chunk-VCTY3W6J.js.map +1 -0
- package/dist/chunk-VF3XSYTI.js +545 -0
- package/dist/chunk-VF3XSYTI.js.map +1 -0
- package/dist/{chunk-TLDB7WRY.js → chunk-YZPO4UHR.js} +28 -31
- package/dist/chunk-YZPO4UHR.js.map +1 -0
- package/dist/cli.js +4 -2
- package/dist/cli.js.map +1 -1
- package/dist/{code-agent-session-CdxteG0y.d.ts → code-agent-session-CjZsVd19.d.ts} +1 -1
- package/dist/contract/index.d.ts +43 -29
- package/dist/contract/index.js +56 -19
- package/dist/contract/index.js.map +1 -1
- package/dist/{control-DbcDxouY.d.ts → control-6vuGfmDH.d.ts} +5 -5
- package/dist/control.d.ts +6 -6
- package/dist/cost-ledger-DWy3XdJc.d.ts +183 -0
- package/dist/{default-registry-DDfv22MQ.d.ts → default-registry-DaK8b3fv.d.ts} +2 -2
- package/dist/{emitter-BRchAAAx.d.ts → emitter-CjD7vUwv.d.ts} +2 -2
- package/dist/{failure-cluster-C48PiReX.d.ts → failure-cluster-DOAcSJ87.d.ts} +2 -2
- package/dist/{feedback-trajectory-pDcz1lQ1.d.ts → feedback-trajectory-BUnM58xL.d.ts} +3 -3
- package/dist/fuzz.d.ts +8 -16
- package/dist/fuzz.js +72 -42
- package/dist/fuzz.js.map +1 -1
- package/dist/{gepa-CQelRtuC.d.ts → gepa-eESocoDi.d.ts} +56 -6
- package/dist/hosted/index.d.ts +13 -10
- package/dist/{index-DbCXJfZ1.d.ts → index-PdX4VnPA.d.ts} +3 -3
- package/dist/index.d.ts +102 -57
- package/dist/index.js +328 -235
- package/dist/index.js.map +1 -1
- package/dist/{insight-report-oMVxDTxl.d.ts → insight-report-DY4nDW9Q.d.ts} +1 -1
- package/dist/{integrity-C6PZ73iC.d.ts → integrity-DqlBiLyK.d.ts} +2 -2
- package/dist/{kind-factory-DWOvXjR_.d.ts → kind-factory-ClZmO25A.d.ts} +2 -2
- package/dist/llm-client-qoDd18Qz.d.ts +289 -0
- package/dist/meta-eval/index.d.ts +8 -7
- package/dist/meta-eval/index.js +1 -1
- package/dist/multishot/index.d.ts +9 -6
- package/dist/openapi.json +1 -1
- package/dist/pipelines/index.d.ts +16 -6
- package/dist/pipelines/index.js +119 -23
- package/dist/pipelines/index.js.map +1 -1
- package/dist/{policy-edit-Clb2v6Oa.d.ts → policy-edit-wG9uFEFm.d.ts} +13 -266
- package/dist/{pre-registration--vU0mMtD.d.ts → pre-registration-BWQhJ3vz.d.ts} +24 -5
- package/dist/{provenance-BbVagC68.d.ts → provenance-DpjwyseI.d.ts} +6 -6
- package/dist/{query-Ck190MOd.d.ts → query-CF7PG61p.d.ts} +5 -3
- package/dist/raw-provider-sink-C46HDghv.d.ts +132 -0
- package/dist/{release-report-CamNDe90.d.ts → release-report-C8G2i5Xi.d.ts} +2 -2
- package/dist/reporting.d.ts +10 -9
- package/dist/{researcher-Dwbo_Fxx.d.ts → researcher-C8XyxQsu.d.ts} +8 -8
- package/dist/rl.d.ts +18 -15
- package/dist/rl.js +2 -2
- package/dist/{rubric-predictive-validity-BIdf9h4R.d.ts → rubric-predictive-validity-p49lLVrE.d.ts} +1 -1
- package/dist/{run-campaign-UADIM77S.js → run-campaign-IM26A6PD.js} +4 -2
- package/dist/{run-record-CZmcpWPo.d.ts → run-record-BDH49H2E.d.ts} +1 -1
- package/dist/{runtime-trajectory-CC0jx9ql.d.ts → runtime-trajectory-DGBIUt4B.d.ts} +1 -1
- package/dist/{schema-SGWcK9wa.d.ts → schema-B3Q3l9Z_.d.ts} +2 -0
- package/dist/{semantic-concept-judge-CKjePUMh.d.ts → semantic-concept-judge-CXnPEJbf.d.ts} +24 -6
- package/dist/{statistics-oUbOJe-S.d.ts → statistics-KUnG73jH.d.ts} +1 -1
- package/dist/{storage-Dw_f7WMt.d.ts → storage-DrX3v_5B.d.ts} +12 -1
- package/dist/{store-9cAScOcb.d.ts → store-C1YxJDEK.d.ts} +1 -132
- package/dist/{store-BsVi7ncX.d.ts → store-DGqD0Pyo.d.ts} +1 -1
- package/dist/storyboard/index.d.ts +1 -1
- package/dist/{summary-report-DTNgQycC.d.ts → summary-report-C5bKFfm-.d.ts} +2 -2
- package/dist/{test-graded-scenario-mzYBKspu.d.ts → test-graded-scenario-B0ybnPY7.d.ts} +3 -3
- package/dist/traces.d.ts +25 -14
- package/dist/traces.js +16 -4
- package/dist/{types-Ca_63YSD.d.ts → types-BSw1rOUB.d.ts} +41 -39
- package/dist/{types-C7DGg5ex.d.ts → types-BkfcQnxV.d.ts} +15 -0
- package/dist/wire/index.d.ts +28 -19
- package/dist/wire/index.js +4 -2
- package/docs/distributed-driver.md +1 -1
- package/package.json +3 -3
- package/dist/chunk-3274WNK7.js.map +0 -1
- package/dist/chunk-4D5RVB3W.js.map +0 -1
- package/dist/chunk-7GKEAIAD.js +0 -205
- package/dist/chunk-7GKEAIAD.js.map +0 -1
- package/dist/chunk-CIUOICJT.js.map +0 -1
- package/dist/chunk-FAOEFFRT.js.map +0 -1
- package/dist/chunk-GSW3OBHK.js.map +0 -1
- package/dist/chunk-GY4SYVPJ.js.map +0 -1
- package/dist/chunk-LNQEP766.js.map +0 -1
- package/dist/chunk-MHNQWM4I.js.map +0 -1
- package/dist/chunk-MPHTT5HE.js +0 -74
- package/dist/chunk-MPHTT5HE.js.map +0 -1
- package/dist/chunk-NBSS5NDZ.js.map +0 -1
- package/dist/chunk-NYFUT3B3.js.map +0 -1
- package/dist/chunk-TLDB7WRY.js.map +0 -1
- package/dist/cost-ledger-DuSqlw5B.d.ts +0 -113
- /package/dist/{chunk-RPDDVKI7.js.map → chunk-4JLWXDYA.js.map} +0 -0
- /package/dist/{chunk-J6P6PK2R.js.map → chunk-FQNLDL4D.js.map} +0 -0
- /package/dist/{chunk-ONM6PEAE.js.map → chunk-GQCZRZ7L.js.map} +0 -0
- /package/dist/{chunk-3LXTCTWL.js.map → chunk-JSDVRFAP.js.map} +0 -0
- /package/dist/{run-campaign-UADIM77S.js.map → run-campaign-IM26A6PD.js.map} +0 -0
|
@@ -0,0 +1,132 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* RawProviderSink — first-class persistence for the actual HTTP-level
|
|
3
|
+
* request/response bodies of every LLM provider call.
|
|
4
|
+
*
|
|
5
|
+
* Why this is a separate sink from the structured `LlmSpan`:
|
|
6
|
+
*
|
|
7
|
+
* - `LlmSpan` records the *intent* — model name, messages, output text,
|
|
8
|
+
* usage. It's what dashboards read; it's NOT enough for forensics.
|
|
9
|
+
* - When a downstream consumer reports "the verifier used the wrong route"
|
|
10
|
+
* or "tokens look right but reasoning was missing," the only way to
|
|
11
|
+
* answer is the raw HTTP body. Span fields can lie (a proxy can echo
|
|
12
|
+
* a different `model` value than what actually answered); the raw
|
|
13
|
+
* response is ground truth.
|
|
14
|
+
*
|
|
15
|
+
* Default behaviour: opt-in. Pass `rawSink` to `LlmClientOptions` (or the
|
|
16
|
+
* matrix runner / BuilderSession sets it up automatically) and every
|
|
17
|
+
* request, response, and error is recorded — including retries, with the
|
|
18
|
+
* attempt index attached so a flaky call's full event chain is recoverable.
|
|
19
|
+
*
|
|
20
|
+
* Redaction is enforced at sink time. The default redactor strips
|
|
21
|
+
* `Authorization`, `X-Api-Key`, `X-Auth-Token`, `Cookie` headers and any
|
|
22
|
+
* payload field whose key matches `apiKey | api_key | bearer | password |
|
|
23
|
+
* secret | token` (case-insensitive). Override via the sink constructor or
|
|
24
|
+
* the per-call `redactor`. The `redactedFields` array on the persisted
|
|
25
|
+
* event lets a reviewer see what was stripped without exposing the values.
|
|
26
|
+
*/
|
|
27
|
+
type RawProviderDirection = 'request' | 'response' | 'error';
|
|
28
|
+
interface RawProviderEvent {
|
|
29
|
+
/** Stable id. Generated by the sink if omitted. */
|
|
30
|
+
eventId: string;
|
|
31
|
+
/** Trace context populated by `LlmClient` when the call is wrapped in a span. */
|
|
32
|
+
runId?: string;
|
|
33
|
+
spanId?: string;
|
|
34
|
+
/**
|
|
35
|
+
* Logical provider name. Free-form so callers can use whatever id matches
|
|
36
|
+
* their topology (`'openai'`, `'anthropic'`, `'tangle-router'`, …). When
|
|
37
|
+
* omitted, derived from `baseUrl` in `LlmClientOptions`.
|
|
38
|
+
*/
|
|
39
|
+
provider: string;
|
|
40
|
+
model: string;
|
|
41
|
+
/** Endpoint path, e.g. `'/v1/chat/completions'`. */
|
|
42
|
+
endpoint: string;
|
|
43
|
+
/** Base URL used for the call (already-normalised — no trailing slash). */
|
|
44
|
+
baseUrl: string;
|
|
45
|
+
/** 0-indexed retry attempt. The first attempt is 0; a retried call gets 1, 2, … */
|
|
46
|
+
attemptIndex: number;
|
|
47
|
+
direction: RawProviderDirection;
|
|
48
|
+
/** Unix ms. */
|
|
49
|
+
timestamp: number;
|
|
50
|
+
/** Wall-clock duration of the call leg. Set on `response` and `error` events; null on `request`. */
|
|
51
|
+
durationMs?: number;
|
|
52
|
+
statusCode?: number;
|
|
53
|
+
requestHeaders?: Record<string, string>;
|
|
54
|
+
requestBody?: unknown;
|
|
55
|
+
responseHeaders?: Record<string, string>;
|
|
56
|
+
responseBody?: unknown;
|
|
57
|
+
/** Set on `direction: 'error'` events. */
|
|
58
|
+
errorMessage?: string;
|
|
59
|
+
/** Field paths the redactor stripped from this event ('header:Authorization', 'body.apiKey', …). */
|
|
60
|
+
redactedFields: string[];
|
|
61
|
+
}
|
|
62
|
+
interface RawProviderSinkFilter {
|
|
63
|
+
runId?: string;
|
|
64
|
+
spanId?: string;
|
|
65
|
+
direction?: RawProviderDirection;
|
|
66
|
+
attemptIndex?: number;
|
|
67
|
+
}
|
|
68
|
+
interface RawProviderSink {
|
|
69
|
+
record(event: RawProviderEvent): Promise<void>;
|
|
70
|
+
/** Optional listing — implementations that durably persist (file, db) should support this. */
|
|
71
|
+
list?(filter?: RawProviderSinkFilter): Promise<RawProviderEvent[]>;
|
|
72
|
+
/** Optional teardown for backed implementations. */
|
|
73
|
+
close?(): Promise<void>;
|
|
74
|
+
}
|
|
75
|
+
type ProviderRedactor = (event: RawProviderEvent) => RawProviderEvent;
|
|
76
|
+
/**
|
|
77
|
+
* Default redactor — strips well-known auth headers and any body field whose
|
|
78
|
+
* key matches the credential pattern. Records every redacted path on
|
|
79
|
+
* `event.redactedFields` so a downstream reviewer can see what was removed.
|
|
80
|
+
*/
|
|
81
|
+
declare function defaultProviderRedactor(event: RawProviderEvent): RawProviderEvent;
|
|
82
|
+
interface InMemoryRawProviderSinkOptions {
|
|
83
|
+
redactor?: ProviderRedactor;
|
|
84
|
+
}
|
|
85
|
+
declare class InMemoryRawProviderSink implements RawProviderSink {
|
|
86
|
+
private events;
|
|
87
|
+
private redactor;
|
|
88
|
+
constructor(opts?: InMemoryRawProviderSinkOptions);
|
|
89
|
+
record(event: RawProviderEvent): Promise<void>;
|
|
90
|
+
list(filter?: RawProviderSinkFilter): Promise<RawProviderEvent[]>;
|
|
91
|
+
size(): number;
|
|
92
|
+
}
|
|
93
|
+
declare class NoopRawProviderSink implements RawProviderSink {
|
|
94
|
+
record(): Promise<void>;
|
|
95
|
+
/**
|
|
96
|
+
* Returns an empty array. Implemented so `assertRunCaptured` does not
|
|
97
|
+
* trip the `no_raw_sink` issue when a caller explicitly opts out of
|
|
98
|
+
* capture by passing this sink — opt-out is a deliberate choice, not a
|
|
99
|
+
* misconfiguration.
|
|
100
|
+
*/
|
|
101
|
+
list(): Promise<RawProviderEvent[]>;
|
|
102
|
+
}
|
|
103
|
+
interface FileSystemRawProviderSinkOptions {
|
|
104
|
+
/** Directory the NDJSON file is written into. Created if missing. */
|
|
105
|
+
dir: string;
|
|
106
|
+
/** File name; default `'raw-provider-events.ndjson'`. */
|
|
107
|
+
fileName?: string;
|
|
108
|
+
/** Bytes after which the writer rolls over to a new file (default 32 MiB). */
|
|
109
|
+
rollAtBytes?: number;
|
|
110
|
+
redactor?: ProviderRedactor;
|
|
111
|
+
}
|
|
112
|
+
declare class FileSystemRawProviderSink implements RawProviderSink {
|
|
113
|
+
private dir;
|
|
114
|
+
private fileName;
|
|
115
|
+
private rollAtBytes;
|
|
116
|
+
private redactor;
|
|
117
|
+
private bytesWritten;
|
|
118
|
+
private rollIndex;
|
|
119
|
+
private initPromise;
|
|
120
|
+
constructor(opts: FileSystemRawProviderSinkOptions);
|
|
121
|
+
private ensureInit;
|
|
122
|
+
private currentPath;
|
|
123
|
+
record(event: RawProviderEvent): Promise<void>;
|
|
124
|
+
list(filter?: RawProviderSinkFilter): Promise<RawProviderEvent[]>;
|
|
125
|
+
}
|
|
126
|
+
/**
|
|
127
|
+
* Best-effort provider id from a base URL. Falls back to the URL host when
|
|
128
|
+
* none of the well-known patterns match.
|
|
129
|
+
*/
|
|
130
|
+
declare function providerFromBaseUrl(baseUrl: string): string;
|
|
131
|
+
|
|
132
|
+
export { FileSystemRawProviderSink as F, InMemoryRawProviderSink as I, NoopRawProviderSink as N, type ProviderRedactor as P, type RawProviderSink as R, type FileSystemRawProviderSinkOptions as a, type InMemoryRawProviderSinkOptions as b, type RawProviderDirection as c, type RawProviderEvent as d, type RawProviderSinkFilter as e, defaultProviderRedactor as f, providerFromBaseUrl as p };
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
import { a as DatasetSplit, b as DatasetManifest, D as DatasetScenario } from './dataset-NENEzRgk.js';
|
|
2
|
-
import { m as GateDecision } from './summary-report-
|
|
3
|
-
import { R as RunRecord, a as RunSplitTag } from './run-record-
|
|
2
|
+
import { m as GateDecision } from './summary-report-C5bKFfm-.js';
|
|
3
|
+
import { R as RunRecord, a as RunSplitTag } from './run-record-BDH49H2E.js';
|
|
4
4
|
|
|
5
5
|
/**
|
|
6
6
|
* Release confidence gate.
|
package/dist/reporting.d.ts
CHANGED
|
@@ -1,16 +1,17 @@
|
|
|
1
|
-
export { R as RubricOutcomePair, a as RubricPredictiveValidityInput, b as RubricPredictiveValidityReport, c as RubricRanking, r as rubricPredictiveValidity } from './rubric-predictive-validity-
|
|
2
|
-
export { B as BootstrapOptions, a as BootstrapResult, J as JudgeReplayGateArgs, R as ReleaseConfidenceAxis, b as ReleaseConfidenceAxisName, c as ReleaseConfidenceInput, d as ReleaseConfidenceIssue, e as ReleaseConfidenceMetrics, f as ReleaseConfidenceScorecard, g as ReleaseConfidenceStatus, h as ReleaseConfidenceThresholds, i as ReleaseTraceEvidence, j as RenderReleaseReportOptions, V as Verdict, k as assertReleaseConfidence, l as bootstrapCi, m as evaluateReleaseConfidence, n as judgeReplayGate, r as renderReleaseReport } from './release-report-
|
|
1
|
+
export { R as RubricOutcomePair, a as RubricPredictiveValidityInput, b as RubricPredictiveValidityReport, c as RubricRanking, r as rubricPredictiveValidity } from './rubric-predictive-validity-p49lLVrE.js';
|
|
2
|
+
export { B as BootstrapOptions, a as BootstrapResult, J as JudgeReplayGateArgs, R as ReleaseConfidenceAxis, b as ReleaseConfidenceAxisName, c as ReleaseConfidenceInput, d as ReleaseConfidenceIssue, e as ReleaseConfidenceMetrics, f as ReleaseConfidenceScorecard, g as ReleaseConfidenceStatus, h as ReleaseConfidenceThresholds, i as ReleaseTraceEvidence, j as RenderReleaseReportOptions, V as Verdict, k as assertReleaseConfidence, l as bootstrapCi, m as evaluateReleaseConfidence, n as judgeReplayGate, r as renderReleaseReport } from './release-report-C8G2i5Xi.js';
|
|
3
3
|
export { I as InterimReleaseConfidence, a as InterimReleaseConfidenceInput, P as PairedEvalueOptions, b as PairedEvalueSequence, c as PairedEvalueStep, S as SequentialDecision, e as evaluateInterimReleaseConfidence, p as pairedEvalueSequence } from './sequential-5iSVfzl2.js';
|
|
4
|
-
export { P as PairedBootstrapOptions, a as PairedBootstrapResult, b as benjaminiHochberg, p as pairedBootstrap, w as wilcoxonSignedRank } from './statistics-
|
|
5
|
-
export { G as GainDistributionBin, a as GainDistributionFigureSpec, b as GainDistributionOptions, P as ParetoFigureSpec, c as ParetoPoint, R as RESEARCH_REPORT_HARD_PAIR_FLOOR, d as ResearchReport, e as ResearchReportCandidate, f as ResearchReportDecision, g as ResearchReportMethodology, h as ResearchReportOptions, i as ResearchReportRecommendation, S as SummaryTable, j as SummaryTableOptions, k as SummaryTableRow, l as gainHistogram, p as paretoChart, r as researchReport, s as summaryTable } from './summary-report-
|
|
6
|
-
import './run-record-
|
|
4
|
+
export { P as PairedBootstrapOptions, a as PairedBootstrapResult, b as benjaminiHochberg, p as pairedBootstrap, w as wilcoxonSignedRank } from './statistics-KUnG73jH.js';
|
|
5
|
+
export { G as GainDistributionBin, a as GainDistributionFigureSpec, b as GainDistributionOptions, P as ParetoFigureSpec, c as ParetoPoint, R as RESEARCH_REPORT_HARD_PAIR_FLOOR, d as ResearchReport, e as ResearchReportCandidate, f as ResearchReportDecision, g as ResearchReportMethodology, h as ResearchReportOptions, i as ResearchReportRecommendation, S as SummaryTable, j as SummaryTableOptions, k as SummaryTableRow, l as gainHistogram, p as paretoChart, r as researchReport, s as summaryTable } from './summary-report-C5bKFfm-.js';
|
|
6
|
+
import './run-record-BDH49H2E.js';
|
|
7
7
|
import '@tangle-network/agent-interface';
|
|
8
8
|
import './errors-oeQrLqXC.js';
|
|
9
|
-
import './schema-
|
|
9
|
+
import './schema-B3Q3l9Z_.js';
|
|
10
10
|
import './outcome-store-rnXLEqSn.js';
|
|
11
11
|
import './dataset-NENEzRgk.js';
|
|
12
12
|
import './judge-calibration-7C-IDmKr.js';
|
|
13
|
-
import './types-
|
|
13
|
+
import './types-BkfcQnxV.js';
|
|
14
|
+
import './cost-ledger-DWy3XdJc.js';
|
|
14
15
|
import '@tangle-network/tcloud';
|
|
15
|
-
import './failure-cluster-
|
|
16
|
-
import './store-
|
|
16
|
+
import './failure-cluster-DOAcSJ87.js';
|
|
17
|
+
import './store-DGqD0Pyo.js';
|
|
@@ -1,11 +1,11 @@
|
|
|
1
|
-
import { a as RunSplitTag, c as RunTokenUsage, d as RunJudgeMetadata, J as JudgeScoresRecord, A as AgentProfileCell, e as AgentProfileCellInput, R as RunRecord } from './run-record-
|
|
2
|
-
import {
|
|
3
|
-
import { h as ResearchReportOptions, d as ResearchReport, m as GateDecision } from './summary-report-
|
|
4
|
-
import { T as TraceEmitter, R as RunCompleteHook } from './emitter-
|
|
5
|
-
import { R as RunIntegrityExpectations, a as RunIntegrityReport } from './integrity-
|
|
6
|
-
import { R as RawProviderSink } from './
|
|
7
|
-
import { F as FailureClass } from './schema-
|
|
8
|
-
import { T as TraceStore } from './store-
|
|
1
|
+
import { a as RunSplitTag, c as RunTokenUsage, d as RunJudgeMetadata, J as JudgeScoresRecord, A as AgentProfileCell, e as AgentProfileCellInput, R as RunRecord } from './run-record-BDH49H2E.js';
|
|
2
|
+
import { a as LlmClientOptions, b as LlmRouteRequirements } from './llm-client-qoDd18Qz.js';
|
|
3
|
+
import { h as ResearchReportOptions, d as ResearchReport, m as GateDecision } from './summary-report-C5bKFfm-.js';
|
|
4
|
+
import { T as TraceEmitter, R as RunCompleteHook } from './emitter-CjD7vUwv.js';
|
|
5
|
+
import { R as RunIntegrityExpectations, a as RunIntegrityReport } from './integrity-DqlBiLyK.js';
|
|
6
|
+
import { R as RawProviderSink } from './raw-provider-sink-C46HDghv.js';
|
|
7
|
+
import { F as FailureClass } from './schema-B3Q3l9Z_.js';
|
|
8
|
+
import { T as TraceStore } from './store-DGqD0Pyo.js';
|
|
9
9
|
|
|
10
10
|
/**
|
|
11
11
|
* EvalCampaign — opinionated matrix runner that wires the four
|
package/dist/rl.d.ts
CHANGED
|
@@ -1,27 +1,30 @@
|
|
|
1
|
-
import { R as RunRecord, a as RunSplitTag } from './run-record-
|
|
1
|
+
import { R as RunRecord, a as RunSplitTag } from './run-record-BDH49H2E.js';
|
|
2
2
|
export { A as AdversarialMutation } from './adversarial-B7loGVVX.js';
|
|
3
|
-
import { S as Span } from './schema-
|
|
4
|
-
import { T as TraceStore } from './store-
|
|
3
|
+
import { S as Span } from './schema-B3Q3l9Z_.js';
|
|
4
|
+
import { T as TraceStore } from './store-DGqD0Pyo.js';
|
|
5
5
|
export { O as OffPolicyEstimate, a as OffPolicyOptions, b as OffPolicyTrajectory, d as doublyRobust, i as inverseProbabilityWeighting, o as offPolicyEstimateAll, s as selfNormalizedImportanceWeighting } from './off-policy-DiwuKKg7.js';
|
|
6
6
|
import { b as OutcomeStore } from './outcome-store-rnXLEqSn.js';
|
|
7
7
|
export { D as DeploymentOutcome, F as FileSystemOutcomeStore, a as FileSystemOutcomeStoreOptions, I as InMemoryOutcomeStore } from './outcome-store-rnXLEqSn.js';
|
|
8
|
-
import { b as RubricPredictiveValidityReport } from './rubric-predictive-validity-
|
|
9
|
-
import { R as Researcher, F as FailureMode, S as SteeringChange, E as ExperimentPlan, a as ExperimentResult, b as EvalCampaignResult, c as EvalCampaignOptions } from './researcher-
|
|
10
|
-
export { r as runEvalCampaign } from './researcher-
|
|
8
|
+
import { b as RubricPredictiveValidityReport } from './rubric-predictive-validity-p49lLVrE.js';
|
|
9
|
+
import { R as Researcher, F as FailureMode, S as SteeringChange, E as ExperimentPlan, a as ExperimentResult, b as EvalCampaignResult, c as EvalCampaignOptions } from './researcher-C8XyxQsu.js';
|
|
10
|
+
export { r as runEvalCampaign } from './researcher-C8XyxQsu.js';
|
|
11
11
|
import { a as VerificationReport } from './multi-layer-verifier-BsqKuLyN.js';
|
|
12
12
|
import { I as InterimReleaseConfidence } from './sequential-5iSVfzl2.js';
|
|
13
|
-
import { C as CampaignResult } from './types-
|
|
13
|
+
import { C as CampaignResult } from './types-BSw1rOUB.js';
|
|
14
14
|
import '@tangle-network/agent-interface';
|
|
15
15
|
import './errors-oeQrLqXC.js';
|
|
16
|
-
import './
|
|
17
|
-
import './
|
|
18
|
-
import './
|
|
19
|
-
import '
|
|
20
|
-
import './
|
|
21
|
-
import './
|
|
22
|
-
import './
|
|
23
|
-
import './integrity-C6PZ73iC.js';
|
|
16
|
+
import './llm-client-qoDd18Qz.js';
|
|
17
|
+
import './cost-ledger-DWy3XdJc.js';
|
|
18
|
+
import './raw-provider-sink-C46HDghv.js';
|
|
19
|
+
import './summary-report-C5bKFfm-.js';
|
|
20
|
+
import './failure-cluster-DOAcSJ87.js';
|
|
21
|
+
import './emitter-CjD7vUwv.js';
|
|
22
|
+
import './integrity-DqlBiLyK.js';
|
|
24
23
|
import './verdict-C9MlYujm.js';
|
|
24
|
+
import './policy-edit-wG9uFEFm.js';
|
|
25
|
+
import './store-C1YxJDEK.js';
|
|
26
|
+
import './types-BkfcQnxV.js';
|
|
27
|
+
import '@tangle-network/tcloud';
|
|
25
28
|
|
|
26
29
|
/**
|
|
27
30
|
* Adaptive curriculum / active scenario selection.
|
package/dist/rl.js
CHANGED
|
@@ -10,7 +10,7 @@ import {
|
|
|
10
10
|
} from "./chunk-3RF76KTD.js";
|
|
11
11
|
import {
|
|
12
12
|
runEvalCampaign
|
|
13
|
-
} from "./chunk-
|
|
13
|
+
} from "./chunk-GQCZRZ7L.js";
|
|
14
14
|
import {
|
|
15
15
|
detectRewardHacking,
|
|
16
16
|
extractVerifiableReward,
|
|
@@ -38,7 +38,7 @@ import "./chunk-TVVP3ZZQ.js";
|
|
|
38
38
|
import "./chunk-5UF54T55.js";
|
|
39
39
|
import "./chunk-XJYR7XFV.js";
|
|
40
40
|
import "./chunk-VSMTAMNK.js";
|
|
41
|
-
import "./chunk-
|
|
41
|
+
import "./chunk-NJC7U437.js";
|
|
42
42
|
import "./chunk-PC4UYEBM.js";
|
|
43
43
|
import {
|
|
44
44
|
ValidationError
|
|
@@ -1,12 +1,14 @@
|
|
|
1
1
|
import {
|
|
2
2
|
planCampaignRun,
|
|
3
3
|
runCampaign
|
|
4
|
-
} from "./chunk-
|
|
4
|
+
} from "./chunk-IDZTTFRR.js";
|
|
5
5
|
import "./chunk-PJQFMIOX.js";
|
|
6
|
+
import "./chunk-VCTY3W6J.js";
|
|
7
|
+
import "./chunk-VI2UW6B6.js";
|
|
6
8
|
import "./chunk-ONWEPEDO.js";
|
|
7
9
|
import "./chunk-PZ5AY32C.js";
|
|
8
10
|
export {
|
|
9
11
|
planCampaignRun,
|
|
10
12
|
runCampaign
|
|
11
13
|
};
|
|
12
|
-
//# sourceMappingURL=run-campaign-
|
|
14
|
+
//# sourceMappingURL=run-campaign-IM26A6PD.js.map
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
import { AgentProfile } from '@tangle-network/agent-interface';
|
|
2
2
|
import { V as ValidationError } from './errors-oeQrLqXC.js';
|
|
3
|
-
import { F as FailureClass } from './schema-
|
|
3
|
+
import { F as FailureClass } from './schema-B3Q3l9Z_.js';
|
|
4
4
|
|
|
5
5
|
type AgentProfileCellSchemaVersion = 'agent-profile-cell/v1';
|
|
6
6
|
type AgentProfileJson = string | number | boolean | null | AgentProfileJson[] | {
|
|
@@ -116,6 +116,8 @@ interface ToolSpan extends SpanBase {
|
|
|
116
116
|
kind: 'tool';
|
|
117
117
|
toolName: string;
|
|
118
118
|
args: unknown;
|
|
119
|
+
/** False when the source observed the call but did not capture its arguments. */
|
|
120
|
+
argsCaptured?: boolean;
|
|
119
121
|
result?: unknown;
|
|
120
122
|
latencyMs?: number;
|
|
121
123
|
}
|
|
@@ -1,10 +1,12 @@
|
|
|
1
1
|
import { AxAIService } from '@ax-llm/ax';
|
|
2
2
|
import { z } from 'zod';
|
|
3
|
-
import {
|
|
4
|
-
import { T as TraceAnalystKindSpec } from './kind-factory-
|
|
5
|
-
import { a as TraceAnalystSpan } from './store-
|
|
6
|
-
import { R as Run, S as Span, e as TraceEvent, A as Artifact, B as BudgetLedgerEntry } from './schema-
|
|
7
|
-
import { T as TraceStore } from './store-
|
|
3
|
+
import { c as AnalystFinding, A as Analyst, a as AnalystContext } from './policy-edit-wG9uFEFm.js';
|
|
4
|
+
import { T as TraceAnalystKindSpec } from './kind-factory-ClZmO25A.js';
|
|
5
|
+
import { a as TraceAnalystSpan } from './store-C1YxJDEK.js';
|
|
6
|
+
import { R as Run, S as Span, e as TraceEvent, A as Artifact, B as BudgetLedgerEntry } from './schema-B3Q3l9Z_.js';
|
|
7
|
+
import { T as TraceStore } from './store-DGqD0Pyo.js';
|
|
8
|
+
import { C as CostLedger } from './cost-ledger-DWy3XdJc.js';
|
|
9
|
+
import { a as LlmClientOptions } from './llm-client-qoDd18Qz.js';
|
|
8
10
|
import { S as Severity } from './multi-layer-verifier-BsqKuLyN.js';
|
|
9
11
|
|
|
10
12
|
interface CreateAnalystAiConfig {
|
|
@@ -499,7 +501,12 @@ interface SuboptimalSignal {
|
|
|
499
501
|
evidence: Record<string, number | string | boolean>;
|
|
500
502
|
}
|
|
501
503
|
interface BehavioralMetrics {
|
|
504
|
+
/** The only trace represented by these metrics; null when spans are empty. */
|
|
505
|
+
traceId: string | null;
|
|
502
506
|
llmCallCount: number;
|
|
507
|
+
/** Causally serial LLM timelines. Parallel branches are never joined. */
|
|
508
|
+
tokenSequences: BehavioralTokenSequence[];
|
|
509
|
+
/** Token values from the longest serial timeline, retained for convenience. */
|
|
503
510
|
inputTokenTrajectory: number[];
|
|
504
511
|
outputTokenTrajectory: number[];
|
|
505
512
|
toolHistogram: Record<string, number>;
|
|
@@ -510,6 +517,12 @@ interface BehavioralMetrics {
|
|
|
510
517
|
hasSelfVerification: boolean;
|
|
511
518
|
signals: SuboptimalSignal[];
|
|
512
519
|
}
|
|
520
|
+
interface BehavioralTokenSequence {
|
|
521
|
+
scopeId: string;
|
|
522
|
+
spanIds: string[];
|
|
523
|
+
inputTokenTrajectory: Array<number | null>;
|
|
524
|
+
outputTokenTrajectory: Array<number | null>;
|
|
525
|
+
}
|
|
513
526
|
/**
|
|
514
527
|
* Reduce a span list to behavioral metrics + fired suboptimality signals.
|
|
515
528
|
* Pure + deterministic: same spans → same output, on any machine, no model.
|
|
@@ -672,6 +685,8 @@ interface SemanticConceptJudgeOptions {
|
|
|
672
685
|
model?: string;
|
|
673
686
|
/** Per-call timeout. Default 300s. */
|
|
674
687
|
timeoutMs?: number;
|
|
688
|
+
/** Provider-enforced output limit. Default 16000. */
|
|
689
|
+
maxTokens?: number;
|
|
675
690
|
/** Pipeline budget for the prompt (source blob truncation). Default 45000. */
|
|
676
691
|
maxSourceChars?: number;
|
|
677
692
|
/** Per-file cap before inclusion. Default 20000. */
|
|
@@ -680,6 +695,9 @@ interface SemanticConceptJudgeOptions {
|
|
|
680
695
|
maxHtmlChars?: number;
|
|
681
696
|
/** LlmClient config (baseUrl, apiKey, authHeader, …). */
|
|
682
697
|
llm?: LlmClientOptions;
|
|
698
|
+
costLedger?: CostLedger;
|
|
699
|
+
costPhase?: string;
|
|
700
|
+
signal?: AbortSignal;
|
|
683
701
|
/**
|
|
684
702
|
* Score aggregation strategy. Default `mean` — uniform average across
|
|
685
703
|
* concepts. Cross-vertical comparisons should use `complexity` to
|
|
@@ -702,4 +720,4 @@ declare function runSemanticConceptJudge(input: SemanticConceptJudgeInput, optio
|
|
|
702
720
|
*/
|
|
703
721
|
declare function createSemanticConceptJudge(options?: SemanticConceptJudgeOptions): (input: SemanticConceptJudgeInput) => Promise<SemanticConceptJudgeResult>;
|
|
704
722
|
|
|
705
|
-
export { type RunScore as A, type BehavioralMetrics as B, type CreateAnalystAiConfig as C, DEFAULT_TRACE_ANALYST_KINDS as D, type RunScoreWeights as E, FAILURE_MODE_KIND_SPEC as F, type
|
|
723
|
+
export { runSemanticConceptJudge as $, type RunScore as A, type BehavioralMetrics as B, type CreateAnalystAiConfig as C, DEFAULT_TRACE_ANALYST_KINDS as D, type RunScoreWeights as E, FAILURE_MODE_KIND_SPEC as F, type BehavioralTokenSequence as G, type ConceptComplexity as H, IMPROVEMENT_KIND_SPEC as I, type ConceptFinding as J, KIND_EXPECTED_SUBJECTS as K, type ConceptSpec as L, type ConceptWeightStrategy as M, DEFAULT_COMPLEXITY_WEIGHTS as N, DEFAULT_RUN_SCORE_WEIGHTS as O, type PersistedFinding as P, type RunCriticOptions as Q, RunCritic as R, type SemanticConceptJudgeOptions as S, SEMANTIC_CONCEPT_JUDGE_VERSION as T, type SemanticConceptJudgeResult as U, type SuboptimalCode as V, type SuboptimalSignal as W, aggregateRunScore as X, clamp01 as Y, computeTraceMetrics as Z, createSemanticConceptJudge as _, type RunTrace as a, type SemanticConceptJudgeInput as b, type DiffPolicy as c, FINDING_SUBJECT_GRAMMAR_PROMPT as d, FINDING_SUBJECT_KINDS as e, FINDING_SUBJECT_SYNTAX as f, type FindingSubject as g, type FindingSubjectKind as h, FindingSubjectStringSchema as i, type FindingsDiff as j, FindingsStore as k, KNOWLEDGE_GAP_KIND_SPEC as l, KNOWLEDGE_POISONING_KIND_SPEC as m, SKILL_USAGE_ANALYST as n, SkillUsageAnalyst as o, type SkillUsageRecord as p, type SkillUsageReport as q, type SkillUsageScanConfig as r, buildSkillUsageReport as s, createAnalystAi as t, defaultIsMaterial as u, diffFindings as v, emitSkillUsageFindings as w, findingSubjectGrammarPromptFor as x, parseFindingSubject as y, renderFindingSubject as z };
|
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
import { d as ContinuousAgreementOptions, C as ContinuousAgreement } from './judge-calibration-7C-IDmKr.js';
|
|
2
|
-
import { J as JudgeScore } from './types-
|
|
2
|
+
import { J as JudgeScore } from './types-BkfcQnxV.js';
|
|
3
3
|
|
|
4
4
|
/** Identity: dimensions already follow "higher = better" by prompt convention
|
|
5
5
|
* (inverted dims like hallucination are scored 10 = best at the source). */
|
|
@@ -1,3 +1,5 @@
|
|
|
1
|
+
import { C as CostLedger } from './cost-ledger-DWy3XdJc.js';
|
|
2
|
+
|
|
1
3
|
/**
|
|
2
4
|
* `CampaignStorage` — the filesystem seam `runCampaign` writes through
|
|
3
5
|
* (run/cell dirs, the resumability cache, per-cell artifacts, trace spans).
|
|
@@ -22,6 +24,9 @@ interface CampaignStorage {
|
|
|
22
24
|
read(path: string): string | undefined;
|
|
23
25
|
/** Write a file (string or bytes). Parent dir is assumed ensured. */
|
|
24
26
|
write(path: string, content: string | Uint8Array): void;
|
|
27
|
+
/** Append only when the current UTF-8 byte length matches `expectedBytes`.
|
|
28
|
+
* Returns the new length, or undefined when another writer won. */
|
|
29
|
+
append?(path: string, content: string, expectedBytes: number): number | undefined;
|
|
25
30
|
}
|
|
26
31
|
/** Node-filesystem storage — the default. Lazily requires `node:fs` so the
|
|
27
32
|
* module imports cleanly in non-Node runtimes (where the caller passes
|
|
@@ -35,5 +40,11 @@ declare function fsCampaignStorage(): CampaignStorage;
|
|
|
35
40
|
* live in a `Map` for the duration of the run; the `CampaignResult` is
|
|
36
41
|
* fully populated, but nothing is persisted to disk. */
|
|
37
42
|
declare function inMemoryCampaignStorage(): CampaignStorage;
|
|
43
|
+
/** Open the durable spend account stored beside a logical run. */
|
|
44
|
+
declare function createRunCostLedger(input: {
|
|
45
|
+
storage: CampaignStorage;
|
|
46
|
+
runDir: string;
|
|
47
|
+
costCeilingUsd?: number;
|
|
48
|
+
}): CostLedger;
|
|
38
49
|
|
|
39
|
-
export { type CampaignStorage as C, fsCampaignStorage as f, inMemoryCampaignStorage as i };
|
|
50
|
+
export { type CampaignStorage as C, createRunCostLedger as c, fsCampaignStorage as f, inMemoryCampaignStorage as i };
|
|
@@ -1,134 +1,3 @@
|
|
|
1
|
-
/**
|
|
2
|
-
* RawProviderSink — first-class persistence for the actual HTTP-level
|
|
3
|
-
* request/response bodies of every LLM provider call.
|
|
4
|
-
*
|
|
5
|
-
* Why this is a separate sink from the structured `LlmSpan`:
|
|
6
|
-
*
|
|
7
|
-
* - `LlmSpan` records the *intent* — model name, messages, output text,
|
|
8
|
-
* usage. It's what dashboards read; it's NOT enough for forensics.
|
|
9
|
-
* - When a downstream consumer reports "the verifier used the wrong route"
|
|
10
|
-
* or "tokens look right but reasoning was missing," the only way to
|
|
11
|
-
* answer is the raw HTTP body. Span fields can lie (a proxy can echo
|
|
12
|
-
* a different `model` value than what actually answered); the raw
|
|
13
|
-
* response is ground truth.
|
|
14
|
-
*
|
|
15
|
-
* Default behaviour: opt-in. Pass `rawSink` to `LlmClientOptions` (or the
|
|
16
|
-
* matrix runner / BuilderSession sets it up automatically) and every
|
|
17
|
-
* request, response, and error is recorded — including retries, with the
|
|
18
|
-
* attempt index attached so a flaky call's full event chain is recoverable.
|
|
19
|
-
*
|
|
20
|
-
* Redaction is enforced at sink time. The default redactor strips
|
|
21
|
-
* `Authorization`, `X-Api-Key`, `X-Auth-Token`, `Cookie` headers and any
|
|
22
|
-
* payload field whose key matches `apiKey | api_key | bearer | password |
|
|
23
|
-
* secret | token` (case-insensitive). Override via the sink constructor or
|
|
24
|
-
* the per-call `redactor`. The `redactedFields` array on the persisted
|
|
25
|
-
* event lets a reviewer see what was stripped without exposing the values.
|
|
26
|
-
*/
|
|
27
|
-
type RawProviderDirection = 'request' | 'response' | 'error';
|
|
28
|
-
interface RawProviderEvent {
|
|
29
|
-
/** Stable id. Generated by the sink if omitted. */
|
|
30
|
-
eventId: string;
|
|
31
|
-
/** Trace context populated by `LlmClient` when the call is wrapped in a span. */
|
|
32
|
-
runId?: string;
|
|
33
|
-
spanId?: string;
|
|
34
|
-
/**
|
|
35
|
-
* Logical provider name. Free-form so callers can use whatever id matches
|
|
36
|
-
* their topology (`'openai'`, `'anthropic'`, `'tangle-router'`, …). When
|
|
37
|
-
* omitted, derived from `baseUrl` in `LlmClientOptions`.
|
|
38
|
-
*/
|
|
39
|
-
provider: string;
|
|
40
|
-
model: string;
|
|
41
|
-
/** Endpoint path, e.g. `'/v1/chat/completions'`. */
|
|
42
|
-
endpoint: string;
|
|
43
|
-
/** Base URL used for the call (already-normalised — no trailing slash). */
|
|
44
|
-
baseUrl: string;
|
|
45
|
-
/** 0-indexed retry attempt. The first attempt is 0; a retried call gets 1, 2, … */
|
|
46
|
-
attemptIndex: number;
|
|
47
|
-
direction: RawProviderDirection;
|
|
48
|
-
/** Unix ms. */
|
|
49
|
-
timestamp: number;
|
|
50
|
-
/** Wall-clock duration of the call leg. Set on `response` and `error` events; null on `request`. */
|
|
51
|
-
durationMs?: number;
|
|
52
|
-
statusCode?: number;
|
|
53
|
-
requestHeaders?: Record<string, string>;
|
|
54
|
-
requestBody?: unknown;
|
|
55
|
-
responseHeaders?: Record<string, string>;
|
|
56
|
-
responseBody?: unknown;
|
|
57
|
-
/** Set on `direction: 'error'` events. */
|
|
58
|
-
errorMessage?: string;
|
|
59
|
-
/** Field paths the redactor stripped from this event ('header:Authorization', 'body.apiKey', …). */
|
|
60
|
-
redactedFields: string[];
|
|
61
|
-
}
|
|
62
|
-
interface RawProviderSinkFilter {
|
|
63
|
-
runId?: string;
|
|
64
|
-
spanId?: string;
|
|
65
|
-
direction?: RawProviderDirection;
|
|
66
|
-
attemptIndex?: number;
|
|
67
|
-
}
|
|
68
|
-
interface RawProviderSink {
|
|
69
|
-
record(event: RawProviderEvent): Promise<void>;
|
|
70
|
-
/** Optional listing — implementations that durably persist (file, db) should support this. */
|
|
71
|
-
list?(filter?: RawProviderSinkFilter): Promise<RawProviderEvent[]>;
|
|
72
|
-
/** Optional teardown for backed implementations. */
|
|
73
|
-
close?(): Promise<void>;
|
|
74
|
-
}
|
|
75
|
-
type ProviderRedactor = (event: RawProviderEvent) => RawProviderEvent;
|
|
76
|
-
/**
|
|
77
|
-
* Default redactor — strips well-known auth headers and any body field whose
|
|
78
|
-
* key matches the credential pattern. Records every redacted path on
|
|
79
|
-
* `event.redactedFields` so a downstream reviewer can see what was removed.
|
|
80
|
-
*/
|
|
81
|
-
declare function defaultProviderRedactor(event: RawProviderEvent): RawProviderEvent;
|
|
82
|
-
interface InMemoryRawProviderSinkOptions {
|
|
83
|
-
redactor?: ProviderRedactor;
|
|
84
|
-
}
|
|
85
|
-
declare class InMemoryRawProviderSink implements RawProviderSink {
|
|
86
|
-
private events;
|
|
87
|
-
private redactor;
|
|
88
|
-
constructor(opts?: InMemoryRawProviderSinkOptions);
|
|
89
|
-
record(event: RawProviderEvent): Promise<void>;
|
|
90
|
-
list(filter?: RawProviderSinkFilter): Promise<RawProviderEvent[]>;
|
|
91
|
-
size(): number;
|
|
92
|
-
}
|
|
93
|
-
declare class NoopRawProviderSink implements RawProviderSink {
|
|
94
|
-
record(): Promise<void>;
|
|
95
|
-
/**
|
|
96
|
-
* Returns an empty array. Implemented so `assertRunCaptured` does not
|
|
97
|
-
* trip the `no_raw_sink` issue when a caller explicitly opts out of
|
|
98
|
-
* capture by passing this sink — opt-out is a deliberate choice, not a
|
|
99
|
-
* misconfiguration.
|
|
100
|
-
*/
|
|
101
|
-
list(): Promise<RawProviderEvent[]>;
|
|
102
|
-
}
|
|
103
|
-
interface FileSystemRawProviderSinkOptions {
|
|
104
|
-
/** Directory the NDJSON file is written into. Created if missing. */
|
|
105
|
-
dir: string;
|
|
106
|
-
/** File name; default `'raw-provider-events.ndjson'`. */
|
|
107
|
-
fileName?: string;
|
|
108
|
-
/** Bytes after which the writer rolls over to a new file (default 32 MiB). */
|
|
109
|
-
rollAtBytes?: number;
|
|
110
|
-
redactor?: ProviderRedactor;
|
|
111
|
-
}
|
|
112
|
-
declare class FileSystemRawProviderSink implements RawProviderSink {
|
|
113
|
-
private dir;
|
|
114
|
-
private fileName;
|
|
115
|
-
private rollAtBytes;
|
|
116
|
-
private redactor;
|
|
117
|
-
private bytesWritten;
|
|
118
|
-
private rollIndex;
|
|
119
|
-
private initPromise;
|
|
120
|
-
constructor(opts: FileSystemRawProviderSinkOptions);
|
|
121
|
-
private ensureInit;
|
|
122
|
-
private currentPath;
|
|
123
|
-
record(event: RawProviderEvent): Promise<void>;
|
|
124
|
-
list(filter?: RawProviderSinkFilter): Promise<RawProviderEvent[]>;
|
|
125
|
-
}
|
|
126
|
-
/**
|
|
127
|
-
* Best-effort provider id from a base URL. Falls back to the URL host when
|
|
128
|
-
* none of the well-known patterns match.
|
|
129
|
-
*/
|
|
130
|
-
declare function providerFromBaseUrl(baseUrl: string): string;
|
|
131
|
-
|
|
132
1
|
/**
|
|
133
2
|
* Shared types for the trace-analyst module.
|
|
134
3
|
*
|
|
@@ -376,4 +245,4 @@ interface TraceAnalysisStore {
|
|
|
376
245
|
}): Promise<SearchSpanResult>;
|
|
377
246
|
}
|
|
378
247
|
|
|
379
|
-
export { DEFAULT_TRACE_ANALYST_BUDGETS as D, type ErrorCluster as E,
|
|
248
|
+
export { DEFAULT_TRACE_ANALYST_BUDGETS as D, type ErrorCluster as E, type QueryTracesPage as Q, type SearchSpanResult as S, type TraceAnalysisStore as T, type ViewSpansResult as V, type TraceAnalystSpan as a, type DatasetOverview as b, type SearchTraceResult as c, type SpanMatchRecord as d, TRACE_ANALYST_TRUNCATION_MARKER_PREFIX as e, type TraceAnalystByteBudgets as f, type TraceAnalystFilters as g, type TraceAnalystSpanKind as h, type TraceAnalystSpanStatus as i, type TraceAnalystTraceSummary as j, type ViewTraceOversized as k, type ViewTraceResult as l };
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import { R as Run, S as Span, e as TraceEvent, A as Artifact, B as BudgetLedgerEntry, f as RunStatus, g as RunLayer, b as SpanKind, E as EventKind } from './schema-
|
|
1
|
+
import { R as Run, S as Span, e as TraceEvent, A as Artifact, B as BudgetLedgerEntry, f as RunStatus, g as RunLayer, b as SpanKind, E as EventKind } from './schema-B3Q3l9Z_.js';
|
|
2
2
|
|
|
3
3
|
interface RunFilter {
|
|
4
4
|
scenarioId?: string;
|
|
@@ -1,5 +1,5 @@
|
|
|
1
|
-
import { R as RunRecord } from './run-record-
|
|
2
|
-
import { F as FailureClusterReport } from './failure-cluster-
|
|
1
|
+
import { R as RunRecord } from './run-record-BDH49H2E.js';
|
|
2
|
+
import { F as FailureClusterReport } from './failure-cluster-DOAcSJ87.js';
|
|
3
3
|
|
|
4
4
|
/**
|
|
5
5
|
* HeldOutGate — first-class held-out paired-delta promotion gate.
|
|
@@ -1,6 +1,6 @@
|
|
|
1
|
-
import { T as TraceEmitter } from './emitter-
|
|
2
|
-
import { R as Run, F as FailureClass } from './schema-
|
|
3
|
-
import { T as TraceStore } from './store-
|
|
1
|
+
import { T as TraceEmitter } from './emitter-CjD7vUwv.js';
|
|
2
|
+
import { R as Run, F as FailureClass } from './schema-B3Q3l9Z_.js';
|
|
3
|
+
import { T as TraceStore } from './store-DGqD0Pyo.js';
|
|
4
4
|
|
|
5
5
|
/**
|
|
6
6
|
* SandboxHarness — executes a scenario in an isolated environment and
|