oh-my-knowledge 0.20.0 → 0.20.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/src/renderer/html-renderer.d.ts.map +1 -1
- package/dist/src/renderer/html-renderer.js +43 -2
- package/dist/src/renderer/html-renderer.js.map +1 -1
- package/dist/src/renderer/layout.d.ts.map +1 -1
- package/dist/src/renderer/layout.js +92 -0
- package/dist/src/renderer/layout.js.map +1 -1
- package/dist/src/renderer/summary.d.ts +3 -0
- package/dist/src/renderer/summary.d.ts.map +1 -1
- package/dist/src/renderer/summary.js +98 -39
- package/dist/src/renderer/summary.js.map +1 -1
- package/dist/src/types/eval.d.ts +239 -0
- package/dist/src/types/eval.d.ts.map +1 -0
- package/dist/src/types/eval.js +2 -0
- package/dist/src/types/eval.js.map +1 -0
- package/dist/src/types/executor.d.ts +65 -0
- package/dist/src/types/executor.d.ts.map +1 -0
- package/dist/src/types/executor.js +2 -0
- package/dist/src/types/executor.js.map +1 -0
- package/dist/src/types/index.d.ts +7 -0
- package/dist/src/types/index.d.ts.map +1 -0
- package/dist/src/types/index.js +7 -0
- package/dist/src/types/index.js.map +1 -0
- package/dist/src/types/judge.d.ts +98 -0
- package/dist/src/types/judge.d.ts.map +1 -0
- package/dist/src/types/judge.js +2 -0
- package/dist/src/types/judge.js.map +1 -0
- package/dist/src/types/report.d.ts +390 -0
- package/dist/src/types/report.d.ts.map +1 -0
- package/dist/src/types/report.js +2 -0
- package/dist/src/types/report.js.map +1 -0
- package/dist/src/types/shared.d.ts +2 -0
- package/dist/src/types/shared.d.ts.map +1 -0
- package/dist/src/types/shared.js +2 -0
- package/dist/src/types/shared.js.map +1 -0
- package/dist/src/types/storage.d.ts +21 -0
- package/dist/src/types/storage.d.ts.map +1 -0
- package/dist/src/types/storage.js +2 -0
- package/dist/src/types/storage.js.map +1 -0
- package/dist/src/types.d.ts +1 -799
- package/dist/src/types.d.ts.map +1 -1
- package/dist/src/types.js +5 -1
- package/dist/src/types.js.map +1 -1
- package/package.json +1 -1
package/dist/src/types.d.ts
CHANGED
|
@@ -1,800 +1,2 @@
|
|
|
1
|
-
export
|
|
2
|
-
tool: string;
|
|
3
|
-
input: unknown;
|
|
4
|
-
output: unknown;
|
|
5
|
-
success: boolean;
|
|
6
|
-
}
|
|
7
|
-
export interface TurnInfo {
|
|
8
|
-
role: 'assistant' | 'tool';
|
|
9
|
-
content: string;
|
|
10
|
-
toolCalls?: ToolCallInfo[];
|
|
11
|
-
durationMs?: number;
|
|
12
|
-
}
|
|
13
|
-
export interface ExecResult {
|
|
14
|
-
ok: boolean;
|
|
15
|
-
output: string | null;
|
|
16
|
-
durationMs: number;
|
|
17
|
-
durationApiMs: number;
|
|
18
|
-
inputTokens: number;
|
|
19
|
-
outputTokens: number;
|
|
20
|
-
cacheReadTokens: number;
|
|
21
|
-
cacheCreationTokens: number;
|
|
22
|
-
costUSD: number;
|
|
23
|
-
stopReason: string;
|
|
24
|
-
numTurns: number;
|
|
25
|
-
fullNumTurns?: number;
|
|
26
|
-
numSubAgents?: number;
|
|
27
|
-
error?: string;
|
|
28
|
-
cached?: boolean;
|
|
29
|
-
turns?: TurnInfo[];
|
|
30
|
-
toolCalls?: ToolCallInfo[];
|
|
31
|
-
}
|
|
32
|
-
export interface ExecutorInput {
|
|
33
|
-
model: string;
|
|
34
|
-
system?: string | null;
|
|
35
|
-
prompt: string;
|
|
36
|
-
cwd?: string | null;
|
|
37
|
-
skillDir?: string | null;
|
|
38
|
-
timeoutMs?: number;
|
|
39
|
-
verbose?: boolean;
|
|
40
|
-
}
|
|
41
|
-
export type ExecutorFn = (input: ExecutorInput) => Promise<ExecResult>;
|
|
42
|
-
export interface Assertion {
|
|
43
|
-
type: string;
|
|
44
|
-
value?: string | number;
|
|
45
|
-
values?: string[];
|
|
46
|
-
pattern?: string;
|
|
47
|
-
flags?: string;
|
|
48
|
-
schema?: Record<string, unknown>;
|
|
49
|
-
weight?: number;
|
|
50
|
-
fn?: string;
|
|
51
|
-
reference?: string;
|
|
52
|
-
threshold?: number;
|
|
53
|
-
/** v0.21 Phase 5a — when true, the assertion's pass/fail is inverted. Works
|
|
54
|
-
* with any type, including legacy `not_contains` (which becomes a redundant
|
|
55
|
-
* but still supported double-negation). */
|
|
56
|
-
not?: boolean;
|
|
57
|
-
/** v0.21 Phase 5a — only used by type='assert-set'. 'any' = at least one
|
|
58
|
-
* child must pass; 'all' = every child must pass. Children may be any
|
|
59
|
-
* assertion type, including nested assert-sets. */
|
|
60
|
-
mode?: 'any' | 'all';
|
|
61
|
-
children?: Assertion[];
|
|
62
|
-
/** v0.21 Phase 5b — for rouge_n_min: which n-gram order (default 1). */
|
|
63
|
-
n?: number;
|
|
64
|
-
}
|
|
65
|
-
export interface Sample {
|
|
66
|
-
sample_id: string;
|
|
67
|
-
prompt: string;
|
|
68
|
-
context?: string;
|
|
69
|
-
cwd?: string;
|
|
70
|
-
rubric?: string;
|
|
71
|
-
assertions?: Assertion[];
|
|
72
|
-
dimensions?: Record<string, string>;
|
|
73
|
-
allowedTools?: string[];
|
|
74
|
-
expectedTools?: string[];
|
|
75
|
-
[key: string]: unknown;
|
|
76
|
-
}
|
|
77
|
-
export type ArtifactKind = 'baseline' | 'skill' | 'prompt' | 'agent' | 'workflow';
|
|
78
|
-
export interface Artifact {
|
|
79
|
-
name: string;
|
|
80
|
-
kind: ArtifactKind;
|
|
81
|
-
source: 'baseline' | 'variant-name' | 'file-path' | 'git' | 'inline' | 'custom';
|
|
82
|
-
content: string | null;
|
|
83
|
-
locator?: string;
|
|
84
|
-
ref?: string;
|
|
85
|
-
cwd?: string;
|
|
86
|
-
experimentRole?: ExperimentRole;
|
|
87
|
-
metadata?: Record<string, unknown>;
|
|
88
|
-
}
|
|
89
|
-
export type ExperimentType = 'baseline' | 'runtime-context-only' | 'artifact-injection';
|
|
90
|
-
export type ExperimentRole = 'control' | 'treatment';
|
|
91
|
-
export interface VariantConfig {
|
|
92
|
-
variant: string;
|
|
93
|
-
artifactKind: ArtifactKind;
|
|
94
|
-
artifactSource: Artifact['source'];
|
|
95
|
-
executionStrategy: ExecutionStrategyKind;
|
|
96
|
-
experimentType: ExperimentType;
|
|
97
|
-
experimentRole: ExperimentRole;
|
|
98
|
-
hasArtifactContent: boolean;
|
|
99
|
-
cwd: string | null;
|
|
100
|
-
locator?: string;
|
|
101
|
-
ref?: string;
|
|
102
|
-
}
|
|
103
|
-
export type ExecutionStrategyKind = 'baseline' | 'system-prompt' | 'user-prompt' | 'agent-session' | 'workflow-session';
|
|
104
|
-
export interface VariantSpec {
|
|
105
|
-
name: string;
|
|
106
|
-
role: ExperimentRole;
|
|
107
|
-
expr: string;
|
|
108
|
-
}
|
|
109
|
-
export interface EvalConfigVariant {
|
|
110
|
-
name: string;
|
|
111
|
-
role: ExperimentRole;
|
|
112
|
-
artifact: string;
|
|
113
|
-
cwd?: string;
|
|
114
|
-
}
|
|
115
|
-
export interface EvalConfig {
|
|
116
|
-
samples: string;
|
|
117
|
-
executor?: string;
|
|
118
|
-
model?: string;
|
|
119
|
-
judgeModel?: string | null;
|
|
120
|
-
judgeExecutor?: string | null;
|
|
121
|
-
concurrency?: number;
|
|
122
|
-
timeoutMs?: number;
|
|
123
|
-
noCache?: boolean;
|
|
124
|
-
blind?: boolean;
|
|
125
|
-
mcpConfig?: string;
|
|
126
|
-
variants: EvalConfigVariant[];
|
|
127
|
-
/** v0.22 — hard budget caps. When any limit is hit during a run, remaining
|
|
128
|
-
* tasks are aborted and the partial report is persisted. CLI flags
|
|
129
|
-
* `--budget-usd` / `--budget-per-sample-usd` / `--budget-per-sample-ms`
|
|
130
|
-
* override the config values. */
|
|
131
|
-
budget?: EvalBudget;
|
|
132
|
-
}
|
|
133
|
-
export interface EvalBudget {
|
|
134
|
-
/** Stop the run if cumulative (exec + judge) cost exceeds this many USD. */
|
|
135
|
-
totalUSD?: number;
|
|
136
|
-
/** Per-sample cost ceiling. Tasks exceeding this fail individually but the run continues. */
|
|
137
|
-
perSampleUSD?: number;
|
|
138
|
-
/** Per-sample wall-clock latency ceiling in milliseconds. */
|
|
139
|
-
perSampleMs?: number;
|
|
140
|
-
}
|
|
141
|
-
export interface EvaluationRequest {
|
|
142
|
-
samplesPath: string;
|
|
143
|
-
skillDir: string;
|
|
144
|
-
artifacts: Artifact[];
|
|
145
|
-
project?: string;
|
|
146
|
-
owner?: string;
|
|
147
|
-
tags?: string[];
|
|
148
|
-
model: string;
|
|
149
|
-
judgeModel: string | null;
|
|
150
|
-
executor: string;
|
|
151
|
-
judgeExecutor?: string | null;
|
|
152
|
-
noJudge: boolean;
|
|
153
|
-
concurrency: number;
|
|
154
|
-
timeoutMs?: number;
|
|
155
|
-
noCache: boolean;
|
|
156
|
-
dryRun: boolean;
|
|
157
|
-
blind: boolean;
|
|
158
|
-
/** --repeat N; 1 表示单次跑,> 1 走 runMultiple 做 variance 分析 */
|
|
159
|
-
repeat?: number;
|
|
160
|
-
/** --each; 默认不传(=false),true 表示 each mode (每个 skill 独立对比 baseline) */
|
|
161
|
-
each?: boolean;
|
|
162
|
-
/** --judge-repeat N; 每条 sample × dimension 用 LLM judge 跑 N 次, 输出 stddev. 默认 1 (单次). */
|
|
163
|
-
judgeRepeat?: number;
|
|
164
|
-
/** --judge-models executor:model,executor:model,... — multi-judge ensemble.
|
|
165
|
-
* 当传入 ≥ 2 个 judge 时, 每条 sample × dimension 由所有 judge 各自打分, 输出
|
|
166
|
-
* inter-judge agreement (Pearson correlation + mean absolute difference) — 反驳
|
|
167
|
-
* "Claude judge Claude 同模态偏差" 的硬证据. 与 judgeRepeat 可组合. */
|
|
168
|
-
judgeModels?: JudgeConfig[];
|
|
169
|
-
/** --bootstrap; true 时 aggregateReport 加跑 bootstrap mean/diff CI, 写入 VariantSummary.
|
|
170
|
-
* 与原 t-interval 共存 (ReportMeta.evaluationFramework='both'), renderer 优先 bootstrap. */
|
|
171
|
-
bootstrap?: boolean;
|
|
172
|
-
/** --bootstrap-samples N; bootstrap 重采样次数, 默认 1000. > 10000 时 stderr 警告. */
|
|
173
|
-
bootstrapSamples?: number;
|
|
174
|
-
/** v0.21 Phase 3a length-debias toggle. Default true (judge prompt v3-cot-length).
|
|
175
|
-
* CLI flag --no-debias-length flips to false (legacy v2-cot prompt). The active
|
|
176
|
-
* value is reflected in ReportMeta.judgePromptHash and ReportMeta.debiasMode. */
|
|
177
|
-
lengthDebias?: boolean;
|
|
178
|
-
/** v0.22 — hard budget caps. See EvalBudget. */
|
|
179
|
-
budget?: EvalBudget;
|
|
180
|
-
}
|
|
181
|
-
/** Single judge configuration: which executor to call and which model alias to pass. */
|
|
182
|
-
export interface JudgeConfig {
|
|
183
|
-
/** Executor name (claude / openai / gemini / anthropic-api / openai-api / shell command). */
|
|
184
|
-
executor: string;
|
|
185
|
-
/** Model alias passed to the executor (e.g. "opus", "haiku", "gpt-4o", "gemini-2.0-pro"). */
|
|
186
|
-
model: string;
|
|
187
|
-
}
|
|
188
|
-
/** Per-judge ensemble entry: which judge gave what score (mean over judge-repeat if N>1). */
|
|
189
|
-
export interface EnsembleJudgeResult {
|
|
190
|
-
/** "executor:model" identifier — e.g. "claude:opus" or "openai:gpt-4o". */
|
|
191
|
-
judge: string;
|
|
192
|
-
/** Mean score from this judge over judge-repeat calls (or single score if repeat=1). */
|
|
193
|
-
score: number;
|
|
194
|
-
/** Stddev across judge-repeat calls for this judge (0 if repeat=1). */
|
|
195
|
-
scoreStddev?: number;
|
|
196
|
-
/** Raw scores per call (length = judgeRepeat). */
|
|
197
|
-
scoreSamples?: number[];
|
|
198
|
-
/** How many of judgeRepeat calls failed (returned score=0). */
|
|
199
|
-
judgeFailureCount?: number;
|
|
200
|
-
/** First-call CoT reasoning from this judge. */
|
|
201
|
-
reasoning?: string;
|
|
202
|
-
/** Cost in USD across all calls from this judge. */
|
|
203
|
-
costUSD?: number;
|
|
204
|
-
}
|
|
205
|
-
/** Inter-judge agreement metrics across an ensemble. Both metrics are pairwise-averaged. */
|
|
206
|
-
export interface JudgeAgreement {
|
|
207
|
-
/** Pairwise Pearson correlation, averaged. 1 = judges fully agree on rank order; 0 = no
|
|
208
|
-
* correlation; -1 = anti-correlated. Note: only defined when at least one judge has
|
|
209
|
-
* variance (constant-score judges produce undefined Pearson). */
|
|
210
|
-
pearson?: number;
|
|
211
|
-
/** Pairwise mean absolute difference of scores. 0 = identical scores. On a 1-5 scale
|
|
212
|
-
* values < 0.5 are tight agreement, > 1.5 is large disagreement. */
|
|
213
|
-
meanAbsDiff: number;
|
|
214
|
-
/** Number of judge pairs the metrics were computed over (= n*(n-1)/2). */
|
|
215
|
-
pairCount: number;
|
|
216
|
-
}
|
|
217
|
-
export type EvaluationJobStatus = 'queued' | 'running' | 'succeeded' | 'failed' | 'cancelled';
|
|
218
|
-
export type EvaluationErrorCategory = 'user' | 'executor' | 'judge' | 'system';
|
|
219
|
-
export interface EvaluationRun {
|
|
220
|
-
runId: string;
|
|
221
|
-
startedAt: string;
|
|
222
|
-
finishedAt?: string;
|
|
223
|
-
status: Extract<EvaluationJobStatus, 'running' | 'succeeded' | 'failed' | 'cancelled'>;
|
|
224
|
-
}
|
|
225
|
-
export interface EvaluationJob {
|
|
226
|
-
jobId: string;
|
|
227
|
-
status: EvaluationJobStatus;
|
|
228
|
-
createdAt: string;
|
|
229
|
-
updatedAt?: string;
|
|
230
|
-
startedAt?: string;
|
|
231
|
-
finishedAt?: string;
|
|
232
|
-
request: EvaluationRequest;
|
|
233
|
-
runId?: string;
|
|
234
|
-
resultReportId?: string;
|
|
235
|
-
error?: string;
|
|
236
|
-
errorCategory?: EvaluationErrorCategory;
|
|
237
|
-
}
|
|
238
|
-
export interface ProgressStart {
|
|
239
|
-
phase: 'start';
|
|
240
|
-
completed: number;
|
|
241
|
-
total: number;
|
|
242
|
-
sample_id: string;
|
|
243
|
-
variant: string;
|
|
244
|
-
}
|
|
245
|
-
export interface ProgressExecDone {
|
|
246
|
-
phase: 'exec_done';
|
|
247
|
-
strategy: string;
|
|
248
|
-
completed: number;
|
|
249
|
-
total: number;
|
|
250
|
-
sample_id: string;
|
|
251
|
-
variant: string;
|
|
252
|
-
durationMs: number;
|
|
253
|
-
inputTokens: number;
|
|
254
|
-
outputTokens: number;
|
|
255
|
-
costUSD: number;
|
|
256
|
-
outputPreview: string | null;
|
|
257
|
-
}
|
|
258
|
-
export interface ProgressGrading {
|
|
259
|
-
phase: 'grading';
|
|
260
|
-
strategy: string;
|
|
261
|
-
completed: number;
|
|
262
|
-
total: number;
|
|
263
|
-
sample_id: string;
|
|
264
|
-
variant: string;
|
|
265
|
-
}
|
|
266
|
-
export interface ProgressDone {
|
|
267
|
-
phase: 'done';
|
|
268
|
-
strategy?: string;
|
|
269
|
-
completed: number;
|
|
270
|
-
total: number;
|
|
271
|
-
sample_id: string;
|
|
272
|
-
variant: string;
|
|
273
|
-
durationMs?: number;
|
|
274
|
-
inputTokens?: number;
|
|
275
|
-
outputTokens?: number;
|
|
276
|
-
costUSD?: number;
|
|
277
|
-
score?: number;
|
|
278
|
-
skipped?: boolean;
|
|
279
|
-
}
|
|
280
|
-
export interface ProgressRetry {
|
|
281
|
-
phase: 'retry';
|
|
282
|
-
completed: number;
|
|
283
|
-
total: number;
|
|
284
|
-
sample_id: string;
|
|
285
|
-
variant: string;
|
|
286
|
-
attempt: number;
|
|
287
|
-
maxAttempts: number;
|
|
288
|
-
}
|
|
289
|
-
export interface ProgressError {
|
|
290
|
-
phase: 'error';
|
|
291
|
-
completed: number;
|
|
292
|
-
total: number;
|
|
293
|
-
sample_id: string;
|
|
294
|
-
variant: string;
|
|
295
|
-
error: string;
|
|
296
|
-
}
|
|
297
|
-
export interface ProgressPreflight {
|
|
298
|
-
phase: 'preflight';
|
|
299
|
-
jobId?: string;
|
|
300
|
-
}
|
|
301
|
-
export type ProgressInfo = ProgressStart | ProgressExecDone | ProgressGrading | ProgressDone | ProgressRetry | ProgressError | ProgressPreflight;
|
|
302
|
-
export type ProgressCallback = (info: ProgressInfo) => void;
|
|
303
|
-
export interface Task {
|
|
304
|
-
sample_id: string;
|
|
305
|
-
variant: string;
|
|
306
|
-
artifact: Artifact;
|
|
307
|
-
prompt: string;
|
|
308
|
-
rubric: string | null;
|
|
309
|
-
assertions: Assertion[] | null;
|
|
310
|
-
dimensions: Record<string, string> | null;
|
|
311
|
-
artifactContent: string | null;
|
|
312
|
-
cwd: string | null;
|
|
313
|
-
_sample: Sample;
|
|
314
|
-
}
|
|
315
|
-
export interface AssertionDetail {
|
|
316
|
-
type: string;
|
|
317
|
-
value: string | number;
|
|
318
|
-
weight: number;
|
|
319
|
-
passed: boolean;
|
|
320
|
-
message?: string;
|
|
321
|
-
}
|
|
322
|
-
export interface AssertionResults {
|
|
323
|
-
passed: number;
|
|
324
|
-
total: number;
|
|
325
|
-
score: number;
|
|
326
|
-
details: AssertionDetail[];
|
|
327
|
-
judgeCostUSD?: number;
|
|
328
|
-
}
|
|
329
|
-
export interface DimensionResult {
|
|
330
|
-
score: number;
|
|
331
|
-
reason: string;
|
|
332
|
-
judgeCostUSD?: number;
|
|
333
|
-
/** When judge-repeat > 1: scores from each judge run (length = repeat count). */
|
|
334
|
-
scoreSamples?: number[];
|
|
335
|
-
/** Standard deviation across scoreSamples (0 when repeat = 1). */
|
|
336
|
-
scoreStddev?: number;
|
|
337
|
-
/** Chain-of-thought reasoning produced by the judge before the final score. */
|
|
338
|
-
reasoning?: string;
|
|
339
|
-
/**
|
|
340
|
-
* Number of judge calls that failed (returned score=0 / non-JSON / executor error).
|
|
341
|
-
* Stddev = 0 + judgeFailureCount > 0 means "looks consistent but actually had failures",
|
|
342
|
-
* NOT "judge agreed perfectly". Always check this before trusting low stddev.
|
|
343
|
-
*/
|
|
344
|
-
judgeFailureCount?: number;
|
|
345
|
-
/** Multi-judge ensemble: per-judge results when judgeModels.length >= 2. */
|
|
346
|
-
ensemble?: EnsembleJudgeResult[];
|
|
347
|
-
/** Multi-judge ensemble: inter-judge agreement metrics. */
|
|
348
|
-
agreement?: JudgeAgreement;
|
|
349
|
-
}
|
|
350
|
-
export interface LayeredScores {
|
|
351
|
-
factScore?: number;
|
|
352
|
-
behaviorScore?: number;
|
|
353
|
-
judgeScore?: number;
|
|
354
|
-
}
|
|
355
|
-
export interface GradeResult {
|
|
356
|
-
compositeScore: number;
|
|
357
|
-
layeredScores?: LayeredScores;
|
|
358
|
-
assertions?: AssertionResults;
|
|
359
|
-
llmScore?: number;
|
|
360
|
-
llmReason?: string;
|
|
361
|
-
/** Single-rubric mode: judge's chain-of-thought reasoning (first call when judgeRepeat > 1). */
|
|
362
|
-
llmReasoning?: string;
|
|
363
|
-
/** When judge-repeat > 1 with single rubric: stddev across N judge calls. */
|
|
364
|
-
llmScoreStddev?: number;
|
|
365
|
-
/** When judge-repeat > 1 with single rubric: raw scores from each judge call. */
|
|
366
|
-
llmScoreSamples?: number[];
|
|
367
|
-
/** When judge-repeat > 1 with single rubric: how many of the N judge calls failed. */
|
|
368
|
-
llmScoreFailures?: number;
|
|
369
|
-
/** Multi-judge ensemble (single rubric): per-judge results when judgeModels.length >= 2. */
|
|
370
|
-
llmEnsemble?: EnsembleJudgeResult[];
|
|
371
|
-
/** Multi-judge ensemble (single rubric): inter-judge agreement metrics. */
|
|
372
|
-
llmAgreement?: JudgeAgreement;
|
|
373
|
-
dimensions?: Record<string, DimensionResult>;
|
|
374
|
-
judgeCostUSD?: number;
|
|
375
|
-
}
|
|
376
|
-
export interface VariantResult {
|
|
377
|
-
ok: boolean;
|
|
378
|
-
durationMs: number;
|
|
379
|
-
durationApiMs: number;
|
|
380
|
-
inputTokens: number;
|
|
381
|
-
outputTokens: number;
|
|
382
|
-
totalTokens: number;
|
|
383
|
-
cacheReadTokens: number;
|
|
384
|
-
cacheCreationTokens: number;
|
|
385
|
-
execCostUSD: number;
|
|
386
|
-
judgeCostUSD: number;
|
|
387
|
-
costUSD: number;
|
|
388
|
-
numTurns: number;
|
|
389
|
-
fullNumTurns?: number;
|
|
390
|
-
numSubAgents?: number;
|
|
391
|
-
assistantTurns?: number;
|
|
392
|
-
toolTurns?: number;
|
|
393
|
-
numToolCalls?: number;
|
|
394
|
-
numToolFailures?: number;
|
|
395
|
-
toolSuccessRate?: number;
|
|
396
|
-
toolNames?: string[];
|
|
397
|
-
traceCoverage?: number;
|
|
398
|
-
error?: string;
|
|
399
|
-
compositeScore?: number;
|
|
400
|
-
layeredScores?: LayeredScores;
|
|
401
|
-
assertions?: AssertionResults;
|
|
402
|
-
llmScore?: number;
|
|
403
|
-
llmReason?: string;
|
|
404
|
-
/** Single-rubric mode: judge's chain-of-thought reasoning (first call when judgeRepeat > 1). */
|
|
405
|
-
llmReasoning?: string;
|
|
406
|
-
/** Single-rubric mode + judgeRepeat > 1: stddev across N judge calls. */
|
|
407
|
-
llmScoreStddev?: number;
|
|
408
|
-
/** Single-rubric mode + judgeRepeat > 1: raw scores from each call. */
|
|
409
|
-
llmScoreSamples?: number[];
|
|
410
|
-
/** Single-rubric mode + judgeRepeat > 1: how many of N calls failed. */
|
|
411
|
-
llmScoreFailures?: number;
|
|
412
|
-
/** Single-rubric mode + judgeModels.length >= 2: per-judge ensemble results. */
|
|
413
|
-
llmEnsemble?: EnsembleJudgeResult[];
|
|
414
|
-
/** Single-rubric mode + judgeModels.length >= 2: inter-judge agreement metrics. */
|
|
415
|
-
llmAgreement?: JudgeAgreement;
|
|
416
|
-
dimensions?: Record<string, DimensionResult>;
|
|
417
|
-
factCheck?: {
|
|
418
|
-
verifiedCount: number;
|
|
419
|
-
totalCount: number;
|
|
420
|
-
verifiedRate: number;
|
|
421
|
-
claims: Array<{
|
|
422
|
-
type: string;
|
|
423
|
-
value: string;
|
|
424
|
-
verified: boolean;
|
|
425
|
-
evidence?: string;
|
|
426
|
-
}>;
|
|
427
|
-
};
|
|
428
|
-
outputPreview: string | null;
|
|
429
|
-
fullOutput?: string;
|
|
430
|
-
turns?: TurnInfo[];
|
|
431
|
-
toolCalls?: ToolCallInfo[];
|
|
432
|
-
timing?: {
|
|
433
|
-
execMs: number;
|
|
434
|
-
gradeMs: number;
|
|
435
|
-
totalMs: number;
|
|
436
|
-
};
|
|
437
|
-
}
|
|
438
|
-
export interface VariantSummary {
|
|
439
|
-
totalSamples: number;
|
|
440
|
-
successCount: number;
|
|
441
|
-
errorCount: number;
|
|
442
|
-
errorRate: number;
|
|
443
|
-
avgDurationMs: number;
|
|
444
|
-
avgInputTokens: number;
|
|
445
|
-
avgOutputTokens: number;
|
|
446
|
-
avgTotalTokens: number;
|
|
447
|
-
totalCostUSD: number;
|
|
448
|
-
totalExecCostUSD: number;
|
|
449
|
-
totalJudgeCostUSD: number;
|
|
450
|
-
avgCostPerSample: number;
|
|
451
|
-
avgNumTurns: number;
|
|
452
|
-
avgFullNumTurns?: number;
|
|
453
|
-
avgNumSubAgents?: number;
|
|
454
|
-
avgAssistantTurns?: number;
|
|
455
|
-
avgToolTurns?: number;
|
|
456
|
-
avgToolCalls?: number;
|
|
457
|
-
avgToolFailures?: number;
|
|
458
|
-
toolSuccessRate?: number;
|
|
459
|
-
toolDistribution?: Record<string, number>;
|
|
460
|
-
traceCoverageRate?: number;
|
|
461
|
-
avgFactScore?: number;
|
|
462
|
-
avgFactVerifiedRate?: number;
|
|
463
|
-
avgBehaviorScore?: number;
|
|
464
|
-
avgJudgeScore?: number;
|
|
465
|
-
avgCompositeScore?: number;
|
|
466
|
-
minCompositeScore?: number;
|
|
467
|
-
maxCompositeScore?: number;
|
|
468
|
-
scoreStddev?: number;
|
|
469
|
-
scoreCV?: number;
|
|
470
|
-
avgAssertionScore?: number;
|
|
471
|
-
avgLlmScore?: number;
|
|
472
|
-
minLlmScore?: number;
|
|
473
|
-
maxLlmScore?: number;
|
|
474
|
-
/** Aggregate-level multi-judge agreement across this variant's samples (single rubric mode).
|
|
475
|
-
* sampleCount = how many samples had complete ensemble data. */
|
|
476
|
-
judgeAgreement?: JudgeAgreement & {
|
|
477
|
-
sampleCount: number;
|
|
478
|
-
};
|
|
479
|
-
/** List of judge identifiers ("executor:model") seen in this variant's ensemble data. */
|
|
480
|
-
judgeModels?: string[];
|
|
481
|
-
/** Bootstrap CI on this variant's compositeScore mean (when --bootstrap enabled).
|
|
482
|
-
* Distribution-free; preferred over t-interval for ordinal LLM scores. */
|
|
483
|
-
bootstrapCI?: {
|
|
484
|
-
low: number;
|
|
485
|
-
high: number;
|
|
486
|
-
estimate: number;
|
|
487
|
-
samples: number;
|
|
488
|
-
};
|
|
489
|
-
}
|
|
490
|
-
/**
|
|
491
|
-
* Pairwise variant comparison stats — used when comparing treatment vs control.
|
|
492
|
-
* Independent from per-variant `bootstrapCI` (which is on each variant alone).
|
|
493
|
-
*/
|
|
494
|
-
export interface VariantPairComparison {
|
|
495
|
-
/** Control variant name (the subtrahend). */
|
|
496
|
-
control: string;
|
|
497
|
-
/** Treatment variant name (the minuend). */
|
|
498
|
-
treatment: string;
|
|
499
|
-
/** Bootstrap CI on (treatment - control) mean diff. `significant` = 0 outside CI. */
|
|
500
|
-
diffBootstrapCI?: {
|
|
501
|
-
low: number;
|
|
502
|
-
high: number;
|
|
503
|
-
estimate: number;
|
|
504
|
-
samples: number;
|
|
505
|
-
significant: boolean;
|
|
506
|
-
};
|
|
507
|
-
}
|
|
508
|
-
export interface GitInfo {
|
|
509
|
-
commit: string;
|
|
510
|
-
commitShort: string;
|
|
511
|
-
branch: string;
|
|
512
|
-
dirty: boolean;
|
|
513
|
-
}
|
|
514
|
-
/** Persisted form of agreement metrics between gold dataset and the LLM judge.
|
|
515
|
-
* Lives on ReportMeta so the renderer can show a "人工锚点" section without
|
|
516
|
-
* re-loading the gold dataset. */
|
|
517
|
-
export interface ReportHumanAgreement {
|
|
518
|
-
/** Krippendorff α (interval weights) — primary metric. */
|
|
519
|
-
alpha: number;
|
|
520
|
-
/** Bootstrap 95% CI on α. */
|
|
521
|
-
alphaCI: {
|
|
522
|
-
low: number;
|
|
523
|
-
high: number;
|
|
524
|
-
estimate: number;
|
|
525
|
-
samples: number;
|
|
526
|
-
};
|
|
527
|
-
/** Quadratic-weighted κ — secondary metric. */
|
|
528
|
-
weightedKappa: number;
|
|
529
|
-
/** Pearson r — tertiary, rank-order only. */
|
|
530
|
-
pearson: number;
|
|
531
|
-
/** Number of (gold, judge) pairs that contributed. */
|
|
532
|
-
sampleCount: number;
|
|
533
|
-
/** Variant whose judge scores were compared. */
|
|
534
|
-
variant: string;
|
|
535
|
-
/** Identifier of the gold annotator (model id, person, or team handle). */
|
|
536
|
-
goldAnnotator: string;
|
|
537
|
-
/** Free-form version string from the gold metadata. */
|
|
538
|
-
goldVersion: string;
|
|
539
|
-
/** Set when annotator id overlapped with judge model id. */
|
|
540
|
-
contaminationWarning?: string;
|
|
541
|
-
/** Sample_ids in the gold set that were absent from the report. */
|
|
542
|
-
missingCount: number;
|
|
543
|
-
/** Sample_ids present in the report but with no judge score (assertion-only etc). */
|
|
544
|
-
unscoredCount: number;
|
|
545
|
-
}
|
|
546
|
-
export interface ReportMeta {
|
|
547
|
-
variants: string[];
|
|
548
|
-
model: string;
|
|
549
|
-
judgeModel: string | null;
|
|
550
|
-
executor: string;
|
|
551
|
-
sampleCount: number;
|
|
552
|
-
taskCount: number;
|
|
553
|
-
totalCostUSD: number;
|
|
554
|
-
timestamp: string;
|
|
555
|
-
cliVersion: string;
|
|
556
|
-
nodeVersion: string;
|
|
557
|
-
artifactHashes: Record<string, string>;
|
|
558
|
-
/** SHA256-12 of every sample's content (sample_id → hash). Same hash = same sample. */
|
|
559
|
-
sampleHashes?: Record<string, string>;
|
|
560
|
-
/** SHA256-12 of the LLM judge prompt template. Different hash = judge changed semantics. */
|
|
561
|
-
judgePromptHash?: string;
|
|
562
|
-
/** Number of times each sample was judged. 1 = single judge (default). */
|
|
563
|
-
judgeRepeat?: number;
|
|
564
|
-
/** Multi-judge ensemble configuration: ["claude:opus", "openai:gpt-4o", ...].
|
|
565
|
-
* When length >= 2, every (sample × dimension) is scored by all judges and
|
|
566
|
-
* agreement metrics are reported per-result. */
|
|
567
|
-
judgeModels?: string[];
|
|
568
|
-
/** Which CI framework was used for this report: 't-test' (legacy default),
|
|
569
|
-
* 'bootstrap' (--bootstrap), or 'both' (some summaries have both). Reports
|
|
570
|
-
* with mismatched frameworks shouldn't be compared blindly on CI bounds. */
|
|
571
|
-
evaluationFramework?: 't-test' | 'bootstrap' | 'both';
|
|
572
|
-
/** Pairwise comparisons (treatment vs control) — populated when --bootstrap and
|
|
573
|
-
* multi-variant. Length = (variants.length - 1). */
|
|
574
|
-
pairComparisons?: VariantPairComparison[];
|
|
575
|
-
/** v0.21 Phase 3 — which judge-bias debias modes were active for this run.
|
|
576
|
-
* Values: 'length' (substance-not-length prompt), 'position' (random ensemble
|
|
577
|
-
* order). Empty / absent means legacy default (no debias). The renderer shows
|
|
578
|
-
* this so readers can tell apples from oranges across reports. */
|
|
579
|
-
debiasMode?: Array<'length' | 'position'>;
|
|
580
|
-
/** v0.22 — set to true when the run was aborted by a budget tracker. The
|
|
581
|
-
* report is partial: only tasks completed before the abort are present. */
|
|
582
|
-
budgetExhausted?: boolean;
|
|
583
|
-
/** v0.22 — budget caps that were active for this run, copied from request.budget
|
|
584
|
-
* for ease of reading without dereferencing request. */
|
|
585
|
-
budget?: EvalBudget;
|
|
586
|
-
/** Human-gold agreement when --gold-dir was passed at run time. Compares the
|
|
587
|
-
* judge's llmScore against the gold annotations on matching sample_ids. See
|
|
588
|
-
* src/grading/human-gold.ts for the metric definitions. */
|
|
589
|
-
humanAgreement?: ReportHumanAgreement;
|
|
590
|
-
variantConfigs?: VariantConfig[];
|
|
591
|
-
request?: EvaluationRequest;
|
|
592
|
-
run?: EvaluationRun;
|
|
593
|
-
job?: EvaluationJob;
|
|
594
|
-
gitInfo?: GitInfo | null;
|
|
595
|
-
blind?: boolean;
|
|
596
|
-
blindMap?: Record<string, string>;
|
|
597
|
-
layeredStats?: boolean;
|
|
598
|
-
}
|
|
599
|
-
export interface ResultEntry {
|
|
600
|
-
sample_id: string;
|
|
601
|
-
variants: Record<string, VariantResult>;
|
|
602
|
-
}
|
|
603
|
-
export interface Report {
|
|
604
|
-
id: string;
|
|
605
|
-
meta: ReportMeta;
|
|
606
|
-
summary: Record<string, VariantSummary>;
|
|
607
|
-
results: ResultEntry[];
|
|
608
|
-
analysis?: AnalysisResult;
|
|
609
|
-
variance?: VarianceData;
|
|
610
|
-
each?: boolean;
|
|
611
|
-
overview?: {
|
|
612
|
-
totalArtifacts: number;
|
|
613
|
-
totalSamples: number;
|
|
614
|
-
totalCostUSD: number;
|
|
615
|
-
artifacts: Array<{
|
|
616
|
-
name: string;
|
|
617
|
-
baselineScore: number | null;
|
|
618
|
-
artifactScore: number | null;
|
|
619
|
-
improvement: string;
|
|
620
|
-
}>;
|
|
621
|
-
};
|
|
622
|
-
artifacts?: Array<{
|
|
623
|
-
name: string;
|
|
624
|
-
sampleCount: number;
|
|
625
|
-
artifactHash: string | null;
|
|
626
|
-
summary: Record<string, VariantSummary>;
|
|
627
|
-
/** --each --repeat N 时由 runMultiple 聚合的三层独立 variance + t 检验 */
|
|
628
|
-
variance?: VarianceData;
|
|
629
|
-
results: ResultEntry[];
|
|
630
|
-
}>;
|
|
631
|
-
}
|
|
632
|
-
export interface Insight {
|
|
633
|
-
type: string;
|
|
634
|
-
severity: 'error' | 'warning' | 'info';
|
|
635
|
-
message: string;
|
|
636
|
-
details: unknown;
|
|
637
|
-
}
|
|
638
|
-
export interface KnowledgeCoverageEntry {
|
|
639
|
-
path: string;
|
|
640
|
-
type: string;
|
|
641
|
-
accessed: boolean;
|
|
642
|
-
accessCount: number;
|
|
643
|
-
lineCount?: number;
|
|
644
|
-
}
|
|
645
|
-
export interface KnowledgeCoverage {
|
|
646
|
-
entries: KnowledgeCoverageEntry[];
|
|
647
|
-
filesCovered: number;
|
|
648
|
-
filesTotal: number;
|
|
649
|
-
fileCoverageRate: number;
|
|
650
|
-
uncoveredFiles: string[];
|
|
651
|
-
grepPatternsUsed: number;
|
|
652
|
-
overallRate: number;
|
|
653
|
-
}
|
|
654
|
-
export interface AnalysisResult {
|
|
655
|
-
summary?: string;
|
|
656
|
-
insights: Insight[];
|
|
657
|
-
suggestions: string[];
|
|
658
|
-
coverage?: Record<string, KnowledgeCoverage>;
|
|
659
|
-
/** Per-variant knowledge gap reports. See docs/knowledge-gap-signal-spec.md */
|
|
660
|
-
gapReports?: Record<string, GapReport>;
|
|
661
|
-
}
|
|
662
|
-
export interface HedgingVerdict {
|
|
663
|
-
isUncertainty: boolean;
|
|
664
|
-
confidence: number;
|
|
665
|
-
reason: string;
|
|
666
|
-
}
|
|
667
|
-
export interface GapSignalRef {
|
|
668
|
-
sampleId: string;
|
|
669
|
-
type: 'failed_search' | 'explicit_marker' | 'hedging' | 'repeated_failure';
|
|
670
|
-
turn?: number;
|
|
671
|
-
context: string;
|
|
672
|
-
evidence?: Record<string, unknown>;
|
|
673
|
-
weight: number;
|
|
674
|
-
classifierVerdict?: HedgingVerdict;
|
|
675
|
-
}
|
|
676
|
-
export interface GapReport {
|
|
677
|
-
variant: string;
|
|
678
|
-
sampleCount: number;
|
|
679
|
-
samplesWithGap: number;
|
|
680
|
-
gapRate: number;
|
|
681
|
-
weightedGapRate: number;
|
|
682
|
-
testSetPath?: string | null;
|
|
683
|
-
testSetHash?: string | null;
|
|
684
|
-
signals: GapSignalRef[];
|
|
685
|
-
byType: {
|
|
686
|
-
failed_search: number;
|
|
687
|
-
explicit_marker: number;
|
|
688
|
-
hedging: number;
|
|
689
|
-
repeated_failure: number;
|
|
690
|
-
};
|
|
691
|
-
}
|
|
692
|
-
export interface VarianceEffectSize {
|
|
693
|
-
cohensD: number;
|
|
694
|
-
hedgesG: number;
|
|
695
|
-
primary: 'd' | 'g' | 'none';
|
|
696
|
-
magnitude: 'negligible' | 'small' | 'medium' | 'large' | 'none';
|
|
697
|
-
pooledStddev: number;
|
|
698
|
-
n1: number;
|
|
699
|
-
n2: number;
|
|
700
|
-
}
|
|
701
|
-
export interface VarianceMetric {
|
|
702
|
-
scores: number[];
|
|
703
|
-
mean: number;
|
|
704
|
-
lower: number;
|
|
705
|
-
upper: number;
|
|
706
|
-
stddev: number;
|
|
707
|
-
}
|
|
708
|
-
export interface VarianceComparisonMetric {
|
|
709
|
-
meanDiff: number;
|
|
710
|
-
tStatistic: number;
|
|
711
|
-
df: number;
|
|
712
|
-
significant: boolean;
|
|
713
|
-
effectSize: VarianceEffectSize;
|
|
714
|
-
}
|
|
715
|
-
export type VarianceMetricKey = 'cost' | 'efficiency';
|
|
716
|
-
export type VarianceLayerKey = 'fact' | 'behavior' | 'judge';
|
|
717
|
-
export interface VariantVariance extends VarianceMetric {
|
|
718
|
-
byMetric?: Partial<Record<VarianceMetricKey, VarianceMetric>>;
|
|
719
|
-
byLayer?: Partial<Record<VarianceLayerKey, VarianceMetric>>;
|
|
720
|
-
}
|
|
721
|
-
export interface VarianceComparison extends VarianceComparisonMetric {
|
|
722
|
-
a: string;
|
|
723
|
-
b: string;
|
|
724
|
-
byMetric?: Partial<Record<VarianceMetricKey, VarianceComparisonMetric>>;
|
|
725
|
-
byLayer?: Partial<Record<VarianceLayerKey, VarianceComparisonMetric>>;
|
|
726
|
-
}
|
|
727
|
-
export interface VarianceData {
|
|
728
|
-
runs: number;
|
|
729
|
-
perVariant: Record<string, VariantVariance>;
|
|
730
|
-
comparisons: VarianceComparison[];
|
|
731
|
-
/** v0.21 Phase 4 — saturation curve data. Populated only when repeat ≥ 2.
|
|
732
|
-
* Per-variant cumulative score arrays at each repeat checkpoint, plus the
|
|
733
|
-
* saturation verdict (only computed when repeat ≥ 5). */
|
|
734
|
-
saturation?: SaturationData;
|
|
735
|
-
}
|
|
736
|
-
/** Per-variant saturation curve data + (optionally) verdict. */
|
|
737
|
-
export interface SaturationData {
|
|
738
|
-
/** Cumulative checkpoint counts (sample-cumulative across runs). */
|
|
739
|
-
checkpointSampleCounts: number[];
|
|
740
|
-
/** Per-variant trace: at each checkpoint, mean and CI bounds.
|
|
741
|
-
* perVariant[variant][i] = { n, mean, ciLow, ciHigh } at checkpoint i. */
|
|
742
|
-
perVariant: Record<string, Array<{
|
|
743
|
-
n: number;
|
|
744
|
-
mean: number;
|
|
745
|
-
ciLow: number;
|
|
746
|
-
ciHigh: number;
|
|
747
|
-
}>>;
|
|
748
|
-
/** Saturation verdict per variant. Only present when repeat ≥ 5. */
|
|
749
|
-
verdicts?: Record<string, {
|
|
750
|
-
saturated: boolean;
|
|
751
|
-
atN: number | null;
|
|
752
|
-
confidence: 'high' | 'medium' | 'low';
|
|
753
|
-
method: 'slope' | 'bootstrap-ci-width' | 'plateau-height';
|
|
754
|
-
threshold: number;
|
|
755
|
-
reason: string;
|
|
756
|
-
}>;
|
|
757
|
-
}
|
|
758
|
-
export interface McpFetchTool {
|
|
759
|
-
name: string;
|
|
760
|
-
urlParam?: string;
|
|
761
|
-
urlTransform?: {
|
|
762
|
-
regex: string;
|
|
763
|
-
params: Record<string, string>;
|
|
764
|
-
};
|
|
765
|
-
contentExtract?: string;
|
|
766
|
-
}
|
|
767
|
-
export interface McpServerDef {
|
|
768
|
-
command: string;
|
|
769
|
-
args?: string[];
|
|
770
|
-
env?: Record<string, string>;
|
|
771
|
-
urlPatterns: string[];
|
|
772
|
-
fetchTool: McpFetchTool;
|
|
773
|
-
}
|
|
774
|
-
export type McpServers = Record<string, McpServerDef>;
|
|
775
|
-
export interface ExecutorCache {
|
|
776
|
-
get(key: string): ExecResult | null;
|
|
777
|
-
set(key: string, value: ExecResult): void;
|
|
778
|
-
save(): void;
|
|
779
|
-
size(): number;
|
|
780
|
-
}
|
|
781
|
-
export interface ReportStore {
|
|
782
|
-
list(): Promise<Report[]>;
|
|
783
|
-
get(id: string): Promise<Report | null>;
|
|
784
|
-
save(id: string, report: Report): Promise<void>;
|
|
785
|
-
update(id: string, mutator: (report: Report) => void): Promise<Report | null>;
|
|
786
|
-
remove(id: string): Promise<boolean>;
|
|
787
|
-
exists(id: string): Promise<boolean>;
|
|
788
|
-
findByVariant(variantName: string): Promise<Report[]>;
|
|
789
|
-
findByArtifactHash(hash: string): Promise<Report[]>;
|
|
790
|
-
}
|
|
791
|
-
export interface JobStore {
|
|
792
|
-
list(): Promise<EvaluationJob[]>;
|
|
793
|
-
get(id: string): Promise<EvaluationJob | null>;
|
|
794
|
-
save(id: string, job: EvaluationJob): Promise<void>;
|
|
795
|
-
update(id: string, mutator: (job: EvaluationJob) => EvaluationJob): Promise<EvaluationJob | null>;
|
|
796
|
-
remove(id: string): Promise<boolean>;
|
|
797
|
-
exists(id: string): Promise<boolean>;
|
|
798
|
-
}
|
|
799
|
-
export type Lang = 'zh' | 'en';
|
|
1
|
+
export * from './types/index.js';
|
|
800
2
|
//# sourceMappingURL=types.d.ts.map
|