oh-my-knowledge 0.20.0 → 0.20.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/src/renderer/html-renderer.d.ts.map +1 -1
- package/dist/src/renderer/html-renderer.js +43 -2
- package/dist/src/renderer/html-renderer.js.map +1 -1
- package/dist/src/renderer/layout.d.ts.map +1 -1
- package/dist/src/renderer/layout.js +92 -0
- package/dist/src/renderer/layout.js.map +1 -1
- package/dist/src/renderer/summary.d.ts +3 -0
- package/dist/src/renderer/summary.d.ts.map +1 -1
- package/dist/src/renderer/summary.js +98 -39
- package/dist/src/renderer/summary.js.map +1 -1
- package/dist/src/types/eval.d.ts +239 -0
- package/dist/src/types/eval.d.ts.map +1 -0
- package/dist/src/types/eval.js +2 -0
- package/dist/src/types/eval.js.map +1 -0
- package/dist/src/types/executor.d.ts +65 -0
- package/dist/src/types/executor.d.ts.map +1 -0
- package/dist/src/types/executor.js +2 -0
- package/dist/src/types/executor.js.map +1 -0
- package/dist/src/types/index.d.ts +7 -0
- package/dist/src/types/index.d.ts.map +1 -0
- package/dist/src/types/index.js +7 -0
- package/dist/src/types/index.js.map +1 -0
- package/dist/src/types/judge.d.ts +98 -0
- package/dist/src/types/judge.d.ts.map +1 -0
- package/dist/src/types/judge.js +2 -0
- package/dist/src/types/judge.js.map +1 -0
- package/dist/src/types/report.d.ts +390 -0
- package/dist/src/types/report.d.ts.map +1 -0
- package/dist/src/types/report.js +2 -0
- package/dist/src/types/report.js.map +1 -0
- package/dist/src/types/shared.d.ts +2 -0
- package/dist/src/types/shared.d.ts.map +1 -0
- package/dist/src/types/shared.js +2 -0
- package/dist/src/types/shared.js.map +1 -0
- package/dist/src/types/storage.d.ts +21 -0
- package/dist/src/types/storage.d.ts.map +1 -0
- package/dist/src/types/storage.js +2 -0
- package/dist/src/types/storage.js.map +1 -0
- package/dist/src/types.d.ts +1 -799
- package/dist/src/types.d.ts.map +1 -1
- package/dist/src/types.js +5 -1
- package/dist/src/types.js.map +1 -1
- package/package.json +1 -1
|
@@ -0,0 +1,98 @@
|
|
|
1
|
+
/** Single judge configuration: which executor to call and which model alias to pass. */
|
|
2
|
+
export interface JudgeConfig {
|
|
3
|
+
/** Executor name (claude / openai / gemini / anthropic-api / openai-api / shell command). */
|
|
4
|
+
executor: string;
|
|
5
|
+
/** Model alias passed to the executor (e.g. "opus", "haiku", "gpt-4o", "gemini-2.0-pro"). */
|
|
6
|
+
model: string;
|
|
7
|
+
}
|
|
8
|
+
/** Per-judge ensemble entry: which judge gave what score (mean over judge-repeat if N>1). */
|
|
9
|
+
export interface EnsembleJudgeResult {
|
|
10
|
+
/** "executor:model" identifier — e.g. "claude:opus" or "openai:gpt-4o". */
|
|
11
|
+
judge: string;
|
|
12
|
+
/** Mean score from this judge over judge-repeat calls (or single score if repeat=1). */
|
|
13
|
+
score: number;
|
|
14
|
+
/** Stddev across judge-repeat calls for this judge (0 if repeat=1). */
|
|
15
|
+
scoreStddev?: number;
|
|
16
|
+
/** Raw scores per call (length = judgeRepeat). */
|
|
17
|
+
scoreSamples?: number[];
|
|
18
|
+
/** How many of judgeRepeat calls failed (returned score=0). */
|
|
19
|
+
judgeFailureCount?: number;
|
|
20
|
+
/** First-call CoT reasoning from this judge. */
|
|
21
|
+
reasoning?: string;
|
|
22
|
+
/** Cost in USD across all calls from this judge. */
|
|
23
|
+
costUSD?: number;
|
|
24
|
+
}
|
|
25
|
+
/** Inter-judge agreement metrics across an ensemble. Both metrics are pairwise-averaged. */
|
|
26
|
+
export interface JudgeAgreement {
|
|
27
|
+
/** Pairwise Pearson correlation, averaged. 1 = judges fully agree on rank order; 0 = no
|
|
28
|
+
* correlation; -1 = anti-correlated. Note: only defined when at least one judge has
|
|
29
|
+
* variance (constant-score judges produce undefined Pearson). */
|
|
30
|
+
pearson?: number;
|
|
31
|
+
/** Pairwise mean absolute difference of scores. 0 = identical scores. On a 1-5 scale
|
|
32
|
+
* values < 0.5 are tight agreement, > 1.5 is large disagreement. */
|
|
33
|
+
meanAbsDiff: number;
|
|
34
|
+
/** Number of judge pairs the metrics were computed over (= n*(n-1)/2). */
|
|
35
|
+
pairCount: number;
|
|
36
|
+
}
|
|
37
|
+
export interface AssertionDetail {
|
|
38
|
+
type: string;
|
|
39
|
+
value: string | number;
|
|
40
|
+
weight: number;
|
|
41
|
+
passed: boolean;
|
|
42
|
+
message?: string;
|
|
43
|
+
}
|
|
44
|
+
export interface AssertionResults {
|
|
45
|
+
passed: number;
|
|
46
|
+
total: number;
|
|
47
|
+
score: number;
|
|
48
|
+
details: AssertionDetail[];
|
|
49
|
+
judgeCostUSD?: number;
|
|
50
|
+
}
|
|
51
|
+
export interface DimensionResult {
|
|
52
|
+
score: number;
|
|
53
|
+
reason: string;
|
|
54
|
+
judgeCostUSD?: number;
|
|
55
|
+
/** When judge-repeat > 1: scores from each judge run (length = repeat count). */
|
|
56
|
+
scoreSamples?: number[];
|
|
57
|
+
/** Standard deviation across scoreSamples (0 when repeat = 1). */
|
|
58
|
+
scoreStddev?: number;
|
|
59
|
+
/** Chain-of-thought reasoning produced by the judge before the final score. */
|
|
60
|
+
reasoning?: string;
|
|
61
|
+
/**
|
|
62
|
+
* Number of judge calls that failed (returned score=0 / non-JSON / executor error).
|
|
63
|
+
* Stddev = 0 + judgeFailureCount > 0 means "looks consistent but actually had failures",
|
|
64
|
+
* NOT "judge agreed perfectly". Always check this before trusting low stddev.
|
|
65
|
+
*/
|
|
66
|
+
judgeFailureCount?: number;
|
|
67
|
+
/** Multi-judge ensemble: per-judge results when judgeModels.length >= 2. */
|
|
68
|
+
ensemble?: EnsembleJudgeResult[];
|
|
69
|
+
/** Multi-judge ensemble: inter-judge agreement metrics. */
|
|
70
|
+
agreement?: JudgeAgreement;
|
|
71
|
+
}
|
|
72
|
+
export interface LayeredScores {
|
|
73
|
+
factScore?: number;
|
|
74
|
+
behaviorScore?: number;
|
|
75
|
+
judgeScore?: number;
|
|
76
|
+
}
|
|
77
|
+
export interface GradeResult {
|
|
78
|
+
compositeScore: number;
|
|
79
|
+
layeredScores?: LayeredScores;
|
|
80
|
+
assertions?: AssertionResults;
|
|
81
|
+
llmScore?: number;
|
|
82
|
+
llmReason?: string;
|
|
83
|
+
/** Single-rubric mode: judge's chain-of-thought reasoning (first call when judgeRepeat > 1). */
|
|
84
|
+
llmReasoning?: string;
|
|
85
|
+
/** When judge-repeat > 1 with single rubric: stddev across N judge calls. */
|
|
86
|
+
llmScoreStddev?: number;
|
|
87
|
+
/** When judge-repeat > 1 with single rubric: raw scores from each judge call. */
|
|
88
|
+
llmScoreSamples?: number[];
|
|
89
|
+
/** When judge-repeat > 1 with single rubric: how many of the N judge calls failed. */
|
|
90
|
+
llmScoreFailures?: number;
|
|
91
|
+
/** Multi-judge ensemble (single rubric): per-judge results when judgeModels.length >= 2. */
|
|
92
|
+
llmEnsemble?: EnsembleJudgeResult[];
|
|
93
|
+
/** Multi-judge ensemble (single rubric): inter-judge agreement metrics. */
|
|
94
|
+
llmAgreement?: JudgeAgreement;
|
|
95
|
+
dimensions?: Record<string, DimensionResult>;
|
|
96
|
+
judgeCostUSD?: number;
|
|
97
|
+
}
|
|
98
|
+
//# sourceMappingURL=judge.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"judge.d.ts","sourceRoot":"","sources":["../../../src/types/judge.ts"],"names":[],"mappings":"AAAA,wFAAwF;AACxF,MAAM,WAAW,WAAW;IAC1B,6FAA6F;IAC7F,QAAQ,EAAE,MAAM,CAAC;IACjB,6FAA6F;IAC7F,KAAK,EAAE,MAAM,CAAC;CACf;AAED,6FAA6F;AAC7F,MAAM,WAAW,mBAAmB;IAClC,2EAA2E;IAC3E,KAAK,EAAE,MAAM,CAAC;IACd,wFAAwF;IACxF,KAAK,EAAE,MAAM,CAAC;IACd,uEAAuE;IACvE,WAAW,CAAC,EAAE,MAAM,CAAC;IACrB,kDAAkD;IAClD,YAAY,CAAC,EAAE,MAAM,EAAE,CAAC;IACxB,+DAA+D;IAC/D,iBAAiB,CAAC,EAAE,MAAM,CAAC;IAC3B,gDAAgD;IAChD,SAAS,CAAC,EAAE,MAAM,CAAC;IACnB,oDAAoD;IACpD,OAAO,CAAC,EAAE,MAAM,CAAC;CAClB;AAED,4FAA4F;AAC5F,MAAM,WAAW,cAAc;IAC7B;;sEAEkE;IAClE,OAAO,CAAC,EAAE,MAAM,CAAC;IACjB;yEACqE;IACrE,WAAW,EAAE,MAAM,CAAC;IACpB,0EAA0E;IAC1E,SAAS,EAAE,MAAM,CAAC;CACnB;AAED,MAAM,WAAW,eAAe;IAC9B,IAAI,EAAE,MAAM,CAAC;IACb,KAAK,EAAE,MAAM,GAAG,MAAM,CAAC;IACvB,MAAM,EAAE,MAAM,CAAC;IACf,MAAM,EAAE,OAAO,CAAC;IAChB,OAAO,CAAC,EAAE,MAAM,CAAC;CAClB;AAED,MAAM,WAAW,gBAAgB;IAC/B,MAAM,EAAE,MAAM,CAAC;IACf,KAAK,EAAE,MAAM,CAAC;IACd,KAAK,EAAE,MAAM,CAAC;IACd,OAAO,EAAE,eAAe,EAAE,CAAC;IAC3B,YAAY,CAAC,EAAE,MAAM,CAAC;CACvB;AAED,MAAM,WAAW,eAAe;IAC9B,KAAK,EAAE,MAAM,CAAC;IACd,MAAM,EAAE,MAAM,CAAC;IACf,YAAY,CAAC,EAAE,MAAM,CAAC;IACtB,iFAAiF;IACjF,YAAY,CAAC,EAAE,MAAM,EAAE,CAAC;IACxB,kEAAkE;IAClE,WAAW,CAAC,EAAE,MAAM,CAAC;IACrB,+EAA+E;IAC/E,SAAS,CAAC,EAAE,MAAM,CAAC;IACnB;;;;OAIG;IACH,iBAAiB,CAAC,EAAE,MAAM,CAAC;IAC3B,4EAA4E;IAC5E,QAAQ,CAAC,EAAE,mBAAmB,EAAE,CAAC;IACjC,2DAA2D;IAC3D,SAAS,CAAC,EAAE,cAAc,CAAC;CAC5B;AAED,MAAM,WAAW,aAAa;IAC5B,SAAS,CAAC,EAAE,MAAM,CAAC;IACnB,aAAa,CAAC,EAAE,MAAM,CAAC;IACvB,UAAU,CAAC,EAAE,MAAM,CAAC;CACrB;AAED,MAAM,WAAW,WAAW;IAC1B,cAAc,EAAE,MAAM,CAAC;IACvB,aAAa,CAAC,EAAE,aAAa,CAAC;IAC9B,UAAU,CAAC,EAAE,gBAAgB,CAAC;IAC9B,QAAQ,CAAC,EAAE,MAAM,CAAC;IAClB,SAAS,CAAC,EAAE,MAAM,CAAC;IACnB,gGAAgG;IAChG,YAAY,CAAC,EAAE,MAAM,CAAC;IACtB,6EAA6E;IAC7E,cAAc,CAAC,EAAE,MAAM,CAAC;IACxB,iFAAiF;IACjF,eAAe,CAAC,EAAE,MAAM,EAAE,CAAC;IAC3B,sFAAsF;IACtF,gBAAgB,CAAC,EAAE,MAAM,CAAC;IAC1B,4FAA4F;IAC5F,WAAW,CAAC,EAAE,mBAAmB,EAAE,CAAC;IACpC,2EAA2E;IAC3E,YAAY,CAAC,EAAE,cAAc,CAAC;IAC9B,UAAU,CAAC,EAAE,MAAM,CAAC,MAAM,EAAE,eAAe,CAAC,CAAC;IAC7C,YAAY,CAAC,EAAE,MAAM,CAAC;CACvB"}
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"judge.js","sourceRoot":"","sources":["../../../src/types/judge.ts"],"names":[],"mappings":""}
|
|
@@ -0,0 +1,390 @@
|
|
|
1
|
+
import type { ToolCallInfo, TurnInfo } from './executor.js';
|
|
2
|
+
import type { AssertionResults, DimensionResult, EnsembleJudgeResult, JudgeAgreement, LayeredScores } from './judge.js';
|
|
3
|
+
import type { EvalBudget, EvaluationJob, EvaluationRequest, EvaluationRun, VariantConfig } from './eval.js';
|
|
4
|
+
export interface VariantResult {
|
|
5
|
+
ok: boolean;
|
|
6
|
+
durationMs: number;
|
|
7
|
+
durationApiMs: number;
|
|
8
|
+
inputTokens: number;
|
|
9
|
+
outputTokens: number;
|
|
10
|
+
totalTokens: number;
|
|
11
|
+
cacheReadTokens: number;
|
|
12
|
+
cacheCreationTokens: number;
|
|
13
|
+
execCostUSD: number;
|
|
14
|
+
judgeCostUSD: number;
|
|
15
|
+
costUSD: number;
|
|
16
|
+
numTurns: number;
|
|
17
|
+
fullNumTurns?: number;
|
|
18
|
+
numSubAgents?: number;
|
|
19
|
+
assistantTurns?: number;
|
|
20
|
+
toolTurns?: number;
|
|
21
|
+
numToolCalls?: number;
|
|
22
|
+
numToolFailures?: number;
|
|
23
|
+
toolSuccessRate?: number;
|
|
24
|
+
toolNames?: string[];
|
|
25
|
+
traceCoverage?: number;
|
|
26
|
+
error?: string;
|
|
27
|
+
compositeScore?: number;
|
|
28
|
+
layeredScores?: LayeredScores;
|
|
29
|
+
assertions?: AssertionResults;
|
|
30
|
+
llmScore?: number;
|
|
31
|
+
llmReason?: string;
|
|
32
|
+
/** Single-rubric mode: judge's chain-of-thought reasoning (first call when judgeRepeat > 1). */
|
|
33
|
+
llmReasoning?: string;
|
|
34
|
+
/** Single-rubric mode + judgeRepeat > 1: stddev across N judge calls. */
|
|
35
|
+
llmScoreStddev?: number;
|
|
36
|
+
/** Single-rubric mode + judgeRepeat > 1: raw scores from each call. */
|
|
37
|
+
llmScoreSamples?: number[];
|
|
38
|
+
/** Single-rubric mode + judgeRepeat > 1: how many of N calls failed. */
|
|
39
|
+
llmScoreFailures?: number;
|
|
40
|
+
/** Single-rubric mode + judgeModels.length >= 2: per-judge ensemble results. */
|
|
41
|
+
llmEnsemble?: EnsembleJudgeResult[];
|
|
42
|
+
/** Single-rubric mode + judgeModels.length >= 2: inter-judge agreement metrics. */
|
|
43
|
+
llmAgreement?: JudgeAgreement;
|
|
44
|
+
dimensions?: Record<string, DimensionResult>;
|
|
45
|
+
factCheck?: {
|
|
46
|
+
verifiedCount: number;
|
|
47
|
+
totalCount: number;
|
|
48
|
+
verifiedRate: number;
|
|
49
|
+
claims: Array<{
|
|
50
|
+
type: string;
|
|
51
|
+
value: string;
|
|
52
|
+
verified: boolean;
|
|
53
|
+
evidence?: string;
|
|
54
|
+
}>;
|
|
55
|
+
};
|
|
56
|
+
outputPreview: string | null;
|
|
57
|
+
fullOutput?: string;
|
|
58
|
+
turns?: TurnInfo[];
|
|
59
|
+
toolCalls?: ToolCallInfo[];
|
|
60
|
+
timing?: {
|
|
61
|
+
execMs: number;
|
|
62
|
+
gradeMs: number;
|
|
63
|
+
totalMs: number;
|
|
64
|
+
};
|
|
65
|
+
}
|
|
66
|
+
export interface VariantSummary {
|
|
67
|
+
totalSamples: number;
|
|
68
|
+
successCount: number;
|
|
69
|
+
errorCount: number;
|
|
70
|
+
errorRate: number;
|
|
71
|
+
avgDurationMs: number;
|
|
72
|
+
avgInputTokens: number;
|
|
73
|
+
avgOutputTokens: number;
|
|
74
|
+
avgTotalTokens: number;
|
|
75
|
+
totalCostUSD: number;
|
|
76
|
+
totalExecCostUSD: number;
|
|
77
|
+
totalJudgeCostUSD: number;
|
|
78
|
+
avgCostPerSample: number;
|
|
79
|
+
avgNumTurns: number;
|
|
80
|
+
avgFullNumTurns?: number;
|
|
81
|
+
avgNumSubAgents?: number;
|
|
82
|
+
avgAssistantTurns?: number;
|
|
83
|
+
avgToolTurns?: number;
|
|
84
|
+
avgToolCalls?: number;
|
|
85
|
+
avgToolFailures?: number;
|
|
86
|
+
toolSuccessRate?: number;
|
|
87
|
+
toolDistribution?: Record<string, number>;
|
|
88
|
+
traceCoverageRate?: number;
|
|
89
|
+
avgFactScore?: number;
|
|
90
|
+
avgFactVerifiedRate?: number;
|
|
91
|
+
avgBehaviorScore?: number;
|
|
92
|
+
avgJudgeScore?: number;
|
|
93
|
+
avgCompositeScore?: number;
|
|
94
|
+
minCompositeScore?: number;
|
|
95
|
+
maxCompositeScore?: number;
|
|
96
|
+
scoreStddev?: number;
|
|
97
|
+
scoreCV?: number;
|
|
98
|
+
avgAssertionScore?: number;
|
|
99
|
+
avgLlmScore?: number;
|
|
100
|
+
minLlmScore?: number;
|
|
101
|
+
maxLlmScore?: number;
|
|
102
|
+
/** Aggregate-level multi-judge agreement across this variant's samples (single rubric mode).
|
|
103
|
+
* sampleCount = how many samples had complete ensemble data. */
|
|
104
|
+
judgeAgreement?: JudgeAgreement & {
|
|
105
|
+
sampleCount: number;
|
|
106
|
+
};
|
|
107
|
+
/** List of judge identifiers ("executor:model") seen in this variant's ensemble data. */
|
|
108
|
+
judgeModels?: string[];
|
|
109
|
+
/** Bootstrap CI on this variant's compositeScore mean (when --bootstrap enabled).
|
|
110
|
+
* Distribution-free; preferred over t-interval for ordinal LLM scores. */
|
|
111
|
+
bootstrapCI?: {
|
|
112
|
+
low: number;
|
|
113
|
+
high: number;
|
|
114
|
+
estimate: number;
|
|
115
|
+
samples: number;
|
|
116
|
+
};
|
|
117
|
+
}
|
|
118
|
+
/**
|
|
119
|
+
* Pairwise variant comparison stats — used when comparing treatment vs control.
|
|
120
|
+
* Independent from per-variant `bootstrapCI` (which is on each variant alone).
|
|
121
|
+
*/
|
|
122
|
+
export interface VariantPairComparison {
|
|
123
|
+
/** Control variant name (the subtrahend). */
|
|
124
|
+
control: string;
|
|
125
|
+
/** Treatment variant name (the minuend). */
|
|
126
|
+
treatment: string;
|
|
127
|
+
/** Bootstrap CI on (treatment - control) mean diff. `significant` = 0 outside CI. */
|
|
128
|
+
diffBootstrapCI?: {
|
|
129
|
+
low: number;
|
|
130
|
+
high: number;
|
|
131
|
+
estimate: number;
|
|
132
|
+
samples: number;
|
|
133
|
+
significant: boolean;
|
|
134
|
+
};
|
|
135
|
+
}
|
|
136
|
+
export interface GitInfo {
|
|
137
|
+
commit: string;
|
|
138
|
+
commitShort: string;
|
|
139
|
+
branch: string;
|
|
140
|
+
dirty: boolean;
|
|
141
|
+
}
|
|
142
|
+
/** Persisted form of agreement metrics between gold dataset and the LLM judge.
|
|
143
|
+
* Lives on ReportMeta so the renderer can show a "人工锚点" section without
|
|
144
|
+
* re-loading the gold dataset. */
|
|
145
|
+
export interface ReportHumanAgreement {
|
|
146
|
+
/** Krippendorff α (interval weights) — primary metric. */
|
|
147
|
+
alpha: number;
|
|
148
|
+
/** Bootstrap 95% CI on α. */
|
|
149
|
+
alphaCI: {
|
|
150
|
+
low: number;
|
|
151
|
+
high: number;
|
|
152
|
+
estimate: number;
|
|
153
|
+
samples: number;
|
|
154
|
+
};
|
|
155
|
+
/** Quadratic-weighted κ — secondary metric. */
|
|
156
|
+
weightedKappa: number;
|
|
157
|
+
/** Pearson r — tertiary, rank-order only. */
|
|
158
|
+
pearson: number;
|
|
159
|
+
/** Number of (gold, judge) pairs that contributed. */
|
|
160
|
+
sampleCount: number;
|
|
161
|
+
/** Variant whose judge scores were compared. */
|
|
162
|
+
variant: string;
|
|
163
|
+
/** Identifier of the gold annotator (model id, person, or team handle). */
|
|
164
|
+
goldAnnotator: string;
|
|
165
|
+
/** Free-form version string from the gold metadata. */
|
|
166
|
+
goldVersion: string;
|
|
167
|
+
/** Set when annotator id overlapped with judge model id. */
|
|
168
|
+
contaminationWarning?: string;
|
|
169
|
+
/** Sample_ids in the gold set that were absent from the report. */
|
|
170
|
+
missingCount: number;
|
|
171
|
+
/** Sample_ids present in the report but with no judge score (assertion-only etc). */
|
|
172
|
+
unscoredCount: number;
|
|
173
|
+
}
|
|
174
|
+
export interface ReportMeta {
|
|
175
|
+
variants: string[];
|
|
176
|
+
model: string;
|
|
177
|
+
judgeModel: string | null;
|
|
178
|
+
executor: string;
|
|
179
|
+
sampleCount: number;
|
|
180
|
+
taskCount: number;
|
|
181
|
+
totalCostUSD: number;
|
|
182
|
+
timestamp: string;
|
|
183
|
+
cliVersion: string;
|
|
184
|
+
nodeVersion: string;
|
|
185
|
+
artifactHashes: Record<string, string>;
|
|
186
|
+
/** v0.21 — Report JSON schema version. Reports without this field are treated as v0
|
|
187
|
+
* (legacy field semantics: pre-v0.21 `gapRate`/`weightedGapRate` map to `evalGapRate`/
|
|
188
|
+
* `evalWeightedGapRate` for eval-side reports). v0.21+ writes 1. */
|
|
189
|
+
schemaVersion?: number;
|
|
190
|
+
/** SHA256-12 of every sample's content (sample_id → hash). Same hash = same sample. */
|
|
191
|
+
sampleHashes?: Record<string, string>;
|
|
192
|
+
/** SHA256-12 of the LLM judge prompt template. Different hash = judge changed semantics. */
|
|
193
|
+
judgePromptHash?: string;
|
|
194
|
+
/** Number of times each sample was judged. 1 = single judge (default). */
|
|
195
|
+
judgeRepeat?: number;
|
|
196
|
+
/** Multi-judge ensemble configuration: ["claude:opus", "openai:gpt-4o", ...].
|
|
197
|
+
* When length >= 2, every (sample × dimension) is scored by all judges and
|
|
198
|
+
* agreement metrics are reported per-result. */
|
|
199
|
+
judgeModels?: string[];
|
|
200
|
+
/** Which CI framework was used for this report: 't-test' (legacy default),
|
|
201
|
+
* 'bootstrap' (--bootstrap), or 'both' (some summaries have both). Reports
|
|
202
|
+
* with mismatched frameworks shouldn't be compared blindly on CI bounds. */
|
|
203
|
+
evaluationFramework?: 't-test' | 'bootstrap' | 'both';
|
|
204
|
+
/** Pairwise comparisons (treatment vs control) — populated when --bootstrap and
|
|
205
|
+
* multi-variant. Length = (variants.length - 1). */
|
|
206
|
+
pairComparisons?: VariantPairComparison[];
|
|
207
|
+
/** v0.21 Phase 3 — which judge-bias debias modes were active for this run.
|
|
208
|
+
* Values: 'length' (substance-not-length prompt), 'position' (random ensemble
|
|
209
|
+
* order). Empty / absent means legacy default (no debias). The renderer shows
|
|
210
|
+
* this so readers can tell apples from oranges across reports. */
|
|
211
|
+
debiasMode?: Array<'length' | 'position'>;
|
|
212
|
+
/** v0.22 — set to true when the run was aborted by a budget tracker. The
|
|
213
|
+
* report is partial: only tasks completed before the abort are present. */
|
|
214
|
+
budgetExhausted?: boolean;
|
|
215
|
+
/** v0.22 — budget caps that were active for this run, copied from request.budget
|
|
216
|
+
* for ease of reading without dereferencing request. */
|
|
217
|
+
budget?: EvalBudget;
|
|
218
|
+
/** Human-gold agreement when --gold-dir was passed at run time. Compares the
|
|
219
|
+
* judge's llmScore against the gold annotations on matching sample_ids. See
|
|
220
|
+
* src/grading/human-gold.ts for the metric definitions. */
|
|
221
|
+
humanAgreement?: ReportHumanAgreement;
|
|
222
|
+
variantConfigs?: VariantConfig[];
|
|
223
|
+
request?: EvaluationRequest;
|
|
224
|
+
run?: EvaluationRun;
|
|
225
|
+
job?: EvaluationJob;
|
|
226
|
+
gitInfo?: GitInfo | null;
|
|
227
|
+
blind?: boolean;
|
|
228
|
+
blindMap?: Record<string, string>;
|
|
229
|
+
layeredStats?: boolean;
|
|
230
|
+
}
|
|
231
|
+
export interface ResultEntry {
|
|
232
|
+
sample_id: string;
|
|
233
|
+
variants: Record<string, VariantResult>;
|
|
234
|
+
}
|
|
235
|
+
export interface Report {
|
|
236
|
+
id: string;
|
|
237
|
+
meta: ReportMeta;
|
|
238
|
+
summary: Record<string, VariantSummary>;
|
|
239
|
+
results: ResultEntry[];
|
|
240
|
+
analysis?: AnalysisResult;
|
|
241
|
+
variance?: VarianceData;
|
|
242
|
+
each?: boolean;
|
|
243
|
+
overview?: {
|
|
244
|
+
totalArtifacts: number;
|
|
245
|
+
totalSamples: number;
|
|
246
|
+
totalCostUSD: number;
|
|
247
|
+
artifacts: Array<{
|
|
248
|
+
name: string;
|
|
249
|
+
baselineScore: number | null;
|
|
250
|
+
artifactScore: number | null;
|
|
251
|
+
improvement: string;
|
|
252
|
+
}>;
|
|
253
|
+
};
|
|
254
|
+
artifacts?: Array<{
|
|
255
|
+
name: string;
|
|
256
|
+
sampleCount: number;
|
|
257
|
+
artifactHash: string | null;
|
|
258
|
+
summary: Record<string, VariantSummary>;
|
|
259
|
+
/** --each --repeat N 时由 runMultiple 聚合的三层独立 variance + t 检验 */
|
|
260
|
+
variance?: VarianceData;
|
|
261
|
+
results: ResultEntry[];
|
|
262
|
+
}>;
|
|
263
|
+
}
|
|
264
|
+
export interface Insight {
|
|
265
|
+
type: string;
|
|
266
|
+
severity: 'error' | 'warning' | 'info';
|
|
267
|
+
message: string;
|
|
268
|
+
details: unknown;
|
|
269
|
+
}
|
|
270
|
+
export interface KnowledgeCoverageEntry {
|
|
271
|
+
path: string;
|
|
272
|
+
type: string;
|
|
273
|
+
accessed: boolean;
|
|
274
|
+
accessCount: number;
|
|
275
|
+
lineCount?: number;
|
|
276
|
+
}
|
|
277
|
+
export interface KnowledgeCoverage {
|
|
278
|
+
entries: KnowledgeCoverageEntry[];
|
|
279
|
+
filesCovered: number;
|
|
280
|
+
filesTotal: number;
|
|
281
|
+
fileCoverageRate: number;
|
|
282
|
+
uncoveredFiles: string[];
|
|
283
|
+
grepPatternsUsed: number;
|
|
284
|
+
overallRate: number;
|
|
285
|
+
}
|
|
286
|
+
export interface AnalysisResult {
|
|
287
|
+
summary?: string;
|
|
288
|
+
insights: Insight[];
|
|
289
|
+
suggestions: string[];
|
|
290
|
+
coverage?: Record<string, KnowledgeCoverage>;
|
|
291
|
+
/** Per-variant knowledge gap reports. See docs/knowledge-gap-signal-spec.md */
|
|
292
|
+
gapReports?: Record<string, GapReport>;
|
|
293
|
+
}
|
|
294
|
+
export interface HedgingVerdict {
|
|
295
|
+
isUncertainty: boolean;
|
|
296
|
+
confidence: number;
|
|
297
|
+
reason: string;
|
|
298
|
+
}
|
|
299
|
+
export interface GapSignalRef {
|
|
300
|
+
sampleId: string;
|
|
301
|
+
type: 'failed_search' | 'explicit_marker' | 'hedging' | 'repeated_failure';
|
|
302
|
+
turn?: number;
|
|
303
|
+
context: string;
|
|
304
|
+
evidence?: Record<string, unknown>;
|
|
305
|
+
weight: number;
|
|
306
|
+
classifierVerdict?: HedgingVerdict;
|
|
307
|
+
}
|
|
308
|
+
export interface GapReport {
|
|
309
|
+
variant: string;
|
|
310
|
+
sampleCount: number;
|
|
311
|
+
samplesWithGap: number;
|
|
312
|
+
gapRate: number;
|
|
313
|
+
weightedGapRate: number;
|
|
314
|
+
testSetPath?: string | null;
|
|
315
|
+
testSetHash?: string | null;
|
|
316
|
+
signals: GapSignalRef[];
|
|
317
|
+
byType: {
|
|
318
|
+
failed_search: number;
|
|
319
|
+
explicit_marker: number;
|
|
320
|
+
hedging: number;
|
|
321
|
+
repeated_failure: number;
|
|
322
|
+
};
|
|
323
|
+
}
|
|
324
|
+
export interface VarianceEffectSize {
|
|
325
|
+
cohensD: number;
|
|
326
|
+
hedgesG: number;
|
|
327
|
+
primary: 'd' | 'g' | 'none';
|
|
328
|
+
magnitude: 'negligible' | 'small' | 'medium' | 'large' | 'none';
|
|
329
|
+
pooledStddev: number;
|
|
330
|
+
n1: number;
|
|
331
|
+
n2: number;
|
|
332
|
+
}
|
|
333
|
+
export interface VarianceMetric {
|
|
334
|
+
scores: number[];
|
|
335
|
+
mean: number;
|
|
336
|
+
lower: number;
|
|
337
|
+
upper: number;
|
|
338
|
+
stddev: number;
|
|
339
|
+
}
|
|
340
|
+
export interface VarianceComparisonMetric {
|
|
341
|
+
meanDiff: number;
|
|
342
|
+
tStatistic: number;
|
|
343
|
+
df: number;
|
|
344
|
+
significant: boolean;
|
|
345
|
+
effectSize: VarianceEffectSize;
|
|
346
|
+
}
|
|
347
|
+
export type VarianceMetricKey = 'cost' | 'efficiency';
|
|
348
|
+
export type VarianceLayerKey = 'fact' | 'behavior' | 'judge';
|
|
349
|
+
export interface VariantVariance extends VarianceMetric {
|
|
350
|
+
byMetric?: Partial<Record<VarianceMetricKey, VarianceMetric>>;
|
|
351
|
+
byLayer?: Partial<Record<VarianceLayerKey, VarianceMetric>>;
|
|
352
|
+
}
|
|
353
|
+
export interface VarianceComparison extends VarianceComparisonMetric {
|
|
354
|
+
a: string;
|
|
355
|
+
b: string;
|
|
356
|
+
byMetric?: Partial<Record<VarianceMetricKey, VarianceComparisonMetric>>;
|
|
357
|
+
byLayer?: Partial<Record<VarianceLayerKey, VarianceComparisonMetric>>;
|
|
358
|
+
}
|
|
359
|
+
export interface VarianceData {
|
|
360
|
+
runs: number;
|
|
361
|
+
perVariant: Record<string, VariantVariance>;
|
|
362
|
+
comparisons: VarianceComparison[];
|
|
363
|
+
/** v0.21 Phase 4 — saturation curve data. Populated only when repeat ≥ 2.
|
|
364
|
+
* Per-variant cumulative score arrays at each repeat checkpoint, plus the
|
|
365
|
+
* saturation verdict (only computed when repeat ≥ 5). */
|
|
366
|
+
saturation?: SaturationData;
|
|
367
|
+
}
|
|
368
|
+
/** Per-variant saturation curve data + (optionally) verdict. */
|
|
369
|
+
export interface SaturationData {
|
|
370
|
+
/** Cumulative checkpoint counts (sample-cumulative across runs). */
|
|
371
|
+
checkpointSampleCounts: number[];
|
|
372
|
+
/** Per-variant trace: at each checkpoint, mean and CI bounds.
|
|
373
|
+
* perVariant[variant][i] = { n, mean, ciLow, ciHigh } at checkpoint i. */
|
|
374
|
+
perVariant: Record<string, Array<{
|
|
375
|
+
n: number;
|
|
376
|
+
mean: number;
|
|
377
|
+
ciLow: number;
|
|
378
|
+
ciHigh: number;
|
|
379
|
+
}>>;
|
|
380
|
+
/** Saturation verdict per variant. Only present when repeat ≥ 5. */
|
|
381
|
+
verdicts?: Record<string, {
|
|
382
|
+
saturated: boolean;
|
|
383
|
+
atN: number | null;
|
|
384
|
+
confidence: 'high' | 'medium' | 'low';
|
|
385
|
+
method: 'slope' | 'bootstrap-ci-width' | 'plateau-height';
|
|
386
|
+
threshold: number;
|
|
387
|
+
reason: string;
|
|
388
|
+
}>;
|
|
389
|
+
}
|
|
390
|
+
//# sourceMappingURL=report.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"report.d.ts","sourceRoot":"","sources":["../../../src/types/report.ts"],"names":[],"mappings":"AAAA,OAAO,KAAK,EAAE,YAAY,EAAE,QAAQ,EAAE,MAAM,eAAe,CAAC;AAC5D,OAAO,KAAK,EAAE,gBAAgB,EAAE,eAAe,EAAE,mBAAmB,EAAE,cAAc,EAAE,aAAa,EAAE,MAAM,YAAY,CAAC;AACxH,OAAO,KAAK,EAAE,UAAU,EAAE,aAAa,EAAE,iBAAiB,EAAE,aAAa,EAAE,aAAa,EAAE,MAAM,WAAW,CAAC;AAE5G,MAAM,WAAW,aAAa;IAC5B,EAAE,EAAE,OAAO,CAAC;IACZ,UAAU,EAAE,MAAM,CAAC;IACnB,aAAa,EAAE,MAAM,CAAC;IACtB,WAAW,EAAE,MAAM,CAAC;IACpB,YAAY,EAAE,MAAM,CAAC;IACrB,WAAW,EAAE,MAAM,CAAC;IACpB,eAAe,EAAE,MAAM,CAAC;IACxB,mBAAmB,EAAE,MAAM,CAAC;IAC5B,WAAW,EAAE,MAAM,CAAC;IACpB,YAAY,EAAE,MAAM,CAAC;IACrB,OAAO,EAAE,MAAM,CAAC;IAChB,QAAQ,EAAE,MAAM,CAAC;IACjB,YAAY,CAAC,EAAE,MAAM,CAAC;IACtB,YAAY,CAAC,EAAE,MAAM,CAAC;IACtB,cAAc,CAAC,EAAE,MAAM,CAAC;IACxB,SAAS,CAAC,EAAE,MAAM,CAAC;IACnB,YAAY,CAAC,EAAE,MAAM,CAAC;IACtB,eAAe,CAAC,EAAE,MAAM,CAAC;IACzB,eAAe,CAAC,EAAE,MAAM,CAAC;IACzB,SAAS,CAAC,EAAE,MAAM,EAAE,CAAC;IACrB,aAAa,CAAC,EAAE,MAAM,CAAC;IACvB,KAAK,CAAC,EAAE,MAAM,CAAC;IACf,cAAc,CAAC,EAAE,MAAM,CAAC;IACxB,aAAa,CAAC,EAAE,aAAa,CAAC;IAC9B,UAAU,CAAC,EAAE,gBAAgB,CAAC;IAC9B,QAAQ,CAAC,EAAE,MAAM,CAAC;IAClB,SAAS,CAAC,EAAE,MAAM,CAAC;IACnB,gGAAgG;IAChG,YAAY,CAAC,EAAE,MAAM,CAAC;IACtB,yEAAyE;IACzE,cAAc,CAAC,EAAE,MAAM,CAAC;IACxB,uEAAuE;IACvE,eAAe,CAAC,EAAE,MAAM,EAAE,CAAC;IAC3B,wEAAwE;IACxE,gBAAgB,CAAC,EAAE,MAAM,CAAC;IAC1B,gFAAgF;IAChF,WAAW,CAAC,EAAE,mBAAmB,EAAE,CAAC;IACpC,mFAAmF;IACnF,YAAY,CAAC,EAAE,cAAc,CAAC;IAC9B,UAAU,CAAC,EAAE,MAAM,CAAC,MAAM,EAAE,eAAe,CAAC,CAAC;IAC7C,SAAS,CAAC,EAAE;QAAE,aAAa,EAAE,MAAM,CAAC;QAAC,UAAU,EAAE,MAAM,CAAC;QAAC,YAAY,EAAE,MAAM,CAAC;QAAC,MAAM,EAAE,KAAK,CAAC;YAAE,IAAI,EAAE,MAAM,CAAC;YAAC,KAAK,EAAE,MAAM,CAAC;YAAC,QAAQ,EAAE,OAAO,CAAC;YAAC,QAAQ,CAAC,EAAE,MAAM,CAAA;SAAE,CAAC,CAAA;KAAE,CAAC;IACtK,aAAa,EAAE,MAAM,GAAG,IAAI,CAAC;IAC7B,UAAU,CAAC,EAAE,MAAM,CAAC;IACpB,KAAK,CAAC,EAAE,QAAQ,EAAE,CAAC;IACnB,SAAS,CAAC,EAAE,YAAY,EAAE,CAAC;IAC3B,MAAM,CAAC,EAAE;QAAE,MAAM,EAAE,MAAM,CAAC;QAAC,OAAO,EAAE,MAAM,CAAC;QAAC,OAAO,EAAE,MAAM,CAAA;KAAE,CAAC;CAC/D;AAED,MAAM,WAAW,cAAc;IAC7B,YAAY,EAAE,MAAM,CAAC;IACrB,YAAY,EAAE,MAAM,CAAC;IACrB,UAAU,EAAE,MAAM,CAAC;IACnB,SAAS,EAAE,MAAM,CAAC;IAClB,aAAa,EAAE,MAAM,CAAC;IACtB,cAAc,EAAE,MAAM,CAAC;IACvB,eAAe,EAAE,MAAM,CAAC;IACxB,cAAc,EAAE,MAAM,CAAC;IACvB,YAAY,EAAE,MAAM,CAAC;IACrB,gBAAgB,EAAE,MAAM,CAAC;IACzB,iBAAiB,EAAE,MAAM,CAAC;IAC1B,gBAAgB,EAAE,MAAM,CAAC;IACzB,WAAW,EAAE,MAAM,CAAC;IACpB,eAAe,CAAC,EAAE,MAAM,CAAC;IACzB,eAAe,CAAC,EAAE,MAAM,CAAC;IACzB,iBAAiB,CAAC,EAAE,MAAM,CAAC;IAC3B,YAAY,CAAC,EAAE,MAAM,CAAC;IACtB,YAAY,CAAC,EAAE,MAAM,CAAC;IACtB,eAAe,CAAC,EAAE,MAAM,CAAC;IACzB,eAAe,CAAC,EAAE,MAAM,CAAC;IACzB,gBAAgB,CAAC,EAAE,MAAM,CAAC,MAAM,EAAE,MAAM,CAAC,CAAC;IAC1C,iBAAiB,CAAC,EAAE,MAAM,CAAC;IAC3B,YAAY,CAAC,EAAE,MAAM,CAAC;IACtB,mBAAmB,CAAC,EAAE,MAAM,CAAC;IAC7B,gBAAgB,CAAC,EAAE,MAAM,CAAC;IAC1B,aAAa,CAAC,EAAE,MAAM,CAAC;IACvB,iBAAiB,CAAC,EAAE,MAAM,CAAC;IAC3B,iBAAiB,CAAC,EAAE,MAAM,CAAC;IAC3B,iBAAiB,CAAC,EAAE,MAAM,CAAC;IAC3B,WAAW,CAAC,EAAE,MAAM,CAAC;IACrB,OAAO,CAAC,EAAE,MAAM,CAAC;IACjB,iBAAiB,CAAC,EAAE,MAAM,CAAC;IAC3B,WAAW,CAAC,EAAE,MAAM,CAAC;IACrB,WAAW,CAAC,EAAE,MAAM,CAAC;IACrB,WAAW,CAAC,EAAE,MAAM,CAAC;IACrB;qEACiE;IACjE,cAAc,CAAC,EAAE,cAAc,GAAG;QAAE,WAAW,EAAE,MAAM,CAAA;KAAE,CAAC;IAC1D,yFAAyF;IACzF,WAAW,CAAC,EAAE,MAAM,EAAE,CAAC;IACvB;+EAC2E;IAC3E,WAAW,CAAC,EAAE;QAAE,GAAG,EAAE,MAAM,CAAC;QAAC,IAAI,EAAE,MAAM,CAAC;QAAC,QAAQ,EAAE,MAAM,CAAC;QAAC,OAAO,EAAE,MAAM,CAAA;KAAE,CAAC;CAChF;AAED;;;GAGG;AACH,MAAM,WAAW,qBAAqB;IACpC,6CAA6C;IAC7C,OAAO,EAAE,MAAM,CAAC;IAChB,4CAA4C;IAC5C,SAAS,EAAE,MAAM,CAAC;IAClB,qFAAqF;IACrF,eAAe,CAAC,EAAE;QAAE,GAAG,EAAE,MAAM,CAAC;QAAC,IAAI,EAAE,MAAM,CAAC;QAAC,QAAQ,EAAE,MAAM,CAAC;QAAC,OAAO,EAAE,MAAM,CAAC;QAAC,WAAW,EAAE,OAAO,CAAA;KAAE,CAAC;CAC1G;AAED,MAAM,WAAW,OAAO;IACtB,MAAM,EAAE,MAAM,CAAC;IACf,WAAW,EAAE,MAAM,CAAC;IACpB,MAAM,EAAE,MAAM,CAAC;IACf,KAAK,EAAE,OAAO,CAAC;CAChB;AAED;;mCAEmC;AACnC,MAAM,WAAW,oBAAoB;IACnC,0DAA0D;IAC1D,KAAK,EAAE,MAAM,CAAC;IACd,6BAA6B;IAC7B,OAAO,EAAE;QAAE,GAAG,EAAE,MAAM,CAAC;QAAC,IAAI,EAAE,MAAM,CAAC;QAAC,QAAQ,EAAE,MAAM,CAAC;QAAC,OAAO,EAAE,MAAM,CAAA;KAAE,CAAC;IAC1E,+CAA+C;IAC/C,aAAa,EAAE,MAAM,CAAC;IACtB,6CAA6C;IAC7C,OAAO,EAAE,MAAM,CAAC;IAChB,sDAAsD;IACtD,WAAW,EAAE,MAAM,CAAC;IACpB,gDAAgD;IAChD,OAAO,EAAE,MAAM,CAAC;IAChB,2EAA2E;IAC3E,aAAa,EAAE,MAAM,CAAC;IACtB,uDAAuD;IACvD,WAAW,EAAE,MAAM,CAAC;IACpB,4DAA4D;IAC5D,oBAAoB,CAAC,EAAE,MAAM,CAAC;IAC9B,mEAAmE;IACnE,YAAY,EAAE,MAAM,CAAC;IACrB,qFAAqF;IACrF,aAAa,EAAE,MAAM,CAAC;CACvB;AAED,MAAM,WAAW,UAAU;IACzB,QAAQ,EAAE,MAAM,EAAE,CAAC;IACnB,KAAK,EAAE,MAAM,CAAC;IACd,UAAU,EAAE,MAAM,GAAG,IAAI,CAAC;IAC1B,QAAQ,EAAE,MAAM,CAAC;IACjB,WAAW,EAAE,MAAM,CAAC;IACpB,SAAS,EAAE,MAAM,CAAC;IAClB,YAAY,EAAE,MAAM,CAAC;IACrB,SAAS,EAAE,MAAM,CAAC;IAClB,UAAU,EAAE,MAAM,CAAC;IACnB,WAAW,EAAE,MAAM,CAAC;IACpB,cAAc,EAAE,MAAM,CAAC,MAAM,EAAE,MAAM,CAAC,CAAC;IACvC;;yEAEqE;IACrE,aAAa,CAAC,EAAE,MAAM,CAAC;IACvB,uFAAuF;IACvF,YAAY,CAAC,EAAE,MAAM,CAAC,MAAM,EAAE,MAAM,CAAC,CAAC;IACtC,4FAA4F;IAC5F,eAAe,CAAC,EAAE,MAAM,CAAC;IACzB,0EAA0E;IAC1E,WAAW,CAAC,EAAE,MAAM,CAAC;IACrB;;qDAEiD;IACjD,WAAW,CAAC,EAAE,MAAM,EAAE,CAAC;IACvB;;iFAE6E;IAC7E,mBAAmB,CAAC,EAAE,QAAQ,GAAG,WAAW,GAAG,MAAM,CAAC;IACtD;yDACqD;IACrD,eAAe,CAAC,EAAE,qBAAqB,EAAE,CAAC;IAC1C;;;uEAGmE;IACnE,UAAU,CAAC,EAAE,KAAK,CAAC,QAAQ,GAAG,UAAU,CAAC,CAAC;IAC1C;gFAC4E;IAC5E,eAAe,CAAC,EAAE,OAAO,CAAC;IAC1B;6DACyD;IACzD,MAAM,CAAC,EAAE,UAAU,CAAC;IACpB;;gEAE4D;IAC5D,cAAc,CAAC,EAAE,oBAAoB,CAAC;IACtC,cAAc,CAAC,EAAE,aAAa,EAAE,CAAC;IACjC,OAAO,CAAC,EAAE,iBAAiB,CAAC;IAC5B,GAAG,CAAC,EAAE,aAAa,CAAC;IACpB,GAAG,CAAC,EAAE,aAAa,CAAC;IACpB,OAAO,CAAC,EAAE,OAAO,GAAG,IAAI,CAAC;IACzB,KAAK,CAAC,EAAE,OAAO,CAAC;IAChB,QAAQ,CAAC,EAAE,MAAM,CAAC,MAAM,EAAE,MAAM,CAAC,CAAC;IAIlC,YAAY,CAAC,EAAE,OAAO,CAAC;CACxB;AAED,MAAM,WAAW,WAAW;IAC1B,SAAS,EAAE,MAAM,CAAC;IAClB,QAAQ,EAAE,MAAM,CAAC,MAAM,EAAE,aAAa,CAAC,CAAC;CACzC;AAED,MAAM,WAAW,MAAM;IACrB,EAAE,EAAE,MAAM,CAAC;IACX,IAAI,EAAE,UAAU,CAAC;IACjB,OAAO,EAAE,MAAM,CAAC,MAAM,EAAE,cAAc,CAAC,CAAC;IACxC,OAAO,EAAE,WAAW,EAAE,CAAC;IACvB,QAAQ,CAAC,EAAE,cAAc,CAAC;IAC1B,QAAQ,CAAC,EAAE,YAAY,CAAC;IACxB,IAAI,CAAC,EAAE,OAAO,CAAC;IACf,QAAQ,CAAC,EAAE;QACT,cAAc,EAAE,MAAM,CAAC;QACvB,YAAY,EAAE,MAAM,CAAC;QACrB,YAAY,EAAE,MAAM,CAAC;QACrB,SAAS,EAAE,KAAK,CAAC;YACf,IAAI,EAAE,MAAM,CAAC;YACb,aAAa,EAAE,MAAM,GAAG,IAAI,CAAC;YAC7B,aAAa,EAAE,MAAM,GAAG,IAAI,CAAC;YAC7B,WAAW,EAAE,MAAM,CAAC;SACrB,CAAC,CAAC;KACJ,CAAC;IACF,SAAS,CAAC,EAAE,KAAK,CAAC;QAChB,IAAI,EAAE,MAAM,CAAC;QACb,WAAW,EAAE,MAAM,CAAC;QACpB,YAAY,EAAE,MAAM,GAAG,IAAI,CAAC;QAC5B,OAAO,EAAE,MAAM,CAAC,MAAM,EAAE,cAAc,CAAC,CAAC;QACxC,+DAA+D;QAC/D,QAAQ,CAAC,EAAE,YAAY,CAAC;QACxB,OAAO,EAAE,WAAW,EAAE,CAAC;KACxB,CAAC,CAAC;CACJ;AAED,MAAM,WAAW,OAAO;IACtB,IAAI,EAAE,MAAM,CAAC;IACb,QAAQ,EAAE,OAAO,GAAG,SAAS,GAAG,MAAM,CAAC;IACvC,OAAO,EAAE,MAAM,CAAC;IAChB,OAAO,EAAE,OAAO,CAAC;CAClB;AAED,MAAM,WAAW,sBAAsB;IACrC,IAAI,EAAE,MAAM,CAAC;IACb,IAAI,EAAE,MAAM,CAAC;IACb,QAAQ,EAAE,OAAO,CAAC;IAClB,WAAW,EAAE,MAAM,CAAC;IACpB,SAAS,CAAC,EAAE,MAAM,CAAC;CACpB;AAED,MAAM,WAAW,iBAAiB;IAChC,OAAO,EAAE,sBAAsB,EAAE,CAAC;IAClC,YAAY,EAAE,MAAM,CAAC;IACrB,UAAU,EAAE,MAAM,CAAC;IACnB,gBAAgB,EAAE,MAAM,CAAC;IACzB,cAAc,EAAE,MAAM,EAAE,CAAC;IACzB,gBAAgB,EAAE,MAAM,CAAC;IACzB,WAAW,EAAE,MAAM,CAAC;CACrB;AAED,MAAM,WAAW,cAAc;IAC7B,OAAO,CAAC,EAAE,MAAM,CAAC;IACjB,QAAQ,EAAE,OAAO,EAAE,CAAC;IACpB,WAAW,EAAE,MAAM,EAAE,CAAC;IACtB,QAAQ,CAAC,EAAE,MAAM,CAAC,MAAM,EAAE,iBAAiB,CAAC,CAAC;IAC7C,+EAA+E;IAC/E,UAAU,CAAC,EAAE,MAAM,CAAC,MAAM,EAAE,SAAS,CAAC,CAAC;CACxC;AAKD,MAAM,WAAW,cAAc;IAC7B,aAAa,EAAE,OAAO,CAAC;IACvB,UAAU,EAAE,MAAM,CAAC;IACnB,MAAM,EAAE,MAAM,CAAC;CAChB;AAED,MAAM,WAAW,YAAY;IAC3B,QAAQ,EAAE,MAAM,CAAC;IACjB,IAAI,EAAE,eAAe,GAAG,iBAAiB,GAAG,SAAS,GAAG,kBAAkB,CAAC;IAC3E,IAAI,CAAC,EAAE,MAAM,CAAC;IACd,OAAO,EAAE,MAAM,CAAC;IAChB,QAAQ,CAAC,EAAE,MAAM,CAAC,MAAM,EAAE,OAAO,CAAC,CAAC;IAInC,MAAM,EAAE,MAAM,CAAC;IAGf,iBAAiB,CAAC,EAAE,cAAc,CAAC;CACpC;AAED,MAAM,WAAW,SAAS;IACxB,OAAO,EAAE,MAAM,CAAC;IAChB,WAAW,EAAE,MAAM,CAAC;IACpB,cAAc,EAAE,MAAM,CAAC;IACvB,OAAO,EAAE,MAAM,CAAC;IAIhB,eAAe,EAAE,MAAM,CAAC;IACxB,WAAW,CAAC,EAAE,MAAM,GAAG,IAAI,CAAC;IAC5B,WAAW,CAAC,EAAE,MAAM,GAAG,IAAI,CAAC;IAC5B,OAAO,EAAE,YAAY,EAAE,CAAC;IACxB,MAAM,EAAE;QACN,aAAa,EAAE,MAAM,CAAC;QACtB,eAAe,EAAE,MAAM,CAAC;QACxB,OAAO,EAAE,MAAM,CAAC;QAChB,gBAAgB,EAAE,MAAM,CAAC;KAC1B,CAAC;CACH;AAED,MAAM,WAAW,kBAAkB;IACjC,OAAO,EAAE,MAAM,CAAC;IAChB,OAAO,EAAE,MAAM,CAAC;IAChB,OAAO,EAAE,GAAG,GAAG,GAAG,GAAG,MAAM,CAAC;IAC5B,SAAS,EAAE,YAAY,GAAG,OAAO,GAAG,QAAQ,GAAG,OAAO,GAAG,MAAM,CAAC;IAChE,YAAY,EAAE,MAAM,CAAC;IACrB,EAAE,EAAE,MAAM,CAAC;IACX,EAAE,EAAE,MAAM,CAAC;CACZ;AAED,MAAM,WAAW,cAAc;IAC7B,MAAM,EAAE,MAAM,EAAE,CAAC;IACjB,IAAI,EAAE,MAAM,CAAC;IACb,KAAK,EAAE,MAAM,CAAC;IACd,KAAK,EAAE,MAAM,CAAC;IACd,MAAM,EAAE,MAAM,CAAC;CAChB;AAED,MAAM,WAAW,wBAAwB;IACvC,QAAQ,EAAE,MAAM,CAAC;IACjB,UAAU,EAAE,MAAM,CAAC;IACnB,EAAE,EAAE,MAAM,CAAC;IACX,WAAW,EAAE,OAAO,CAAC;IACrB,UAAU,EAAE,kBAAkB,CAAC;CAChC;AAKD,MAAM,MAAM,iBAAiB,GAAG,MAAM,GAAG,YAAY,CAAC;AAStD,MAAM,MAAM,gBAAgB,GAAG,MAAM,GAAG,UAAU,GAAG,OAAO,CAAC;AAE7D,MAAM,WAAW,eAAgB,SAAQ,cAAc;IACrD,QAAQ,CAAC,EAAE,OAAO,CAAC,MAAM,CAAC,iBAAiB,EAAE,cAAc,CAAC,CAAC,CAAC;IAC9D,OAAO,CAAC,EAAE,OAAO,CAAC,MAAM,CAAC,gBAAgB,EAAE,cAAc,CAAC,CAAC,CAAC;CAC7D;AAED,MAAM,WAAW,kBAAmB,SAAQ,wBAAwB;IAClE,CAAC,EAAE,MAAM,CAAC;IACV,CAAC,EAAE,MAAM,CAAC;IACV,QAAQ,CAAC,EAAE,OAAO,CAAC,MAAM,CAAC,iBAAiB,EAAE,wBAAwB,CAAC,CAAC,CAAC;IACxE,OAAO,CAAC,EAAE,OAAO,CAAC,MAAM,CAAC,gBAAgB,EAAE,wBAAwB,CAAC,CAAC,CAAC;CACvE;AAED,MAAM,WAAW,YAAY;IAC3B,IAAI,EAAE,MAAM,CAAC;IACb,UAAU,EAAE,MAAM,CAAC,MAAM,EAAE,eAAe,CAAC,CAAC;IAC5C,WAAW,EAAE,kBAAkB,EAAE,CAAC;IAClC;;8DAE0D;IAC1D,UAAU,CAAC,EAAE,cAAc,CAAC;CAC7B;AAED,gEAAgE;AAChE,MAAM,WAAW,cAAc;IAC7B,oEAAoE;IACpE,sBAAsB,EAAE,MAAM,EAAE,CAAC;IACjC;+EAC2E;IAC3E,UAAU,EAAE,MAAM,CAAC,MAAM,EAAE,KAAK,CAAC;QAAE,CAAC,EAAE,MAAM,CAAC;QAAC,IAAI,EAAE,MAAM,CAAC;QAAC,KAAK,EAAE,MAAM,CAAC;QAAC,MAAM,EAAE,MAAM,CAAA;KAAE,CAAC,CAAC,CAAC;IAC9F,oEAAoE;IACpE,QAAQ,CAAC,EAAE,MAAM,CAAC,MAAM,EAAE;QACxB,SAAS,EAAE,OAAO,CAAC;QACnB,GAAG,EAAE,MAAM,GAAG,IAAI,CAAC;QACnB,UAAU,EAAE,MAAM,GAAG,QAAQ,GAAG,KAAK,CAAC;QACtC,MAAM,EAAE,OAAO,GAAG,oBAAoB,GAAG,gBAAgB,CAAC;QAC1D,SAAS,EAAE,MAAM,CAAC;QAClB,MAAM,EAAE,MAAM,CAAC;KAChB,CAAC,CAAC;CACJ"}
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"report.js","sourceRoot":"","sources":["../../../src/types/report.ts"],"names":[],"mappings":""}
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"shared.d.ts","sourceRoot":"","sources":["../../../src/types/shared.ts"],"names":[],"mappings":"AACA,MAAM,MAAM,IAAI,GAAG,IAAI,GAAG,IAAI,CAAC"}
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"shared.js","sourceRoot":"","sources":["../../../src/types/shared.ts"],"names":[],"mappings":""}
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
import type { EvaluationJob } from './eval.js';
|
|
2
|
+
import type { Report } from './report.js';
|
|
3
|
+
export interface ReportStore {
|
|
4
|
+
list(): Promise<Report[]>;
|
|
5
|
+
get(id: string): Promise<Report | null>;
|
|
6
|
+
save(id: string, report: Report): Promise<void>;
|
|
7
|
+
update(id: string, mutator: (report: Report) => void): Promise<Report | null>;
|
|
8
|
+
remove(id: string): Promise<boolean>;
|
|
9
|
+
exists(id: string): Promise<boolean>;
|
|
10
|
+
findByVariant(variantName: string): Promise<Report[]>;
|
|
11
|
+
findByArtifactHash(hash: string): Promise<Report[]>;
|
|
12
|
+
}
|
|
13
|
+
export interface JobStore {
|
|
14
|
+
list(): Promise<EvaluationJob[]>;
|
|
15
|
+
get(id: string): Promise<EvaluationJob | null>;
|
|
16
|
+
save(id: string, job: EvaluationJob): Promise<void>;
|
|
17
|
+
update(id: string, mutator: (job: EvaluationJob) => EvaluationJob): Promise<EvaluationJob | null>;
|
|
18
|
+
remove(id: string): Promise<boolean>;
|
|
19
|
+
exists(id: string): Promise<boolean>;
|
|
20
|
+
}
|
|
21
|
+
//# sourceMappingURL=storage.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"storage.d.ts","sourceRoot":"","sources":["../../../src/types/storage.ts"],"names":[],"mappings":"AAAA,OAAO,KAAK,EAAE,aAAa,EAAE,MAAM,WAAW,CAAC;AAC/C,OAAO,KAAK,EAAE,MAAM,EAAE,MAAM,aAAa,CAAC;AAE1C,MAAM,WAAW,WAAW;IAC1B,IAAI,IAAI,OAAO,CAAC,MAAM,EAAE,CAAC,CAAC;IAC1B,GAAG,CAAC,EAAE,EAAE,MAAM,GAAG,OAAO,CAAC,MAAM,GAAG,IAAI,CAAC,CAAC;IACxC,IAAI,CAAC,EAAE,EAAE,MAAM,EAAE,MAAM,EAAE,MAAM,GAAG,OAAO,CAAC,IAAI,CAAC,CAAC;IAChD,MAAM,CAAC,EAAE,EAAE,MAAM,EAAE,OAAO,EAAE,CAAC,MAAM,EAAE,MAAM,KAAK,IAAI,GAAG,OAAO,CAAC,MAAM,GAAG,IAAI,CAAC,CAAC;IAC9E,MAAM,CAAC,EAAE,EAAE,MAAM,GAAG,OAAO,CAAC,OAAO,CAAC,CAAC;IACrC,MAAM,CAAC,EAAE,EAAE,MAAM,GAAG,OAAO,CAAC,OAAO,CAAC,CAAC;IACrC,aAAa,CAAC,WAAW,EAAE,MAAM,GAAG,OAAO,CAAC,MAAM,EAAE,CAAC,CAAC;IACtD,kBAAkB,CAAC,IAAI,EAAE,MAAM,GAAG,OAAO,CAAC,MAAM,EAAE,CAAC,CAAC;CACrD;AAED,MAAM,WAAW,QAAQ;IACvB,IAAI,IAAI,OAAO,CAAC,aAAa,EAAE,CAAC,CAAC;IACjC,GAAG,CAAC,EAAE,EAAE,MAAM,GAAG,OAAO,CAAC,aAAa,GAAG,IAAI,CAAC,CAAC;IAC/C,IAAI,CAAC,EAAE,EAAE,MAAM,EAAE,GAAG,EAAE,aAAa,GAAG,OAAO,CAAC,IAAI,CAAC,CAAC;IACpD,MAAM,CAAC,EAAE,EAAE,MAAM,EAAE,OAAO,EAAE,CAAC,GAAG,EAAE,aAAa,KAAK,aAAa,GAAG,OAAO,CAAC,aAAa,GAAG,IAAI,CAAC,CAAC;IAClG,MAAM,CAAC,EAAE,EAAE,MAAM,GAAG,OAAO,CAAC,OAAO,CAAC,CAAC;IACrC,MAAM,CAAC,EAAE,EAAE,MAAM,GAAG,OAAO,CAAC,OAAO,CAAC,CAAC;CACtC"}
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"storage.js","sourceRoot":"","sources":["../../../src/types/storage.ts"],"names":[],"mappings":""}
|