@hmharness/evaluation 0.8.1 → 0.8.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/index.d.ts +1 -0
- package/dist/index.js +1 -0
- package/dist/score.d.ts +82 -0
- package/dist/score.js +168 -0
- package/package.json +1 -1
package/dist/index.d.ts
CHANGED
package/dist/index.js
CHANGED
package/dist/score.d.ts
ADDED
|
@@ -0,0 +1,82 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @hmharness/evaluation - HMH Score 1.0 (P1-03)
|
|
3
|
+
*
|
|
4
|
+
* The audit called for: "统一 0-100 多维总分体系" with dimensions:
|
|
5
|
+
* correctness, test, reliability, repair, security, cost, latency,
|
|
6
|
+
* human, maintainability.
|
|
7
|
+
*
|
|
8
|
+
* HMH Score is the single number that tracks whether the harness is
|
|
9
|
+
* actually getting better. Every dimension maps to [0,100]; the composite
|
|
10
|
+
* is a weighted mean. Weights are versioned (changing weights = version bump).
|
|
11
|
+
*/
|
|
12
|
+
export declare const HMH_SCORE_VERSION = "1.0.0";
|
|
13
|
+
export type ScoreDimension = 'correctness' | 'test' | 'reliability' | 'repair' | 'security' | 'cost' | 'latency' | 'human' | 'maintainability';
|
|
14
|
+
/** All dimensions in canonical order */
|
|
15
|
+
export declare const ALL_DIMENSIONS: ScoreDimension[];
|
|
16
|
+
/** Default weights (sum to 1.0) - versioned, changing requires bump */
|
|
17
|
+
export declare const DEFAULT_WEIGHTS: Record<ScoreDimension, number>;
|
|
18
|
+
export interface DimensionScore {
|
|
19
|
+
dimension: ScoreDimension;
|
|
20
|
+
/** raw score [0,100] */
|
|
21
|
+
score: number;
|
|
22
|
+
/** how this score was computed (for audit) */
|
|
23
|
+
method: string;
|
|
24
|
+
/** sample size used to compute this score */
|
|
25
|
+
sampleSize: number;
|
|
26
|
+
}
|
|
27
|
+
export interface HMHScore {
|
|
28
|
+
version: string;
|
|
29
|
+
/** composite score [0,100] */
|
|
30
|
+
composite: number;
|
|
31
|
+
/** per-dimension breakdown */
|
|
32
|
+
dimensions: DimensionScore[];
|
|
33
|
+
/** when this score was computed */
|
|
34
|
+
computedAt: string;
|
|
35
|
+
/** any warnings about data quality */
|
|
36
|
+
warnings: string[];
|
|
37
|
+
}
|
|
38
|
+
/**
|
|
39
|
+
* Compute the composite HMH Score from dimension scores.
|
|
40
|
+
* Pure - testable.
|
|
41
|
+
*/
|
|
42
|
+
export declare function computeHMHScore(dimensions: DimensionScore[], weights?: Record<ScoreDimension, number>): HMHScore;
|
|
43
|
+
/**
|
|
44
|
+
* Compute correctness from pass rate.
|
|
45
|
+
* Pure - testable.
|
|
46
|
+
*/
|
|
47
|
+
export declare function correctnessFromPassRate(passRate: number, sampleSize: number): DimensionScore;
|
|
48
|
+
/**
|
|
49
|
+
* Compute test coverage score.
|
|
50
|
+
* Pure - testable.
|
|
51
|
+
*/
|
|
52
|
+
export declare function testFromCoverage(coveragePercent: number): DimensionScore;
|
|
53
|
+
/**
|
|
54
|
+
* Compute reliability from success rate.
|
|
55
|
+
* Pure - testable.
|
|
56
|
+
*/
|
|
57
|
+
export declare function reliabilityFromSuccessRate(successRate: number, sessions: number): DimensionScore;
|
|
58
|
+
/**
|
|
59
|
+
* Compute security score from red-team results.
|
|
60
|
+
* Pure - testable.
|
|
61
|
+
*/
|
|
62
|
+
export declare function securityFromRedTeam(blocked: number, total: number): DimensionScore;
|
|
63
|
+
/**
|
|
64
|
+
* Compute cost efficiency (inverse of cost, normalized).
|
|
65
|
+
* Pure - testable.
|
|
66
|
+
*/
|
|
67
|
+
export declare function costFromTokensPerTask(avgTokens: number, budgetTokens: number): DimensionScore;
|
|
68
|
+
/**
|
|
69
|
+
* Compute latency score (inverse of duration, normalized).
|
|
70
|
+
* Pure - testable.
|
|
71
|
+
*/
|
|
72
|
+
export declare function latencyFromDuration(avgMs: number, budgetMs: number): DimensionScore;
|
|
73
|
+
/**
|
|
74
|
+
* Compute human satisfaction from judge scores.
|
|
75
|
+
* Pure - testable.
|
|
76
|
+
*/
|
|
77
|
+
export declare function humanFromJudgeScores(scores: number[]): DimensionScore;
|
|
78
|
+
/**
|
|
79
|
+
* Format an HMH Score as a human-readable string.
|
|
80
|
+
* Pure - testable.
|
|
81
|
+
*/
|
|
82
|
+
export declare function formatHMHScore(score: HMHScore): string;
|
package/dist/score.js
ADDED
|
@@ -0,0 +1,168 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @hmharness/evaluation - HMH Score 1.0 (P1-03)
|
|
3
|
+
*
|
|
4
|
+
* The audit called for: "统一 0-100 多维总分体系" with dimensions:
|
|
5
|
+
* correctness, test, reliability, repair, security, cost, latency,
|
|
6
|
+
* human, maintainability.
|
|
7
|
+
*
|
|
8
|
+
* HMH Score is the single number that tracks whether the harness is
|
|
9
|
+
* actually getting better. Every dimension maps to [0,100]; the composite
|
|
10
|
+
* is a weighted mean. Weights are versioned (changing weights = version bump).
|
|
11
|
+
*/
|
|
12
|
+
export const HMH_SCORE_VERSION = '1.0.0';
|
|
13
|
+
/** All dimensions in canonical order */
|
|
14
|
+
export const ALL_DIMENSIONS = [
|
|
15
|
+
'correctness', 'test', 'reliability', 'repair', 'security',
|
|
16
|
+
'cost', 'latency', 'human', 'maintainability',
|
|
17
|
+
];
|
|
18
|
+
/** Default weights (sum to 1.0) - versioned, changing requires bump */
|
|
19
|
+
export const DEFAULT_WEIGHTS = {
|
|
20
|
+
correctness: 0.25,
|
|
21
|
+
test: 0.15,
|
|
22
|
+
reliability: 0.12,
|
|
23
|
+
repair: 0.08,
|
|
24
|
+
security: 0.10,
|
|
25
|
+
cost: 0.08,
|
|
26
|
+
latency: 0.07,
|
|
27
|
+
human: 0.10,
|
|
28
|
+
maintainability: 0.05,
|
|
29
|
+
};
|
|
30
|
+
/**
|
|
31
|
+
* Compute the composite HMH Score from dimension scores.
|
|
32
|
+
* Pure - testable.
|
|
33
|
+
*/
|
|
34
|
+
export function computeHMHScore(dimensions, weights = DEFAULT_WEIGHTS) {
|
|
35
|
+
const warnings = [];
|
|
36
|
+
const dimMap = new Map(dimensions.map(d => [d.dimension, d]));
|
|
37
|
+
const missing = ALL_DIMENSIONS.filter(d => !dimMap.has(d));
|
|
38
|
+
if (missing.length > 0)
|
|
39
|
+
warnings.push(`missing dimensions: ${missing.join(', ')} (treated as 0)`);
|
|
40
|
+
let composite = 0;
|
|
41
|
+
let totalWeight = 0;
|
|
42
|
+
for (const dim of ALL_DIMENSIONS) {
|
|
43
|
+
const ds = dimMap.get(dim);
|
|
44
|
+
const w = weights[dim] ?? 0;
|
|
45
|
+
if (ds) {
|
|
46
|
+
composite += Math.max(0, Math.min(100, ds.score)) * w;
|
|
47
|
+
totalWeight += w;
|
|
48
|
+
}
|
|
49
|
+
}
|
|
50
|
+
if (totalWeight > 0)
|
|
51
|
+
composite = composite / totalWeight;
|
|
52
|
+
return {
|
|
53
|
+
version: HMH_SCORE_VERSION,
|
|
54
|
+
composite: Math.round(composite * 100) / 100,
|
|
55
|
+
dimensions,
|
|
56
|
+
computedAt: new Date().toISOString(),
|
|
57
|
+
warnings,
|
|
58
|
+
};
|
|
59
|
+
}
|
|
60
|
+
/**
|
|
61
|
+
* Compute correctness from pass rate.
|
|
62
|
+
* Pure - testable.
|
|
63
|
+
*/
|
|
64
|
+
export function correctnessFromPassRate(passRate, sampleSize) {
|
|
65
|
+
return {
|
|
66
|
+
dimension: 'correctness',
|
|
67
|
+
score: Math.round(passRate * 100),
|
|
68
|
+
method: `pass rate × 100 (${sampleSize} samples)`,
|
|
69
|
+
sampleSize,
|
|
70
|
+
};
|
|
71
|
+
}
|
|
72
|
+
/**
|
|
73
|
+
* Compute test coverage score.
|
|
74
|
+
* Pure - testable.
|
|
75
|
+
*/
|
|
76
|
+
export function testFromCoverage(coveragePercent) {
|
|
77
|
+
return {
|
|
78
|
+
dimension: 'test',
|
|
79
|
+
score: Math.round(Math.min(100, coveragePercent)),
|
|
80
|
+
method: `test coverage % (${coveragePercent.toFixed(1)}%)`,
|
|
81
|
+
sampleSize: 1,
|
|
82
|
+
};
|
|
83
|
+
}
|
|
84
|
+
/**
|
|
85
|
+
* Compute reliability from success rate.
|
|
86
|
+
* Pure - testable.
|
|
87
|
+
*/
|
|
88
|
+
export function reliabilityFromSuccessRate(successRate, sessions) {
|
|
89
|
+
return {
|
|
90
|
+
dimension: 'reliability',
|
|
91
|
+
score: Math.round(successRate * 100),
|
|
92
|
+
method: `session success rate × 100 (${sessions} sessions)`,
|
|
93
|
+
sampleSize: sessions,
|
|
94
|
+
};
|
|
95
|
+
}
|
|
96
|
+
/**
|
|
97
|
+
* Compute security score from red-team results.
|
|
98
|
+
* Pure - testable.
|
|
99
|
+
*/
|
|
100
|
+
export function securityFromRedTeam(blocked, total) {
|
|
101
|
+
const rate = total > 0 ? blocked / total : 0;
|
|
102
|
+
return {
|
|
103
|
+
dimension: 'security',
|
|
104
|
+
score: Math.round(rate * 100),
|
|
105
|
+
method: `red-team block rate (${blocked}/${total} attacks blocked)`,
|
|
106
|
+
sampleSize: total,
|
|
107
|
+
};
|
|
108
|
+
}
|
|
109
|
+
/**
|
|
110
|
+
* Compute cost efficiency (inverse of cost, normalized).
|
|
111
|
+
* Pure - testable.
|
|
112
|
+
*/
|
|
113
|
+
export function costFromTokensPerTask(avgTokens, budgetTokens) {
|
|
114
|
+
const ratio = budgetTokens > 0 ? avgTokens / budgetTokens : 1;
|
|
115
|
+
const score = Math.round(Math.max(0, Math.min(100, (1 - ratio) * 100)));
|
|
116
|
+
return {
|
|
117
|
+
dimension: 'cost',
|
|
118
|
+
score,
|
|
119
|
+
method: `token efficiency (avg ${avgTokens} / budget ${budgetTokens})`,
|
|
120
|
+
sampleSize: 1,
|
|
121
|
+
};
|
|
122
|
+
}
|
|
123
|
+
/**
|
|
124
|
+
* Compute latency score (inverse of duration, normalized).
|
|
125
|
+
* Pure - testable.
|
|
126
|
+
*/
|
|
127
|
+
export function latencyFromDuration(avgMs, budgetMs) {
|
|
128
|
+
const ratio = budgetMs > 0 ? avgMs / budgetMs : 1;
|
|
129
|
+
const score = Math.round(Math.max(0, Math.min(100, (1 - ratio) * 100)));
|
|
130
|
+
return {
|
|
131
|
+
dimension: 'latency',
|
|
132
|
+
score,
|
|
133
|
+
method: `latency efficiency (avg ${avgMs}ms / budget ${budgetMs}ms)`,
|
|
134
|
+
sampleSize: 1,
|
|
135
|
+
};
|
|
136
|
+
}
|
|
137
|
+
/**
|
|
138
|
+
* Compute human satisfaction from judge scores.
|
|
139
|
+
* Pure - testable.
|
|
140
|
+
*/
|
|
141
|
+
export function humanFromJudgeScores(scores) {
|
|
142
|
+
if (scores.length === 0) {
|
|
143
|
+
return { dimension: 'human', score: 0, method: 'no judge scores', sampleSize: 0 };
|
|
144
|
+
}
|
|
145
|
+
const avg = scores.reduce((a, b) => a + b, 0) / scores.length;
|
|
146
|
+
return {
|
|
147
|
+
dimension: 'human',
|
|
148
|
+
score: Math.round((avg / 5) * 100), // 5-point scale → 0-100
|
|
149
|
+
method: `avg judge score (${avg.toFixed(2)}/5 across ${scores.length} sessions)`,
|
|
150
|
+
sampleSize: scores.length,
|
|
151
|
+
};
|
|
152
|
+
}
|
|
153
|
+
/**
|
|
154
|
+
* Format an HMH Score as a human-readable string.
|
|
155
|
+
* Pure - testable.
|
|
156
|
+
*/
|
|
157
|
+
export function formatHMHScore(score) {
|
|
158
|
+
const lines = [`HMH Score v${score.version}: ${score.composite}/100`];
|
|
159
|
+
const sorted = [...score.dimensions].sort((a, b) => b.score - a.score);
|
|
160
|
+
for (const d of sorted) {
|
|
161
|
+
const bar = '█'.repeat(Math.round(d.score / 10)) + '░'.repeat(10 - Math.round(d.score / 10));
|
|
162
|
+
lines.push(` ${d.dimension.padEnd(16)} ${bar} ${String(d.score).padStart(3)} (${d.method})`);
|
|
163
|
+
}
|
|
164
|
+
if (score.warnings.length > 0) {
|
|
165
|
+
lines.push(` ⚠ ${score.warnings.join('; ')}`);
|
|
166
|
+
}
|
|
167
|
+
return lines.join('\n');
|
|
168
|
+
}
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@hmharness/evaluation",
|
|
3
|
-
"version": "0.8.
|
|
3
|
+
"version": "0.8.3",
|
|
4
4
|
"description": "hmharness evaluation: the Evaluator/Judge contract (V2 blueprint M2). Hard evidence outranks LLM judgment - build results, exit codes, exact/regex assertions first; the LLM judge is a last resort and is labeled as such. Evaluations attach to trajectories (judge.completed events).",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"main": "dist/index.js",
|