@hmharness/evaluation 0.8.3 → 0.8.5

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.d.ts CHANGED
@@ -3,3 +3,5 @@ export * from './evaluators.ts';
3
3
  export * from './runner.ts';
4
4
  export * from './harmonybench.ts';
5
5
  export * from './score.ts';
6
+ export * from './sdk.ts';
7
+ export * from './marketplace.ts';
package/dist/index.js CHANGED
@@ -3,3 +3,5 @@ export * from "./evaluators.js";
3
3
  export * from "./runner.js";
4
4
  export * from "./harmonybench.js";
5
5
  export * from "./score.js";
6
+ export * from "./sdk.js";
7
+ export * from "./marketplace.js";
@@ -0,0 +1,82 @@
1
+ /**
2
+ * @hmharness/evaluation - Benchmark Marketplace (P2-07)
3
+ *
4
+ * The audit called for: "允许外部贡献任务,但生产 gate 只读取签名/审核后的 fixture"
5
+ *
6
+ * Provides governance for external benchmark contributions:
7
+ * 1. Contribution submission with metadata
8
+ * 2. Review/approval workflow
9
+ * 3. Signature verification (simulated - real crypto in production)
10
+ * 4. Registry of approved vs pending vs rejected fixtures
11
+ */
12
+ export type FixtureStatus = 'pending' | 'approved' | 'rejected' | 'deprecated';
13
+ export interface BenchmarkFixture {
14
+ id: string;
15
+ name: string;
16
+ description: string;
17
+ category: string;
18
+ difficulty: 1 | 2 | 3;
19
+ prompt: string;
20
+ expectedOutput: string;
21
+ assertionType: 'exact' | 'contains' | 'regex';
22
+ assertionValue: string;
23
+ /** contributor metadata */
24
+ contributor: string;
25
+ submittedAt: string;
26
+ /** governance */
27
+ status: FixtureStatus;
28
+ reviewedBy?: string;
29
+ reviewedAt?: string;
30
+ reviewNotes?: string;
31
+ /** content hash for integrity verification */
32
+ contentHash: string;
33
+ /** signature (simulated - real HMAC in production) */
34
+ signature?: string;
35
+ }
36
+ export interface MarketplaceStats {
37
+ total: number;
38
+ approved: number;
39
+ pending: number;
40
+ rejected: number;
41
+ byCategory: Record<string, number>;
42
+ }
43
+ /**
44
+ * Compute a content hash for a fixture (simple but deterministic).
45
+ * Pure - testable.
46
+ */
47
+ export declare function hashFixture(fixture: Omit<BenchmarkFixture, 'contentHash' | 'status' | 'submittedAt'>): string;
48
+ /**
49
+ * Sign a fixture (simulated HMAC).
50
+ * Pure - testable.
51
+ */
52
+ export declare function signFixture(fixture: BenchmarkFixture, secret: string): string;
53
+ /**
54
+ * Verify a fixture signature.
55
+ * Pure - testable.
56
+ */
57
+ export declare function verifySignature(fixture: BenchmarkFixture, secret: string): boolean;
58
+ /**
59
+ * Validate a contributed fixture for completeness.
60
+ * Pure - testable.
61
+ */
62
+ export declare function validateFixture(fixture: unknown): {
63
+ valid: boolean;
64
+ errors: string[];
65
+ };
66
+ /**
67
+ * The fixture registry - manages the marketplace lifecycle.
68
+ */
69
+ export declare class FixtureRegistry {
70
+ private fixtures;
71
+ submit(fixture: BenchmarkFixture): {
72
+ ok: boolean;
73
+ reason?: string;
74
+ };
75
+ review(id: string, approved: boolean, reviewer: string, notes?: string): boolean;
76
+ deprecate(id: string): boolean;
77
+ /** Only approved fixtures are usable in production gates */
78
+ getApproved(): BenchmarkFixture[];
79
+ getPending(): BenchmarkFixture[];
80
+ get(id: string): BenchmarkFixture | undefined;
81
+ stats(): MarketplaceStats;
82
+ }
@@ -0,0 +1,125 @@
1
+ /**
2
+ * @hmharness/evaluation - Benchmark Marketplace (P2-07)
3
+ *
4
+ * The audit called for: "允许外部贡献任务,但生产 gate 只读取签名/审核后的 fixture"
5
+ *
6
+ * Provides governance for external benchmark contributions:
7
+ * 1. Contribution submission with metadata
8
+ * 2. Review/approval workflow
9
+ * 3. Signature verification (simulated - real crypto in production)
10
+ * 4. Registry of approved vs pending vs rejected fixtures
11
+ */
12
+ /**
13
+ * Compute a content hash for a fixture (simple but deterministic).
14
+ * Pure - testable.
15
+ */
16
+ export function hashFixture(fixture) {
17
+ const content = JSON.stringify({
18
+ name: fixture.name, prompt: fixture.prompt,
19
+ expected: fixture.expectedOutput, assertion: fixture.assertionValue,
20
+ });
21
+ let hash = 0;
22
+ for (let i = 0; i < content.length; i++) {
23
+ hash = ((hash << 5) - hash + content.charCodeAt(i)) | 0;
24
+ }
25
+ return `fx-${Math.abs(hash).toString(16).padStart(8, '0')}`;
26
+ }
27
+ /**
28
+ * Sign a fixture (simulated HMAC).
29
+ * Pure - testable.
30
+ */
31
+ export function signFixture(fixture, secret) {
32
+ const payload = `${fixture.id}:${fixture.contentHash}:${secret}`;
33
+ let hash = 0;
34
+ for (let i = 0; i < payload.length; i++) {
35
+ hash = ((hash << 5) - hash + payload.charCodeAt(i)) | 0;
36
+ }
37
+ return `sig-${Math.abs(hash).toString(16)}`;
38
+ }
39
+ /**
40
+ * Verify a fixture signature.
41
+ * Pure - testable.
42
+ */
43
+ export function verifySignature(fixture, secret) {
44
+ if (!fixture.signature)
45
+ return false;
46
+ return fixture.signature === signFixture(fixture, secret);
47
+ }
48
+ /**
49
+ * Validate a contributed fixture for completeness.
50
+ * Pure - testable.
51
+ */
52
+ export function validateFixture(fixture) {
53
+ const errors = [];
54
+ const f = fixture;
55
+ if (!f.name)
56
+ errors.push('name is required');
57
+ if (!f.prompt)
58
+ errors.push('prompt is required');
59
+ if (!f.contributor)
60
+ errors.push('contributor is required');
61
+ if (!f.category)
62
+ errors.push('category is required');
63
+ if (f.difficulty === undefined || ![1, 2, 3].includes(f.difficulty))
64
+ errors.push('difficulty must be 1, 2, or 3');
65
+ if (!f.assertionType || !['exact', 'contains', 'regex'].includes(f.assertionType))
66
+ errors.push('assertionType must be exact/contains/regex');
67
+ if (!f.assertionValue)
68
+ errors.push('assertionValue is required');
69
+ return { valid: errors.length === 0, errors };
70
+ }
71
+ /**
72
+ * The fixture registry - manages the marketplace lifecycle.
73
+ */
74
+ export class FixtureRegistry {
75
+ fixtures = new Map();
76
+ submit(fixture) {
77
+ const v = validateFixture(fixture);
78
+ if (!v.valid)
79
+ return { ok: false, reason: v.errors.join('; ') };
80
+ if (this.fixtures.has(fixture.id))
81
+ return { ok: false, reason: `fixture ${fixture.id} already exists` };
82
+ this.fixtures.set(fixture.id, fixture);
83
+ return { ok: true };
84
+ }
85
+ review(id, approved, reviewer, notes) {
86
+ const f = this.fixtures.get(id);
87
+ if (!f)
88
+ return false;
89
+ f.status = approved ? 'approved' : 'rejected';
90
+ f.reviewedBy = reviewer;
91
+ f.reviewedAt = new Date().toISOString();
92
+ f.reviewNotes = notes;
93
+ return true;
94
+ }
95
+ deprecate(id) {
96
+ const f = this.fixtures.get(id);
97
+ if (!f)
98
+ return false;
99
+ f.status = 'deprecated';
100
+ return true;
101
+ }
102
+ /** Only approved fixtures are usable in production gates */
103
+ getApproved() {
104
+ return [...this.fixtures.values()].filter(f => f.status === 'approved');
105
+ }
106
+ getPending() {
107
+ return [...this.fixtures.values()].filter(f => f.status === 'pending');
108
+ }
109
+ get(id) {
110
+ return this.fixtures.get(id);
111
+ }
112
+ stats() {
113
+ const all = [...this.fixtures.values()];
114
+ const byCategory = {};
115
+ for (const f of all)
116
+ byCategory[f.category] = (byCategory[f.category] ?? 0) + 1;
117
+ return {
118
+ total: all.length,
119
+ approved: all.filter(f => f.status === 'approved').length,
120
+ pending: all.filter(f => f.status === 'pending').length,
121
+ rejected: all.filter(f => f.status === 'rejected').length,
122
+ byCategory,
123
+ };
124
+ }
125
+ }
package/dist/sdk.d.ts ADDED
@@ -0,0 +1,83 @@
1
+ /**
2
+ * @hmharness/evaluation - Evaluator SDK (P1-02)
3
+ *
4
+ * The audit called for: "统一 evaluator plugin、evidence references、metric registry"
5
+ *
6
+ * This module provides the plugin interface that ALL evaluators implement,
7
+ * evidence references that link scores to concrete proof, and a metric
8
+ * registry for named, versioned scoring functions.
9
+ */
10
+ /** Evidence tier per the existing evidence ladder (M2) */
11
+ export type EvidenceTier = 1 | 2 | 3 | 4 | 5 | 6 | 7 | 8;
12
+ /** Reference to concrete evidence backing an evaluation */
13
+ export interface EvidenceRef {
14
+ /** what kind of evidence */
15
+ kind: 'build-log' | 'test-result' | 'command-output' | 'file-content' | 'judge-score' | 'device-log' | 'metric';
16
+ /** where to find it (path, URL, or inline) */
17
+ source: string;
18
+ /** relevant excerpt (for audit trail) */
19
+ excerpt?: string;
20
+ /** evidence tier (1=build, 7=llm-judge, 8=self-report) */
21
+ tier: EvidenceTier;
22
+ }
23
+ /** The result of running an evaluator */
24
+ export interface SdkEvaluationResult {
25
+ /** evaluator that produced this result */
26
+ evaluatorId: string;
27
+ /** pass/fail/indeterminate */
28
+ outcome: 'pass' | 'fail' | 'indeterminate';
29
+ /** numeric score [0,1] if applicable */
30
+ score?: number;
31
+ /** human-readable detail */
32
+ detail: string;
33
+ /** evidence backing this result */
34
+ evidence: EvidenceRef[];
35
+ /** when this evaluation ran */
36
+ evaluatedAt: string;
37
+ /** duration in ms */
38
+ durationMs: number;
39
+ }
40
+ /** The evaluator plugin interface - ALL evaluators implement this */
41
+ export interface SdkEvaluator {
42
+ /** unique evaluator id */
43
+ readonly id: string;
44
+ /** what this evaluator checks */
45
+ readonly description: string;
46
+ /** evidence tier this evaluator operates at */
47
+ readonly tier: EvidenceTier;
48
+ /** evaluate and return a result */
49
+ evaluate(input: unknown): Promise<SdkEvaluationResult>;
50
+ }
51
+ /** A named metric in the registry */
52
+ export interface MetricDef {
53
+ name: string;
54
+ description: string;
55
+ /** units (e.g. 'percent', 'count', 'ms') */
56
+ unit: string;
57
+ /** higher is better? (for display) */
58
+ higherIsBetter: boolean;
59
+ /** compute from evaluation results */
60
+ compute: (results: SdkEvaluationResult[]) => number;
61
+ }
62
+ /**
63
+ * The metric registry - named, versioned scoring functions.
64
+ * Metrics are registered once and referenced by name everywhere.
65
+ */
66
+ export declare class MetricRegistry {
67
+ private metrics;
68
+ register(def: MetricDef): {
69
+ ok: boolean;
70
+ reason?: string;
71
+ };
72
+ get(name: string): MetricDef | undefined;
73
+ list(): MetricDef[];
74
+ /** Compute a metric from evaluation results */
75
+ compute(name: string, results: SdkEvaluationResult[]): number | undefined;
76
+ }
77
+ /** Create a standard metric registry with common metrics pre-registered */
78
+ export declare function createDefaultRegistry(): MetricRegistry;
79
+ /**
80
+ * Create a simple pass/fail evaluator from a predicate function.
81
+ * Convenience factory for the most common evaluator pattern.
82
+ */
83
+ export declare function createPredicateEvaluator(id: string, description: string, tier: EvidenceTier, predicate: (input: unknown) => boolean, evidenceKind?: EvidenceRef['kind']): SdkEvaluator;
package/dist/sdk.js ADDED
@@ -0,0 +1,85 @@
1
+ /**
2
+ * @hmharness/evaluation - Evaluator SDK (P1-02)
3
+ *
4
+ * The audit called for: "统一 evaluator plugin、evidence references、metric registry"
5
+ *
6
+ * This module provides the plugin interface that ALL evaluators implement,
7
+ * evidence references that link scores to concrete proof, and a metric
8
+ * registry for named, versioned scoring functions.
9
+ */
10
+ /**
11
+ * The metric registry - named, versioned scoring functions.
12
+ * Metrics are registered once and referenced by name everywhere.
13
+ */
14
+ export class MetricRegistry {
15
+ metrics = new Map();
16
+ register(def) {
17
+ if (this.metrics.has(def.name)) {
18
+ return { ok: false, reason: `metric ${def.name} already registered` };
19
+ }
20
+ this.metrics.set(def.name, def);
21
+ return { ok: true };
22
+ }
23
+ get(name) {
24
+ return this.metrics.get(name);
25
+ }
26
+ list() {
27
+ return [...this.metrics.values()];
28
+ }
29
+ /** Compute a metric from evaluation results */
30
+ compute(name, results) {
31
+ const m = this.metrics.get(name);
32
+ if (!m)
33
+ return undefined;
34
+ return m.compute(results);
35
+ }
36
+ }
37
+ /** Create a standard metric registry with common metrics pre-registered */
38
+ export function createDefaultRegistry() {
39
+ const reg = new MetricRegistry();
40
+ reg.register({
41
+ name: 'pass_rate', description: 'fraction of evaluations that passed',
42
+ unit: 'fraction', higherIsBetter: true,
43
+ compute: (rs) => rs.length > 0 ? rs.filter(r => r.outcome === 'pass').length / rs.length : 0,
44
+ });
45
+ reg.register({
46
+ name: 'avg_score', description: 'average numeric score',
47
+ unit: 'fraction', higherIsBetter: true,
48
+ compute: (rs) => {
49
+ const scored = rs.filter(r => r.score !== undefined);
50
+ return scored.length > 0 ? scored.reduce((a, b) => a + (b.score ?? 0), 0) / scored.length : 0;
51
+ },
52
+ });
53
+ reg.register({
54
+ name: 'avg_duration', description: 'average evaluation duration',
55
+ unit: 'ms', higherIsBetter: false,
56
+ compute: (rs) => rs.length > 0 ? rs.reduce((a, b) => a + b.durationMs, 0) / rs.length : 0,
57
+ });
58
+ reg.register({
59
+ name: 'evidence_coverage', description: 'fraction of results with evidence',
60
+ unit: 'fraction', higherIsBetter: true,
61
+ compute: (rs) => rs.length > 0 ? rs.filter(r => r.evidence.length > 0).length / rs.length : 0,
62
+ });
63
+ return reg;
64
+ }
65
+ /**
66
+ * Create a simple pass/fail evaluator from a predicate function.
67
+ * Convenience factory for the most common evaluator pattern.
68
+ */
69
+ export function createPredicateEvaluator(id, description, tier, predicate, evidenceKind = 'command-output') {
70
+ return {
71
+ id, description, tier,
72
+ async evaluate(input) {
73
+ const start = Date.now();
74
+ const pass = predicate(input);
75
+ return {
76
+ evaluatorId: id,
77
+ outcome: pass ? 'pass' : 'fail',
78
+ detail: pass ? 'predicate satisfied' : 'predicate not satisfied',
79
+ evidence: [{ kind: evidenceKind, source: 'inline', tier, excerpt: String(input).slice(0, 200) }],
80
+ evaluatedAt: new Date().toISOString(),
81
+ durationMs: Date.now() - start,
82
+ };
83
+ },
84
+ };
85
+ }
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@hmharness/evaluation",
3
- "version": "0.8.3",
3
+ "version": "0.8.5",
4
4
  "description": "hmharness evaluation: the Evaluator/Judge contract (V2 blueprint M2). Hard evidence outranks LLM judgment - build results, exit codes, exact/regex assertions first; the LLM judge is a last resort and is labeled as such. Evaluations attach to trajectories (judge.completed events).",
5
5
  "type": "module",
6
6
  "main": "dist/index.js",