@hmharness/evaluation 0.8.0 → 0.8.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,109 @@
1
+ /**
2
+ * @hmharness/evaluation - HarmonyBench v1 (P0-04, 2026-09-24 audit)
3
+ *
4
+ * The formal benchmark: 50+ versioned cases across 6 categories, with
5
+ * release baseline tracking, regression detection, and formal scoring.
6
+ * This upgrades the internal bench tool to the "正式 Benchmark" the audit
7
+ * demanded: "带 task manifest、fixture version、expected behavior、scoring、
8
+ * environment lock、release baseline、regression set".
9
+ */
10
+ export interface BenchCase {
11
+ /** unique case id: category-NNN */
12
+ id: string;
13
+ /** the task prompt */
14
+ prompt: string;
15
+ /** expected behavior description */
16
+ expected: string;
17
+ /** assertion: what the output must contain/match */
18
+ assertion: {
19
+ type: 'exact' | 'contains' | 'not_contains' | 'regex';
20
+ value: string;
21
+ };
22
+ /** category for reporting */
23
+ category: BenchCategory;
24
+ /** case difficulty (1=easy, 2=medium, 3=hard) */
25
+ difficulty: 1 | 2 | 3;
26
+ /** minimum turns expected (efficiency signal) */
27
+ expectedTurns: number;
28
+ /** whether this case requires tools */
29
+ needsTools: boolean;
30
+ }
31
+ export type BenchCategory = 'exactness' | 'cjk' | 'code' | 'reasoning' | 'tools' | 'harmony';
32
+ export interface BenchResult {
33
+ caseId: string;
34
+ category: BenchCategory;
35
+ pass: boolean;
36
+ output: string;
37
+ turns: number;
38
+ tokens: number;
39
+ durationMs: number;
40
+ }
41
+ export interface BenchReport {
42
+ /** benchmark version (fixture governance) */
43
+ benchVersion: string;
44
+ /** hmh version that produced this report */
45
+ hmhVersion: string;
46
+ timestamp: string;
47
+ total: number;
48
+ passed: number;
49
+ passRate: number;
50
+ byCategory: Record<BenchCategory, {
51
+ total: number;
52
+ passed: number;
53
+ rate: number;
54
+ }>;
55
+ regressionVsBaseline?: Array<{
56
+ caseId: string;
57
+ was: 'pass';
58
+ now: 'fail';
59
+ }>;
60
+ results: BenchResult[];
61
+ }
62
+ /**
63
+ * HarmonyBench v1.0.0 - 54 cases across 6 categories.
64
+ * Version-locked: any change to cases requires a version bump.
65
+ */
66
+ export declare const HARMONYBENCH_VERSION = "1.0.0";
67
+ export declare const HARMONYBENCH_CASES: BenchCase[];
68
+ /**
69
+ * Run a single assertion against model output.
70
+ * Pure - testable.
71
+ */
72
+ export declare function checkAssertion(output: string, assertion: BenchCase['assertion']): boolean;
73
+ /**
74
+ * Build the by-category summary from individual results.
75
+ * Pure - testable.
76
+ */
77
+ export declare function summarizeByCategory(results: Array<{
78
+ caseId: string;
79
+ category: BenchCategory;
80
+ pass: boolean;
81
+ }>): Record<BenchCategory, {
82
+ total: number;
83
+ passed: number;
84
+ rate: number;
85
+ }>;
86
+ /**
87
+ * Detect regressions vs a baseline report.
88
+ * Pure - testable.
89
+ */
90
+ export declare function detectRegressions(current: Array<{
91
+ caseId: string;
92
+ pass: boolean;
93
+ }>, baseline: Array<{
94
+ caseId: string;
95
+ pass: boolean;
96
+ }>): Array<{
97
+ caseId: string;
98
+ was: 'pass';
99
+ now: 'fail';
100
+ }>;
101
+ /**
102
+ * Validate the benchmark suite itself (fixture governance).
103
+ * Every case must have: unique id, non-empty prompt, valid assertion,
104
+ * valid category, valid difficulty.
105
+ */
106
+ export declare function validateBenchSuite(cases: BenchCase[]): {
107
+ valid: boolean;
108
+ errors: string[];
109
+ };
@@ -0,0 +1,149 @@
1
+ /**
2
+ * @hmharness/evaluation - HarmonyBench v1 (P0-04, 2026-09-24 audit)
3
+ *
4
+ * The formal benchmark: 50+ versioned cases across 6 categories, with
5
+ * release baseline tracking, regression detection, and formal scoring.
6
+ * This upgrades the internal bench tool to the "正式 Benchmark" the audit
7
+ * demanded: "带 task manifest、fixture version、expected behavior、scoring、
8
+ * environment lock、release baseline、regression set".
9
+ */
10
+ /**
11
+ * HarmonyBench v1.0.0 - 54 cases across 6 categories.
12
+ * Version-locked: any change to cases requires a version bump.
13
+ */
14
+ export const HARMONYBENCH_VERSION = '1.0.0';
15
+ export const HARMONYBENCH_CASES = [
16
+ // ===== EXACTNESS (10) =====
17
+ { id: 'exact-001', prompt: 'reply with exactly: HELLO', expected: 'exact string HELLO', assertion: { type: 'exact', value: 'HELLO' }, category: 'exactness', difficulty: 1, expectedTurns: 1, needsTools: false },
18
+ { id: 'exact-002', prompt: 'reply with exactly: 42', expected: 'exact string 42', assertion: { type: 'exact', value: '42' }, category: 'exactness', difficulty: 1, expectedTurns: 1, needsTools: false },
19
+ { id: 'exact-003', prompt: 'reply with exactly: {"ok":true}', expected: 'exact JSON', assertion: { type: 'exact', value: '{"ok":true}' }, category: 'exactness', difficulty: 1, expectedTurns: 1, needsTools: false },
20
+ { id: 'exact-004', prompt: 'reply with exactly: 3.14159', expected: 'exact number', assertion: { type: 'exact', value: '3.14159' }, category: 'exactness', difficulty: 1, expectedTurns: 1, needsTools: false },
21
+ { id: 'exact-005', prompt: 'reply with exactly: A', expected: 'single letter', assertion: { type: 'exact', value: 'A' }, category: 'exactness', difficulty: 1, expectedTurns: 1, needsTools: false },
22
+ { id: 'exact-006', prompt: 'reply with exactly: done', expected: 'lowercase done', assertion: { type: 'exact', value: 'done' }, category: 'exactness', difficulty: 1, expectedTurns: 1, needsTools: false },
23
+ { id: 'exact-007', prompt: 'reply with exactly: null', expected: 'null literal', assertion: { type: 'exact', value: 'null' }, category: 'exactness', difficulty: 1, expectedTurns: 1, needsTools: false },
24
+ { id: 'exact-008', prompt: 'reply with exactly: true', expected: 'boolean true', assertion: { type: 'exact', value: 'true' }, category: 'exactness', difficulty: 1, expectedTurns: 1, needsTools: false },
25
+ { id: 'exact-009', prompt: 'reply with exactly: [1,2,3]', expected: 'array literal', assertion: { type: 'exact', value: '[1,2,3]' }, category: 'exactness', difficulty: 2, expectedTurns: 1, needsTools: false },
26
+ { id: 'exact-010', prompt: 'reply with exactly: OK-DONE', expected: 'hyphenated', assertion: { type: 'exact', value: 'OK-DONE' }, category: 'exactness', difficulty: 1, expectedTurns: 1, needsTools: false },
27
+ // ===== CJK / Chinese (10) =====
28
+ { id: 'cjk-001', prompt: '用中文回答:1+1等于几?只输出数字', expected: 'Chinese digit 2', assertion: { type: 'regex', value: '2|二' }, category: 'cjk', difficulty: 1, expectedTurns: 1, needsTools: false },
29
+ { id: 'cjk-002', prompt: '把"hello"翻译成中文,只输出翻译结果', expected: '你好', assertion: { type: 'contains', value: '你好' }, category: 'cjk', difficulty: 1, expectedTurns: 1, needsTools: false },
30
+ { id: 'cjk-003', prompt: '把"世界"翻译成英文,只输出翻译结果', expected: 'world', assertion: { type: 'contains', value: 'world' }, category: 'cjk', difficulty: 1, expectedTurns: 1, needsTools: false },
31
+ { id: 'cjk-004', prompt: '用不超过10个汉字解释什么是JSON', expected: 'short Chinese explanation', assertion: { type: 'regex', value: '[\\u4e00-\\u9fff]{2,10}' }, category: 'cjk', difficulty: 2, expectedTurns: 1, needsTools: false },
32
+ { id: 'cjk-005', prompt: '用不超过10个汉字解释什么是API', expected: 'short Chinese explanation', assertion: { type: 'regex', value: '[\\u4e00-\\u9fff]{2,10}' }, category: 'cjk', difficulty: 2, expectedTurns: 1, needsTools: false },
33
+ { id: 'cjk-006', prompt: '列出3个编程语言,逗号分隔,不要其他内容', expected: '3 languages', assertion: { type: 'regex', value: '\\w+,\\s*\\w+,\\s*\\w+' }, category: 'cjk', difficulty: 1, expectedTurns: 1, needsTools: false },
34
+ { id: 'cjk-007', prompt: '用中文写一句关于春天的诗,不要解释', expected: 'Chinese poem line', assertion: { type: 'regex', value: '[\\u4e00-\\u9fff]{4,}' }, category: 'cjk', difficulty: 2, expectedTurns: 1, needsTools: false },
35
+ { id: 'cjk-008', prompt: '把"开源改变世界"翻译成英文,只输出翻译', expected: 'Open source changes the world', assertion: { type: 'regex', value: '(?i)open.?source' }, category: 'cjk', difficulty: 1, expectedTurns: 1, needsTools: false },
36
+ { id: 'cjk-009', prompt: '一年有多少天?只输出数字', expected: '365', assertion: { type: 'contains', value: '365' }, category: 'cjk', difficulty: 1, expectedTurns: 1, needsTools: false },
37
+ { id: 'cjk-010', prompt: '列出3个欧洲国家,逗号分隔,不要其他内容', expected: '3 European countries', assertion: { type: 'regex', value: '\\w+,\\s*\\w+,\\s*\\w+' }, category: 'cjk', difficulty: 1, expectedTurns: 1, needsTools: false },
38
+ // ===== CODE (10) =====
39
+ { id: 'code-001', prompt: '写一个Python函数 is_palindrome(s),判断回文,只输出代码', expected: 'Python function', assertion: { type: 'contains', value: 'def is_palindrome' }, category: 'code', difficulty: 2, expectedTurns: 1, needsTools: false },
40
+ { id: 'code-002', prompt: '写一个JavaScript函数 sum(arr),返回数组和,只输出代码', expected: 'JS function', assertion: { type: 'contains', value: 'function sum' }, category: 'code', difficulty: 2, expectedTurns: 1, needsTools: false },
41
+ { id: 'code-003', prompt: '写一个SQL查询:从users表选取age>18的name,只输出SQL', expected: 'SELECT statement', assertion: { type: 'regex', value: '(?i)SELECT.*FROM.*users.*WHERE.*age' }, category: 'code', difficulty: 2, expectedTurns: 1, needsTools: false },
42
+ { id: 'code-004', prompt: '用正则表达式匹配中国大陆手机号(1开头11位),只输出正则', expected: 'phone regex', assertion: { type: 'contains', value: '1' }, category: 'code', difficulty: 3, expectedTurns: 1, needsTools: false },
43
+ { id: 'code-005', prompt: '写一个Bash命令:列出当前目录下所有.ts文件,只输出命令', expected: 'find/ls command', assertion: { type: 'regex', value: '(find|ls).*\\.ts' }, category: 'code', difficulty: 2, expectedTurns: 1, needsTools: false },
44
+ { id: 'code-006', prompt: '写一个 TypeScript interface User,包含 name:string 和 age:number,只输出代码', expected: 'TS interface', assertion: { type: 'contains', value: 'interface User' }, category: 'code', difficulty: 2, expectedTurns: 1, needsTools: false },
45
+ { id: 'code-007', prompt: '写一个 Python 列表推导式:生成1到10的平方数列表,只输出代码', expected: 'list comprehension', assertion: { type: 'contains', value: '**2' }, category: 'code', difficulty: 2, expectedTurns: 1, needsTools: false },
46
+ { id: 'code-008', prompt: '写一个 Git 命令:撤销最后一次 commit 但保留更改,只输出命令', expected: 'git reset --soft', assertion: { type: 'contains', value: 'reset' }, category: 'code', difficulty: 2, expectedTurns: 1, needsTools: false },
47
+ { id: 'code-009', prompt: '把 JSON {"a":1} 压缩成一行,只输出结果', expected: 'compressed JSON', assertion: { type: 'exact', value: '{"a":1}' }, category: 'code', difficulty: 1, expectedTurns: 1, needsTools: false },
48
+ { id: 'code-010', prompt: '写一个 CSS 规则:将文字颜色设为红色,只输出CSS', expected: 'color: red', assertion: { type: 'regex', value: 'color:\\s*red' }, category: 'code', difficulty: 1, expectedTurns: 1, needsTools: false },
49
+ // ===== REASONING (8) =====
50
+ { id: 'reason-001', prompt: '一个矩形长12宽8,求面积和周长。格式:面积=X 周长=Y', expected: '面积=96 周长=40', assertion: { type: 'contains', value: '96' }, category: 'reasoning', difficulty: 2, expectedTurns: 1, needsTools: false },
51
+ { id: 'reason-002', prompt: '1到100的和是多少?只输出数字', expected: '5050', assertion: { type: 'contains', value: '5050' }, category: 'reasoning', difficulty: 2, expectedTurns: 1, needsTools: false },
52
+ { id: 'reason-003', prompt: '二分查找的时间复杂度是什么?只输出大O表示', expected: 'O(log n)', assertion: { type: 'regex', value: 'O\\(.*log' }, category: 'reasoning', difficulty: 1, expectedTurns: 1, needsTools: false },
53
+ { id: 'reason-004', prompt: '17+25等于多少?只输出数字', expected: '42', assertion: { type: 'contains', value: '42' }, category: 'reasoning', difficulty: 1, expectedTurns: 1, needsTools: false },
54
+ { id: 'reason-005', prompt: '100除以4等于多少?只输出数字', expected: '25', assertion: { type: 'contains', value: '25' }, category: 'reasoning', difficulty: 1, expectedTurns: 1, needsTools: false },
55
+ { id: 'reason-006', prompt: '7乘以8等于多少?只输出数字', expected: '56', assertion: { type: 'contains', value: '56' }, category: 'reasoning', difficulty: 1, expectedTurns: 1, needsTools: false },
56
+ { id: 'reason-007', prompt: '50减去23等于多少?只输出数字', expected: '27', assertion: { type: 'contains', value: '27' }, category: 'reasoning', difficulty: 1, expectedTurns: 1, needsTools: false },
57
+ { id: 'reason-008', prompt: '把 "hello world foo bar" 按空格分割,输出数组,只输出结果', expected: 'array of 4 strings', assertion: { type: 'regex', value: 'hello.*world.*foo.*bar' }, category: 'reasoning', difficulty: 2, expectedTurns: 1, needsTools: false },
58
+ // ===== TOOLS (8) =====
59
+ { id: 'tool-001', prompt: '读取 package.json 文件并告诉我 name 字段的值', expected: 'reads file and extracts name', assertion: { type: 'regex', value: 'hmharness|name' }, category: 'tools', difficulty: 2, expectedTurns: 3, needsTools: true },
60
+ { id: 'tool-002', prompt: '列出当前目录下的所有文件', expected: 'uses list_directory tool', assertion: { type: 'regex', value: '.+' }, category: 'tools', difficulty: 1, expectedTurns: 2, needsTools: true },
61
+ { id: 'tool-003', prompt: '检查 package.json 的 version 字段', expected: 'reads and reports version', assertion: { type: 'regex', value: '\\d+\\.\\d+' }, category: 'tools', difficulty: 2, expectedTurns: 3, needsTools: true },
62
+ { id: 'tool-004', prompt: '创建一个文件 test-output.txt 内容为 "bench test",然后读回来验证', expected: 'writes and reads back', assertion: { type: 'contains', value: 'bench test' }, category: 'tools', difficulty: 2, expectedTurns: 4, needsTools: true },
63
+ { id: 'tool-005', prompt: '运行 echo hello 并返回输出', expected: 'executes command', assertion: { type: 'contains', value: 'hello' }, category: 'tools', difficulty: 2, expectedTurns: 3, needsTools: true },
64
+ { id: 'tool-006', prompt: '搜索代码中的 "function" 关键词', expected: 'searches code', assertion: { type: 'regex', value: '.+' }, category: 'tools', difficulty: 2, expectedTurns: 3, needsTools: true },
65
+ { id: 'tool-007', prompt: '读取 tsconfig.json 并报告 target 值', expected: 'reads config', assertion: { type: 'regex', value: '(?i)(es\\d+|target)' }, category: 'tools', difficulty: 2, expectedTurns: 3, needsTools: true },
66
+ { id: 'tool-008', prompt: '获取当前工作目录路径', expected: 'returns cwd', assertion: { type: 'regex', value: '[A-Z]:|/' }, category: 'tools', difficulty: 1, expectedTurns: 2, needsTools: true },
67
+ // ===== HARMONY (8) =====
68
+ { id: 'harm-001', prompt: '什么是 HarmonyOS?用一句话回答', expected: 'describes HarmonyOS', assertion: { type: 'contains', value: 'HarmonyOS' }, category: 'harmony', difficulty: 1, expectedTurns: 1, needsTools: false },
69
+ { id: 'harm-002', prompt: 'ArkTS 和 TypeScript 的关系是什么?用一句话回答', expected: 'superset/extension relationship', assertion: { type: 'regex', value: '(?i)(扩展|超集|superset|extension|基于)' }, category: 'harmony', difficulty: 2, expectedTurns: 1, needsTools: false },
70
+ { id: 'harm-003', prompt: '写出鸿蒙 UIAbility 的生命周期回调名称,逗号分隔', expected: 'lifecycle callbacks', assertion: { type: 'regex', value: 'onCreate' }, category: 'harmony', difficulty: 3, expectedTurns: 1, needsTools: false },
71
+ { id: 'harm-004', prompt: '什么是 Stage 模型?用一句话回答', expected: 'describes Stage model', assertion: { type: 'regex', value: '(?i)(stage|模型|model)' }, category: 'harmony', difficulty: 2, expectedTurns: 1, needsTools: false },
72
+ { id: 'harm-005', prompt: 'DevEco Studio 是什么?用一句话回答', expected: 'IDE description', assertion: { type: 'regex', value: '(?i)(IDE|开发|develop)' }, category: 'harmony', difficulty: 1, expectedTurns: 1, needsTools: false },
73
+ { id: 'harm-006', prompt: 'hdc 是什么工具?用一句话回答', expected: 'device connector', assertion: { type: 'regex', value: '(?i)(设备|device|调试|debug|连接)' }, category: 'harmony', difficulty: 1, expectedTurns: 1, needsTools: false },
74
+ { id: 'harm-007', prompt: 'ArkUI 使用什么声明式语法?只输出语法名称', expected: 'declarative UI', assertion: { type: 'regex', value: '(?i)(declar|声明)' }, category: 'harmony', difficulty: 2, expectedTurns: 1, needsTools: false },
75
+ { id: 'harm-008', prompt: 'HAP 是什么文件格式?用一句话回答', expected: 'Harmony Application Package', assertion: { type: 'regex', value: '(?i)(包|package|应用|app)' }, category: 'harmony', difficulty: 1, expectedTurns: 1, needsTools: false },
76
+ ];
77
+ /**
78
+ * Run a single assertion against model output.
79
+ * Pure - testable.
80
+ */
81
+ export function checkAssertion(output, assertion) {
82
+ const trimmed = output.trim();
83
+ switch (assertion.type) {
84
+ case 'exact':
85
+ return trimmed === assertion.value;
86
+ case 'contains':
87
+ return trimmed.includes(assertion.value);
88
+ case 'not_contains':
89
+ return !trimmed.includes(assertion.value);
90
+ case 'regex':
91
+ try {
92
+ return new RegExp(assertion.value, 'i').test(trimmed);
93
+ }
94
+ catch {
95
+ return trimmed.includes(assertion.value);
96
+ }
97
+ default:
98
+ return false;
99
+ }
100
+ }
101
+ /**
102
+ * Build the by-category summary from individual results.
103
+ * Pure - testable.
104
+ */
105
+ export function summarizeByCategory(results) {
106
+ const cats = ['exactness', 'cjk', 'code', 'reasoning', 'tools', 'harmony'];
107
+ const out = {};
108
+ for (const c of cats) {
109
+ const rows = results.filter(r => r.category === c);
110
+ const passed = rows.filter(r => r.pass).length;
111
+ out[c] = { total: rows.length, passed, rate: rows.length ? passed / rows.length : 0 };
112
+ }
113
+ return out;
114
+ }
115
+ /**
116
+ * Detect regressions vs a baseline report.
117
+ * Pure - testable.
118
+ */
119
+ export function detectRegressions(current, baseline) {
120
+ const baseMap = new Map(baseline.map(r => [r.caseId, r.pass]));
121
+ return current
122
+ .filter(r => !r.pass && baseMap.get(r.caseId) === true)
123
+ .map(r => ({ caseId: r.caseId, was: 'pass', now: 'fail' }));
124
+ }
125
+ /**
126
+ * Validate the benchmark suite itself (fixture governance).
127
+ * Every case must have: unique id, non-empty prompt, valid assertion,
128
+ * valid category, valid difficulty.
129
+ */
130
+ export function validateBenchSuite(cases) {
131
+ const errors = [];
132
+ const ids = new Set();
133
+ for (const c of cases) {
134
+ if (ids.has(c.id))
135
+ errors.push(`duplicate id: ${c.id}`);
136
+ ids.add(c.id);
137
+ if (!c.prompt.trim())
138
+ errors.push(`empty prompt: ${c.id}`);
139
+ if (!c.assertion.value)
140
+ errors.push(`empty assertion: ${c.id}`);
141
+ if (!['exact', 'contains', 'not_contains', 'regex'].includes(c.assertion.type))
142
+ errors.push(`invalid assertion type: ${c.id}`);
143
+ if (![1, 2, 3].includes(c.difficulty))
144
+ errors.push(`invalid difficulty: ${c.id}`);
145
+ }
146
+ if (cases.length < 50)
147
+ errors.push(`only ${cases.length} cases, need >=50 for HarmonyBench v1`);
148
+ return { valid: errors.length === 0, errors };
149
+ }
package/dist/index.d.ts CHANGED
@@ -1,3 +1,4 @@
1
1
  export * from './types.ts';
2
2
  export * from './evaluators.ts';
3
3
  export * from './runner.ts';
4
+ export * from './harmonybench.ts';
package/dist/index.js CHANGED
@@ -1,3 +1,4 @@
1
1
  export * from "./types.js";
2
2
  export * from "./evaluators.js";
3
3
  export * from "./runner.js";
4
+ export * from "./harmonybench.js";
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@hmharness/evaluation",
3
- "version": "0.8.0",
3
+ "version": "0.8.1",
4
4
  "description": "hmharness evaluation: the Evaluator/Judge contract (V2 blueprint M2). Hard evidence outranks LLM judgment - build results, exit codes, exact/regex assertions first; the LLM judge is a last resort and is labeled as such. Evaluations attach to trajectories (judge.completed events).",
5
5
  "type": "module",
6
6
  "main": "dist/index.js",