@hmharness/evaluation 0.8.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/evaluators.d.ts +36 -0
- package/dist/evaluators.js +125 -0
- package/dist/index.d.ts +3 -0
- package/dist/index.js +3 -0
- package/dist/runner.d.ts +35 -0
- package/dist/runner.js +67 -0
- package/dist/types.d.ts +55 -0
- package/dist/types.js +29 -0
- package/package.json +32 -0
|
@@ -0,0 +1,36 @@
|
|
|
1
|
+
import type { Evaluator } from './types.ts';
|
|
2
|
+
export interface TextAssertionInput {
|
|
3
|
+
output: string;
|
|
4
|
+
/** ALL substrings must appear (case-insensitive). */
|
|
5
|
+
expect?: string[];
|
|
6
|
+
/** Output must equal this (trimmed). */
|
|
7
|
+
expectExact?: string;
|
|
8
|
+
/** Output must match this regex. */
|
|
9
|
+
expectRegex?: string;
|
|
10
|
+
/** NONE of these may appear. */
|
|
11
|
+
expectNone?: string[];
|
|
12
|
+
/** At least ONE must appear. */
|
|
13
|
+
expectAny?: string[];
|
|
14
|
+
}
|
|
15
|
+
export declare const textAssertionEvaluator: Evaluator<TextAssertionInput>;
|
|
16
|
+
export interface CommandInput {
|
|
17
|
+
/** Executable to run (NO shell string - argv array style only, per shellgate doctrine). */
|
|
18
|
+
command: string;
|
|
19
|
+
args?: string[];
|
|
20
|
+
cwd?: string;
|
|
21
|
+
timeoutMs?: number;
|
|
22
|
+
/** Substrings expected in stdout when the command is considered successful. */
|
|
23
|
+
expectOut?: string[];
|
|
24
|
+
/** Forbid these in stdout+stderr (failure markers). */
|
|
25
|
+
forbidOut?: string[];
|
|
26
|
+
}
|
|
27
|
+
export declare const commandEvaluator: Evaluator<CommandInput>;
|
|
28
|
+
export interface LlmJudgeInput {
|
|
29
|
+
task: string;
|
|
30
|
+
output: string;
|
|
31
|
+
criteria: string[];
|
|
32
|
+
/** Injected provider call - tests substitute; production passes kernel chat(). */
|
|
33
|
+
call: (system: string, user: string) => Promise<string>;
|
|
34
|
+
}
|
|
35
|
+
export declare const llmJudgeEvaluator: Evaluator<LlmJudgeInput>;
|
|
36
|
+
export declare const allEvaluators: Evaluator<never>[];
|
|
@@ -0,0 +1,125 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @hmharness/evaluation - concrete evaluators (hard evidence first).
|
|
3
|
+
*/
|
|
4
|
+
import { execFile } from 'node:child_process';
|
|
5
|
+
import { promisify } from 'node:util';
|
|
6
|
+
import { scoreFromEvidence } from "./types.js";
|
|
7
|
+
const execCb = promisify(execFile);
|
|
8
|
+
function textEvidence(input) {
|
|
9
|
+
const evidence = [];
|
|
10
|
+
const failures = [];
|
|
11
|
+
const out = input.output ?? '';
|
|
12
|
+
const lower = out.toLowerCase();
|
|
13
|
+
if (input.expectExact !== undefined) {
|
|
14
|
+
const ok = out.trim() === input.expectExact.trim();
|
|
15
|
+
evidence.push({ kind: 'static', detail: `exact match ${ok ? 'hit' : 'miss'}`, passed: ok });
|
|
16
|
+
if (!ok)
|
|
17
|
+
failures.push({ reason: `exact mismatch: got "${out.trim().slice(0, 80)}"` });
|
|
18
|
+
}
|
|
19
|
+
if (input.expectRegex) {
|
|
20
|
+
let ok = false;
|
|
21
|
+
try {
|
|
22
|
+
ok = new RegExp(input.expectRegex).test(out);
|
|
23
|
+
}
|
|
24
|
+
catch {
|
|
25
|
+
failures.push({ reason: `invalid expectRegex: ${input.expectRegex.slice(0, 60)}` });
|
|
26
|
+
}
|
|
27
|
+
evidence.push({ kind: 'static', detail: `regex ${ok ? 'matched' : 'no match'}`, passed: ok });
|
|
28
|
+
if (!ok)
|
|
29
|
+
failures.push({ reason: `regex did not match: ${input.expectRegex.slice(0, 60)}` });
|
|
30
|
+
}
|
|
31
|
+
if (input.expect?.length) {
|
|
32
|
+
const missing = input.expect.filter((e) => !lower.includes(e.toLowerCase()));
|
|
33
|
+
evidence.push({ kind: 'static', detail: `all-substrings ${input.expect.length - missing.length}/${input.expect.length}`, passed: missing.length === 0 });
|
|
34
|
+
if (missing.length)
|
|
35
|
+
failures.push({ reason: `missing substrings: ${missing.join(' && ').slice(0, 120)}` });
|
|
36
|
+
}
|
|
37
|
+
if (input.expectNone?.length) {
|
|
38
|
+
const leaked = input.expectNone.filter((e) => lower.includes(e.toLowerCase()));
|
|
39
|
+
evidence.push({ kind: 'static', detail: `forbidden-substrings clean: ${leaked.length === 0}`, passed: leaked.length === 0 });
|
|
40
|
+
if (leaked.length)
|
|
41
|
+
failures.push({ reason: `forbidden substrings present: ${leaked.join(' && ').slice(0, 120)}` });
|
|
42
|
+
}
|
|
43
|
+
if (input.expectAny?.length) {
|
|
44
|
+
const hit = input.expectAny.find((e) => lower.includes(e.toLowerCase()));
|
|
45
|
+
evidence.push({ kind: 'static', detail: `any-substring ${hit ? `hit: ${hit.slice(0, 40)}` : 'none'}`, passed: Boolean(hit) });
|
|
46
|
+
if (!hit)
|
|
47
|
+
failures.push({ reason: `none of the any-of substrings appeared` });
|
|
48
|
+
}
|
|
49
|
+
return { evidence, failures };
|
|
50
|
+
}
|
|
51
|
+
export const textAssertionEvaluator = {
|
|
52
|
+
id: 'text-assertion',
|
|
53
|
+
version: '1.0.0',
|
|
54
|
+
evidenceKind: 'static',
|
|
55
|
+
description: 'Structured text assertions: exact / regex / all-substrings / forbidden / any-of (the bench gate semantics, as an Evaluator).',
|
|
56
|
+
async evaluate(input) {
|
|
57
|
+
const t0 = Date.now();
|
|
58
|
+
const { evidence, failures } = textEvidence(input);
|
|
59
|
+
const { score, passed } = scoreFromEvidence(evidence, failures);
|
|
60
|
+
return { score, passed, evidence, failures, evaluatorId: this.id, evaluatorVersion: this.version, durationMs: Date.now() - t0 };
|
|
61
|
+
},
|
|
62
|
+
};
|
|
63
|
+
export const commandEvaluator = {
|
|
64
|
+
id: 'command-exit',
|
|
65
|
+
version: '1.0.0',
|
|
66
|
+
evidenceKind: 'build',
|
|
67
|
+
description: 'Run a command (execFile, no shell) and treat exit code + output as hard evidence - the build/test tier of the ladder.',
|
|
68
|
+
async evaluate(input) {
|
|
69
|
+
const t0 = Date.now();
|
|
70
|
+
const evidence = [];
|
|
71
|
+
const failures = [];
|
|
72
|
+
try {
|
|
73
|
+
const r = await execCb(input.command, input.args ?? [], { cwd: input.cwd, timeout: input.timeoutMs ?? 120_000, windowsHide: true, maxBuffer: 4 * 1024 * 1024 });
|
|
74
|
+
const out = String(r.stdout ?? '') + String(r.stderr ?? '');
|
|
75
|
+
evidence.push({ kind: 'build', detail: `exit 0 in ${Date.now() - t0}ms; output head: ${out.replace(/\s+/g, ' ').slice(0, 120)}`, passed: true });
|
|
76
|
+
if (input.expectOut?.length) {
|
|
77
|
+
const missing = input.expectOut.filter((e) => !out.toLowerCase().includes(e.toLowerCase()));
|
|
78
|
+
evidence.push({ kind: 'build', detail: `expected output ${input.expectOut.length - missing.length}/${input.expectOut.length}`, passed: missing.length === 0 });
|
|
79
|
+
if (missing.length)
|
|
80
|
+
failures.push({ reason: `expected output missing: ${missing.join(' && ')}` });
|
|
81
|
+
}
|
|
82
|
+
if (input.forbidOut?.length) {
|
|
83
|
+
const leaked = input.forbidOut.filter((e) => out.toLowerCase().includes(e.toLowerCase()));
|
|
84
|
+
evidence.push({ kind: 'build', detail: `forbidden markers clean: ${leaked.length === 0}`, passed: leaked.length === 0 });
|
|
85
|
+
if (leaked.length)
|
|
86
|
+
failures.push({ reason: `forbidden markers present: ${leaked.join(' && ')}` });
|
|
87
|
+
}
|
|
88
|
+
}
|
|
89
|
+
catch (err) {
|
|
90
|
+
const e = err;
|
|
91
|
+
const out = String(e.stdout ?? '') + String(e.stderr ?? '');
|
|
92
|
+
evidence.push({ kind: 'build', detail: `exit ${e.code ?? '?'}${e.killed ? ' (timed out)' : ''}; output head: ${out.replace(/\s+/g, ' ').slice(0, 120)}`, passed: false });
|
|
93
|
+
failures.push({ reason: `command failed with exit ${e.code ?? '?'}: ${out.replace(/\s+/g, ' ').slice(0, 140)}` });
|
|
94
|
+
}
|
|
95
|
+
const { score, passed } = scoreFromEvidence(evidence, failures);
|
|
96
|
+
return { score, passed, evidence, failures, evaluatorId: this.id, evaluatorVersion: this.version, durationMs: Date.now() - t0 };
|
|
97
|
+
},
|
|
98
|
+
};
|
|
99
|
+
export const llmJudgeEvaluator = {
|
|
100
|
+
id: 'llm-judge',
|
|
101
|
+
version: '1.0.0',
|
|
102
|
+
evidenceKind: 'llmJudge',
|
|
103
|
+
description: 'LLM judge - LAST resort on the evidence ladder. Its verdict alone can never mark a run fully passed (score hard-capped at 0.7).',
|
|
104
|
+
async evaluate(input) {
|
|
105
|
+
const t0 = Date.now();
|
|
106
|
+
const evidence = [];
|
|
107
|
+
const failures = [];
|
|
108
|
+
try {
|
|
109
|
+
const verdict = await input.call('You are an independent evaluator. Judge ONLY from the task, the output, and the criteria. Never trust the agent\'s self-assessment. Reply with exactly PASS or FAIL on the first line, then one short reason.', `Task: ${input.task.slice(0, 500)}\nCriteria:\n${input.criteria.map((c) => `- ${c}`).join('\n')}\nOutput:\n${input.output.slice(0, 4000)}`);
|
|
110
|
+
const pass = /^\s*PASS\b/i.test(verdict);
|
|
111
|
+
evidence.push({ kind: 'llmJudge', detail: verdict.replace(/\s+/g, ' ').slice(0, 160), passed: pass });
|
|
112
|
+
if (!pass)
|
|
113
|
+
failures.push({ reason: `judge: ${verdict.replace(/\s+/g, ' ').slice(0, 140)}` });
|
|
114
|
+
}
|
|
115
|
+
catch (err) {
|
|
116
|
+
failures.push({ reason: `judge call failed: ${String(err).slice(0, 120)}` });
|
|
117
|
+
evidence.push({ kind: 'llmJudge', detail: 'judge unavailable', passed: false });
|
|
118
|
+
}
|
|
119
|
+
const { score, passed } = scoreFromEvidence(evidence, failures);
|
|
120
|
+
// judge-only pass is capped by scoreFromEvidence (0.7) and can still be
|
|
121
|
+
// 'passed' for advisory purposes; promotion gates must combine with harder evidence
|
|
122
|
+
return { score, passed, evidence, failures, evaluatorId: this.id, evaluatorVersion: this.version, durationMs: Date.now() - t0 };
|
|
123
|
+
},
|
|
124
|
+
};
|
|
125
|
+
export const allEvaluators = [textAssertionEvaluator, commandEvaluator, llmJudgeEvaluator];
|
package/dist/index.d.ts
ADDED
package/dist/index.js
ADDED
package/dist/runner.d.ts
ADDED
|
@@ -0,0 +1,35 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @hmharness/evaluation - trajectory linkage + bench bridge (M2).
|
|
3
|
+
*
|
|
4
|
+
* evaluateRun(): judge a finished run from its TRAJECTORY RECORD - the
|
|
5
|
+
* outcome, tool completion ratio and error observations - never from the
|
|
6
|
+
* agent's word for itself. The verdict is appended back as judge.completed
|
|
7
|
+
* events on a linked judge-run, so every promotion decision is auditable.
|
|
8
|
+
* runBenchCase(): the existing bench gate semantics through the Evaluator
|
|
9
|
+
* contract, so evolution and evaluation share one assertion core.
|
|
10
|
+
*/
|
|
11
|
+
import { type Trajectory } from '@hmharness/observability';
|
|
12
|
+
import { type TextAssertionInput, type EvaluationResult } from './index.ts';
|
|
13
|
+
/** Score a run from its recorded trajectory metrics (hard evidence only). */
|
|
14
|
+
export declare function evaluateTrajectory(trajectory: Trajectory): EvaluationResult;
|
|
15
|
+
/** Evaluate a completed run from its trajectory and record the verdict back
|
|
16
|
+
* onto a linked judge-run (part of the auditable record). */
|
|
17
|
+
export declare function evaluateRun(home: string, runId: string, _assertions?: TextAssertionInput): Promise<EvaluationResult & {
|
|
18
|
+
runId: string;
|
|
19
|
+
}>;
|
|
20
|
+
/** Bench gate semantics through the Evaluator contract (shared assertion core). */
|
|
21
|
+
export declare function runBenchCase(c: {
|
|
22
|
+
name: string;
|
|
23
|
+
prompt: string;
|
|
24
|
+
output: string;
|
|
25
|
+
expect?: string[];
|
|
26
|
+
expectExact?: string;
|
|
27
|
+
expectRegex?: string;
|
|
28
|
+
expectNone?: string[];
|
|
29
|
+
expectAny?: string[];
|
|
30
|
+
}): Promise<{
|
|
31
|
+
name: string;
|
|
32
|
+
pass: boolean;
|
|
33
|
+
detail: string;
|
|
34
|
+
score: number;
|
|
35
|
+
}>;
|
package/dist/runner.js
ADDED
|
@@ -0,0 +1,67 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @hmharness/evaluation - trajectory linkage + bench bridge (M2).
|
|
3
|
+
*
|
|
4
|
+
* evaluateRun(): judge a finished run from its TRAJECTORY RECORD - the
|
|
5
|
+
* outcome, tool completion ratio and error observations - never from the
|
|
6
|
+
* agent's word for itself. The verdict is appended back as judge.completed
|
|
7
|
+
* events on a linked judge-run, so every promotion decision is auditable.
|
|
8
|
+
* runBenchCase(): the existing bench gate semantics through the Evaluator
|
|
9
|
+
* contract, so evolution and evaluation share one assertion core.
|
|
10
|
+
*/
|
|
11
|
+
import { jsonlTrajectoryStore, createTrajectoryRecorder } from '@hmharness/observability';
|
|
12
|
+
import { textAssertionEvaluator } from "./index.js";
|
|
13
|
+
/** Score a run from its recorded trajectory metrics (hard evidence only). */
|
|
14
|
+
export function evaluateTrajectory(trajectory) {
|
|
15
|
+
const t0 = Date.now();
|
|
16
|
+
const evidence = [];
|
|
17
|
+
const failures = [];
|
|
18
|
+
const outcome = trajectory.outcome;
|
|
19
|
+
if (outcome) {
|
|
20
|
+
evidence.push({ kind: 'runtime', detail: `outcome ${outcome.success ? 'success' : 'failed'} (${outcome.reason ?? '?'})`, passed: outcome.success });
|
|
21
|
+
if (!outcome.success)
|
|
22
|
+
failures.push({ reason: `run outcome: ${outcome.reason ?? 'failed'}${outcome.error ? ` - ${outcome.error}` : ''}` });
|
|
23
|
+
}
|
|
24
|
+
else {
|
|
25
|
+
failures.push({ reason: 'run has no outcome recorded (unfinished?)' });
|
|
26
|
+
}
|
|
27
|
+
const toolDone = trajectory.events.filter((e) => e.type === 'tool.completed');
|
|
28
|
+
const toolFailed = toolDone.filter((e) => e.payload?.isError === true);
|
|
29
|
+
if (toolDone.length > 0) {
|
|
30
|
+
const ratio = (toolDone.length - toolFailed.length) / toolDone.length;
|
|
31
|
+
evidence.push({ kind: 'runtime', detail: `tool completion ${toolDone.length - toolFailed.length}/${toolDone.length}`, passed: ratio >= 0.5 });
|
|
32
|
+
if (ratio < 0.5)
|
|
33
|
+
failures.push({ reason: `majority of tool calls failed (${toolFailed.length}/${toolDone.length})` });
|
|
34
|
+
}
|
|
35
|
+
const errors = trajectory.events.filter((e) => e.type === 'error.observed').length;
|
|
36
|
+
evidence.push({ kind: 'runtime', detail: `${errors} error observation(s)`, passed: errors === 0 });
|
|
37
|
+
const score = failures.length === 0 ? 1 : Math.max(0, 1 - failures.length * 0.34);
|
|
38
|
+
return {
|
|
39
|
+
score, passed: failures.length === 0, evidence, failures,
|
|
40
|
+
evaluatorId: 'trajectory-metrics', evaluatorVersion: '1.0.0', durationMs: Date.now() - t0,
|
|
41
|
+
};
|
|
42
|
+
}
|
|
43
|
+
/** Evaluate a completed run from its trajectory and record the verdict back
|
|
44
|
+
* onto a linked judge-run (part of the auditable record). */
|
|
45
|
+
export async function evaluateRun(home, runId, _assertions) {
|
|
46
|
+
const store = jsonlTrajectoryStore(home);
|
|
47
|
+
const trajectory = await store.getRun(runId);
|
|
48
|
+
const result = evaluateTrajectory(trajectory);
|
|
49
|
+
const rec = createTrajectoryRecorder(home, { task: `(judge) ${trajectory.task.slice(0, 80)}`, cwd: trajectory.cwd });
|
|
50
|
+
rec.emit('judge.started', 'judge', { subjectRunId: runId, evaluator: 'trajectory-metrics' });
|
|
51
|
+
rec.emit('judge.completed', 'judge', {
|
|
52
|
+
subjectRunId: runId, evaluator: 'trajectory-metrics',
|
|
53
|
+
passed: result.passed, score: result.score, failures: result.failures.map((f) => f.reason),
|
|
54
|
+
});
|
|
55
|
+
rec.finish({ success: result.passed, reason: 'final' });
|
|
56
|
+
return { ...result, runId };
|
|
57
|
+
}
|
|
58
|
+
/** Bench gate semantics through the Evaluator contract (shared assertion core). */
|
|
59
|
+
export async function runBenchCase(c) {
|
|
60
|
+
void textAssertionEvaluator; // assertion core shared via evaluators.ts
|
|
61
|
+
const r = await textAssertionEvaluator.evaluate({
|
|
62
|
+
output: c.output,
|
|
63
|
+
expect: c.expect, expectExact: c.expectExact, expectRegex: c.expectRegex,
|
|
64
|
+
expectNone: c.expectNone, expectAny: c.expectAny,
|
|
65
|
+
});
|
|
66
|
+
return { name: c.name, pass: r.passed, detail: r.failures.map((f) => f.reason).join('; ') || 'ok', score: r.score };
|
|
67
|
+
}
|
package/dist/types.d.ts
ADDED
|
@@ -0,0 +1,55 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @hmharness/evaluation - the Evaluator contract (V2 blueprint M2).
|
|
3
|
+
*
|
|
4
|
+
* Evidence outranks narrative. The evidence ladder (ADR-0001 rule 4):
|
|
5
|
+
* a build failure is FACT, an LLM opinion is a guess with good manners.
|
|
6
|
+
* Every evaluation carries machine-checkable evidence references so a
|
|
7
|
+
* promotion decision can be audited without re-running anything.
|
|
8
|
+
*/
|
|
9
|
+
/** Where a piece of evidence sits on the trust ladder - LOWER number = harder. */
|
|
10
|
+
export declare const EVIDENCE_RANK: {
|
|
11
|
+
readonly build: 1;
|
|
12
|
+
readonly tests: 2;
|
|
13
|
+
readonly static: 3;
|
|
14
|
+
readonly runtime: 4;
|
|
15
|
+
readonly deviceLogs: 5;
|
|
16
|
+
readonly screenshot: 6;
|
|
17
|
+
readonly llmJudge: 7;
|
|
18
|
+
readonly selfReport: 8;
|
|
19
|
+
};
|
|
20
|
+
export type EvidenceKind = keyof typeof EVIDENCE_RANK;
|
|
21
|
+
export interface Evidence {
|
|
22
|
+
kind: EvidenceKind;
|
|
23
|
+
/** What was observed, verbatim where possible (bounded). */
|
|
24
|
+
detail: string;
|
|
25
|
+
passed?: boolean;
|
|
26
|
+
}
|
|
27
|
+
export interface Failure {
|
|
28
|
+
reason: string;
|
|
29
|
+
evidence?: Evidence;
|
|
30
|
+
}
|
|
31
|
+
export interface EvaluationResult<E = Evidence> {
|
|
32
|
+
/** 0..1 */
|
|
33
|
+
score: number;
|
|
34
|
+
passed: boolean;
|
|
35
|
+
evidence: E[];
|
|
36
|
+
failures: Failure[];
|
|
37
|
+
evaluatorId: string;
|
|
38
|
+
evaluatorVersion: string;
|
|
39
|
+
durationMs: number;
|
|
40
|
+
/** The run this evaluation judged, when tied to a trajectory. */
|
|
41
|
+
runId?: string;
|
|
42
|
+
}
|
|
43
|
+
export interface Evaluator<Input = unknown> {
|
|
44
|
+
id: string;
|
|
45
|
+
version: string;
|
|
46
|
+
/** Where this evaluator's evidence sits on the trust ladder. */
|
|
47
|
+
evidenceKind: EvidenceKind;
|
|
48
|
+
description: string;
|
|
49
|
+
evaluate(input: Input): Promise<EvaluationResult>;
|
|
50
|
+
}
|
|
51
|
+
/** Shared scoring helper: all-or-nothing evidence -> 1 or 0; partial allowed. */
|
|
52
|
+
export declare function scoreFromEvidence(evidence: Evidence[], failures: Failure[]): {
|
|
53
|
+
score: number;
|
|
54
|
+
passed: boolean;
|
|
55
|
+
};
|
package/dist/types.js
ADDED
|
@@ -0,0 +1,29 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @hmharness/evaluation - the Evaluator contract (V2 blueprint M2).
|
|
3
|
+
*
|
|
4
|
+
* Evidence outranks narrative. The evidence ladder (ADR-0001 rule 4):
|
|
5
|
+
* a build failure is FACT, an LLM opinion is a guess with good manners.
|
|
6
|
+
* Every evaluation carries machine-checkable evidence references so a
|
|
7
|
+
* promotion decision can be audited without re-running anything.
|
|
8
|
+
*/
|
|
9
|
+
/** Where a piece of evidence sits on the trust ladder - LOWER number = harder. */
|
|
10
|
+
export const EVIDENCE_RANK = {
|
|
11
|
+
build: 1,
|
|
12
|
+
tests: 2,
|
|
13
|
+
static: 3,
|
|
14
|
+
runtime: 4,
|
|
15
|
+
deviceLogs: 5,
|
|
16
|
+
screenshot: 6,
|
|
17
|
+
llmJudge: 7,
|
|
18
|
+
selfReport: 8,
|
|
19
|
+
};
|
|
20
|
+
/** Shared scoring helper: all-or-nothing evidence -> 1 or 0; partial allowed. */
|
|
21
|
+
export function scoreFromEvidence(evidence, failures) {
|
|
22
|
+
const ranked = [...evidence].sort((a, b) => EVIDENCE_RANK[a.kind] - EVIDENCE_RANK[b.kind]);
|
|
23
|
+
const passing = ranked.filter((e) => e.passed !== false).length;
|
|
24
|
+
// an llmJudge/selfReport-only pass is never a full pass (evidence-first rule)
|
|
25
|
+
const weakestPass = ranked.length > 0 ? EVIDENCE_RANK[ranked[ranked.length - 1].kind] : EVIDENCE_RANK.selfReport;
|
|
26
|
+
const hardCap = weakestPass >= EVIDENCE_RANK.llmJudge ? 0.7 : 1;
|
|
27
|
+
const ratio = ranked.length > 0 ? passing / ranked.length : 0;
|
|
28
|
+
return { score: Math.min(hardCap, ratio), passed: failures.length === 0 && ratio === 1 };
|
|
29
|
+
}
|
package/package.json
ADDED
|
@@ -0,0 +1,32 @@
|
|
|
1
|
+
{
|
|
2
|
+
"name": "@hmharness/evaluation",
|
|
3
|
+
"version": "0.8.0",
|
|
4
|
+
"description": "hmharness evaluation: the Evaluator/Judge contract (V2 blueprint M2). Hard evidence outranks LLM judgment - build results, exit codes, exact/regex assertions first; the LLM judge is a last resort and is labeled as such. Evaluations attach to trajectories (judge.completed events).",
|
|
5
|
+
"type": "module",
|
|
6
|
+
"main": "dist/index.js",
|
|
7
|
+
"exports": {
|
|
8
|
+
".": {
|
|
9
|
+
"types": "./dist/index.d.ts",
|
|
10
|
+
"default": "./dist/index.js"
|
|
11
|
+
}
|
|
12
|
+
},
|
|
13
|
+
"types": "dist/index.d.ts",
|
|
14
|
+
"scripts": {
|
|
15
|
+
"build": "tsc -p tsconfig.build.json"
|
|
16
|
+
},
|
|
17
|
+
"dependencies": {
|
|
18
|
+
"@hmharness/kernel": "0.8.0",
|
|
19
|
+
"@hmharness/observability": "0.7.0"
|
|
20
|
+
},
|
|
21
|
+
"files": [
|
|
22
|
+
"dist"
|
|
23
|
+
],
|
|
24
|
+
"license": "MIT",
|
|
25
|
+
"repository": {
|
|
26
|
+
"type": "git",
|
|
27
|
+
"url": "git+https://github.com/swsgbl/hmharness.git"
|
|
28
|
+
},
|
|
29
|
+
"engines": {
|
|
30
|
+
"node": ">=22"
|
|
31
|
+
}
|
|
32
|
+
}
|