@hamza1331/aieval 0.0.0-stage → 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +28 -0
- package/LICENSE +21 -0
- package/README.md +225 -2
- package/dist/checks.d.ts +39 -0
- package/dist/checks.js +396 -0
- package/dist/cli.d.ts +2 -0
- package/dist/cli.js +20 -0
- package/dist/cliMain.d.ts +7 -0
- package/dist/cliMain.js +170 -0
- package/dist/index.d.ts +13 -0
- package/dist/index.js +83 -0
- package/dist/registry.d.ts +46 -0
- package/dist/registry.js +82 -0
- package/dist/runner.d.ts +47 -0
- package/dist/runner.js +143 -0
- package/dist/semantic.d.ts +36 -0
- package/dist/semantic.js +74 -0
- package/dist/types.d.ts +42 -0
- package/dist/types.js +1 -0
- package/package.json +77 -4
package/CHANGELOG.md
ADDED
|
@@ -0,0 +1,28 @@
|
|
|
1
|
+
# Changelog
|
|
2
|
+
|
|
3
|
+
All notable changes to this project are documented here. The format is based on
|
|
4
|
+
[Keep a Changelog](https://keepachangelog.com/en/1.1.0/), and this project adheres to
|
|
5
|
+
[Semantic Versioning](https://semver.org/spec/v2.0.0.html).
|
|
6
|
+
|
|
7
|
+
## [Unreleased]
|
|
8
|
+
|
|
9
|
+
## [0.1.0]
|
|
10
|
+
|
|
11
|
+
First release.
|
|
12
|
+
|
|
13
|
+
### Added
|
|
14
|
+
|
|
15
|
+
- Core API: `evaluate`, `validate`, `assert`, and the `Check`, `Failure`, `Severity` and `EvaluationResult` model.
|
|
16
|
+
- Severity-based routing with a `failOn` threshold (default `"error"`); lower severities are reported as warnings.
|
|
17
|
+
- Checks: `schemaCheck` (Zod), `requiredFields`, `noExtraFields`, `enumCheck`, `fieldTypeCheck`, `jsonCheck`,
|
|
18
|
+
`nonEmpty`, `regexCheck`, `rangeCheck`, `toolCallCheck`, `toolCallCheck.oneOf`, `customCheck`.
|
|
19
|
+
- Tool-call validation for `{ name, args }` and OpenAI-style `{ name, arguments: "<json>" }` calls.
|
|
20
|
+
- Optional semantic judging: `semanticCheck`, the `SemanticJudge` interface and `mockJudge`.
|
|
21
|
+
- JSON check registry (`buildChecks`) and fixture runner (`loadFixtures`, `runFixture`, `runAll`).
|
|
22
|
+
- CLI: `aieval run` and `aieval check`, with `--fail-on`, `--format text|json` and stable exit codes.
|
|
23
|
+
- Per-check metadata in `result.metadata.checks` and `durationMs`.
|
|
24
|
+
- `EvaluationError` thrown by `assert`, carrying the full result.
|
|
25
|
+
- Checks that throw are converted to `critical` failures instead of crashing the run.
|
|
26
|
+
|
|
27
|
+
[Unreleased]: https://github.com/hamza1331/aieval/compare/v0.1.0...HEAD
|
|
28
|
+
[0.1.0]: https://github.com/hamza1331/aieval/releases/tag/v0.1.0
|
package/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
package/README.md
CHANGED
|
@@ -1,3 +1,226 @@
|
|
|
1
|
-
#
|
|
1
|
+
# aieval
|
|
2
2
|
|
|
3
|
-
|
|
3
|
+
[](https://github.com/hamza1331/aieval/actions/workflows/ci.yml)
|
|
4
|
+
[](https://www.npmjs.com/package/@hamza1331/aieval)
|
|
5
|
+
[](https://opensource.org/licenses/MIT)
|
|
6
|
+
|
|
7
|
+
TypeScript-first, deterministic-first validation for LLM outputs and agent tool calls.
|
|
8
|
+
|
|
9
|
+
Catch the failures you can detect with code — malformed JSON, missing fields, enum drift, bad tool-call arguments,
|
|
10
|
+
policy violations — **before** you spend time and money on LLM-as-a-judge scoring. Failures are structured and
|
|
11
|
+
machine-readable, so they are easy to log, assert on, or feed back to a model for a retry.
|
|
12
|
+
|
|
13
|
+
- Small and composable: plain functions, no framework, no hosted service
|
|
14
|
+
- Strong typing and no runtime dependencies (`zod` is a peer dependency, used by `schemaCheck`)
|
|
15
|
+
- A fixture-driven CLI for local checks and CI
|
|
16
|
+
- Optional, clearly-labelled semantic (LLM) judging behind a provider-agnostic interface
|
|
17
|
+
|
|
18
|
+
## Install
|
|
19
|
+
|
|
20
|
+
```bash
|
|
21
|
+
npm install @hamza1331/aieval zod
|
|
22
|
+
```
|
|
23
|
+
|
|
24
|
+
Requires Node.js 20+. `zod` (v4) is a peer dependency used by `schemaCheck`.
|
|
25
|
+
|
|
26
|
+
## Quick start
|
|
27
|
+
|
|
28
|
+
```ts
|
|
29
|
+
import { z } from "zod";
|
|
30
|
+
import { evaluate, requiredFields, schemaCheck } from "@hamza1331/aieval";
|
|
31
|
+
|
|
32
|
+
const schema = z.object({ name: z.string(), email: z.string() });
|
|
33
|
+
|
|
34
|
+
const result = await evaluate({
|
|
35
|
+
output: { name: "Ada Lovelace", email: "ada@example.com" },
|
|
36
|
+
checks: [schemaCheck(schema), requiredFields(["name", "email"])],
|
|
37
|
+
});
|
|
38
|
+
|
|
39
|
+
console.log(result.passed); // true
|
|
40
|
+
console.log(result.failures); // []
|
|
41
|
+
```
|
|
42
|
+
|
|
43
|
+
Other entry points:
|
|
44
|
+
|
|
45
|
+
```ts
|
|
46
|
+
import { assert, validate } from "@hamza1331/aieval";
|
|
47
|
+
|
|
48
|
+
const result = await validate(output, checks); // same as evaluate({ output, checks })
|
|
49
|
+
await assert(output, checks); // throws EvaluationError (with `.result`) if validation fails
|
|
50
|
+
```
|
|
51
|
+
|
|
52
|
+
## Results and severity
|
|
53
|
+
|
|
54
|
+
Every issue is a `Failure`:
|
|
55
|
+
|
|
56
|
+
```ts
|
|
57
|
+
interface Failure {
|
|
58
|
+
code: FailureCode; // "missing_field" | "invalid_type" | "invalid_enum" | "schema_violation" | "invalid_json" | ...
|
|
59
|
+
message: string;
|
|
60
|
+
severity: "info" | "warning" | "error" | "critical";
|
|
61
|
+
path?: string; // e.g. "user.age" or "args.city"
|
|
62
|
+
metadata?: Record<string, unknown>; // always includes `checkName`
|
|
63
|
+
suggestedRepair?: string;
|
|
64
|
+
}
|
|
65
|
+
```
|
|
66
|
+
|
|
67
|
+
**Severity decides what fails.** Issues at or above `failOn` (default `"error"`) go to `result.failures` and make
|
|
68
|
+
`result.passed` false. Lower-severity issues go to `result.warnings` and do not fail the run.
|
|
69
|
+
|
|
70
|
+
```ts
|
|
71
|
+
await evaluate({ output, checks }); // warnings are reported, run still passes
|
|
72
|
+
await evaluate({ output, checks, failOn: "warning" }); // strict mode: warnings fail the run
|
|
73
|
+
```
|
|
74
|
+
|
|
75
|
+
If a check throws, `evaluate` records a `critical` failure and keeps going. `result.metadata` carries `durationMs`
|
|
76
|
+
and, for checks that report data (like semantic scores), `checks[checkName]`.
|
|
77
|
+
|
|
78
|
+
## Built-in checks
|
|
79
|
+
|
|
80
|
+
| Check | What it verifies | Failure code(s) |
|
|
81
|
+
| --------------------------------------- | ---------------------------------------------------------- | ------------------------------------------ |
|
|
82
|
+
| `schemaCheck(zodSchema)` | Output matches a Zod schema | `schema_violation` |
|
|
83
|
+
| `requiredFields(fields)` | Object has the given keys | `missing_field` |
|
|
84
|
+
| `noExtraFields(allowed, { severity? })` | No keys outside the allowed set (default severity `error`) | `schema_violation` |
|
|
85
|
+
| `enumCheck(field, values)` | Field is one of the allowed values | `invalid_enum`, `missing_field` |
|
|
86
|
+
| `fieldTypeCheck(field, type)` | Field has the runtime type | `invalid_type` |
|
|
87
|
+
| `jsonCheck()` | String output parses as JSON | `invalid_json` |
|
|
88
|
+
| `nonEmpty(field)` | Field is not blank/empty | `policy_violation`, `missing_field` |
|
|
89
|
+
| `regexCheck(field, pattern)` | String field matches a regex | `policy_violation`, `invalid_type` |
|
|
90
|
+
| `rangeCheck(field, { min?, max? })` | Number within bounds | `policy_violation`, `invalid_type` |
|
|
91
|
+
| `toolCallCheck(definition)` | Tool call matches a contract | `tool_call_mismatch`, `missing_field`, ... |
|
|
92
|
+
| `toolCallCheck.oneOf(definitions)` | Tool call matches one of several allowed tools | `tool_call_mismatch`, ... |
|
|
93
|
+
| `semanticCheck(judge, options)` | Optional LLM-judged score (advisory by default) | `unsupported_claim` |
|
|
94
|
+
| `customCheck(name, fn)` | Your own sync or async logic | anything |
|
|
95
|
+
|
|
96
|
+
### Tool-call validation
|
|
97
|
+
|
|
98
|
+
Validate a model's tool call before executing it. A call is `{ name, args }`, or `{ name, arguments: "<json string>" }`
|
|
99
|
+
as returned by OpenAI-style APIs.
|
|
100
|
+
|
|
101
|
+
```ts
|
|
102
|
+
import { evaluate, toolCallCheck } from "@hamza1331/aieval";
|
|
103
|
+
|
|
104
|
+
const result = await evaluate({
|
|
105
|
+
output: { name: "get_weather", args: { city: "London", unit: "celsius" } },
|
|
106
|
+
checks: [
|
|
107
|
+
toolCallCheck({
|
|
108
|
+
name: "get_weather",
|
|
109
|
+
requiredArgs: ["city"],
|
|
110
|
+
allowedArgs: ["city", "unit"],
|
|
111
|
+
argTypes: { city: "string", unit: "string" },
|
|
112
|
+
}),
|
|
113
|
+
],
|
|
114
|
+
});
|
|
115
|
+
```
|
|
116
|
+
|
|
117
|
+
Unknown arguments are errors unless you set `allowExtraArgs: true`. To allow several tools and reject everything
|
|
118
|
+
else, use `toolCallCheck.oneOf([...definitions])`.
|
|
119
|
+
|
|
120
|
+
### Custom checks
|
|
121
|
+
|
|
122
|
+
```ts
|
|
123
|
+
import { customCheck } from "@hamza1331/aieval";
|
|
124
|
+
|
|
125
|
+
const evenScore = customCheck<{ score: number }>(
|
|
126
|
+
"evenScore",
|
|
127
|
+
async (value) => ({
|
|
128
|
+
passed: value.score % 2 === 0,
|
|
129
|
+
failures:
|
|
130
|
+
value.score % 2 === 0
|
|
131
|
+
? []
|
|
132
|
+
: [
|
|
133
|
+
{
|
|
134
|
+
code: "policy_violation",
|
|
135
|
+
message: "score must be even",
|
|
136
|
+
severity: "error",
|
|
137
|
+
path: "score",
|
|
138
|
+
},
|
|
139
|
+
],
|
|
140
|
+
warnings: [],
|
|
141
|
+
}),
|
|
142
|
+
);
|
|
143
|
+
```
|
|
144
|
+
|
|
145
|
+
## Optional: semantic judging
|
|
146
|
+
|
|
147
|
+
Deterministic checks cover what code can verify. For subjective quality you can plug in an LLM judge — a plain
|
|
148
|
+
function you provide, so `aieval` has no provider SDK or network dependency.
|
|
149
|
+
|
|
150
|
+
```ts
|
|
151
|
+
import { evaluate, semanticCheck, type SemanticJudge } from "@hamza1331/aieval";
|
|
152
|
+
|
|
153
|
+
const judge: SemanticJudge = async ({ input, output, criteria }) => {
|
|
154
|
+
// call your model here and return { score: 0..1, reasoning? }
|
|
155
|
+
return { score: 0.9, reasoning: "Grounded in the context." };
|
|
156
|
+
};
|
|
157
|
+
|
|
158
|
+
const result = await evaluate({
|
|
159
|
+
input: question,
|
|
160
|
+
output: answer,
|
|
161
|
+
checks: [
|
|
162
|
+
semanticCheck(judge, {
|
|
163
|
+
criteria: "The answer is grounded in the context",
|
|
164
|
+
threshold: 0.7,
|
|
165
|
+
}),
|
|
166
|
+
],
|
|
167
|
+
});
|
|
168
|
+
```
|
|
169
|
+
|
|
170
|
+
Semantic scores are **advisory by default**: a low score is a `warning`, not a failure. Raise `severity` or use
|
|
171
|
+
`failOn: "warning"` to enforce them. Scores appear in `result.metadata.checks`. Use `mockJudge` in tests. See
|
|
172
|
+
[`examples/openai-judge.ts`](examples/openai-judge.ts) for a real provider via `fetch`.
|
|
173
|
+
|
|
174
|
+
## CLI
|
|
175
|
+
|
|
176
|
+
```bash
|
|
177
|
+
# Run fixture files or directories of them
|
|
178
|
+
aieval run examples/fixtures
|
|
179
|
+
aieval run fixtures/ --fail-on warning --format json
|
|
180
|
+
|
|
181
|
+
# Validate one output (file or stdin) against a list of check specs
|
|
182
|
+
echo '{"name":"Ada"}' | aieval check --checks checks.json
|
|
183
|
+
```
|
|
184
|
+
|
|
185
|
+
A fixture is JSON (a single object or an array):
|
|
186
|
+
|
|
187
|
+
```json
|
|
188
|
+
{
|
|
189
|
+
"name": "model invented a priority",
|
|
190
|
+
"output": { "title": "Login fails", "priority": "urgent" },
|
|
191
|
+
"checks": [
|
|
192
|
+
{ "type": "enum", "field": "priority", "values": ["low", "medium", "high"] }
|
|
193
|
+
],
|
|
194
|
+
"expected": { "passed": false, "codes": ["invalid_enum"] }
|
|
195
|
+
}
|
|
196
|
+
```
|
|
197
|
+
|
|
198
|
+
- With `expected`, the case is a **regression test**: the result must match. Without it, the case passes if validation
|
|
199
|
+
passes.
|
|
200
|
+
- Check types: `requiredFields`, `noExtraFields`, `enum`, `fieldType`, `toolCall`, `toolCallOneOf`, `json`,
|
|
201
|
+
`nonEmpty`, `regex`, `range`. (`schemaCheck` and `semanticCheck` are code-only.)
|
|
202
|
+
- Exit codes: `0` all passed, `1` at least one case failed, `2` usage or input error.
|
|
203
|
+
|
|
204
|
+
The same machinery is available programmatically: `buildChecks`, `loadFixtures`, `runFixture`, `runAll`.
|
|
205
|
+
|
|
206
|
+
## Design principles
|
|
207
|
+
|
|
208
|
+
- Fail fast on what you can check deterministically; keep LLM evaluation optional and separate.
|
|
209
|
+
- Make failures explainable and machine-readable.
|
|
210
|
+
- Stay small: no dashboard, hosted service, agent framework or dataset platform.
|
|
211
|
+
|
|
212
|
+
See [`PROJECT_DISCOVERY.md`](PROJECT_DISCOVERY.md) for the research and rationale.
|
|
213
|
+
|
|
214
|
+
## Roadmap
|
|
215
|
+
|
|
216
|
+
- Repair and retry strategy helpers
|
|
217
|
+
- Adapter packages for OpenAI and LangChain (only if there is demand)
|
|
218
|
+
- Trace metadata export (OpenTelemetry / Langfuse)
|
|
219
|
+
|
|
220
|
+
## Contributing
|
|
221
|
+
|
|
222
|
+
See [CONTRIBUTING.md](CONTRIBUTING.md).
|
|
223
|
+
|
|
224
|
+
## License
|
|
225
|
+
|
|
226
|
+
MIT
|
package/dist/checks.d.ts
ADDED
|
@@ -0,0 +1,39 @@
|
|
|
1
|
+
import { z } from "zod";
|
|
2
|
+
import type { Check, CheckContext, CheckResult, Failure, Severity } from "./types.js";
|
|
3
|
+
export declare const makeFailure: (code: Failure["code"], message: string, severity?: Severity, options?: Partial<Pick<Failure, "path" | "metadata" | "suggestedRepair">>) => Failure;
|
|
4
|
+
export declare const requiredFields: (fields: string[]) => Check<Record<string, unknown>>;
|
|
5
|
+
export declare const noExtraFields: (allowedFields: string[], options?: {
|
|
6
|
+
severity?: Severity;
|
|
7
|
+
}) => Check<Record<string, unknown>>;
|
|
8
|
+
export declare const enumCheck: <T extends string | number>(field: string, allowedValues: readonly T[]) => Check<Record<string, unknown>>;
|
|
9
|
+
export declare const fieldTypeCheck: (field: string, expectedType: "string" | "number" | "boolean" | "object" | "array") => Check<Record<string, unknown>>;
|
|
10
|
+
export declare const schemaCheck: <T>(schema: z.ZodType<T>) => Check<T>;
|
|
11
|
+
type ValueType = "string" | "number" | "boolean" | "object" | "array";
|
|
12
|
+
export interface ToolCallDefinition {
|
|
13
|
+
name: string;
|
|
14
|
+
requiredArgs?: string[];
|
|
15
|
+
allowedArgs?: string[];
|
|
16
|
+
argTypes?: Record<string, ValueType>;
|
|
17
|
+
/** When true, arguments not listed in allowedArgs/argTypes are tolerated. Defaults to false. */
|
|
18
|
+
allowExtraArgs?: boolean;
|
|
19
|
+
}
|
|
20
|
+
/**
|
|
21
|
+
* Validates a tool call shaped `{ name, args }` (or `{ name, arguments: "<json>" }`).
|
|
22
|
+
*/
|
|
23
|
+
export declare const toolCallCheck: {
|
|
24
|
+
<T = unknown>(definition: ToolCallDefinition): Check<T>;
|
|
25
|
+
oneOf<T = unknown>(definitions: ToolCallDefinition[]): Check<T>;
|
|
26
|
+
};
|
|
27
|
+
/**
|
|
28
|
+
* Parses a JSON string output and emits `invalid_json` on failure.
|
|
29
|
+
* Use to guard string outputs before running object-level checks.
|
|
30
|
+
*/
|
|
31
|
+
export declare const jsonCheck: () => Check<unknown>;
|
|
32
|
+
export declare const nonEmpty: (field: string) => Check<Record<string, unknown>>;
|
|
33
|
+
export declare const regexCheck: (field: string, pattern: RegExp) => Check<Record<string, unknown>>;
|
|
34
|
+
export declare const rangeCheck: (field: string, bounds: {
|
|
35
|
+
min?: number;
|
|
36
|
+
max?: number;
|
|
37
|
+
}) => Check<Record<string, unknown>>;
|
|
38
|
+
export declare const customCheck: <T>(name: string, fn: (value: T, context: CheckContext) => CheckResult | Promise<CheckResult>) => Check<T>;
|
|
39
|
+
export {};
|